{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T16:38:54Z","timestamp":1777567134516,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","funder":[{"name":"State Key Laboratory of General Artificial Intelligence"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755031","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:47:42Z","timestamp":1761371262000},"page":"9638-9647","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Behave Your Motion: Habit-preserved Cross-category Animal Motion Transfer"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7148-1129","authenticated-orcid":false,"given":"Zhimin","family":"Zhang","sequence":"first","affiliation":[{"name":"Wangxuan Institute of Computer Technology, Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4707-0099","authenticated-orcid":false,"given":"Bi'an","family":"Du","sequence":"additional","affiliation":[{"name":"Wangxuan Institute of Computer Technology, Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2963-6244","authenticated-orcid":false,"given":"Caoyuan","family":"Ma","sequence":"additional","affiliation":[{"name":"School of Computer Science, Wuhan University, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3846-9157","authenticated-orcid":false,"given":"Zheng","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computer Science, Wuhan University, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9860-0922","authenticated-orcid":false,"given":"Wei","family":"Hu","sequence":"additional","affiliation":[{"name":"Wangxuan Institute of Computer Technology, Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3386569.3392462"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3386569.3392469"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592097"},{"key":"e_1_3_2_1_4_1","volume-title":"Automatic rigging and animation of 3d characters. ACM Transactions on graphics (TOG)","author":"Baran Ilya","year":"2007","unstructured":"Ilya Baran and Jovan Popovi\u0107. 2007. Automatic rigging and animation of 3d characters. ACM Transactions on graphics (TOG), Vol. 26, 3 (2007), 72-es."},{"key":"e_1_3_2_1_5_1","first-page":"3","volume-title":"Perth","author":"Biggs Benjamin","year":"2019","unstructured":"Benjamin Biggs, Thomas Roddick, Andrew Fitzgibbon, and Roberto Cipolla. 2019. Creatures great and smal: Recovering the shape and motion of animals from video. In Computer Vision-ACCV 2018: 14th Asian Conference on Computer Vision, Perth, Australia, December 2-6, 2018, Revised Selected Papers, Part V 14. Springer, 3-19."},{"key":"e_1_3_2_1_6_1","volume-title":"Variational lossy autoencoder. arXiv preprint arXiv:1611.02731","author":"Chen Xi","year":"2016","unstructured":"Xi Chen, Diederik P Kingma, Tim Salimans, Yan Duan, Prafulla Dhariwal, John Schulman, Ilya Sutskever, and Pieter Abbeel. 2016. Variational lossy autoencoder. arXiv preprint arXiv:1611.02731 (2016)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00702"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00941"},{"key":"e_1_3_2_1_9_1","volume-title":"Density estimation using real nvp. arXiv preprint arXiv:1605.08803","author":"Dinh Laurent","year":"2016","unstructured":"Laurent Dinh, Jascha Sohl-Dickstein, and Samy Bengio. 2016. Density estimation using real nvp. arXiv preprint arXiv:1605.08803 (2016)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01970"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/280814.280820"},{"key":"e_1_3_2_1_12_1","volume-title":"Joseph Liu, Yizhak Ben-Shabat, Victor Zordan, and Mubbasir Kapadia.","author":"Guo Ziyu","year":"2025","unstructured":"Ziyu Guo, Young Yoon Lee, Joseph Liu, Yizhak Ben-Shabat, Victor Zordan, and Mubbasir Kapadia. 2025. StyleMotif: Multi-Modal Motion Stylization using Style-Content Cross Fusion. arXiv preprint arXiv:2503.21775 (2025)."},{"key":"e_1_3_2_1_13_1","volume-title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems","author":"Heusel Martin","year":"2017","unstructured":"Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_14_1","volume-title":"Denoising diffusion probabilistic models. Advances in neural information processing systems","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems, Vol. 33 (2020), 6840-6851."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2897824.2925975"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01607"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3516429"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00168"},{"key":"e_1_3_2_1_19_1","volume-title":"Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114","author":"Kingma Diederik P","year":"2013","unstructured":"Diederik P Kingma. 2013. Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00838"},{"key":"e_1_3_2_1_21_1","volume-title":"Dancing to music. Advances in neural information processing systems","author":"Lee Hsin-Ying","year":"2019","unstructured":"Hsin-Ying Lee, Xiaodong Yang, Ming-Yu Liu, Ting-Chun Wang, Yu-Ding Lu, Ming-Hsuan Yang, and Jan Kautz. 2019. Dancing to music. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/311535.311539"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i2.20014"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cag.2022.04.001"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01247"},{"key":"e_1_3_2_1_26_1","volume-title":"Laurence Tianruo Yang, and Weihao Yuan","author":"Li Zhe","year":"2024","unstructured":"Zhe Li, Yisheng He, Lei Zhong, Weichao Shen, Qi Zuo, Lingteng Qiu, Zilong Dong, Laurence Tianruo Yang, and Weihao Yuan. 2024. Mulsmo: Multimodal stylized motion generation by bidirectional control flow. arXiv preprint arXiv:2412.09901 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"Revisiting classifier two-sample tests. arXiv preprint arXiv:1610.06545","author":"Lopez-Paz David","year":"2016","unstructured":"David Lopez-Paz and Maxime Oquab. 2016. Revisiting classifier two-sample tests. arXiv preprint arXiv:1610.06545 (2016)."},{"key":"e_1_3_2_1_28_1","volume-title":"cGANs with projection discriminator. arXiv preprint arXiv:1802.05637","author":"Miyato Takeru","year":"2018","unstructured":"Takeru Miyato and Masanori Koyama. 2018. cGANs with projection discriminator. arXiv preprint arXiv:1802.05637 (2018)."},{"key":"e_1_3_2_1_29_1","volume-title":"https:\/\/openai.com\/index\/hello-gpt-4o\/ Accessed on","author":"AI.","year":"2024","unstructured":"OpenAI. 2024. GPT-4o. https:\/\/openai.com\/index\/hello-gpt-4o\/ Accessed on October 25, 2024."},{"key":"e_1_3_2_1_30_1","unstructured":"Alec Radford Karthik Narasimhan Tim Salimans Ilya Sutskever et al. 2018. Improving language understanding by generative pre-training. (2018)."},{"key":"e_1_3_2_1_31_1","volume-title":"Aaron Van den Oord, and Oriol Vinyals","author":"Razavi Ali","year":"2019","unstructured":"Ali Razavi, Aaron Van den Oord, and Oriol Vinyals. 2019. Generating diverse high-fidelity images with vq-vae-2. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01077"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00084"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/1037957.1037963"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20047-2_21"},{"key":"e_1_3_2_1_36_1","unstructured":"Aaron Van Den Oord Oriol Vinyals et al. 2017. Neural discrete representation learning. Advances in neural information processing systems Vol. 30 (2017)."},{"key":"e_1_3_2_1_37_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_38_1","volume-title":"Leader and Follower: Interactive Motion Generation under Trajectory Constraints. arXiv preprint arXiv:2502.11563","author":"Wang Runqi","year":"2025","unstructured":"Runqi Wang, Caoyuan Ma, Jian Zhao, Hanrui Xu, Dongfang Sun, Haoyang Chen, Lin Xiong, Zheng Wang, and Xuelong Li. 2025. Leader and Follower: Interactive Motion Generation under Trajectory Constraints. arXiv preprint arXiv:2502.11563 (2025)."},{"key":"e_1_3_2_1_39_1","first-page":"14959","article-title":"Humanise: Language-conditioned human motion generation in 3d scenes","volume":"35","author":"Wang Zan","year":"2022","unstructured":"Zan Wang, Yixin Chen, Tengyu Liu, Yixin Zhu, Wei Liang, and Siyuan Huang. 2022. Humanise: Language-conditioned human motion generation in 3d scenes. Advances in Neural Information Processing Systems, Vol. 35 (2022), 14959-14971.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01340"},{"key":"e_1_3_2_1_41_1","unstructured":"Shitao Xiao Zheng Liu Peitian Zhang and Niklas Muennighoff. 2023a. C-Pack: Packaged Resources To Advance General Chinese Embedding. arXiv:2309.07597 [cs.CL]"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"crossref","unstructured":"Shitao Xiao Zheng Liu Peitian Zhang and Xingrun Xing. 2023b. LM-Cocktail: Resilient Tuning of Language Models via Model Merging. arXiv:2311.13534 [cs.CL]","DOI":"10.18653\/v1\/2024.findings-acl.145"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00125"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00053"},{"key":"e_1_3_2_1_45_1","unstructured":"Hao Zhang Di Chang Fang Li Mohammad Soleymani and Narendra Ahuja. 2024a. MagicPose4D: Crafting Articulated Models with Appearance and Motion Control. arXiv:2405.14017 [cs.CV]"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01332"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01415"},{"key":"e_1_3_2_1_48_1","unstructured":"Peitian Zhang Shitao Xiao Zheng Liu Zhicheng Dou and Jian-Yun Nie. 2023b. Retrieve Anything To Augment Large Language Models. arXiv:2310.07554 [cs.IR]"},{"key":"e_1_3_2_1_49_1","volume-title":"Motion Avatar: Generate Human and Animal Avatars with Arbitrary Motion. arXiv preprint arXiv:2405.11286","author":"Zhang Zeyu","year":"2024","unstructured":"Zeyu Zhang, Yiran Wang, Biao Wu, Shuo Chen, Zhiyuan Zhang, Shiya Huang, Wenbo Zhang, Meng Fang, Ling Chen, and Yang Zhao. 2024b. Motion Avatar: Generate Human and Animal Avatars with Arbitrary Motion. arXiv preprint arXiv:2405.11286 (2024)."},{"key":"e_1_3_2_1_50_1","volume-title":"European Conference on Computer Vision. Springer, 405-421","author":"Zhong Lei","year":"2024","unstructured":"Lei Zhong, Yiming Xie, Varun Jampani, Deqing Sun, and Huaizu Jiang. 2024. Smoodi: Stylized motion diffusion model. In European Conference on Computer Vision. Springer, 405-421."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00545"},{"key":"e_1_3_2_1_52_1","volume-title":"Human motion generation: A survey","author":"Zhu Wentao","year":"2023","unstructured":"Wentao Zhu, Xiaoxuan Ma, Dongwoo Ro, Hai Ci, Jinlu Zhang, Jiaxin Shi, Feng Gao, Qi Tian, and Yizhou Wang. 2023. Human motion generation: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence (2023)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755031","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:16:35Z","timestamp":1765307795000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755031"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":52,"alternative-id":["10.1145\/3746027.3755031","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755031","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}