{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T19:05:40Z","timestamp":1784228740403,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":111,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62421003"],"award-info":[{"award-number":["62421003"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,19]]},"DOI":"10.1145\/3799902.3811066","type":"proceedings-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:15:27Z","timestamp":1784218527000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["EchoAvatar: Real-time Generative Avatar Animation from Audio Streams"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-1036-7737","authenticated-orcid":false,"given":"Bohong","family":"Chen","sequence":"first","affiliation":[{"name":"State Key Lab of CAD&amp;CG, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6558-4165","authenticated-orcid":false,"given":"Yumeng","family":"Li","sequence":"additional","affiliation":[{"name":"State Key Lab of CAD&amp;CG, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9804-3872","authenticated-orcid":false,"given":"Yinglin","family":"Xu","sequence":"additional","affiliation":[{"name":"State Key Lab of CAD&amp;CG, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9120-9592","authenticated-orcid":false,"given":"Youyi","family":"Zheng","sequence":"additional","affiliation":[{"name":"State Key Lab of CAD&amp;CG, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7043-4410","authenticated-orcid":false,"given":"Yanlin","family":"Weng","sequence":"additional","affiliation":[{"name":"State Key Lab of CAD&amp;CG, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4243-6112","authenticated-orcid":false,"given":"Kun","family":"Zhou","sequence":"additional","affiliation":[{"name":"State Key Lab of CAD&amp;CG, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"crossref","unstructured":"Kfir Aberman Yijia Weng Dani Lischinski Daniel Cohen-Or and Baoquan Chen. 2020. Unpaired Motion Style Transfer from Video to Animation. ACM Transactions on Graphics (TOG) 39 4 (2020) 64.","DOI":"10.1145\/3386569.3392469"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"crossref","unstructured":"Simon Alexanderson Gustav\u00a0Eje Henter Taras Kucherenko and Jonas Beskow. 2020. Style-controllable speech-driven gesture synthesis using normalising flows. Computer Graphics Forum 39 2 (2020) 487\u2013496.","DOI":"10.1111\/cgf.13946"},{"key":"e_1_3_3_2_4_1","doi-asserted-by":"crossref","unstructured":"Simon Alexanderson Rajmund Nagy Jonas Beskow and Gustav\u00a0Eje Henter. 2023. Listen Denoise Action! Audio-Driven Motion Synthesis with Diffusion Models. ACM Trans. Graph. 42 4 Article 44 (July 2023) 20\u00a0pages.","DOI":"10.1145\/3592458"},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"crossref","unstructured":"Tenglong Ao Qingzhe Gao Yuke Lou Baoquan Chen and Libin Liu. 2022. Rhythmic gesticulator: Rhythm-aware co-speech gesture synthesis with hierarchical neural embeddings. ACM Transactions on Graphics (TOG) 41 6 (2022) 1\u201319.","DOI":"10.1145\/3550454.3555435"},{"key":"e_1_3_3_2_6_1","unstructured":"Tenglong Ao Zeyi Zhang and Libin Liu. 2023. GestureDiffuCLIP: Gesture Diffusion Model with CLIP Latents. ACM Trans. Graph. (2023) 18\u00a0pages."},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01030"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3757377.3763992"},{"key":"e_1_3_3_2_9_1","volume-title":"Proceedings of the 3rd Conference on Robot Learning","author":"Brown Daniel\u00a0S.","year":"2019","unstructured":"Daniel\u00a0S. Brown, Wonjoon Goo, and Scott Niekum. 2019. Better-than-Demonstrator Imitation Learning via Automatically-Ranked Demonstrations. In Proceedings of the 3rd Conference on Robot Learning."},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/192161.192272"},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/383259.383315"},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680847"},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3721238.3730611"},{"key":"e_1_3_3_2_14_1","unstructured":"Bohong Chen and Haiyang Liu. 2025. DyStream: Streaming Dyadic Talking Heads Generation via Flow Matching-based Autoregressive Model. arxiv:https:\/\/arXiv.org\/abs\/2512.24408\u00a0[cs.CV]"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00581"},{"key":"e_1_3_3_2_16_1","unstructured":"Junyu Chen Han Cai Junsong Chen Enze Xie Shang Yang Haotian Tang Muyang Li Yao Lu and Song Han. 2024a. Deep Compression Autoencoder for Efficient High-Resolution Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.10733 (2024)."},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00702"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"crossref","unstructured":"Ling-Hao Chen Shunlin Lu Ailing Zeng Hao Zhang Benyou Wang Ruimao Zhang and Lei Zhang. 2025b. Motionllm: Understanding human behaviors from human motions and videos. IEEE Transactions on Pattern Analysis and Machine Intelligence (2025).","DOI":"10.1109\/TPAMI.2025.3627546"},{"key":"e_1_3_3_2_19_1","series-title":"Proceedings of Machine Learning Research","first-page":"5178","volume-title":"Proceedings of the 40th International Conference on Machine Learning","volume":"202","author":"Chen Sanyuan","year":"2023","unstructured":"Sanyuan Chen, Yu Wu, Chengyi Wang, Shujie Liu, Daniel Tompkins, Zhuo Chen, Wanxiang Che, Xiangzhan Yu, and Furu Wei. 2023. BEATs: Audio Pre-Training with Acoustic Tokenizers. In Proceedings of the 40th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0202), Andreas Krause, Emma Brunskill, Kyunghyun Cho, Barbara Engelhardt, Sivan Sabato, and Jonathan Scarlett (Eds.). PMLR, 5178\u20135193."},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687677"},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"crossref","unstructured":"Jade Copet Felix Kreuk Itai Gat Tal Remez David Kant Gabriel Synnaeve Yossi Adi and Alexandre D\u00e9fossez. 2023. Simple and controllable music generation. Advances in Neural Information Processing Systems 36 (2023) 47704\u201347720.","DOI":"10.52202\/075280-2066"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","unstructured":"Abe Davis and Maneesh Agrawala. 2018. Visual rhythm and beat. ACM Trans. Graph. 37 4 Article 122 (July 2018) 11\u00a0pages. 10.1145\/3197517.3201371","DOI":"10.1145\/3197517.3201371"},{"key":"e_1_3_3_2_23_1","unstructured":"Alexandre D\u00e9fossez Jade Copet Gabriel Synnaeve and Yossi Adi. 2022. High fidelity neural audio compression. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2210.13438 (2022)."},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01239"},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3757377.3763951"},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"crossref","unstructured":"Saeed Ghorbani Ylva Ferstl Daniel Holden Nikolaus\u00a0F. Troje and Marc-Andr\u00e9 Carbonneau. 2023. ZeroEGGS: Zero-shot Example-based Gesture Generation from Speech. Computer Graphics Forum 42 1 (2023) 206\u2013216. arXiv:https:\/\/onlinelibrary.wiley.com\/doi\/pdf\/10.1111\/cgf.14734","DOI":"10.1111\/cgf.14734"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3721238.3730741"},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00361"},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"crossref","unstructured":"Chuan Guo Yuxuan Mu Muhammad\u00a0Gohar Javed Sen Wang and Li Cheng. 2024. MoMask: Generative Masked Modeling of 3D Human Motions. (June 2024) 1900\u20131910.","DOI":"10.1109\/CVPR52733.2024.00186"},{"key":"e_1_3_3_2_30_1","unstructured":"Tuomas Haarnoja et\u00a0al. 2018. Soft actor-critic algorithms and applications. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1812.05905 (2018)."},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3528233.3530750"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"crossref","unstructured":"Ikhsanul Habibie Weipeng Xu Dushyant Mehta Lingjie Liu Hans-Peter Seidel Gerard Pons-Moll Mohamed Elgharib and Christian Theobalt. 2021. Learning Speech-driven 3D Conversational Gestures from Video. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2102.06837 (2021).","DOI":"10.1145\/3472306.3478335"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01091"},{"key":"e_1_3_3_2_34_1","volume-title":"Motorica-Retarget","author":"Holden Daniel","year":"2024","unstructured":"Daniel Holden. 2024a. Motorica-Retarget. https:\/\/github.com\/orangeduck\/motorica-retarget"},{"key":"e_1_3_3_2_35_1","volume-title":"ZeroEGGs-Retarget","author":"Holden Daniel","year":"2024","unstructured":"Daniel Holden. 2024b. ZeroEGGs-Retarget. https:\/\/github.com\/orangeduck\/zeroeggs-retarget"},{"key":"e_1_3_3_2_36_1","unstructured":"Ruibing Hou Mingshuang Luo Hongyu Pan Hong Chang and Shiguang Shan. 2025. Motionverse: A unified multimodal framework for motion comprehension generation and editing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2509.23635 (2025)."},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"publisher","unstructured":"IEEE Subcommittee oe\u2019en Subjective Measurements. 1969. IEEE Recommended Practice for Speech Quality Measurements. IEEE Transactions on Audio and Electroacoustics 17 3 (1969) 225\u2013246. 10.1109\/TAU.1969.1162058","DOI":"10.1109\/TAU.1969.1162058"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW69036.2025.00214"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02119"},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/11821830_17"},{"key":"e_1_3_3_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3382507.3418815"},{"key":"e_1_3_3_2_42_1","volume-title":"NeurIPS","author":"Kumar Aviral","year":"2020","unstructured":"Aviral Kumar et\u00a0al. 2020. Conservative Q-Learning for Offline Reinforcement Learning. In NeurIPS."},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"crossref","unstructured":"Jina Lee and Stacy Marsella. 2006. Nonverbal Behavior Generator for Embodied Conversational Agents(IVA \u201906). Springer 243\u2013255.","DOI":"10.1007\/11821830_20"},{"key":"e_1_3_3_2_44_1","doi-asserted-by":"crossref","unstructured":"Margot Lhommet Yuyu Xu and Stacy Marsella. 2015. Cerebella: Automatic Generation of Nonverbal Behavior for Virtual Humans(AAAI \u201915 1).","DOI":"10.1609\/aaai.v29i1.9778"},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01110"},{"key":"e_1_3_3_2_46_1","unstructured":"Lei Li et\u00a0al. 2023b. Silkie: Preference distillation for large visual language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.10665 (2023)."},{"key":"e_1_3_3_2_47_1","volume-title":"CVPR","author":"Li Siyao","year":"2022","unstructured":"Siyao Li et\u00a0al. 2022. Bailando: 3D Dance Generation by Actor-Critic GPT with Choreographic Memory. In CVPR."},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"crossref","unstructured":"Weiyu Li Xuelin Chen Peizhuo Li Olga Sorkine-Hornung and Baoquan Chen. 2023a. Example-Based Motion Synthesis via Generative Motion Matching. ACM Transactions on Graphics (TOG) 42 4 Article 94 (2023).","DOI":"10.1145\/3592395"},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"publisher","unstructured":"Hung\u00a0Yu Ling Fabio Zinno George Cheng and Michiel Van De\u00a0Panne. 2020. Character controllers using motion VAEs. ACM Trans. Graph. 39 4 Article 40 (Aug. 2020) 12\u00a0pages. 10.1145\/3386569.3392422","DOI":"10.1145\/3386569.3392422"},{"key":"e_1_3_3_2_50_1","unstructured":"Binjie Liu Lina Liu Sanyi Zhang Songen Gu Yihao Zhi Tianyi Zhu Lei Yang and Long Ye. 2025b. MAG: Multi-Modal Aligned Autoregressive Co-Speech Gesture Generation without Vector Quantization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.14040 (2025)."},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548400"},{"key":"e_1_3_3_2_52_1","unstructured":"Haiyang Liu Zihao Zhu Giorgio Becherini Yichen Peng Mingyang Su You Zhou Naoya Iwamoto Bo Zheng and Michael\u00a0J. Black. 2024b. EMAGE: Towards Unified Holistic Co-Speech Gesture Generation via Masked Audio Gesture Modeling. arxiv:https:\/\/arXiv.org\/abs\/2401.00374\u00a0[cs.CV]"},{"key":"e_1_3_3_2_53_1","unstructured":"Haiyang Liu Zihao Zhu Naoya Iwamoto Yichen Peng Zhengqing Li You Zhou Elif Bozkurt and Bo Zheng. 2022d. BEAT: A Large-Scale Semantic and Emotional Multi-Modal Dataset for Conversational Gestures Synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2203.05297 (2022)."},{"key":"e_1_3_3_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01017"},{"key":"e_1_3_3_2_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3757377.3763954"},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1554"},{"key":"e_1_3_3_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01021"},{"key":"e_1_3_3_2_58_1","unstructured":"Zixuan Liu et\u00a0al. 2024a. Enhancing llm safety via constrained direct preference optimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.02475 (2024)."},{"key":"e_1_3_3_2_59_1","doi-asserted-by":"crossref","unstructured":"Jintao Lu He Zhang Yuting Ye Takaaki Shiratori Sebastian Starke and Taku Komura. 2025b. CHOICE: Coordinated human-object interaction in cluttered environments for pick-and-place actions. ACM Transactions on Graphics 45 2 (2025) 1\u201318.","DOI":"10.1145\/3770746"},{"key":"e_1_3_3_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02595"},{"key":"e_1_3_3_2_61_1","doi-asserted-by":"crossref","unstructured":"Shuhong Lu Youngwoo Yoon and Andrew\u00a0W. Feng. 2023. Co-Speech Gesture Synthesis using Discrete Gesture Token Learning. 2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS) (2023) 9808\u20139815.","DOI":"10.1109\/IROS55552.2023.10342027"},{"key":"e_1_3_3_2_62_1","unstructured":"Jacob Menick et\u00a0al. 2022. Teaching language models to support answers with verified quotes. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2203.11147 (2022)."},{"key":"e_1_3_3_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00138"},{"key":"e_1_3_3_2_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01545"},{"key":"e_1_3_3_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01975"},{"key":"e_1_3_3_2_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3757377.3763948"},{"key":"e_1_3_3_2_67_1","unstructured":"Long Ouyang et\u00a0al. 2022a. Training language models to follow instructions with human feedback. NeurIPS (2022)."},{"key":"e_1_3_3_2_68_1","doi-asserted-by":"crossref","unstructured":"Long Ouyang Jeffrey Wu Xu Jiang Diogo Almeida Carroll Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray et\u00a0al. 2022b. Training language models to follow instructions with human feedback. Advances in neural information processing systems 35 (2022) 27730\u201327744.","DOI":"10.52202\/068431-2011"},{"key":"e_1_3_3_2_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00870"},{"key":"e_1_3_3_2_70_1","unstructured":"Andr\u00e9\u00a0Susano Pinto et\u00a0al. 2023. Tuning computer vision models with task rewards. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.08242 (2023)."},{"key":"e_1_3_3_2_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3721238.3730756"},{"key":"e_1_3_3_2_72_1","unstructured":"Alec Radford Jong\u00a0Wook Kim Chris Hallacy Aditya Ramesh Gabriel Goh Sandhini Agarwal Girish Sastry Amanda Askell Pamela Mishkin Jack Clark Gretchen Krueger and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. arxiv:https:\/\/arXiv.org\/abs\/2103.00020\u00a0[cs.CV]"},{"key":"e_1_3_3_2_73_1","doi-asserted-by":"crossref","unstructured":"Rafael Rafailov et\u00a0al. 2023. Direct preference optimization: Your language model is secretly a reward model. NeurIPS (2023).","DOI":"10.52202\/075280-2338"},{"key":"e_1_3_3_2_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01155"},{"key":"e_1_3_3_2_75_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Shafir Yoni","year":"2024","unstructured":"Yoni Shafir, Guy Tevet, Roy Kapon, and Amit\u00a0Haim Bermano. 2024. Human Motion Diffusion as a Generative Prior. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_3_2_76_1","unstructured":"Zhihong Shao Peiyi Wang Qihao Zhu Runxin Xu Alan Song Mingchuan Xiao Y.\u00a0K. Li Y. Zhang Ins Zhang Y. Wang et\u00a0al. 2024. DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.03300 (2024)."},{"key":"e_1_3_3_2_77_1","unstructured":"Shuaijie She et\u00a0al. 2024. Mapo: Advancing multilingual reasoning through multilingual alignment-as-preference optimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.06838 (2024)."},{"key":"e_1_3_3_2_78_1","doi-asserted-by":"crossref","unstructured":"Min Shi Wenke Feng Lin Gao and Dengming Gao. 2024a. Generating diverse clothed 3D human animations via a generative model. Computational Visual Media 10 2 (2024) 261\u2013277.","DOI":"10.1007\/s41095-022-0324-2"},{"key":"e_1_3_3_2_79_1","doi-asserted-by":"crossref","unstructured":"Yi Shi Jingbo Wang Xuekun Jiang Bingkun Lin Bo Dai and Xue\u00a0Bin Peng. 2024b. Interactive Character Control with Auto-Regressive Motion Diffusion Models. ACM Trans. Graph. 43 (jul 2024).","DOI":"10.1145\/3658140"},{"key":"e_1_3_3_2_80_1","unstructured":"Mingyang Sun et\u00a0al. 2023. Co-speech Gesture Synthesis by Reinforcement Learning with Contrastive Pre-trained Rewards. CVPR (2023)."},{"key":"e_1_3_3_2_81_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"e_1_3_3_2_82_1","unstructured":"Richard\u00a0S Sutton David McAllester Satinder Singh and Yishay Mansour. 1999. Policy gradient methods for reinforcement learning with function approximation. NeurIPS 12 (1999)."},{"key":"e_1_3_3_2_83_1","volume-title":"The Eleventh International Conference on Learning Representations","author":"Tevet Guy","year":"2023","unstructured":"Guy Tevet, Sigal Raab, Brian Gordon, Yoni Shafir, Daniel Cohen-or, and Amit\u00a0Haim Bermano. 2023. Human Motion Diffusion Model. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_3_2_84_1","unstructured":"Aaron van\u00a0den Oord Yazhe Li and Oriol Vinyals. 2018. Representation Learning with Contrastive Predictive Coding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1807.03748 (2018)."},{"key":"e_1_3_3_2_85_1","unstructured":"Weilin Wan Zhiyang Dou Taku Komura Wenping Wang Dinesh Jayaraman and Lingjie Liu. 2023. TLControl: Trajectory and Language Control for Human Motion Synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.17135 (2023)."},{"key":"e_1_3_3_2_86_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i1.27783"},{"key":"e_1_3_3_2_87_1","doi-asserted-by":"crossref","unstructured":"Ronald\u00a0J Williams. 1992. Simple statistical gradient-following algorithms for connectionist reinforcement learning. Reinforcement learning (1992) 5\u201332.","DOI":"10.1007\/978-1-4615-3618-5_2"},{"key":"e_1_3_3_2_88_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.00940"},{"key":"e_1_3_3_2_89_1","unstructured":"Yiming Xie Varun Jampani Lei Zhong Deqing Sun and Huaizu Jiang. 2023. OmniControl: Control Any Joint at Any Time for Human Motion Generation. arxiv:https:\/\/arXiv.org\/abs\/2310.08580"},{"key":"e_1_3_3_2_90_1","volume-title":"Advances in Neural Information Processing Systems","author":"Xu Shuyang","year":"2025","unstructured":"Shuyang Xu, Zhiyang Dou, Mingyi Shi, Liang Pan, Leo Ho, Jingbo Wang, Yuan Liu, Cheng Lin, Yuexin Ma, Wenping Wang, and Taku Komura. 2025. MOSPA: Human Motion Generation Driven by Spatial Audio. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_3_2_91_1","unstructured":"An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei Huan Lin Jian Yang Jianhong Tu Jianwei Zhang Jianxin Yang Jiaxi Yang Jingren Zhou Junyang Lin Kai Dang Keming Lu Keqin Bao Kexin Yang Le Yu Mei Li Mingfeng Xue Pei Zhang Qin Zhu Rui Men Runji Lin Tianhao Li Tingyu Xia Xingzhang Ren Xuancheng Ren Yang Fan Yang Su Yichang Zhang Yu Wan Yuqiong Liu Zeyu Cui Zhenru Zhang and Zihan Qiu. 2024. Qwen2.5 Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.15115 (2024)."},{"key":"e_1_3_3_2_92_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01172"},{"key":"e_1_3_3_2_93_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612503"},{"key":"e_1_3_3_2_94_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/650"},{"key":"e_1_3_3_2_95_1","doi-asserted-by":"crossref","unstructured":"Heyuan Yao Zhenhua Song Yuyang Zhou Tenglong Ao Baoquan Chen and Libin Liu. 2024. MoConVQ: Unified Physics-Based Motion Control via Scalable Discrete Representations. ACM Trans. Graph. 43 4 Article 144 (July 2024) 21\u00a0pages.","DOI":"10.1145\/3658137"},{"key":"e_1_3_3_2_96_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9981117"},{"key":"e_1_3_3_2_97_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20065-6_41"},{"key":"e_1_3_3_2_98_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00053"},{"key":"e_1_3_3_2_99_1","doi-asserted-by":"crossref","unstructured":"Youngwoo Yoon Bok Cha Joo-Haeng Lee Minsu Jang Jaeyeon Lee Jaehong Kim and Geehyuk Lee. 2020. Speech gesture generation from the trimodal context of text audio and speaker identity. ACM Transactions on Graphics (TOG) 39 6 (2020) 1\u201316.","DOI":"10.1145\/3414685.3417838"},{"key":"e_1_3_3_2_100_1","doi-asserted-by":"crossref","unstructured":"Hongyi Yuan et\u00a0al. 2023. Rrhf: Rank responses to align language models with human feedback. NeurIPS (2023).","DOI":"10.52202\/075280-0482"},{"key":"e_1_3_3_2_101_1","doi-asserted-by":"publisher","unstructured":"Neil Zeghidour Alejandro Luebs Ahmed Omran Jan Skoglund and Marco Tagliasacchi. 2022. SoundStream: An End-to-End Neural Audio Codec. IEEE\/ACM Transactions on Audio Speech and Language Processing 30 (2022) 495\u2013507. 10.1109\/TASLP.2021.3129994","DOI":"10.1109\/TASLP.2021.3129994"},{"key":"e_1_3_3_2_102_1","unstructured":"Mingyuan Zhang Zhongang Cai Liang Pan Fangzhou Hong Xinying Guo Lei Yang and Ziwei Liu. 2022. MotionDiffuse: Text-Driven Human Motion Generation with Diffusion Model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2208.15001 (2022)."},{"key":"e_1_3_3_2_103_1","first-page":"397","volume-title":"Computer Vision \u2013 ECCV 2024: 18th European Conference, Milan, Italy, September 29\u2013October 4, 2024, Proceedings, Part XIII","author":"Zhang Mingyuan","year":"2024","unstructured":"Mingyuan Zhang, Daisheng Jin, Chenyang Gu, Fangzhou Hong, Zhongang Cai, Jingfang Huang, Chongzhi Zhang, Xinying Guo, Lei Yang, Ying He, and Ziwei Liu. 2024b. Large Motion Model for Unified Multi-modal Motion Generation. In Computer Vision \u2013 ECCV 2024: 18th European Conference, Milan, Italy, September 29\u2013October 4, 2024, Proceedings, Part XIII (Milan, Italy). Springer-Verlag, Berlin, Heidelberg, 397\u2013421."},{"key":"e_1_3_3_2_104_1","doi-asserted-by":"crossref","unstructured":"Zeyi Zhang Tenglong Ao Yuyao Zhang Qingzhe Gao Chuan Lin Baoquan Chen and Libin Liu. 2024a. Semantic Gesticulator: Semantics-Aware Co-Speech Gesture Synthesis. ACM Trans. Graph. (2024) 17\u00a0pages.","DOI":"10.1145\/3658134"},{"key":"e_1_3_3_2_105_1","unstructured":"Li Zhao and Zhengmin Lu. 2024. DanceFusion: A Spatio-Temporal Skeleton Diffusion Transformer for Audio-Driven Dance Motion Reconstruction. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.04646 (2024)."},{"key":"e_1_3_3_2_106_1","unstructured":"Zhiyuan Zhao et\u00a0al. 2023. Beyond hallucinations: Enhancing lvlms through hallucination-aware direct preference optimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.16839 (2023)."},{"key":"e_1_3_3_2_107_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01963"},{"key":"e_1_3_3_2_108_1","doi-asserted-by":"publisher","DOI":"10.1145\/3721238.3730664"},{"key":"e_1_3_3_2_109_1","doi-asserted-by":"publisher","DOI":"10.1145\/3536221.3558063"},{"key":"e_1_3_3_2_110_1","unstructured":"Yiyang Zhou et\u00a0al. 2024. Aligning modalities in vision large language models via preference fine-tuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.11411 (2024)."},{"key":"e_1_3_3_2_111_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00589"},{"key":"e_1_3_3_2_112_1","volume-title":"International Conference on Learning Representations","author":"Zhou Yi","year":"2018","unstructured":"Yi Zhou, Zimo Li, Shuangjiu Xiao, Chong He, Zeng Huang, and Hao Li. 2018. Auto-Conditioned Recurrent Networks for Extended Complex Human Motion Synthesis. In International Conference on Learning Representations."}],"event":{"name":"SIGGRAPH Conference Papers '26: Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers","location":"Los Angeles CA USA","acronym":"SIGGRAPH Conference Papers '26","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T18:19:08Z","timestamp":1784225948000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3799902.3811066"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":111,"alternative-id":["10.1145\/3799902.3811066","10.1145\/3799902"],"URL":"https:\/\/doi.org\/10.1145\/3799902.3811066","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}