{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:09:13Z","timestamp":1784268553008,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680625","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:27Z","timestamp":1729925967000},"page":"10794-10803","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["MambaGesture: Enhancing Co-Speech Gesture Generation with Mamba and Disentangled Multi-Modality Fusion"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-3732-3035","authenticated-orcid":false,"given":"Chencan","family":"Fu","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6592-8411","authenticated-orcid":false,"given":"Yabiao","family":"Wang","sequence":"additional","affiliation":[{"name":"Zhejiang University &amp; Tencent Youtu Lab, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8891-6766","authenticated-orcid":false,"given":"Jiangning","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tencent Youtu Lab, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4064-994X","authenticated-orcid":false,"given":"Zhengkai","family":"Jiang","sequence":"additional","affiliation":[{"name":"Tencent Youtu Lab, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2666-753X","authenticated-orcid":false,"given":"Xiaofeng","family":"Mao","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1036-5076","authenticated-orcid":false,"given":"Jiafu","family":"Wu","sequence":"additional","affiliation":[{"name":"Tencent Youtu Lab, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2554-1294","authenticated-orcid":false,"given":"Weijian","family":"Cao","sequence":"additional","affiliation":[{"name":"Tencent Youtu Lab, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4216-8090","authenticated-orcid":false,"given":"Chengjie","family":"Wang","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University &amp; Tencent Youtu Lab, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5650-5118","authenticated-orcid":false,"given":"Yanhao","family":"Ge","sequence":"additional","affiliation":[{"name":"VIVO, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4822-8939","authenticated-orcid":false,"given":"Yong","family":"Liu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"denoise, action! audio-driven motion synthesis with diffusion models. ACM Transactions on Graphics (TOG)","author":"Alexanderson Simon","year":"2023","unstructured":"Simon Alexanderson, Rajmund Nagy, Jonas Beskow, and Gustav Eje Henter. 2023. Listen, denoise, action! audio-driven motion synthesis with diffusion models. ACM Transactions on Graphics (TOG) (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"Gesturediffuclip: Gesture diffusion model with clip latents. ACM Transactions on Graphics (TOG)","author":"Ao Tenglong","year":"2023","unstructured":"Tenglong Ao, Zeyi Zhang, and Libin Liu. 2023. Gesturediffuclip: Gesture diffusion model with clip latents. ACM Transactions on Graphics (TOG) (2023)."},{"key":"e_1_3_2_1_3_1","unstructured":"Kirsten Bergmann and Stefan Kopp. 2009. Increasing the expressiveness of virtual agents: autonomous generation of speech and gesture for spatial description tasks.. In AAMAS (1)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/192161.192272"},{"key":"e_1_3_2_1_5_1","volume-title":"Diffsheg: A diffusion-based approach for real-time speech-driven holistic 3d expression and gesture generation.","author":"Chen Junming","year":"2024","unstructured":"Junming Chen, Yunfei Liu, Jianan Wang, Ailing Zeng, Yu Li, and Qifeng Chen. 2024. Diffsheg: A diffusion-based approach for real-time speech-driven holistic 3d expression and gesture generation. (2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"WavLM: Large-Scale Self-Supervised Pre-training for Full Stack Speech Processing. arXiv preprint arXiv:2110.13900","author":"Chen Sanyuan","year":"2021","unstructured":"Sanyuan Chen, Chengyi Wang, Zhengyang Chen, Yu Wu, Shujie Liu, Zhuo Chen, Jinyu Li, Naoyuki Kanda, Takuya Yoshioka, Xiong Xiao, Jian Wu, Long Zhou, Shuo Ren, Yanmin Qian, Yao Qian, Jian Wu, Michael Zeng, and Furu Wei. 2021. WavLM: Large-Scale Self-Supervised Pre-training for Full Stack Speech Processing. arXiv preprint arXiv:2110.13900 (2021)."},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Chhatre Kiran","year":"2024","unstructured":"Kiran Chhatre, Nikos Athanasiou, Giorgio Becherini, Christopher Peters, Michael J. Black, and Timo Bolkart. 2024. AMUSE: Emotional Speech-driven 3D Body Animation via Disentangled Latent Diffusion. In Proceedings IEEE Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-23974-8_14"},{"key":"e_1_3_2_1_9_1","volume-title":"Hungry hungry hippos: Towards language modeling with state space models. arXiv preprint arXiv:2212.14052","author":"Fu Daniel Y","year":"2022","unstructured":"Daniel Y Fu, Tri Dao, Khaled K Saab, Armin W Thomas, Atri Rudra, and Christopher R\u00e9. 2022. Hungry hungry hippos: Towards language modeling with state space models. arXiv preprint arXiv:2212.14052 (2022)."},{"key":"e_1_3_2_1_10_1","volume-title":"Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752","author":"Gu Albert","year":"2023","unstructured":"Albert Gu and Tri Dao. 2023. Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752 (2023)."},{"key":"e_1_3_2_1_11_1","volume-title":"Efficiently Modeling Long Sequences with Structured State Spaces. In The International Conference on Learning Representations (ICLR).","author":"Gu Albert","year":"2022","unstructured":"Albert Gu, Karan Goel, and Christopher R\u00e9. 2022. Efficiently Modeling Long Sequences with Structured State Spaces. In The International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_12_1","volume-title":"Combining recurrent, convolutional, and continuous-time models with linear state space layers. Advances in neural information processing systems","author":"Gu Albert","year":"2021","unstructured":"Albert Gu, Isys Johnson, Karan Goel, Khaled Saab, Tri Dao, Atri Rudra, and Christopher R\u00e9. 2021. Combining recurrent, convolutional, and continuous-time models with linear state space layers. Advances in neural information processing systems (2021)."},{"key":"e_1_3_2_1_13_1","volume-title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems","author":"Heusel Martin","year":"2017","unstructured":"Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems (2017)."},{"key":"e_1_3_2_1_14_1","volume-title":"Denoising diffusion probabilistic models. Advances in neural information processing systems","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems (2020)."},{"key":"e_1_3_2_1_15_1","volume-title":"International conference on machine learning.","author":"Katharopoulos Angelos","year":"2020","unstructured":"Angelos Katharopoulos, Apoorv Vyas, Nikolaos Pappas, and Franccois Fleuret. 2020. Transformers are rnns: Fast autoregressive transformers with linear attention. In International conference on machine learning."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i7.25996"},{"key":"e_1_3_2_1_17_1","unstructured":"Michael Kipp. 2005. Gesture generation by imitation: From human behavior to computer character animation."},{"key":"e_1_3_2_1_18_1","volume-title":"DiffWave: A Versatile Diffusion Model for Audio Synthesis. In International Conference on Learning Representations.","author":"Kong Zhifeng","year":"2020","unstructured":"Zhifeng Kong, Wei Ping, Jiaji Huang, Kexin Zhao, and Bryan Catanzaro. 2020. DiffWave: A Versatile Diffusion Model for Audio Synthesis. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/11821830_17"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CA.2002.1017547"},{"key":"e_1_3_2_1_21_1","volume-title":"Dancing to music. Advances in neural information processing systems","author":"Lee Hsin-Ying","year":"2019","unstructured":"Hsin-Ying Lee, Xiaodong Yang, Ming-Yu Liu, Ting-Chun Wang, Yu-Ding Lu, Ming-Hsuan Yang, and Jan Kautz. 2019. Dancing to music. Advances in neural information processing systems (2019)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Sergey Levine Christian Theobalt and Vladlen Koltun. 2009. Real-time prosody-driven synthesis of body language. In ACM SIGGRAPH Asia 2009 papers.","DOI":"10.1145\/1661412.1618518"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01110"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01315"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548400"},{"key":"e_1_3_2_1_26_1","volume":"202","author":"Liu Haiyang","unstructured":"Haiyang Liu, Zihao Zhu, Giorgio Becherini, Yichen Peng, Mingyang Su, You Zhou, Naoya Iwamoto, Bo Zheng, and Michael J Black. 2023. EMAGE: Towards Unified Holistic Co-Speech Gesture Generation via Masked Audio Gesture Modeling. arXiv preprint arXiv:2401.00374 (2023).","journal-title":"Michael J Black."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20071-7_36"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01021"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/2485895.2485900"},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the International Conference on Language Resources and Evaluation (LREC","author":"Mikolov Tomas","year":"2018","unstructured":"Tomas Mikolov, Edouard Grave, Piotr Bojanowski, Christian Puhrsch, and Armand Joulin. 2018. Advances in Pre-Training Distributed Word Representations. In Proceedings of the International Conference on Language Resources and Evaluation (LREC 2018)."},{"key":"e_1_3_2_1_31_1","volume-title":"Gesture modeling and animation based on a probabilistic re-creation of speaker style. ACM Transactions On Graphics (TOG)","author":"Neff Michael","year":"2008","unstructured":"Michael Neff, Michael Kipp, Irene Albrecht, and Hans-Peter Seidel. 2008. Gesture modeling and animation based on a probabilistic re-creation of speaker style. ACM Transactions On Graphics (TOG) (2008)."},{"key":"e_1_3_2_1_32_1","volume-title":"Gustav Eje Henter, and Michael Neff","author":"Nyatsanga Simbarashe","year":"2023","unstructured":"Simbarashe Nyatsanga, Taras Kucherenko, Chaitanya Ahuja, Gustav Eje Henter, and Michael Neff. 2023. A Comprehensive Review of Data-Driven Co-Speech Gesture Generation. In Computer Graphics Forum."},{"key":"e_1_3_2_1_33_1","volume-title":"International Conference on Machine Learning.","author":"Poli Michael","year":"2023","unstructured":"Michael Poli, Stefano Massaroli, Eric Nguyen, Daniel Y Fu, Tri Dao, Stephen Baccus, Yoshua Bengio, Stefano Ermon, and Christopher R\u00e9. 2023. Hyena hierarchy: Towards larger convolutional language models. In International Conference on Machine Learning."},{"key":"e_1_3_2_1_34_1","volume-title":"Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125","author":"Ramesh Aditya","year":"2022","unstructured":"Aditya Ramesh, Prafulla Dhariwal, Alex Nichol, Casey Chu, and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 (2022)."},{"key":"e_1_3_2_1_35_1","volume-title":"Automating the production of communicative gestures in embodied characters. Frontiers in psychology","author":"Ravenet Brian","year":"2018","unstructured":"Brian Ravenet, Catherine Pelachaud, Chlo\u00e9 Clavel, and Stacy Marsella. 2018. Automating the production of communicative gestures in embodied characters. Frontiers in psychology (2018)."},{"key":"e_1_3_2_1_36_1","volume-title":"Efficient content-based sparse attention with routing transformers. Transactions of the Association for Computational Linguistics","author":"Roy Aurko","year":"2021","unstructured":"Aurko Roy, Mohammad Saffar, Ashish Vaswani, and David Grangier. 2021. Efficient content-based sparse attention with routing transformers. Transactions of the Association for Computational Linguistics (2021)."},{"key":"e_1_3_2_1_37_1","volume-title":"Retentive network: A successor to transformer for large language models. arXiv preprint arXiv:2307.08621","author":"Sun Yutao","year":"2023","unstructured":"Yutao Sun, Li Dong, Shaohan Huang, Shuming Ma, Yuqing Xia, Jilong Xue, Jianyong Wang, and Furu Wei. 2023. Retentive network: A successor to transformer for large language models. arXiv preprint arXiv:2307.08621 (2023)."},{"key":"e_1_3_2_1_38_1","volume-title":"Human Motion Diffusion Model. In The Eleventh International Conference on Learning Representations.","author":"Tevet Guy","year":"2022","unstructured":"Guy Tevet, Sigal Raab, Brian Gordon, Yoni Shafir, Daniel Cohen-or, and Amit Haim Bermano. 2022. Human Motion Diffusion Model. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_39_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems (2017)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-74997-4_10"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Sen Wang Jiangning Zhang Weijian Cao Xiaobin Hu Moran Li Xiaozhong Ji Xin Tan Mengtian Li Zhifeng Xie Chengjie Wang et al. 2024. MMoFusion: Multi-modal Co-Speech Motion Generation with Diffusion Model. arXiv preprint arXiv:2403.02905 (2024).","DOI":"10.2139\/ssrn.4965116"},{"key":"e_1_3_2_1_42_1","volume-title":"MambaTalk: Efficient Holistic Gesture Synthesis with Selective State Space Models. arXiv preprint arXiv:2403.09471","author":"Xu Zunnan","year":"2024","unstructured":"Zunnan Xu, Yukang Lin, Haonan Han, Sicheng Yang, Ronghui Li, Yachao Zhang, and Xiu Li. 2024. MambaTalk: Efficient Holistic Gesture Synthesis with Selective State Space Models. arXiv preprint arXiv:2403.09471 (2024)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612503"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/650"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Sicheng Yang Zunnan Xu Haiwei Xue Yongkang Cheng Shaoli Huang Mingming Gong and Zhiyong Wu. 2024. Freetalker: Controllable Speech and Text-Driven Gesture Generation Based on Diffusion Models for Enhanced Speaker Naturalness. In ICASSP 2024 - 2024 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP).","DOI":"10.1109\/ICASSP48485.2024.10447978"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3577190.3616114"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"Yanzhe Yang Jimei Yang and Jessica Hodgins. 2020. Statistics-based motion synthesis for social conversations. In Computer Graphics Forum.","DOI":"10.1111\/cgf.14114"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00053"},{"key":"e_1_3_2_1_49_1","volume-title":"Speech gesture generation from the trimodal context of text, audio, and speaker identity. ACM Transactions on Graphics (TOG)","author":"Yoon Youngwoo","year":"2020","unstructured":"Youngwoo Yoon, Bok Cha, Joo-Haeng Lee, Minsu Jang, Jaeyeon Lee, Jaehong Kim, and Geehyuk Lee. 2020. Speech gesture generation from the trimodal context of text, audio, and speaker identity. ACM Transactions on Graphics (TOG) (2020)."},{"key":"e_1_3_2_1_50_1","volume-title":"Motiondiffuse: Text-driven human motion generation with diffusion model","author":"Zhang Mingyuan","year":"2024","unstructured":"Mingyuan Zhang, Zhongang Cai, Liang Pan, Fangzhou Hong, Xinying Guo, Lei Yang, and Ziwei Liu. 2024. Motiondiffuse: Text-driven human motion generation with diffusion model. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024)."},{"key":"e_1_3_2_1_51_1","volume-title":"Motion Mamba: Efficient and Long Sequence Motion Generation with Hierarchical and Bidirectional Selective SSM. arXiv preprint arXiv:2403.07487","author":"Zhang Zeyu","year":"2024","unstructured":"Zeyu Zhang, Akide Liu, Ian Reid, Richard Hartley, Bohan Zhuang, and Hao Tang. 2024. Motion Mamba: Efficient and Long Sequence Motion Generation with Hierarchical and Bidirectional Selective SSM. arXiv preprint arXiv:2403.07487 (2024)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01902"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01016"},{"key":"e_1_3_2_1_54_1","volume-title":"Human motion generation: A survey","author":"Zhu Wentao","year":"2023","unstructured":"Wentao Zhu, Xiaoxuan Ma, Dongwoo Ro, Hai Ci, Jinlu Zhang, Jiaxin Shi, Feng Gao, Qi Tian, and Yizhou Wang. 2023. Human motion generation: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence (2023)."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680625","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680625","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:57Z","timestamp":1750295877000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680625"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":54,"alternative-id":["10.1145\/3664647.3680625","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680625","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}