{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T05:16:38Z","timestamp":1783142198733,"version":"3.54.6"},"reference-count":60,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62507037"],"award-info":[{"award-number":["62507037"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.eswa.2026.132711","type":"journal-article","created":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T06:25:03Z","timestamp":1777875903000},"page":"132711","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Holistic co-speech motion generation via cross-gated attention and cross-limb interaction"],"prefix":"10.1016","volume":"326","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2743-2017","authenticated-orcid":false,"given":"Zixiang","family":"Lu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhixiang","family":"Sheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-9466-3604","authenticated-orcid":false,"given":"Zhitong","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ping","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2872-388X","authenticated-orcid":false,"given":"Qiguang","family":"Miao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"3","key":"10.1016\/j.eswa.2026.132711_bib0001","doi-asserted-by":"crossref","first-page":"483","DOI":"10.1145\/566654.566606","article-title":"Interactive motion generation from examples","volume":"21","author":"Arikan","year":"2002","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"10.1016\/j.eswa.2026.132711_bib0002","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132711_bib0003","doi-asserted-by":"crossref","first-page":"135","DOI":"10.1162\/tacl_a_00051","article-title":"Enriching word vectors with subword information","volume":"5","author":"Bojanowski","year":"2017","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"10.1016\/j.eswa.2026.132711_bib0004","series-title":"European conference on computer vision","first-page":"356","article-title":"Implicit neural representations for variable length human motion generation","author":"Cervantes","year":"2022"},{"key":"10.1016\/j.eswa.2026.132711_bib0005","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"6200","article-title":"The language of motion: Unifying verbal and non-verbal language of 3D human motion","author":"Chen","year":"2025"},{"key":"10.1016\/j.eswa.2026.132711_bib0006","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"7352","article-title":"DiffSheg: A diffusion-based approach for real-time speech-driven holistic 3D expression and gesture generation","author":"Chen","year":"2024"},{"key":"10.1016\/j.eswa.2026.132711_bib0007","series-title":"Proceedings of the 32nd ACM international conference on multimedia","first-page":"4024","article-title":"Realistic full-body motion generation from sparse tracking with state space model","author":"Dong","year":"2024"},{"key":"10.1016\/j.eswa.2026.132711_bib0008","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"18770","article-title":"FaceFormer: Speech-driven 3D facial animation with transformers","author":"Fan","year":"2022"},{"key":"10.1016\/j.eswa.2026.132711_bib0009","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"1396","article-title":"Synthesis of compositional animations from textual descriptions","author":"Ghosh","year":"2021"},{"key":"10.1016\/j.eswa.2026.132711_bib0010","series-title":"First conference on language modeling","article-title":"Mamba: Linear-time sequence modeling with selective state spaces","author":"Gu","year":"2024"},{"key":"10.1016\/j.eswa.2026.132711_bib0011","series-title":"Proceedings of the 21st ACM international conference on intelligent virtual agents","first-page":"101","article-title":"Learning speech-driven 3D conversational gestures from video","author":"Habibie","year":"2021"},{"key":"10.1016\/j.eswa.2026.132711_bib0012","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"2024","article-title":"Causal motion tokenizer for streaming motion generation","author":"Jiang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132711_bib0013","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4401","article-title":"A style-based generator architecture for generative adversarial networks","author":"Karras","year":"2019"},{"key":"10.1016\/j.eswa.2026.132711_bib0014","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8110","article-title":"Analyzing and improving the image quality of stylegan","author":"Karras","year":"2020"},{"key":"10.1016\/j.eswa.2026.132711_bib0015","series-title":"SIGGRAPH asia 2024 conference papers","first-page":"1","article-title":"Body gesture generation for multimodal conversational agents","author":"Kim","year":"2024"},{"key":"10.1016\/j.eswa.2026.132711_bib0016","unstructured":"Kingma, D. P. (2014). Adam: A method for stochastic optimization. arXiv: 1412.6980."},{"key":"10.1016\/j.eswa.2026.132711_bib0017","series-title":"Proceedings of the 19th ACM international conference on intelligent virtual agents","first-page":"97","article-title":"Analyzing input and output representations for speech-driven gesture generation","author":"Kucherenko","year":"2019"},{"key":"10.1016\/j.eswa.2026.132711_bib0018","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8673","article-title":"Music-driven group choreography","author":"Le","year":"2023"},{"key":"10.1016\/j.eswa.2026.132711_bib0019","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"11293","article-title":"Audio2Gestures: Generating diverse gestures from speech audio with conditional variational autoencoders","author":"Li","year":"2021"},{"key":"10.1016\/j.eswa.2026.132711_bib0020","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"13401","article-title":"AI choreographer: Music conditioned 3D dance generation with AIST++","author":"Li","year":"2021"},{"key":"10.1016\/j.eswa.2026.132711_bib0021","unstructured":"Li, Z., Wang, S., Zhang, Z., & Tang, H. (2025). ReMoMask: Retrieval-augmented masked motion generation. arXiv: 2508.02605."},{"key":"10.1016\/j.eswa.2026.132711_bib0022","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"3764","article-title":"DisCo: Disentangled implicit content and rhythm learning for diverse co-speech gestures synthesis","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132711_bib0023","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"1144","article-title":"EMAGE: Towards unified holistic co-speech gesture generation via expressive masked audio gesture modeling","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.132711_bib0024","series-title":"European conference on computer vision","first-page":"612","article-title":"BEAT: A large-scale semantic and emotional multi-modal dataset for conversational gestures synthesis","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132711_bib0025","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"13963","article-title":"SemGes: Semantics-aware co-speech gesture generation using semantic coherence and relevance learning","author":"Liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132711_bib0026","unstructured":"Liu, L., He, Y., Chu, Z., Xing, X., & Xu, X. (2025b). MimicParts: Part-aware style injection for speech-driven 3D motion generation. arXiv: 2510.13208."},{"key":"10.1016\/j.eswa.2026.132711_bib0027","series-title":"Companion of the 2023\u202fACM\/IEEE international conference on human-robot interaction","first-page":"548","article-title":"Human gesture recognition with a flow-based model for human robot interaction","author":"Liu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132711_bib0028","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10462","article-title":"Learning hierarchical cross-modal association for co-speech gesture generation","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132711_bib0029","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"1566","article-title":"Towards variable and coordinated holistic co-speech motion generation","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.132711_bib0030","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"20406","article-title":"Speech driven tongue animation","author":"Medina","year":"2022"},{"key":"10.1016\/j.eswa.2026.132711_bib0031","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"16578","article-title":"Retrieving semantics from the deep: An RAG solution for gesture synthesis","author":"Mughal","year":"2025"},{"key":"10.1016\/j.eswa.2026.132711_bib0032","doi-asserted-by":"crossref","unstructured":"Peng, W., Zhang, K., & Zhang, S. Q. (2024). T3m: Text guided 3D human motion synthesis from speech. arXiv: 2408.12885.","DOI":"10.18653\/v1\/2024.findings-naacl.74"},{"key":"10.1016\/j.eswa.2026.132711_bib0033","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"20687","article-title":"EmoTalk: Speech-driven emotional disentanglement for 3d face animation","author":"Peng","year":"2023"},{"issue":"6619","key":"10.1016\/j.eswa.2026.132711_bib0034","doi-asserted-by":"crossref","first-page":"505","DOI":"10.1126\/science.abq2591","article-title":"The emergent properties of the connected brain","volume":"378","author":"Thiebaut de Schotten","year":"2022","journal-title":"Science"},{"key":"10.1016\/j.eswa.2026.132711_bib0035","unstructured":"Somvanshi, S., Monzurul Islam, M., Sultana Mimi, M., Bashar Polock, S. B., Chhetri, G., & Das, S. (2025). A survey on structured state space sequence (s4) models. (arXive-printsarXiv.2503)."},{"key":"10.1016\/j.eswa.2026.132711_bib0036","series-title":"European conference on computer vision","first-page":"358","article-title":"MotionCLIP: Exposing human motion generation to clip space","author":"Tevet","year":"2022"},{"key":"10.1016\/j.eswa.2026.132711_bib0037","unstructured":"Tevet, G., Raab, S., Gordon, B., Shafir, Y., Cohen-Or, D., & Bermano, A. H. (2022b). Human motion diffusion model. arXiv: 2209.14916."},{"issue":"6","key":"10.1016\/j.eswa.2026.132711_bib0038","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3478513.3480570","article-title":"TransFlower: Probabilistic autoregressive dance generation with multimodal attention","volume":"40","author":"Valle-P\u00e9rez","year":"2021","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"10.1016\/j.eswa.2026.132711_bib0039","series-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","first-page":"6309","article-title":"Neural discrete representation learning","author":"Van Den Oord","year":"2017"},{"key":"10.1016\/j.eswa.2026.132711_bib0040","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"7169","article-title":"TIMotion: Temporal and interactive framework for efficient human-human motion generation","author":"Wang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132711_bib0041","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"12780","article-title":"CodeTalker: Speech-driven 3D facial animation with discrete motion prior","author":"Xing","year":"2023"},{"key":"10.1016\/j.eswa.2026.132711_bib0042","first-page":"1","article-title":"Combo: Co-speech holistic 3D human m otion generation and efficient customiza b le adaptation in harmony","author":"Xu","year":"2025","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132711_bib0043","doi-asserted-by":"crossref","first-page":"20055","DOI":"10.52202\/079017-0633","article-title":"MambaTalk: Efficient holistic gesture synthesis with selective state space models","volume":"37","author":"Xu","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132711_bib0044","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"6387","article-title":"Chain of generation: Multi-modal gesture synthesis via cascaded conditional control","volume":"vol. 38","author":"Xu","year":"2024"},{"key":"10.1016\/j.eswa.2026.132711_bib0045","doi-asserted-by":"crossref","first-page":"10709","DOI":"10.1109\/TPAMI.2025.3594034","article-title":"Human motion video generation: A survey","volume":"47","author":"Xue","year":"2025","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132711_bib0046","doi-asserted-by":"crossref","unstructured":"Yang, S., Wu, Z., Li, M., Zhang, Z., Hao, L., Bao, W., Cheng, M., & Xiao, L. (2023a). DiffuseStyleGesture: Stylized audio-driven co-speech gesture generation with diffusion models. arXiv: 2305.04919.","DOI":"10.24963\/ijcai.2023\/650"},{"key":"10.1016\/j.eswa.2026.132711_bib0047","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"2321","article-title":"QPGesture: Quantization-based and phase-guided motion matching for natural speech-driven gesture generation","author":"Yang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132711_bib0048","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"469","article-title":"Generating holistic 3D human motion from speech","author":"Yi","year":"2023"},{"key":"10.1016\/j.eswa.2026.132711_bib0049","series-title":"The thirty-ninth annual conference on neural information processing systems","article-title":"PyraMotion: Attentional pyramid-structured motion integration for co-speech 3D gesture synthesis","author":"Yin","year":"2025"},{"issue":"6","key":"10.1016\/j.eswa.2026.132711_bib0050","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3414685.3417838","article-title":"Speech gesture generation from the trimodal context of text, audio, and speaker identity","volume":"39","author":"Yoon","year":"2020","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"10.1016\/j.eswa.2026.132711_bib0051","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"9797","article-title":"Light-T2M: A lightweight and fast model for text-to-motion generation","volume":"vol. 39","author":"Zeng","year":"2025"},{"key":"10.1016\/j.eswa.2026.132711_bib0052","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"17592","article-title":"EnerGymogen: Compositional human motion generation with energy-based diffusion model in latent space","author":"Zhang","year":"2025"},{"issue":"6","key":"10.1016\/j.eswa.2026.132711_bib0053","doi-asserted-by":"crossref","first-page":"4115","DOI":"10.1109\/TPAMI.2024.3355414","article-title":"MotionDiffuse: Text-driven human motion generation with diffusion model","volume":"46","author":"Zhang","year":"2024","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132711_bib0054","series-title":"Proceedings of the 2025\u202fCHI conference on human factors in computing systems","first-page":"1","article-title":"Prompting an embodied AI agent: How embodiment and multimodal signaling affects prompting behaviour","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132711_bib0055","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"13761","article-title":"SemTalk: Holistic co-speech motion generation with frame-level semantic emphasis","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132711_bib0056","series-title":"Proceedings of the 33rd ACM international conference on multimedia","first-page":"10827","article-title":"EchoMask: Speech-queried attention-based mask modeling for holistic co-speech motion generation","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132711_bib0057","series-title":"European conference on computer vision","first-page":"265","article-title":"Motion mamba: Efficient and long sequence motion generation","author":"Zhang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132711_bib0058","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"20807","article-title":"LivelySpeaker: Towards semantic-aware co-speech gesture generation","author":"Zhi","year":"2023"},{"key":"10.1016\/j.eswa.2026.132711_bib0059","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"509","article-title":"ATTT2M: Text-driven human motion generation with multi-perspective attention mechanism","author":"Zhong","year":"2023"},{"issue":"4","key":"10.1016\/j.eswa.2026.132711_bib0060","doi-asserted-by":"crossref","first-page":"2430","DOI":"10.1109\/TPAMI.2023.3330935","article-title":"Human motion generation: A survey","volume":"46","author":"Zhu","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426016246?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426016246?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T04:32:59Z","timestamp":1783139579000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426016246"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":60,"alternative-id":["S0957417426016246"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132711","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Holistic co-speech motion generation via cross-gated attention and cross-limb interaction","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132711","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132711"}}