{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T15:45:57Z","timestamp":1782920757198,"version":"3.54.5"},"reference-count":60,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100018919","name":"Pengcheng Laboratory","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100018919","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.eswa.2026.133415","type":"journal-article","created":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T23:54:53Z","timestamp":1782345293000},"page":"133415","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PD","title":["GranuMamba: A multi-granularity state space model for co-speech gesture generation"],"prefix":"10.1016","volume":"331","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9886-8814","authenticated-orcid":false,"given":"Jiye","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0196-3730","authenticated-orcid":false,"given":"Gaolin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9762-4826","authenticated-orcid":false,"given":"Jinhan","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0728-2442","authenticated-orcid":false,"given":"Yifan","family":"Ge","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2422-942X","authenticated-orcid":false,"given":"Xiuhua","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1567-0931","authenticated-orcid":false,"given":"Jiangbo","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133415_bib0001","series-title":"Findings of the association for computational linguistics: EMNLP 2020","first-page":"1884","article-title":"No gestures left behind: Learning relationships between spoken language and freeform gestures","author":"Ahuja","year":"2020"},{"key":"10.1016\/j.eswa.2026.133415_bib0002","series-title":"Computer graphics forum","first-page":"487","article-title":"Style-controllable speech-driven gesture synthesis using normalising flows","volume":"vol. 39","author":"Alexanderson","year":"2020"},{"key":"10.1016\/j.eswa.2026.133415_bib0003","first-page":"12449","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"1","key":"10.1016\/j.eswa.2026.133415_bib0004","doi-asserted-by":"crossref","first-page":"A553","DOI":"10.1137\/22M1518396","article-title":"Randomized block gram\u2013schmidt process for the solution of linear systems and eigenvalue problems","volume":"47","author":"Balabanov","year":"2025","journal-title":"SIAM Journal on Scientific Computing"},{"key":"10.1016\/j.eswa.2026.133415_bib0005","series-title":"International conference on machine learning","first-page":"685","article-title":"Frequency bias in neural networks for input of non-uniform density","author":"Basri","year":"2020"},{"key":"10.1016\/j.eswa.2026.133415_bib0006","series-title":"2021\u202fIEEE Virtual reality and 3d user interfaces (VR)","first-page":"1","article-title":"Text2Gestures: A transformer-based network for generating emotive body gestures for virtual agents","author":"Bhattacharya","year":"2021"},{"key":"10.1016\/j.eswa.2026.133415_bib0007","first-page":"19","article-title":"Methodology for the subjective assessment of the quality of television pictures","volume":"4","author":"BT","year":"2002","journal-title":"International Telecommunication Union"},{"key":"10.1016\/j.eswa.2026.133415_bib0008","series-title":"European conference on computer vision","first-page":"356","article-title":"Implicit neural representations for variable length human motion generation","author":"Cervantes","year":"2022"},{"key":"10.1016\/j.eswa.2026.133415_bib0009","series-title":"Proceedings of the 30th asia and south pacific design automation conference","first-page":"183","article-title":"A practical randomized GMRES algorithm for solving linear equation system in circuit simulation","author":"Chen","year":"2025"},{"key":"10.1016\/j.eswa.2026.133415_bib0010","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"6200","article-title":"The language of motion: Unifying verbal and non-verbal language of 3d human motion","author":"Chen","year":"2025"},{"key":"10.1016\/j.eswa.2026.133415_bib0011","series-title":"European conference on computer vision","first-page":"20","article-title":"Monocular expressive body regression through body-driven attention","author":"Choutas","year":"2020"},{"key":"10.1016\/j.eswa.2026.133415_bib0012","series-title":"Proceedings of the 2019 conference of the north american chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.eswa.2026.133415_bib0013","doi-asserted-by":"crossref","first-page":"114122","DOI":"10.52202\/079017-3625","article-title":"Addressing spectral bias of deep neural networks by multi-grade deep learning","volume":"37","author":"Fang","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133415_bib0014","series-title":"Proceedings of the 32nd ACM international conference on multimedia","first-page":"10794","article-title":"MambaGesture: Enhancing co-speech gesture generation with mamba and disentangled multi-modality fusion","author":"Fu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133415_bib0015","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3497","article-title":"Learning individual styles of conversational gesture","author":"Ginosar","year":"2019"},{"key":"10.1016\/j.eswa.2026.133415_bib0016","unstructured":"Gu, A., & Dao, T. (2023). Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv: 2312.00752."},{"key":"10.1016\/j.eswa.2026.133415_bib0017","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2026.115261","article-title":"MAR-GCN: A meta-action refinement graph convolutional network for skeleton-based human action recognition","volume":"336","author":"Guo","year":"2026","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.133415_bib0018","doi-asserted-by":"crossref","first-page":"208","DOI":"10.1109\/TIP.2025.3645572","article-title":"Toward unified co-speech gesture generation via hierarchical implicit periodicity learning","volume":"35","author":"Guo","year":"2025","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133415_bib0019","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"2263","article-title":"Co-speech gesture video generation via motion-decoupled diffusion model","author":"He","year":"2024"},{"key":"10.1016\/j.eswa.2026.133415_bib0020","unstructured":"He, Y., Tiwari, G., Zhang, X., Bora, P., Birdal, T., Lenssen, J. E., & Pons-Moll, G. (2025). MoLingo: Motion-language alignment for text-to-motion generation. arXiv preprint arXiv: 2512.13840."},{"key":"10.1016\/j.eswa.2026.133415_bib0021","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"11293","article-title":"Audio2Gestures: Generating diverse gestures from speech audio with conditional variational autoencoders","author":"Li","year":"2021"},{"key":"10.1016\/j.eswa.2026.133415_bib0022","series-title":"Proceedings of the IEEE\/CVF CVPR","first-page":"13401","article-title":"Ai choreographer: Music conditioned 3d dance generation with aist++","author":"Li","year":"2021"},{"issue":"6","key":"10.1016\/j.eswa.2026.133415_bib0023","doi-asserted-by":"crossref","first-page":"194","DOI":"10.1145\/3130800.3130813","article-title":"Learning a model of facial shape and expression from 4d scans","volume":"36","author":"Li","year":"2017","journal-title":"ACM Transactions on Graphics"},{"key":"10.1016\/j.eswa.2026.133415_bib0024","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"11384","article-title":"Co-speech gesture video generation with implicit motion-audio entanglement","author":"Li","year":"2025"},{"key":"10.1016\/j.eswa.2026.133415_bib0025","unstructured":"Liu, B., Liu, L., Zhang, S., Gu, S., Zhi, Y., Zhu, T., Yang, L., & Ye, L. (2025). MAG: Multi-modal aligned autoregressive co-speech gesture generation without vector quantization. arXiv preprint arXiv: 2503.14040."},{"key":"10.1016\/j.eswa.2026.133415_bib0026","series-title":"International conference on representation learning","first-page":"52220","article-title":"TANGO: Co-speech gesture video reenactment with hierarchical audio motion embedding and diffusion interpolation","author":"liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133415_bib0027","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"1144","article-title":"EMAGE: Towards unified holistic co-speech gesture generation via expressive masked audio gesture modeling","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133415_bib0028","series-title":"European conference on computer vision","first-page":"612","article-title":"BEAT: A large-scale semantic and emotional multi-modal dataset for conversational gestures synthesis","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133415_bib0029","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision (ICCV)","first-page":"10929","article-title":"GestureLSM: Latent shortcut based co-speech gesture generation with spatial-temporal modeling","author":"Liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133415_bib0030","series-title":"Proceedings of the IEEE\/CVF CVPR","first-page":"10462","article-title":"Learning hierarchical cross-modal association for co-speech gesture generation","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133415_bib0031","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"1566","article-title":"Towards variable and coordinated holistic co-speech motion generation","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133415_bib0032","unstructured":"Mughal, M. H., Dabral, R., Demberg, V., & Theobalt, C. (2026). Miburi: Towards expressive interactive gesture synthesis. arXiv preprint arXiv: 2603.03282."},{"key":"10.1016\/j.eswa.2026.133415_bib0033","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"16578","article-title":"Retrieving semantics from the deep: An rag solution for gesture synthesis","author":"Mughal","year":"2025"},{"key":"10.1016\/j.eswa.2026.133415_bib0034","series-title":"Computer graphics forum","first-page":"569","article-title":"A comprehensive review of data-driven co-speech gesture generation","volume":"vol. 42","author":"Nyatsanga","year":"2023"},{"key":"10.1016\/j.eswa.2026.133415_bib0035","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10975","article-title":"Expressive body capture: 3d hands, face, and body from a single image","author":"Pavlakos","year":"2019"},{"key":"10.1016\/j.eswa.2026.133415_bib0036","article-title":"CoCoGesture: Towards coherent co-speech 3d gesture generation in the wild","volume":"126","author":"Qi","year":"2025","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.133415_bib0037","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"1173","article-title":"Meshtalk: 3D face animation from speech using cross-modality disentanglement","author":"Richard","year":"2021"},{"key":"10.1016\/j.eswa.2026.133415_bib0038","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/s41701-025-00197-2","article-title":"Multidimensional labeling of gesture in communication: The m3d proposal","volume":"9","author":"Rohrer","year":"2025","journal-title":"Corpus Pragmatics"},{"key":"10.1016\/j.eswa.2026.133415_bib0039","series-title":"International conference on machine learning","first-page":"2256","article-title":"Deep unsupervised learning using nonequilibrium thermodynamics","author":"Sohl-Dickstein","year":"2015"},{"key":"10.1016\/j.eswa.2026.133415_bib0040","series-title":"Icassp 2024-2024 ieee international conference on acoustics, speech and signal processing (icassp)","first-page":"4115","article-title":"Adaptive super resolution for one-shot talking-head generation","author":"Song","year":"2024"},{"issue":"2","key":"10.1016\/j.eswa.2026.133415_bib0041","doi-asserted-by":"crossref","first-page":"203","DOI":"10.1177\/002383099403700208","article-title":"Hand and mind: What gestures reveal about thought","volume":"37","author":"Studdert-Kennedy","year":"1994","journal-title":"Language and Speech"},{"issue":"5","key":"10.1016\/j.eswa.2026.133415_bib0042","doi-asserted-by":"crossref","first-page":"2910","DOI":"10.1007\/s11263-024-02300-7","article-title":"Beyond talking\u2013generating holistic 3d human dyadic motion for communication","volume":"133","author":"Sun","year":"2025","journal-title":"International Journal of Computer Vision"},{"issue":"4","key":"10.1016\/j.eswa.2026.133415_bib0043","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3658221","article-title":"DiffPoseTalk: Speech-driven stylistic 3d facial animation and head pose generation via diffusion models","volume":"43","author":"Sun","year":"2024","journal-title":"ACM Transactions on Graphics (ToG)"},{"key":"10.1016\/j.eswa.2026.133415_bib0044","doi-asserted-by":"crossref","unstructured":"Tang, Y., Guo, J., Liu, P., Wang, Z., Hua, H., Zhong, J.-X., Xiao, Y., Huang, C., Song, L., Liang, S. et al. (2025). Generative ai for cel-animation: A survey. arXiv preprint arXiv: 2501.06250.","DOI":"10.1109\/ICCVW69036.2025.00400"},{"key":"10.1016\/j.eswa.2026.133415_bib0045","series-title":"2015 Ieee information theory workshop (itw)","first-page":"1","article-title":"Deep learning and the information bottleneck principle","author":"Tishby","year":"2015"},{"key":"10.1016\/j.eswa.2026.133415_bib0046","unstructured":"Unterthiner, T., Van Steenkiste, S., Kurach, K., Marinier, R., Michalski, M., & Gelly, S. (2018). Towards accurate generative models of video: A new metric & challenges. arXiv preprint arXiv: 1812.01717."},{"key":"10.1016\/j.eswa.2026.133415_bib0047","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"6092","article-title":"Multi-scale control signal-aware transformer for motion synthesis without phase","volume":"vol. 37","author":"Wang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133415_bib0048","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"12780","article-title":"CodeTalker: Speech-driven 3d facial animation with discrete motion prior","author":"Xing","year":"2023"},{"key":"10.1016\/j.eswa.2026.133415_bib0049","doi-asserted-by":"crossref","first-page":"20055","DOI":"10.52202\/079017-0633","article-title":"MambaTalk: Efficient holistic gesture synthesis with selective state space models","volume":"37","author":"Xu","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133415_bib0050","series-title":"Proceedings of the AAAI conference on artificial intelligence","article-title":"Spatial temporal graph convolutional networks for skeleton-based action recognition","volume":"vol. 32","author":"Yan","year":"2018"},{"key":"10.1016\/j.eswa.2026.133415_bib0051","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"469","article-title":"Generating holistic 3D human motion from speech","author":"Yi","year":"2023"},{"key":"10.1016\/j.eswa.2026.133415_bib0052","unstructured":"Yin, Z., Tsui, Y. H., & Hui, P. (2025). M3g: Multi-granular gesture generator for audio-driven full-body human motion synthesis. arXiv preprint arXiv: 2505.08293."},{"issue":"6","key":"10.1016\/j.eswa.2026.133415_bib0053","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3414685.3417838","article-title":"Speech gesture generation from the trimodal context of text, audio, and speaker identity","volume":"39","author":"Yoon","year":"2020","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"10.1016\/j.eswa.2026.133415_bib0054","series-title":"2019 International conference on robotics and automation (ICRA)","first-page":"4303","article-title":"Robots learn social skills: End-to-end learning of co-speech gesture generation for humanoid robots","author":"Yoon","year":"2019"},{"key":"10.1016\/j.eswa.2026.133415_bib0055","unstructured":"Zhan, X., Fu, X., Yang, C., Zhang, X., Fu, D., Fang, P., Sun, T., Cai, X., Kim, H., Li, Y. et al. (2026). UMO: Unified sparse motion modeling for real-time co-speech avatars. arXiv preprint arXiv: 2605.14731."},{"key":"10.1016\/j.eswa.2026.133415_bib0056","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.130833","article-title":"Co-speech video generation via motion transfer based on diffusion models","volume":"650","author":"Zhang","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.133415_bib0057","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"13761","article-title":"SemTalk: Holistic co-speech motion generation with frame-level semantic emphasis","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133415_bib0058","series-title":"Proceedings of the 42nd international conference on machine learning","first-page":"74896","article-title":"MimicMotion: High-quality human motion video generation with confidence-aware pose guidance","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133415_bib0059","doi-asserted-by":"crossref","DOI":"10.1109\/TVCG.2026.3679469","article-title":"ExGes: Expressive human motion retrieval and modulation for audio-driven gesture synthesis","volume":"32","author":"Zhou","year":"2026","journal-title":"IEEE Transactions on Visualization and Computer Graphics"},{"key":"10.1016\/j.eswa.2026.133415_bib0060","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10544","article-title":"Taming diffusion models for audio-driven co-speech gesture generation","author":"Zhu","year":"2023"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426023249?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426023249?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T14:57:39Z","timestamp":1782917859000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426023249"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":60,"alternative-id":["S0957417426023249"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133415","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"GranuMamba: A multi-granularity state space model for co-speech gesture generation","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133415","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133415"}}