{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T17:01:37Z","timestamp":1784998897008,"version":"3.55.0"},"reference-count":63,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T00:00:00Z","timestamp":1778803200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100012401","name":"Beijing Science and Technology Planning Project","doi-asserted-by":"publisher","award":["Z231100005923029"],"award-info":[{"award-number":["Z231100005923029"]}],"id":[{"id":"10.13039\/501100012401","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62176020"],"award-info":[{"award-number":["62176020"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62436001"],"award-info":[{"award-number":["62436001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62536001"],"award-info":[{"award-number":["62536001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Graphical Models"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.gmod.2026.101332","type":"journal-article","created":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T20:10:58Z","timestamp":1779999058000},"page":"101332","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Co-speech holistic 3D motion generation with style from video"],"prefix":"10.1016","volume":"146","author":[{"given":"Yayu","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6195-9782","authenticated-orcid":false,"given":"Yu-Hui","family":"Wen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenguang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liping","family":"Jing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jian","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.gmod.2026.101332_b1","series-title":"Hand and mind: What gestures reveal about thought","author":"McNeill","year":"1992"},{"key":"10.1016\/j.gmod.2026.101332_b2","series-title":"SemTalk: Holistic co-speech motion generation with frame-level semantic emphasis","author":"Zhang","year":"2024"},{"key":"10.1016\/j.gmod.2026.101332_b3","first-page":"569","article-title":"A comprehensive review of data-driven co-speech gesture generation","volume":"vol. 42","author":"Nyatsanga","year":"2023"},{"issue":"4","key":"10.1016\/j.gmod.2026.101332_b4","first-page":"1","article-title":"Semantic gesticulator: Semantics-aware co-speech gesture synthesis","volume":"43","author":"Zhang","year":"2024","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.gmod.2026.101332_b5","first-page":"206","article-title":"Zeroeggs: Zero-shot example-based gesture generation from speech","volume":"vol. 42","author":"Ghorbani","year":"2023"},{"key":"10.1016\/j.gmod.2026.101332_b6","series-title":"European Conference on Computer Vision","first-page":"712","article-title":"Audio-driven stylized gesture generation with flow-based model","author":"Ye","year":"2022"},{"issue":"3","key":"10.1016\/j.gmod.2026.101332_b7","doi-asserted-by":"crossref","first-page":"374","DOI":"10.1016\/j.jrp.2010.04.002","article-title":"Motion patterns in political speech and their influence on personality ratings","volume":"44","author":"Koppensteiner","year":"2010","journal-title":"J. Res. Pers."},{"issue":"4","key":"10.1016\/j.gmod.2026.101332_b8","first-page":"1","article-title":"Understanding the impact of animated gesture performance on personality perceptions","volume":"36","author":"Smith","year":"2017","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.gmod.2026.101332_b9","series-title":"Diffusestylegesture: Stylized audio-driven co-speech gesture generation with diffusion models","author":"Yang","year":"2023"},{"key":"10.1016\/j.gmod.2026.101332_b10","doi-asserted-by":"crossref","unstructured":"J. Chen, Y. Liu, J. Wang, A. Zeng, Y. Li, Q. Chen, Diffsheg: A diffusion-based approach for real-time speech-driven holistic 3d expression and gesture generation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 7352\u20137361.","DOI":"10.1109\/CVPR52733.2024.00702"},{"key":"10.1016\/j.gmod.2026.101332_b11","doi-asserted-by":"crossref","unstructured":"Y. Liu, Q. Cao, Y. Wen, H. Jiang, C. Ding, Towards variable and coordinated holistic co-speech motion generation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 1566\u20131576.","DOI":"10.1109\/CVPR52733.2024.00155"},{"key":"10.1016\/j.gmod.2026.101332_b12","series-title":"European Conference on Computer Vision","first-page":"612","article-title":"Beat: A large-scale semantic and emotional multi-modal dataset for conversational gestures synthesis","author":"Liu","year":"2022"},{"issue":"4","key":"10.1016\/j.gmod.2026.101332_b13","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3592097","article-title":"Gesturediffuclip: Gesture diffusion model with clip latents","volume":"42","author":"Ao","year":"2023","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.gmod.2026.101332_b14","series-title":"International Conference on Machine Learning","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","author":"Chen","year":"2020"},{"key":"10.1016\/j.gmod.2026.101332_b15","doi-asserted-by":"crossref","unstructured":"J. Cassell, C. Pelachaud, N. Badler, M. Steedman, B. Achorn, T. Becket, B. Douville, S. Prevost, M. Stone, Animated conversation: rule-based generation of facial expression, gesture & spoken intonation for multiple conversational agents, in: Proceedings of the 21st Annual Conference on Computer Graphics and Interactive Techniques, 1994, pp. 413\u2013420.","DOI":"10.1145\/192161.192272"},{"key":"10.1016\/j.gmod.2026.101332_b16","doi-asserted-by":"crossref","unstructured":"J. Cassell, H.H. Vilhj\u00e1lmsson, T. Bickmore, Beat: the behavior expression animation toolkit, in: Proceedings of the 28th Annual Conference on Computer Graphics and Interactive Techniques, 2001, pp. 477\u2013486.","DOI":"10.1145\/383259.383315"},{"key":"10.1016\/j.gmod.2026.101332_b17","series-title":"Gesture generation by imitation: From human behavior to computer character animation","author":"Kipp","year":"2005"},{"key":"10.1016\/j.gmod.2026.101332_b18","series-title":"International Workshop on Intelligent Virtual Agents","first-page":"205","article-title":"Towards a common framework for multimodal generation: The behavior markup language","author":"Kopp","year":"2006"},{"key":"10.1016\/j.gmod.2026.101332_b19","doi-asserted-by":"crossref","unstructured":"T. Kucherenko, D. Hasegawa, G.E. Henter, N. Kaneko, H. Kjellstr\u00f6m, Analyzing input and output representations for speech-driven gesture generation, in: Proceedings of the 19th ACM International Conference on Intelligent Virtual Agents, 2019, pp. 97\u2013104.","DOI":"10.1145\/3308532.3329472"},{"key":"10.1016\/j.gmod.2026.101332_b20","doi-asserted-by":"crossref","unstructured":"K. Takeuchi, D. Hasegawa, S. Shirakawa, N. Kaneko, H. Sakuta, K. Sumi, Speech-to-gesture generation: A challenge in deep learning approach with bi-directional LSTM, in: Proceedings of the 5th International Conference on Human Agent Interaction, 2017, pp. 365\u2013369.","DOI":"10.1145\/3125739.3132594"},{"issue":"11","key":"10.1016\/j.gmod.2026.101332_b21","doi-asserted-by":"crossref","first-page":"139","DOI":"10.1145\/3422622","article-title":"Generative adversarial networks","volume":"63","author":"Goodfellow","year":"2020","journal-title":"Commun. ACM"},{"key":"10.1016\/j.gmod.2026.101332_b22","doi-asserted-by":"crossref","unstructured":"S. Ginosar, A. Bar, G. Kohavi, C. Chan, A. Owens, J. Malik, Learning individual styles of conversational gesture, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 3497\u20133506.","DOI":"10.1109\/CVPR.2019.00361"},{"key":"10.1016\/j.gmod.2026.101332_b23","doi-asserted-by":"crossref","unstructured":"H. Liu, N. Iwamoto, Z. Zhu, Z. Li, Y. Zhou, E. Bozkurt, B. Zheng, Disco: Disentangled implicit content and rhythm learning for diverse co-speech gestures synthesis, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 3764\u20133773.","DOI":"10.1145\/3503161.3548400"},{"key":"10.1016\/j.gmod.2026.101332_b24","doi-asserted-by":"crossref","unstructured":"J. Li, D. Kang, W. Pei, X. Zhe, Y. Zhang, Z. He, L. Bao, Audio2gestures: Generating diverse gestures from speech audio with conditional variational autoencoders, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 11293\u201311302.","DOI":"10.1109\/ICCV48922.2021.01110"},{"key":"10.1016\/j.gmod.2026.101332_b25","article-title":"Neural discrete representation learning","volume":"30","author":"Van Den Oord","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.gmod.2026.101332_b26","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.gmod.2026.101332_b27","doi-asserted-by":"crossref","unstructured":"L. Zhu, X. Liu, X. Liu, R. Qian, Z. Liu, L. Yu, Taming diffusion models for audio-driven co-speech gesture generation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 10544\u201310553.","DOI":"10.1109\/CVPR52729.2023.01016"},{"issue":"4","key":"10.1016\/j.gmod.2026.101332_b28","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3592458","article-title":"Listen, denoise, action! audio-driven motion synthesis with diffusion models","volume":"42","author":"Alexanderson","year":"2023","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.gmod.2026.101332_b29","series-title":"Diffwave: A versatile diffusion model for audio synthesis","author":"Kong","year":"2020"},{"key":"10.1016\/j.gmod.2026.101332_b30","doi-asserted-by":"crossref","unstructured":"S. Yang, Z. Wu, M. Li, Z. Zhang, L. Hao, W. Bao, H. Zhuang, Qpgesture: Quantization-based and phase-guided motion matching for natural speech-driven gesture generation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 2321\u20132330.","DOI":"10.1109\/CVPR52729.2023.00230"},{"key":"10.1016\/j.gmod.2026.101332_b31","doi-asserted-by":"crossref","unstructured":"Y. Fan, Z. Lin, J. Saito, W. Wang, T. Komura, Faceformer: Speech-driven 3d facial animation with transformers, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 18770\u201318780.","DOI":"10.1109\/CVPR52688.2022.01821"},{"issue":"4","key":"10.1016\/j.gmod.2026.101332_b32","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3658221","article-title":"Diffposetalk: Speech-driven stylistic 3d facial animation and head pose generation via diffusion models","volume":"43","author":"Sun","year":"2024","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.gmod.2026.101332_b33","doi-asserted-by":"crossref","unstructured":"Y. Pan, K. Singh, L.G. Hafemann, Model See Model Do: Speech-Driven Facial Animation with Style Control, in: Proceedings of the Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers, 2025, pp. 1\u201310.","DOI":"10.1145\/3721238.3730672"},{"issue":"10","key":"10.1016\/j.gmod.2026.101332_b34","doi-asserted-by":"crossref","first-page":"12287","DOI":"10.1109\/TPAMI.2023.3271691","article-title":"Pymaf-x: Towards well-aligned full-body model regression from monocular images","volume":"45","author":"Zhang","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.gmod.2026.101332_b35","series-title":"2021 International Conference on 3D Vision (3DV)","first-page":"792","article-title":"Collaborative regression of expressive bodies using moderation","author":"Feng","year":"2021"},{"issue":"8","key":"10.1016\/j.gmod.2026.101332_b36","doi-asserted-by":"crossref","first-page":"9469","DOI":"10.1109\/TPAMI.2023.3247907","article-title":"Consistent 3d hand reconstruction in video via self-supervised learning","volume":"45","author":"Tu","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.gmod.2026.101332_b37","doi-asserted-by":"crossref","unstructured":"H. Liu, Z. Zhu, G. Becherini, Y. Peng, M. Su, Y. Zhou, X. Zhe, N. Iwamoto, B. Zheng, M.J. Black, Emage: Towards unified holistic co-speech gesture generation via expressive masked audio gesture modeling, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 1144\u20131154.","DOI":"10.1109\/CVPR52733.2024.00115"},{"key":"10.1016\/j.gmod.2026.101332_b38","doi-asserted-by":"crossref","unstructured":"H. Yi, H. Liang, Y. Liu, Q. Cao, Y. Wen, T. Bolkart, D. Tao, M.J. Black, Generating holistic 3d human motion from speech, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 469\u2013480.","DOI":"10.1109\/CVPR52729.2023.00053"},{"key":"10.1016\/j.gmod.2026.101332_b39","article-title":"CoCoGesture: Towards coherent co-speech 3D gesture generation in the wild","author":"Qi","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.gmod.2026.101332_b40","doi-asserted-by":"crossref","unstructured":"I. Habibie, W. Xu, D. Mehta, L. Liu, H.-P. Seidel, G. Pons-Moll, M. Elgharib, C. Theobalt, Learning speech-driven 3d conversational gestures from video, in: Proceedings of the 21st ACM International Conference on Intelligent Virtual Agents, 2021, pp. 101\u2013108.","DOI":"10.1145\/3472306.3478335"},{"key":"10.1016\/j.gmod.2026.101332_b41","doi-asserted-by":"crossref","unstructured":"P. Liu, L. Song, J. Huang, H. Liu, C. Xu, GestureLSM: Latent Shortcut based Co-Speech Gesture Generation with Spatial-Temporal Modeling, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2025, pp. 10929\u201310939.","DOI":"10.1109\/ICCV51701.2025.01017"},{"key":"10.1016\/j.gmod.2026.101332_b42","series-title":"Flow matching for generative modeling","author":"Lipman","year":"2022"},{"key":"10.1016\/j.gmod.2026.101332_b43","series-title":"Flow straight and fast: Learning to generate and transfer data with rectified flow","author":"Liu","year":"2022"},{"key":"10.1016\/j.gmod.2026.101332_b44","doi-asserted-by":"crossref","unstructured":"X. Zhang, J. Li, J. Zhang, J. Ren, L. Bo, Z. Tu, Echomask: Speech-queried attention-based mask modeling for holistic co-speech motion generation, in: Proceedings of the 33rd ACM International Conference on Multimedia, 2025, pp. 10827\u201310836.","DOI":"10.1145\/3746027.3754847"},{"key":"10.1016\/j.gmod.2026.101332_b45","doi-asserted-by":"crossref","unstructured":"X. Zhang, J. Li, J. Zhang, Z. Dang, J. Ren, L. Bo, Z. Tu, Semtalk: Holistic co-speech motion generation with frame-level semantic emphasis, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2025, pp. 13761\u201313771.","DOI":"10.1109\/ICCV51701.2025.01277"},{"key":"10.1016\/j.gmod.2026.101332_b46","series-title":"European Conference on Computer Vision","first-page":"172","article-title":"Co-speech gesture video generation with 3d human meshes","author":"Mahapatra","year":"2024"},{"key":"10.1016\/j.gmod.2026.101332_b47","doi-asserted-by":"crossref","unstructured":"D. Lee, C. Kim, S. Kim, M. Cho, W.-S. Han, Autoregressive image generation using residual quantization, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 11523\u201311532.","DOI":"10.1109\/CVPR52688.2022.01123"},{"key":"10.1016\/j.gmod.2026.101332_b48","doi-asserted-by":"crossref","unstructured":"X. Huang, S. Belongie, Arbitrary style transfer in real-time with adaptive instance normalization, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 1501\u20131510.","DOI":"10.1109\/ICCV.2017.167"},{"key":"10.1016\/j.gmod.2026.101332_b49","doi-asserted-by":"crossref","unstructured":"K. Chhatre, N. Athanasiou, G. Becherini, C. Peters, M.J. Black, T. Bolkart, et al., Emotional speech-driven 3d body animation via disentangled latent diffusion, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 1942\u20131953.","DOI":"10.1109\/CVPR52733.2024.00190"},{"key":"10.1016\/j.gmod.2026.101332_b50","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.gmod.2026.101332_b51","doi-asserted-by":"crossref","unstructured":"B. Chen, Y. Li, Y. Zheng, Y.-X. Ding, K. Zhou, Motion-example-controlled Co-speech Gesture Generation Leveraging Large Language Models, in: Proceedings of the Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers, 2025, pp. 1\u201312.","DOI":"10.1145\/3721238.3730611"},{"key":"10.1016\/j.gmod.2026.101332_b52","doi-asserted-by":"crossref","unstructured":"G. Pavlakos, V. Choutas, N. Ghorbani, T. Bolkart, A.A. Osman, D. Tzionas, M.J. Black, Expressive body capture: 3d hands, face, and body from a single image, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 10975\u201310985.","DOI":"10.1109\/CVPR.2019.01123"},{"key":"10.1016\/j.gmod.2026.101332_b53","doi-asserted-by":"crossref","unstructured":"Y. Zhou, C. Barnes, J. Lu, J. Yang, H. Li, On the continuity of rotation representations in neural networks, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 5745\u20135753.","DOI":"10.1109\/CVPR.2019.00589"},{"key":"10.1016\/j.gmod.2026.101332_b54","doi-asserted-by":"crossref","unstructured":"C. Guo, Y. Mu, M.G. Javed, S. Wang, L. Cheng, Momask: Generative masked modeling of 3d human motions, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 1900\u20131910.","DOI":"10.1109\/CVPR52733.2024.00186"},{"key":"10.1016\/j.gmod.2026.101332_b55","doi-asserted-by":"crossref","unstructured":"B. Chen, Y. Li, Y.-X. Ding, T. Shao, K. Zhou, Enabling synergistic full-body control in prompt-based co-speech motion generation, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 6774\u20136783.","DOI":"10.1145\/3664647.3680847"},{"key":"10.1016\/j.gmod.2026.101332_b56","article-title":"SpeechAct: Towards generating whole-body motion from speech","author":"Zhang","year":"2025","journal-title":"IEEE Trans. Vis. Comput. Graphics"},{"issue":"4","key":"10.1016\/j.gmod.2026.101332_b57","doi-asserted-by":"crossref","first-page":"64:1","DOI":"10.1145\/3386569.3392469","article-title":"Unpaired motion style transfer from video to animation","volume":"39","author":"Aberman","year":"2020","journal-title":"ACM Trans. Graph."},{"issue":"4","key":"10.1016\/j.gmod.2026.101332_b58","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3731167","article-title":"Sketch2anim: Towards transferring sketch storyboards into 3d animation","volume":"44","author":"Zhong","year":"2025","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.gmod.2026.101332_b59","series-title":"Classifier-free diffusion guidance","author":"Ho","year":"2022"},{"issue":"Nov","key":"10.1016\/j.gmod.2026.101332_b60","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."},{"issue":"6","key":"10.1016\/j.gmod.2026.101332_b61","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3414685.3417838","article-title":"Speech gesture generation from the trimodal context of text, audio, and speaker identity","volume":"39","author":"Yoon","year":"2020","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.gmod.2026.101332_b62","doi-asserted-by":"crossref","unstructured":"R. Li, S. Yang, D.A. Ross, A. Kanazawa, Ai choreographer: Music conditioned 3d dance generation with aist++, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 13401\u201313412.","DOI":"10.1109\/ICCV48922.2021.01315"},{"key":"10.1016\/j.gmod.2026.101332_b63","doi-asserted-by":"crossref","unstructured":"J. Xing, M. Xia, Y. Zhang, X. Cun, J. Wang, T.-T. Wong, Codetalker: Speech-driven 3d facial animation with discrete motion prior, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 12780\u201312790.","DOI":"10.1109\/CVPR52729.2023.01229"}],"container-title":["Graphical Models"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1524070326000135?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1524070326000135?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T16:23:17Z","timestamp":1784996597000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1524070326000135"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":63,"alternative-id":["S1524070326000135"],"URL":"https:\/\/doi.org\/10.1016\/j.gmod.2026.101332","relation":{},"ISSN":["1524-0703"],"issn-type":[{"value":"1524-0703","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Co-speech holistic 3D motion generation with style from video","name":"articletitle","label":"Article Title"},{"value":"Graphical Models","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.gmod.2026.101332","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Inc.","name":"copyright","label":"Copyright"}],"article-number":"101332"}}