{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T11:36:06Z","timestamp":1782387366211,"version":"3.54.5"},"reference-count":77,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2024,12,17]],"date-time":"2024-12-17T00:00:00Z","timestamp":1734393600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,17]],"date-time":"2024-12-17T00:00:00Z","timestamp":1734393600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62171256"],"award-info":[{"award-number":["62171256"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1007\/s11263-024-02300-7","type":"journal-article","created":{"date-parts":[[2024,12,17]],"date-time":"2024-12-17T11:28:34Z","timestamp":1734434914000},"page":"2910-2926","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Beyond Talking \u2013 Generating Holistic 3D Human Dyadic Motion for Communication"],"prefix":"10.1007","volume":"133","author":[{"given":"Mingze","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chao","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinyu","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Baigui","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5942-3671","authenticated-orcid":false,"given":"Ruqi","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,17]]},"reference":[{"key":"2300_CR1","doi-asserted-by":"crossref","unstructured":"Ahuja, C., Ma, S., Morency, LP., & Sheikh, Y. (2019). To react or not to react: End-to-end visual pose forecasting for personalized avatar during dyadic conversations. In 2019 International conference on multimodal interaction (pp. 74\u201384).","DOI":"10.1145\/3340555.3353725"},{"key":"2300_CR2","doi-asserted-by":"crossref","unstructured":"Ahuja, C., Joshi, P., Ishii, R., & Morency, L.P. (2023). Continual learning for personalized co-speech gesture generation. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 20893\u201320903).","DOI":"10.1109\/ICCV51070.2023.01910"},{"issue":"6","key":"2300_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3550454.3555435","volume":"41","author":"T Ao","year":"2022","unstructured":"Ao, T., Gao, Q., Lou, Y., Chen, B., & Liu, L. (2022). Rhythmic gesticulator: Rhythm-aware co-speech gesture synthesis with hierarchical neural embeddings. ACM Transactions on Graphics (TOG), 41(6), 1\u201319.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"2300_CR4","first-page":"12449","volume":"33","author":"A Baevski","year":"2020","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., & Auli, M. (2020). wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in Neural Information Processing Systems, 33, 12449\u201312460.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2300_CR5","doi-asserted-by":"crossref","unstructured":"Bhattacharya, U., Childs, E., Rewkowski, N., & Manocha, D. (2021). Speech2affectivegestures: Synthesizing co-speech gestures with generative adversarial affective expression learning. In: Proceedings of the 29th ACM International Conference on Multimedia, pp 2027\u20132036.","DOI":"10.1145\/3474085.3475223"},{"key":"2300_CR6","unstructured":"Birdwhistell, R. (1952). Introduction to Kenesics."},{"key":"2300_CR7","first-page":"157","volume":"2","author":"V Blanz","year":"2023","unstructured":"Blanz, V., & Vetter, T. (2023). A morphable model for the synthesis of 3d faces. Seminal Graphics Papers: Pushing the Boundaries, 2, 157\u2013164.","journal-title":"Seminal Graphics Papers: Pushing the Boundaries"},{"issue":"3","key":"2300_CR8","doi-asserted-by":"publisher","first-page":"338","DOI":"10.1037\/1082-989X.7.3.338","volume":"7","author":"SM Boker","year":"2002","unstructured":"Boker, S. M., Rotondo, J. L., Xu, M., & King, K. (2002). Windowed cross-correlation and peak picking for the analysis of variability in the association between behavioral time series. Psychology Methods, 7(3), 338.","journal-title":"Psychology Methods"},{"key":"2300_CR9","doi-asserted-by":"publisher","first-page":"857","DOI":"10.1007\/s10579-016-9377-0","volume":"51","author":"E Bozkurt","year":"2017","unstructured":"Bozkurt, E., Khaki, H., Ke\u00e7eci, S., T\u00fcrker, B. B., Yemez, Y., & Erzin, E. (2017). The jestkod database: An affective multimodal database of dyadic interactions. Language Resources and Evaluation, 51, 857\u2013872.","journal-title":"Language Resources and Evaluation"},{"key":"2300_CR10","doi-asserted-by":"crossref","unstructured":"Chang, Z., Hu, W., Yang, Q., & Zheng, S. (2023). Hierarchical semantic perceptual listener head video generation: A high-performance pipeline. In Proceedings of the 31st ACM International Conference on Multimedia (pp. 9581\u20139585).","DOI":"10.1145\/3581783.3612869"},{"key":"2300_CR11","doi-asserted-by":"crossref","unstructured":"Cudeiro, D., Bolkart, T., Laidlaw, C., Ranjan, A., & Black, M.J. (2019). Capture, learning, and synthesis of 3d speaking styles. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 10101\u201310111).","DOI":"10.1109\/CVPR.2019.01034"},{"issue":"4","key":"2300_CR12","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1109\/TASSP.1980.1163420","volume":"28","author":"S Davis","year":"1980","unstructured":"Davis, S., & Mermelstein, P. (1980). Comparison of parametric representations for monosyllabic word recognition in continuously spoken sentences. IEEE Transactions on Acoustics, Speech, and Signal Processing, 28(4), 357\u2013366.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"2300_CR13","unstructured":"Devlin, J., Chang, M.W., Lee, K., & Toutanova, K. (2018). Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805."},{"key":"2300_CR14","doi-asserted-by":"crossref","unstructured":"Doukas, M.C., Zafeiriou, S., & Sharmanska, V. (2021). Headgan: One-shot neural head synthesis and editing. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 14398\u201314407).","DOI":"10.1109\/ICCV48922.2021.01413"},{"key":"2300_CR15","doi-asserted-by":"crossref","unstructured":"Fan, Y., Lin, Z., Saito, J., Wang, W., & Komura, T. (2022). Faceformer: Speech-driven 3d facial animation with transformers. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 18770\u201318780).","DOI":"10.1109\/CVPR52688.2022.01821"},{"key":"2300_CR16","doi-asserted-by":"crossref","unstructured":"Ferstl, Y., & McDonnell, R. (2018). Investigating the use of recurrent motion modelling for speech gesture generation. In Proceedings of the 18th international conference on intelligent virtual agents (pp. 93\u201398).","DOI":"10.1145\/3267851.3267898"},{"issue":"5","key":"2300_CR17","doi-asserted-by":"publisher","first-page":"579","DOI":"10.1002\/cav.267","volume":"19","author":"M Gillies","year":"2008","unstructured":"Gillies, M., Pan, X., Slater, M., & Shawe-Taylor, J. (2008). Responsive listening behavior. Computer Animation and Virtual Worlds, 19(5), 579\u2013589.","journal-title":"Computer Animation and Virtual Worlds"},{"key":"2300_CR18","doi-asserted-by":"crossref","unstructured":"Ginosar, S., Bar, A., Kohavi, G., Chan, C., Owens, A., & Malik, J. (2019). Learning individual styles of conversational gesture. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 3497\u20133506).","DOI":"10.1109\/CVPR.2019.00361"},{"key":"2300_CR19","doi-asserted-by":"crossref","unstructured":"Guo, Y., Chen, K., Liang, S., Liu, Y.J., Bao, H., & Zhang, J. (2021). Ad-nerf: Audio driven neural radiance fields for talking head synthesis. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 5784\u20135794).","DOI":"10.1109\/ICCV48922.2021.00573"},{"key":"2300_CR20","doi-asserted-by":"crossref","unstructured":"Habibie, I., Xu, W., Mehta, D., Liu, L., Seidel, HP., Pons-Moll, G., Elgharib, M., & Theobalt, C. (2021). Learning speech-driven 3d conversational gestures from video. In Proceedings of the 21st ACM international conference on intelligent virtual agents (pp. 101\u2013108).","DOI":"10.1145\/3472306.3478335"},{"key":"2300_CR21","doi-asserted-by":"crossref","unstructured":"Huang, CM., & Mutlu, B. (2014). Learning-based modeling of multimodal behaviors for humanlike robots. In Proceedings of the 2014 ACM\/IEEE international conference on human-robot interaction (pp. 57\u201364).","DOI":"10.1145\/2559636.2559668"},{"key":"2300_CR22","unstructured":"Jocelyn Scheirer RWP. (1999). Affective objects. MIT Media Laboratory: Tech. rep."},{"key":"2300_CR23","doi-asserted-by":"crossref","unstructured":"Jonell, P., Kucherenko, T., Henter, GE., & Beskow, J. (2020). Let\u2019s face it: Probabilistic multi-modal interlocutor-aware generation of facial gestures in dyadic settings. In Proceedings of the 20th ACM international conference on intelligent virtual agents (pp. 1\u20138).","DOI":"10.1145\/3383652.3423911"},{"key":"2300_CR24","doi-asserted-by":"crossref","unstructured":"Joo, H., Simon, T., Cikara, M., & Sheikh, Y. (2019). Towards social artificial intelligence: Nonverbal social signal prediction in a triadic interaction. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 10873\u201310883).","DOI":"10.1109\/CVPR.2019.01113"},{"key":"2300_CR25","unstructured":"Kipp, M. (2005). Gesture generation by imitation: From human behavior to computer character animation. Universal-Publishers."},{"key":"2300_CR26","doi-asserted-by":"crossref","unstructured":"Kopp, S., Krenn, B., Marsella, S., Marshall, AN., Pelachaud, C., Pirker, H., Th\u00f3risson, KR., & Vilhj\u00e1lmsson, H. (2006). Towards a common framework for multimodal generation: The behavior markup language. In Intelligent virtual agents: 6th international conference, IVA 2006, Marina Del Rey, CA, USA, August 21\u201323, 2006. Proceedings 6, Springer (pp. 205\u2013217).","DOI":"10.1007\/11821830_17"},{"key":"2300_CR27","doi-asserted-by":"crossref","unstructured":"Kucherenko, T., Jonell, P., Van\u00a0Waveren, S., Henter, GE., Alexandersson, S., Leite, I., & Kjellstr\u00f6m, H. (2020). Gesticulator: A framework for semantically-aware speech-driven gesture generation. In Proceedings of the 2020 international conference on multimodal interaction (pp. 242\u2013250).","DOI":"10.1145\/3382507.3418815"},{"key":"2300_CR28","doi-asserted-by":"crossref","unstructured":"Kucherenko, T., Nagy, R., Yoon, Y., Woo, J., Nikolov, T., Tsakov, M., & Henter, GE. (2023). The genea challenge 2023: A large-scale evaluation of gesture generation models in monadic and dyadic settings. In Proceedings of the 25th international conference on multimodal interaction (pp. 792\u2013801).","DOI":"10.1145\/3577190.3616120"},{"key":"2300_CR29","doi-asserted-by":"crossref","unstructured":"Lee, G., Deng, Z., Ma, S., Shiratori, T., Srinivasa, SS., & Sheikh, Y. (2019). Talking with hands 16.2 m: A large-scale dataset of synchronized body-finger motion and audio for conversational motion analysis and synthesis. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 763\u2013772).","DOI":"10.1109\/ICCV.2019.00085"},{"key":"2300_CR30","doi-asserted-by":"crossref","unstructured":"Levine, S., Kr\u00e4henb\u00fchl, P., Thrun, S., & Koltun, V. (2010). Gesture controllers. In Acm siggraph 2010 papers (pp. 1\u201311).","DOI":"10.1145\/1833349.1778861"},{"key":"2300_CR31","unstructured":"Li, YA., Han, C., & Mesgarani ,N. (2022). Styletts: A style-based generative model for natural and diverse text-to-speech synthesis. arXiv preprint arXiv:2205.15439."},{"key":"2300_CR32","doi-asserted-by":"crossref","unstructured":"Liu, J., Wang, X., Fu, X., Chai, Y., Yu, C., Dai, J., & Han, J. (2023). Mfr-net: Multi-faceted responsive listening head generation via denoising diffusion model. In Proceedings of the 31st ACM international conference on multimedia (pp. 6734\u20136743).","DOI":"10.1145\/3581783.3612123"},{"key":"2300_CR33","doi-asserted-by":"crossref","unstructured":"Liu, X., Wu, Q., Zhou, H., Xu, Y., Qian, R., Lin, X., Zhou, X., Wu, W., Dai, B., & Zhou, B. (2022a). Learning hierarchical cross-modal association for co-speech gesture generation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 10462\u201310472).","DOI":"10.1109\/CVPR52688.2022.01021"},{"key":"2300_CR34","doi-asserted-by":"crossref","unstructured":"Liu, X., Xu, Y., Wu, Q., Zhou, H., Wu, W., & Zhou, B. (2022b). Semantic-aware implicit neural audio-driven video portrait generation. In European conference on computer vision (Springer, pp. 106\u2013125).","DOI":"10.1007\/978-3-031-19836-6_7"},{"issue":"6","key":"2300_CR35","first-page":"248:1","volume":"34","author":"M Loper","year":"2015","unstructured":"Loper, M., Mahmood, N., Romero, J., Pons-Moll, G., & Black, M. J. (2015). SMPL: A skinned multi-person linear model. ACM Transactions on Graphics, (Proc SIGGRAPH Asia), 34(6), 248:1-248:16.","journal-title":"ACM Transactions on Graphics, (Proc SIGGRAPH Asia)"},{"issue":"6","key":"2300_CR36","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3478513.3480484","volume":"40","author":"Y Lu","year":"2021","unstructured":"Lu, Y., Chai, J., & Cao, X. (2021). Live speech portraits: Real-time photorealistic talking-head animation. ACM Transactions on Graphics (TOG), 40(6), 1\u201317.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"2300_CR37","doi-asserted-by":"publisher","first-page":"498","DOI":"10.21437\/Interspeech.2017-1386","volume":"2017","author":"M McAuliffe","year":"2017","unstructured":"McAuliffe, M., Socolof, M., Mihuc, S., Wagner, M., & Sonderegger, M. (2017). Montreal forced aligner: Trainable text-speech alignment using kaldi. Interspeech, 2017, 498\u2013502.","journal-title":"Interspeech"},{"key":"2300_CR38","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Chung, J.S., & Zisserman, A. (2017). Voxceleb: A large-scale speaker identification dataset. arXiv preprint arXiv:1706.08612.","DOI":"10.21437\/Interspeech.2017-950"},{"key":"2300_CR39","doi-asserted-by":"crossref","unstructured":"Ng, E., Joo, H., Hu, L., Li, H., Darrell, T., Kanazawa, A., & Ginosar, S. (2022). Learning to listen: Modeling non-deterministic dyadic facial motion. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 20395\u201320405).","DOI":"10.1109\/CVPR52688.2022.01975"},{"key":"2300_CR40","doi-asserted-by":"crossref","unstructured":"Palmero, C., Selva, J., Smeureanu, S., Junior, J., Jacques, C., Clap\u00e9s, A., Mosegu\u00ed, A., Zhang, Z., Gallardo, D., Guilera, G., et\u00a0al. (2021). Context-aware personality inference in dyadic scenarios: Introducing the udiva dataset. In Proceedings of the IEEE\/CVF winter conference on applications of computer vision (pp. 1\u201312).","DOI":"10.1109\/WACVW52041.2021.00005"},{"key":"2300_CR41","unstructured":"Palmero, C., Barquero, G., Junior, JCJ., Clap\u00e9s, A., N\u00fanez, J., Curto, D., Smeureanu, S., Selva, J., Zhang, Z., Saeteros, D., et\u00a0al. (2022) Chalearn lap challenges on self-reported personality recognition and non-verbal behavior forecasting during social dyadic interactions: Dataset, design, and results. In Understanding social behavior in dyadic and small group interactions, PMLR (pp. 4\u201352)."},{"key":"2300_CR42","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Choutas, V., Ghorbani, N., Bolkart, T., Osman, AAA., Tzionas, D., & Black, M.J. (2019) Expressive body capture: 3d hands, face, and body from a single image. In Proceedings IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2019.01123"},{"key":"2300_CR43","doi-asserted-by":"crossref","unstructured":"Paysan, P., Knothe, R., Amberg, B., Romdhani, S., & Vetter, T. (2009). A 3d face model for pose and illumination invariant face recognition. In 2009 sixth IEEE international conference on advanced video and signal based surveillance (Ieee, pp. 296\u2013301).","DOI":"10.1109\/AVSS.2009.58"},{"key":"2300_CR44","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, MJ., & Varol, G. (2021a). Action-conditioned 3D human motion synthesis with transformer VAE. In International Conference on Computer Vision (ICCV).","DOI":"10.1109\/ICCV48922.2021.01080"},{"key":"2300_CR45","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, M.J., & Varol, G. (2021b). Action-conditioned 3d human motion synthesis with transformer vae. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 10985\u201310995).","DOI":"10.1109\/ICCV48922.2021.01080"},{"key":"2300_CR46","doi-asserted-by":"crossref","unstructured":"Qian, S., Tu, Z., Zhi, Y., Liu, W., & Gao, S. (2021). Speech drives templates: Co-speech gesture synthesis with learned templates. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 11077\u201311086).","DOI":"10.1109\/ICCV48922.2021.01089"},{"key":"2300_CR47","doi-asserted-by":"crossref","unstructured":"Richard, A., Zollh\u00f6fer, M., Wen, Y., De\u00a0la Torre, F., & Sheikh, Y. (2021). Meshtalk: 3d face animation from speech using cross-modality disentanglement. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 1173\u20131182).","DOI":"10.1109\/ICCV48922.2021.00121"},{"issue":"4","key":"2300_CR48","doi-asserted-by":"publisher","first-page":"543","DOI":"10.1016\/j.specom.2011.11.004","volume":"54","author":"M Sahidullah","year":"2012","unstructured":"Sahidullah, M., & Saha, G. (2012). Design, analysis and experimental evaluation of block based transformation in mfcc computation for speaker recognition. Speech Communication, 54(4), 543\u2013565.","journal-title":"Speech Communication"},{"issue":"8","key":"2300_CR49","doi-asserted-by":"publisher","first-page":"1330","DOI":"10.1109\/TPAMI.2007.70797","volume":"30","author":"ME Sargin","year":"2008","unstructured":"Sargin, M. E., Yemez, Y., Erzin, E., & Tekalp, A. M. (2008). Analysis of head gesture and prosody patterns for prosody-driven head-gesture animation. IEEE Transactions on Pattern Analysis and Machine Intelligence, 30(8), 1330\u20131345.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2300_CR50","doi-asserted-by":"crossref","unstructured":"Song, L., Yin, G., Jin, Z., Dong, X., & Xu, C. (2023a). Emotional listener portrait: Realistic listener motion simulation in conversation. arXiv preprint arXiv:2310.00068.","DOI":"10.1109\/ICCV51070.2023.01905"},{"key":"2300_CR51","doi-asserted-by":"crossref","unstructured":"Song, S., Spitale, M., Luo, C., Barquero, G., Palmero, C., Escalera, S., Valstar, M., Baur, T., Ringeval, F., Andre, E., et\u00a0al. (2023b). React2023: the first multi-modal multiple appropriate facial reaction generation challenge. arXiv preprint arXiv:2306.06583.","DOI":"10.1145\/3581783.3612832"},{"key":"2300_CR52","doi-asserted-by":"crossref","unstructured":"Sun, Q., Wang, Y., Zeng, A., Yin, W., Wei, C., Wang, W., Mei, H., Leung, CS., Liu, Z., Yang, L., et\u00a0al. (2024). Aios: All-in-one-stage expressive human pose and shape estimation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 1834\u20131843).","DOI":"10.1109\/CVPR52733.2024.00180"},{"issue":"4","key":"2300_CR53","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073640","volume":"36","author":"S Suwajanakorn","year":"2017","unstructured":"Suwajanakorn, S., Seitz, S. M., & Kemelmacher-Shlizerman, I. (2017). Synthesizing obama: Learning lip sync from audio. ACM Transactions on Graphics (ToG), 36(4), 1\u201313.","journal-title":"ACM Transactions on Graphics (ToG)"},{"key":"2300_CR54","doi-asserted-by":"crossref","unstructured":"Takeuchi, K., Kubota, S., Suzuki, K., Hasegawa, D., & Sakuta, H. (2017). Creating a gesture-speech dataset for speech-based automatic gesture generation. In: HCI International 2017\u2013Posters\u2019 Extended Abstracts: 19th International Conference, HCI International 2017, Vancouver, BC, Canada, July 9\u201314, 2017, Proceedings, Part I 19, Springer, pp 198\u2013202.","DOI":"10.1007\/978-3-319-58750-9_28"},{"key":"2300_CR55","doi-asserted-by":"crossref","unstructured":"Tuyen, NTV., & Celiktutan, O. (2022). Agree or disagree\u00c6\u2019 generating body gestures from affective contextual cues during dyadic interactions. In 2022 31st IEEE international conference on robot and human interactive communication (RO-MAN) (IEEE, pp. 1542\u20131547).","DOI":"10.1109\/RO-MAN53752.2022.9900760"},{"issue":"24","key":"2300_CR56","doi-asserted-by":"publisher","first-page":"1552","DOI":"10.1080\/01691864.2023.2279595","volume":"37","author":"NTV Tuyen","year":"2023","unstructured":"Tuyen, N. T. V., & Celiktutan, O. (2023). It takes two, not one: Context-aware nonverbal behaviour generation in dyadic interactions. Advanced Robotics, 37(24), 1552\u20131565.","journal-title":"Advanced Robotics"},{"key":"2300_CR57","doi-asserted-by":"crossref","unstructured":"Tuyen, N.T.V., Georgescu, A.L., Di\u00a0Giulio, I., & Celiktutan, O. (2023). A multimodal dataset for robot learning to imitate social human-human interaction. In Companion of the 2023 ACM\/IEEE international conference on human-robot interaction (pp. 238\u2013242).","DOI":"10.1145\/3568294.3580080"},{"key":"2300_CR58","unstructured":"Van Den\u00a0Oord, A., Vinyals, O., et\u00a0al. (2017). Neural discrete representation learning. Advances in Neural Information Processing Systems 30."},{"key":"2300_CR59","doi-asserted-by":"crossref","unstructured":"Wang, K., Wu, Q., Song, L., Yang, Z., Wu, W., Qian, C., He, R., Qiao, Y., & Loy, CC. (2020). Mead: A large-scale audio-visual dataset for emotional talking-face generation. In ECCV.","DOI":"10.1007\/978-3-030-58589-1_42"},{"key":"2300_CR60","doi-asserted-by":"crossref","unstructured":"Wang, T.C., Mallya, A., Liu, M.Y. (2021). One-shot free-view neural talking-head synthesis for video conferencing. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 10039\u201310049).","DOI":"10.1109\/CVPR46437.2021.00991"},{"key":"2300_CR61","doi-asserted-by":"crossref","unstructured":"Wu, X., Hu, P., Wu, Y., Lyu, X., Cao, Y.P., Shan, Y., Yang, W., Sun, Z., & Qi, X. (2023). Speech2lip: High-fidelity speech to lip generation by learning from a short video. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 22168\u201322177).","DOI":"10.1109\/ICCV51070.2023.02026"},{"key":"2300_CR62","unstructured":"Wuu, Ch., Zheng, N., Ardisson, S., Bali, R., Belko, D., Brockmeyer, E., Evans, L., Godisart, T., Ha, H., Huang, X., et\u00a0al. (2022). Multiface: A dataset for neural face rendering. arXiv preprint arXiv:2207.11243."},{"key":"2300_CR63","doi-asserted-by":"crossref","unstructured":"Xu, C., Zhu, J., Zhang, J., Han, Y., Chu, W., Tai, Y., Wang, C., Xie, Z., & Liu, Y. (2023). High-fidelity generalized emotional talking face generation with multi-modal emotion space learning. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 6609\u20136619).","DOI":"10.1109\/CVPR52729.2023.00639"},{"key":"2300_CR64","unstructured":"Ye, Z., Jiang, Z., Ren, Y., Liu, J., He, J., & Zhao, Z. (2023). Geneface: Generalized and high-fidelity audio-driven 3d talking face synthesis. arXiv preprint arXiv:2301.13430."},{"key":"2300_CR65","doi-asserted-by":"crossref","unstructured":"Yi, H., Liang, H., Liu, Y., Cao, Q., Wen, Y., Bolkart, T., Tao, D., & Black, M.J. (2023). Generating holistic 3d human motion from speech. In CVPR.","DOI":"10.1109\/CVPR52729.2023.00053"},{"key":"2300_CR66","doi-asserted-by":"crossref","unstructured":"Yin, L., Wang, Y., He, T., Liu, J., Zhao, W., Li, B., Jin, X., & Lin, J. (2023). Emog: Synthesizing emotive co-speech 3d gesture with diffusion model. arXiv preprint arXiv:2306.11496.","DOI":"10.2139\/ssrn.4818829"},{"key":"2300_CR67","doi-asserted-by":"crossref","unstructured":"Yoon, Y., Ko, WR., Jang, M., Lee, J., Kim, J., & Lee, G. (2019). Robots learn social skills: End-to-end learning of co-speech gesture generation for humanoid robots. In 2019 international conference on robotics and automation (ICRA) (IEEE, pp. 4303\u20134309).","DOI":"10.1109\/ICRA.2019.8793720"},{"issue":"6","key":"2300_CR68","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3414685.3417838","volume":"39","author":"Y Yoon","year":"2020","unstructured":"Yoon, Y., Cha, B., Lee, J. H., Jang, M., Lee, J., Kim, J., & Lee, G. (2020). Speech gesture generation from the trimodal context of text, audio, and speaker identity. ACM Transactions on Graphics (TOG), 39(6), 1\u201316.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"2300_CR69","doi-asserted-by":"crossref","unstructured":"Yoon, Y., Wolfert, P., Kucherenko, T., Viegas, C., Nikolov, T., Tsakov, M., & Henter, GE. (2022). The genea challenge 2022: A large evaluation of data-driven co-speech gesture generation. In Proceedings of the 2022 international conference on multimodal interaction (pp. 736\u2013747).","DOI":"10.1145\/3536221.3558058"},{"key":"2300_CR70","doi-asserted-by":"crossref","unstructured":"Zhang, H., Tian, Y., Zhang, Y., Li, M., An, L., Sun, Z., & Liu, Y. (2023a). Pymaf-x: Towards well-aligned full-body model regression from monocular images. IEEE Transactions on Pattern Analysis and Machine Intelligence.","DOI":"10.1109\/TPAMI.2023.3271691"},{"key":"2300_CR71","doi-asserted-by":"crossref","unstructured":"Zhang, J., Zhang, Y., Cun, X., Huang, S., Zhang, Y., Zhao, H., Lu, H., & Shen, X. (2023b). T2m-gpt: Generating human motion from textual descriptions with discrete representations. arXiv preprint arXiv:2301.06052.","DOI":"10.1109\/CVPR52729.2023.01415"},{"key":"2300_CR72","doi-asserted-by":"crossref","unstructured":"Zhang, W., Cun, X., Wang, X., Zhang, Y., Shen, X., Guo, Y., Shan, Y., & Wang, F. (2023c). Sadtalker: Learning realistic 3d motion coefficients for stylized audio-driven single image talking face animation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 8652\u20138661).","DOI":"10.1109\/CVPR52729.2023.00836"},{"key":"2300_CR73","doi-asserted-by":"crossref","unstructured":"Zhao, Q., Long, P., Zhang, Q., Qin, D., Liang, H., Zhang, L., Zhang, Y., Yu, J., & Xu, L. (2024). Media2face: Co-speech facial animation generation with multi-modality guidance. In ACM SIGGRAPH 2024 conference papers (pp. 1\u201313).","DOI":"10.1145\/3641519.3657413"},{"key":"2300_CR74","doi-asserted-by":"publisher","first-page":"582","DOI":"10.1007\/BF02943243","volume":"16","author":"F Zheng","year":"2001","unstructured":"Zheng, F., Zhang, G., & Song, Z. (2001). Comparison of different implementations of mfcc. Journal of Computer Science and Technology, 16, 582\u2013589.","journal-title":"Journal of Computer Science and Technology"},{"key":"2300_CR75","doi-asserted-by":"crossref","unstructured":"Zhi, Y., Cun, X., Chen, X., Shen, X., Guo, W., Huang, S., & Gao, S. (2023). Livelyspeaker: Towards semantic-aware co-speech gesture generation. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 20807\u201320817).","DOI":"10.1109\/ICCV51070.2023.01902"},{"key":"2300_CR76","doi-asserted-by":"crossref","unstructured":"Zhou, M., Bai, Y., Zhang, W., Yao, T., Zhao, T., & Mei, T. (2022). Responsive listening head generation: a benchmark dataset and baseline. In European conference on computer vision (Springer, pp. 124\u2013142).","DOI":"10.1007\/978-3-031-19839-7_8"},{"key":"2300_CR77","doi-asserted-by":"crossref","unstructured":"Zhu, L., Liu, X., Liu, X., Qian, R., Liu, Z., & Yu, L. (2023). Taming diffusion models for audio-driven co-speech gesture generation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 10544\u201310553).","DOI":"10.1109\/CVPR52729.2023.01016"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02300-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02300-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02300-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,17]],"date-time":"2025-04-17T06:02:28Z","timestamp":1744869748000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02300-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,17]]},"references-count":77,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2025,5]]}},"alternative-id":["2300"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02300-7","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,17]]},"assertion":[{"value":"29 March 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 November 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 December 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}