{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T12:50:56Z","timestamp":1784551856276,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,5,6]],"date-time":"2025-05-06T00:00:00Z","timestamp":1746489600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Marie Sk?odowska-Curie grant (CLIPE project)","award":["860768"],"award-info":[{"award-number":["860768"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,7]]},"DOI":"10.1145\/3722564.3728374","type":"proceedings-article","created":{"date-parts":[[2025,5,5]],"date-time":"2025-05-05T11:20:22Z","timestamp":1746444022000},"page":"1-3","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Evaluating Speech and Video Models for Face-Body Congruence"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7414-845X","authenticated-orcid":false,"given":"Kiran","family":"Chhatre","sequence":"first","affiliation":[{"name":"KTH Royal Institute of Technolgy, Stockholm, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1206-5701","authenticated-orcid":false,"given":"Renan","family":"Guarese","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technolgy, Stockholm, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6571-0623","authenticated-orcid":false,"given":"Andrii","family":"Matviienko","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technolgy, Stockholm, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7257-0761","authenticated-orcid":false,"given":"Christopher","family":"Peters","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technolgy, Stockholm, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,5,6]]},"reference":[{"key":"e_1_3_3_3_2_1","doi-asserted-by":"publisher","unstructured":"Simon Alexanderson Rajmund Nagy Jonas Beskow and Gustav\u00a0Eje Henter. 2023. Listen Denoise Action! Audio-Driven Motion Synthesis with Diffusion Models. ACM Trans. Graph. 42 4 (2023) 1\u201320. 10.1145\/3592458","DOI":"10.1145\/3592458"},{"key":"e_1_3_3_3_3_1","unstructured":"Aggelina Chatziagapi Louis-Philippe Morency Hongyu Gong Michael Zollh\u00f6fer Dimitris Samaras and Alexander Richard. 2025. AV-Flow: Transforming Text to Audio-Visual Human-like Interactions. arXiv preprint arXiv:2502.13133 (2025)."},{"key":"e_1_3_3_3_4_1","first-page":"1942","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Chhatre Kiran","year":"2024","unstructured":"Kiran Chhatre, Radek Dan\u011b\u010dek, Nikos Athanasiou, Giorgio Becherini, Christopher Peters, Michael\u00a0J. Black, and Timo Bolkart. 2024. AMUSE: Emotional Speech-driven 3D Body Animation via Disentangled Latent Diffusion. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 1942\u20131953. https:\/\/amuse.is.tue.mpg.de"},{"key":"e_1_3_3_3_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01034"},{"key":"e_1_3_3_3_6_1","doi-asserted-by":"publisher","unstructured":"Radek Dan\u011b\u010dek Kiran Chhatre Shashank Tripathi Yandong Wen Michael Black and Timo Bolkart. 2023. Emotional Speech-Driven Animation with Content-Emotion Disentanglement. ACM. 10.1145\/3610548.3618183","DOI":"10.1145\/3610548.3618183"},{"key":"e_1_3_3_3_7_1","volume-title":"GENEA: Generation and Evaluation of Non-verbal Behaviour for Embodied Agents Workshop 2024","author":"Deichler Anna","year":"2024","unstructured":"Anna Deichler, Jonas Beskow, and Axel\u00a0Wiebe Werner. 2024. Gesture Evaluation in Virtual Reality. In GENEA: Generation and Evaluation of Non-verbal Behaviour for Embodied Agents Workshop 2024. https:\/\/openreview.net\/forum?id=1G4fIPocY2"},{"key":"e_1_3_3_3_8_1","unstructured":"Alexey Dosovitskiy. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_3_3_9_1","unstructured":"Yingruo Fan Zhaojiang Lin Jun Saito Wenping Wang and Taku Komura. 2021. FaceFormer: Speech-Driven 3D Facial Animation with Transformers. arXiv preprint arXiv:2112.05329 (2021)."},{"key":"e_1_3_3_3_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01821"},{"key":"e_1_3_3_3_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV53792.2021.00088"},{"key":"e_1_3_3_3_12_1","doi-asserted-by":"crossref","unstructured":"Yao Feng Haiwen Feng Michael\u00a0J. Black and Timo Bolkart. 2021b. Learning an Animatable Detailed 3D Face Model from In-the-Wild Images. ACM Transactions on Graphics (ToG) Proc. SIGGRAPH 40 4 (Aug. 2021) 88:1\u201388:13.","DOI":"10.1145\/3450626.3459936"},{"key":"e_1_3_3_3_13_1","doi-asserted-by":"publisher","unstructured":"Alan Fraser Isabella Branson Ross Hollett Craig Speelman and Shane Rogers. 2022. Expressiveness of real-time motion captured avatars influences perceived animation realism and perceived quality of social interaction in virtual reality. Frontiers in Virtual Reality 3 (12 2022) 981400. 10.3389\/frvir.2022.981400","DOI":"10.3389\/frvir.2022.981400"},{"key":"e_1_3_3_3_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00361"},{"key":"e_1_3_3_3_15_1","doi-asserted-by":"crossref","unstructured":"Ikhsanul Habibie Mohamed\u00a0A. Elgharib Kripasindhu Sarkar Ahsan Abdullah Simbarashe\u00a0Linval Nyatsanga Michael Neff and Christian Theobalt. 2022. A Motion Matching-based Framework for Controllable Gesture Synthesis from Speech. International Conference on Computer Graphics and Interactive Techniques (SIGGRAPH) (2022).","DOI":"10.1145\/3528233.3530750"},{"key":"e_1_3_3_3_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3577190.3616120"},{"key":"e_1_3_3_3_17_1","unstructured":"Jing Li Di Kang Wenjie Pei Xuefei Zhe Ying Zhang Linchao Bao and Zhenyu He. 2023. Audio2Gestures: Generating Diverse Gestures from Audio. arxiv:2301.06690\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2301.06690"},{"key":"e_1_3_3_3_18_1","unstructured":"Ruilong Li Shan Yang David\u00a0A. Ross and Angjoo Kanazawa. 2021. AI Choreographer: Music Conditioned 3D Dance Generation with AIST++. arxiv:2101.08779\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2101.08779"},{"key":"e_1_3_3_3_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01022"},{"key":"e_1_3_3_3_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00115"},{"key":"e_1_3_3_3_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00138"},{"key":"e_1_3_3_3_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00101"},{"key":"e_1_3_3_3_23_1","doi-asserted-by":"publisher","unstructured":"Miles\u00a0L. Patterson Alan\u00a0J. Fridlund and Carlos Crivelli. 2023. Four Misconceptions About Nonverbal Communication. Perspectives on Psychological Science 18 6 (2023) 1388\u20131411. 10.1177\/17456916221148142 arXiv:https:\/\/doi.org\/10.1177\/17456916221148142 PMID: 36791676.","DOI":"10.1177\/17456916221148142"},{"key":"e_1_3_3_3_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01123"},{"key":"e_1_3_3_3_25_1","unstructured":"Hai\u00a0Xuan Pham Yuting Wang and Vladimir Pavlovic. 2017. End-to-end Learning for 3D Facial Animation from Raw Waveforms of Speech. CoRR abs\/1710.00920 (2017). arXiv:1710.00920http:\/\/arxiv.org\/abs\/1710.00920"},{"key":"e_1_3_3_3_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00121"},{"key":"e_1_3_3_3_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_3_28_1","doi-asserted-by":"publisher","unstructured":"Felix Sharkov V. Silkin and O. Kireeva. 2022. Non-verbal signs of personality: Communicative meanings of facial expressions. RUDN Journal of Sociology 22 (06 2022) 387\u2013403. 10.22363\/2313-2272-2022-22-2-387-403","DOI":"10.22363\/2313-2272-2022-22-2-387-403"},{"key":"e_1_3_3_3_29_1","unstructured":"Mingyi Shi Dafei Qin Leo Ho Zhouyingcheng Liao Yinghao Huang Junichi Yamagishi and Taku Komura. 2024. It Takes Two: Real-time Co-Speech Two-person\u2019s Interaction Generation via Reactive Auto-regressive Diffusion Model. arxiv:2412.02419\u00a0[cs.SD] https:\/\/arxiv.org\/abs\/2412.02419"},{"key":"e_1_3_3_3_30_1","doi-asserted-by":"publisher","unstructured":"Noa Simhi and Galit Yovel. 2020. Independent contributions of the face body and gait to the representation of the whole person. Attention Perception & Psychophysics 83 (10 2020) 1\u201316. 10.3758\/s13414-020-02110-2","DOI":"10.3758\/s13414-020-02110-2"},{"key":"e_1_3_3_3_31_1","doi-asserted-by":"publisher","unstructured":"Chloe Stewart Derek Mitchell Stephen Pasternak Paul Tremblay and Elizabeth Finger. 2024. The nonverbal expression of guilt in healthy adults. Scientific Reports 14 (05 2024). 10.1038\/s41598-024-60980-0","DOI":"10.1038\/s41598-024-60980-0"},{"key":"e_1_3_3_3_32_1","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems (2017)."},{"key":"e_1_3_3_3_33_1","doi-asserted-by":"publisher","unstructured":"Jinbo Xing Menghan Xia Yuechen Zhang Xiaodong Cun Jue Wang and Tien-Tsin Wong. 2023. CodeTalker: Speech-Driven 3D Facial Animation with Discrete Motion Prior. (2023) 12780\u201312790. 10.1109\/CVPR52729.2023.01229","DOI":"10.1109\/CVPR52729.2023.01229"},{"key":"e_1_3_3_3_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00230"},{"key":"e_1_3_3_3_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00053"},{"key":"e_1_3_3_3_36_1","doi-asserted-by":"crossref","unstructured":"Youngwoo Yoon Bok Cha Joo-Haeng Lee Minsu Jang Jaeyeon Lee Jaehong Kim and Geehyuk Lee. 2020. Speech Gesture Generation from the Trimodal Context of Text Audio and Speaker Identity. ACM Transactions on Graphics 39 6 (2020).","DOI":"10.1145\/3414685.3417838"}],"event":{"name":"I3D '25: Companion Proceedings of the ACM SIGGRAPH Symposium on Interactive 3D Graphics and Games","location":"Jersey City NJ USA","acronym":"I3D '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Companion Proceedings of the ACM SIGGRAPH Symposium on Interactive 3D Graphics and Games"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3722564.3728374","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:40Z","timestamp":1750295920000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3722564.3728374"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,6]]},"references-count":35,"alternative-id":["10.1145\/3722564.3728374","10.1145\/3722564"],"URL":"https:\/\/doi.org\/10.1145\/3722564.3728374","relation":{},"subject":[],"published":{"date-parts":[[2025,5,6]]},"assertion":[{"value":"2025-05-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}