{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:42:35Z","timestamp":1777653755071,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":65,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,11,4]],"date-time":"2024-11-04T00:00:00Z","timestamp":1730678400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Dutch Research Council (NWO) through a Gravitation grant to the Language in Interaction consortium.","award":["024.001.006"],"award-info":[{"award-number":["024.001.006"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,4]]},"DOI":"10.1145\/3678957.3685707","type":"proceedings-article","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T04:35:53Z","timestamp":1730262953000},"page":"274-283","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Learning Co-Speech Gesture Representations in Dialogue through Contrastive Learning: An Intrinsic Evaluation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0603-9817","authenticated-orcid":false,"given":"Esam","family":"Ghaleb","sequence":"first","affiliation":[{"name":"Institute for Logic, Language and Computation (ILLC), University of Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1651-0657","authenticated-orcid":false,"given":"Bulat","family":"Khaertdinov","sequence":"additional","affiliation":[{"name":"Department of Advanced Computing Sciences, Maastricht University, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2729-6502","authenticated-orcid":false,"given":"Wim","family":"Pouw","sequence":"additional","affiliation":[{"name":"Radboud University, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1812-6907","authenticated-orcid":false,"given":"Marlou","family":"Rasenberg","sequence":"additional","affiliation":[{"name":"Meertens Institute, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0671-6651","authenticated-orcid":false,"given":"Judith","family":"Holler","sequence":"additional","affiliation":[{"name":"Radboud University, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4914-2643","authenticated-orcid":false,"given":"Asli","family":"Ozyurek","sequence":"additional","affiliation":[{"name":"Radboud University, Donders Institute, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5540-5943","authenticated-orcid":false,"given":"Raquel","family":"Fernandez","sequence":"additional","affiliation":[{"name":"Institute for Logic, Language and Computationa, University of Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,11,4]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Proceedings of the Annual Meeting of the Cognitive Science Society, Vol.\u00a046","author":"Akamine Sho","year":"2024","unstructured":"Sho Akamine, Esam Ghaleb, Marlou Rasenberg, Raquel Fern\u00e1ndez, Antje Meyer, and Asl\u0131 \u00d6zy\u00fcrek. 2024. Speakers align both their gestures and words not only to establish but also to maintain reference to create shared labels for novel objects in interaction. In Proceedings of the Annual Meeting of the Cognitive Science Society, Vol.\u00a046."},{"key":"e_1_3_2_2_2_1","volume-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems 33","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems 33 (2020), 12449\u201312460."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/82"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00422"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1080"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9412317"},{"key":"e_1_3_2_2_7_1","unstructured":"Kirsten Bergmann and Stefan Kopp. 2009. Increasing the expressiveness of virtual agents: autonomous generation of speech and gesture for spatial description tasks.. In AAMAS (1). 361\u2013368."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-12553-9_16"},{"key":"e_1_3_2_2_9_1","volume-title":"Proceedings of the Annual Meeting of the Cognitive Science Society, Vol.\u00a034","author":"Bergmann Kirsten","year":"2012","unstructured":"Kirsten Bergmann and Stefan Kopp. 2012. Gestural alignment in natural dialogue. In Proceedings of the Annual Meeting of the Cognitive Science Society, Vol.\u00a034."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475223"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN55064.2022.9892522"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.5555\/2655713.2655714"},{"key":"e_1_3_2_2_13_1","volume-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing. IEEE, 1\u20135.","author":"Chen W.","unstructured":"L.\u00a0W. Chen and A. Rudnicky. 2023. Exploring Wav2vec 2.0 Fine Tuning for Improved Speech Emotion Recognition. In ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing. IEEE, 1\u20135."},{"key":"e_1_3_2_2_14_1","volume-title":"International conference on machine learning. PMLR, 1597\u20131607","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. In International conference on machine learning. PMLR, 1597\u20131607."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1198"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","unstructured":"Maureen de Seyssel Marvin Lavechin Yossi Adi Emmanuel Dupoux and Guillaume Wisniewski. 2022. Probing phoneme language and speaker information in unsupervised speech representations. In Interspeech 2022-23rd INTERSPEECH Conference.","DOI":"10.21437\/Interspeech.2022-373"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuroimage.2022.119734"},{"key":"e_1_3_2_2_18_1","volume-title":"Leveraging Speech for Gesture Detection in Multimodal Communication. arXiv:2404.14952v1","author":"Ghaleb Esam","year":"2024","unstructured":"Esam Ghaleb, Ilya Burenko, Marlou Rasenberg, Wim Pouw, Ivan Toni, Peter Uhrig, Anna Wilson, Judith Holler, Asl\u0131 \u00d6zy\u00fcrek, and Raquel Fern\u00e1ndez. 2024. Leveraging Speech for Gesture Detection in Multimodal Communication. arXiv:2404.14952v1 (2024)."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00396"},{"key":"e_1_3_2_2_20_1","volume-title":"Proceedings of the Annual Meeting of the Cognitive Science Society, Vol.\u00a046","author":"Ghaleb Esam","year":"2024","unstructured":"Esam Ghaleb, Marlou Rasenberg, Wim Pouw, Ivan Toni, Judith Holler, Asl\u0131 \u00d6zy\u00fcrek, and Raquel Fern\u00e1ndez. 2024. Analysing Cross-Speaker Convergence in Face-to-Face Dialogue through the Lens of Automatically Detected Shared Linguistic Constructions. In Proceedings of the Annual Meeting of the Cognitive Science Society, Vol.\u00a046."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/THMS.2021.3086003"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19957"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10919-011-0105-6"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01568"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.3390\/technologies9010002"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW53098.2021.00380"},{"key":"e_1_3_2_2_30_1","volume-title":"MultiFacet: A Multi-Tasking Framework for Speech-to-Sign Language Generation. In Companion Publication of the 25th International Conference on Multimodal Interaction. 205\u2013213","author":"Kanakanti Mounika","year":"2023","unstructured":"Mounika Kanakanti, Shantanu Singh, and Manish Shrivastava. 2023. MultiFacet: A Multi-Tasking Framework for Speech-to-Sign Language Generation. In Companion Publication of the 25th International Conference on Multimodal Interaction. 205\u2013213."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511807572.007"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2013.09.014"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TBIOM.2020.2968216"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3308532.3329472"},{"key":"e_1_3_2_2_35_1","volume-title":"21st International Conference on Autonomous Agents and Multiagent Systems, AAMAS 2022","author":"Kucherenko Taras","year":"2022","unstructured":"Taras Kucherenko, Rajmund Nagy, Michael Neff, Hedvig Kjellstr\u00f6m, and Gustav\u00a0Eje Henter. 2022. Multimodal analysis of the predictability of hand-gesture properties. In 21st International Conference on Autonomous Agents and Multiagent Systems, AAMAS 2022, Auckland, New Zealand, May 9-13, 2022. ACM Press, 770\u2013779."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3461615.3485408"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00471"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT54892.2023.10023428"},{"key":"e_1_3_2_2_39_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 3304\u20133312","author":"Liu D.","unstructured":"D. Liu, L. Zhang, and Y. Wu. 2022. LD-ConGR: A large RGB-D video dataset for long-distance continuous gesture recognition. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 3304\u20133312."},{"key":"e_1_3_2_2_40_1","volume-title":"Beat: A large-scale semantic and emotional multi-modal dataset for conversational gestures synthesis. In European Conference on Computer Vision","author":"Liu H.","year":"2022","unstructured":"H. Liu, Z. Zhu, N. Iwamoto, Y. Peng, Z. Li, Y. Zhou, ..., and B. Zheng. 2022. Beat: A large-scale semantic and emotional multi-modal dataset for conversational gestures synthesis. In European Conference on Computer Vision. Springer Nature Switzerland, Cham, 612\u2013630."},{"key":"e_1_3_2_2_41_1","first-page":"11s","article-title":"BEAT","volume":"9","author":"Liu Haiyang","year":"2022","unstructured":"Haiyang Liu, Zihao Zhu, Naoya Iwamoto, Yichen Peng, Zhengqing Li, You Zhou, Elif Bozkurt, and Bo Zheng. 2022. BEAT: A Large-Scale Semantic and Emotional Multi-Modal Dataset for Conversational Gestures Synthesis: Supplementary Materials. Gesture 9, 10s (2022), 11s.","journal-title":"Gesture"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2974061"},{"key":"e_1_3_2_2_43_1","volume-title":"Hand and mind. Advances in Visual Semiotics 351","author":"McNeill David","year":"1992","unstructured":"David McNeill. 1992. Hand and mind. Advances in Visual Semiotics 351 (1992)."},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1017\/S1351324919000305"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"S. Nyatsanga T. Kucherenko C. Ahuja G.\u00a0E. Henter and M. Neff. 2023. A Comprehensive Review of Data\u2010Driven Co\u2010Speech Gesture Generation. In Computer Graphics Forum Vol.\u00a042. 569\u2013596.","DOI":"10.1111\/cgf.14776"},{"key":"e_1_3_2_2_46_1","volume-title":"What do self-supervised speech models know about words?Transactions of the Association for Computational Linguistics 12","author":"Pasad Ankita","year":"2024","unstructured":"Ankita Pasad, Chung-Ming Chien, Shane Settle, and Karen Livescu. 2024. What do self-supervised speech models know about words?Transactions of the Association for Computational Linguistics 12 (2024), 372\u2013391."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-703"},{"key":"e_1_3_2_2_48_1","volume-title":"Word Representation Learning in Multimodal Pre-Trained Transformers: An Intrinsic Evaluation. Transactions of the Association for Computational Linguistics (TACL)","author":"Pezzelle Sandro","year":"2021","unstructured":"Sandro Pezzelle, Ece Takmaz, and Raquel Fern\u00e1ndez. 2021. Word Representation Learning in Multimodal Pre-Trained Transformers: An Intrinsic Evaluation. Transactions of the Association for Computational Linguistics (TACL) (2021). https:\/\/direct.mit.edu\/tacl\/article-pdf\/doi\/10.1162\/tacl_a_00443\/1979754\/tacl_a_00443.pdf"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-77817-0_20"},{"key":"e_1_3_2_2_50_1","volume-title":"The quantification of gesture\u2013speech synchrony: A tutorial and validation of multimodal data acquisition using device-based and video-based motion tracking. Behavior research methods 52","author":"Pouw Wim","year":"2020","unstructured":"Wim Pouw, James\u00a0P Trujillo, and James\u00a0A Dixon. 2020. The quantification of gesture\u2013speech synchrony: A tutorial and validation of multimodal data acquisition using device-based and video-based motion tracking. Behavior research methods 52 (2020), 723\u2013740."},{"key":"e_1_3_2_2_51_1","volume-title":"International conference on machine learning. PMLR, 8748\u20138763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1080\/0163853X.2021.1992235"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-012-9356-9"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSEN.2020.2991741"},{"key":"e_1_3_2_2_56_1","volume-title":"What all do audio transformer models hear? probing acoustic representations for language delivery and its structure. arXiv preprint arXiv:2101.00387","author":"Shah Jui","year":"2021","unstructured":"Jui Shah, Yaman\u00a0Kumar Singla, Changyou Chen, and Rajiv\u00a0Ratn Shah. 2021. What all do audio transformer models hear? probing acoustic representations for language delivery and its structure. arXiv preprint arXiv:2101.00387 (2021)."},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475307"},{"key":"e_1_3_2_2_58_1","volume-title":"Proceedings, Part XI 16","author":"Tian Yonglong","year":"2020","unstructured":"Yonglong Tian, Dilip Krishnan, and Phillip Isola. 2020. Contrastive multiview coding. In Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XI 16. Springer, 776\u2013794."},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.3389\/fpsyg.2021.628728"},{"key":"e_1_3_2_2_60_1","volume-title":"Toward the markerless and automatic analysis of kinematic features: A toolkit for gesture and movement research. Behavior research methods 51","author":"Trujillo P","year":"2019","unstructured":"James\u00a0P Trujillo, Julija Vaitonyte, Irina Simanova, and Asli \u00d6zy\u00fcrek. 2019. Toward the markerless and automatic analysis of kinematic features: A toolkit for gesture and movement research. Behavior research methods 51 (2019), 769\u2013777."},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2017.371"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i6.28458"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/3414685.3417838"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2869278"}],"event":{"name":"ICMI '24: INTERNATIONAL CONFERENCE ON MULTIMODAL INTERACTION","location":"San Jose Costa Rica","acronym":"ICMI '24"},"container-title":["International Conference on Multimodel Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3678957.3685707","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3678957.3685707","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:10:12Z","timestamp":1750295412000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3678957.3685707"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,4]]},"references-count":65,"alternative-id":["10.1145\/3678957.3685707","10.1145\/3678957"],"URL":"https:\/\/doi.org\/10.1145\/3678957.3685707","relation":{},"subject":[],"published":{"date-parts":[[2024,11,4]]},"assertion":[{"value":"2024-11-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}