{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:04:14Z","timestamp":1750309454180,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,11,4]],"date-time":"2024-11-04T00:00:00Z","timestamp":1730678400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,4]]},"DOI":"10.1145\/3678957.3689332","type":"proceedings-article","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T04:35:53Z","timestamp":1730262953000},"page":"684-689","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Multimodal Emotion Recognition Harnessing the Complementarity of Speech, Language, and Vision"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8953-7872","authenticated-orcid":false,"given":"Thomas","family":"Thebaud","sequence":"first","affiliation":[{"name":"Center for Language and Speech Processing, Johns Hopkins University, USA and Human Language Technology Center of Excellence, Johns Hopkins University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8159-9841","authenticated-orcid":false,"given":"Anna","family":"Favaro","sequence":"additional","affiliation":[{"name":"Center for Language and Speech Processing, Johns Hopkins University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6263-3232","authenticated-orcid":false,"given":"Yaohan","family":"Guan","sequence":"additional","affiliation":[{"name":"Center for Language and Speech Processing, Johns Hopkins University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7697-5184","authenticated-orcid":false,"given":"Yuchen","family":"Yang","sequence":"additional","affiliation":[{"name":"Center for Language and Speech Processing, Johns Hopkins University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3779-3588","authenticated-orcid":false,"given":"Prabhav","family":"Singh","sequence":"additional","affiliation":[{"name":"Center for Language and Speech Processing, Johns Hopkins University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9459-8426","authenticated-orcid":false,"given":"Jesus","family":"Villalba","sequence":"additional","affiliation":[{"name":"Center for Language and Speech Processing, Johns Hopkins University, USA and Human Language Technology Center of Excellence, Johns Hopkins University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3033-7005","authenticated-orcid":false,"given":"Laureano","family":"Mono-Velazquez","sequence":"additional","affiliation":[{"name":"Center for Language and Speech Processing, Johns Hopkins University, USA and Human Language Technology Center of Excellence, Johns Hopkins University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4489-5753","authenticated-orcid":false,"given":"Najim","family":"Dehak","sequence":"additional","affiliation":[{"name":"Center for Language and Speech Processing, Johns Hopkins University, USA and Human Language Technology Center of Excellence, Johns Hopkins University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,11,4]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-021-09958-2"},{"key":"e_1_3_2_1_2_1","volume-title":"Common voice: A massively-multilingual speech corpus. arXiv preprint arXiv:1912.06670","author":"Ardila Rosana","year":"2019","unstructured":"Rosana Ardila, Megan Branson, Kelly Davis, Michael Henretty, Michael Kohler, Josh Meyer, Reuben Morais, Lindsay Saunders, Francis\u00a0M Tyers, and Gregor Weber. 2019. Common voice: A massively-multilingual speech corpus. arXiv preprint arXiv:1912.06670 (2019)."},{"key":"e_1_3_2_1_3_1","volume-title":"XLS-R: Self-supervised cross-lingual speech representation learning at scale. arXiv preprint arXiv:2111.09296","author":"Babu Arun","year":"2021","unstructured":"Arun Babu, Changhan Wang, Andros Tjandra, Kushal Lakhotia, Qiantong Xu, Naman Goyal, Kritika Singh, Patrick Von\u00a0Platen, Yatharth Saraf, Juan Pino, 2021. XLS-R: Self-supervised cross-lingual speech representation learning at scale. arXiv preprint arXiv:2111.09296 (2021)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2006.11477"},{"key":"e_1_3_2_1_5_1","volume-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems 33","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems 33 (2020), 12449\u201312460."},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML).","author":"Bertasius Gedas","year":"2021","unstructured":"Gedas Bertasius, Heng Wang, and Lorenzo Torresani. 2021. Is Space-Time Attention All You Need for Video Understanding?. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Qiong Cao Li Shen Weidi Xie Omkar\u00a0M. Parkhi and Andrew Zisserman. 2018. VGGFace2: A dataset for recognising faces across pose and age. arxiv:1710.08092\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1710.08092","DOI":"10.1109\/FG.2018.00020"},{"key":"e_1_3_2_1_8_1","unstructured":"Joao Carreira and Andrew Zisserman. 2018. Quo Vadis Action Recognition? A New Model and the Kinetics Dataset. arxiv:1705.07750\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1705.07750"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"e_1_3_2_1_10_1","volume-title":"Unsupervised cross-lingual representation learning for speech recognition. arXiv preprint arXiv:2006.13979","author":"Conneau Alexis","year":"2020","unstructured":"Alexis Conneau, Alexei Baevski, Ronan Collobert, Abdelrahman Mohamed, and Michael Auli. 2020. Unsupervised cross-lingual representation learning for speech recognition. arXiv preprint arXiv:2006.13979 (2020)."},{"key":"e_1_3_2_1_11_1","volume-title":"Unsupervised Cross-lingual Representation Learning at Scale. CoRR abs\/1911.02116","author":"Conneau Alexis","year":"2019","unstructured":"Alexis Conneau, Kartikay Khandelwal, Naman Goyal, Vishrav Chaudhary, Guillaume Wenzek, Francisco Guzm\u00e1n, Edouard Grave, Myle Ott, Luke Zettlemoyer, and Veselin Stoyanov. 2019. Unsupervised Cross-lingual Representation Learning at Scale. CoRR abs\/1911.02116 (2019). arXiv:1911.02116http:\/\/arxiv.org\/abs\/1911.02116"},{"key":"e_1_3_2_1_12_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. CoRR abs\/1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. CoRR abs\/1810.04805 (2018). arxiv:1810.04805http:\/\/arxiv.org\/abs\/1810.04805"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compbiomed.2023.107559"},{"key":"e_1_3_2_1_14_1","volume-title":"Language-agnostic BERT sentence embedding. arXiv preprint arXiv:2007.01852","author":"Feng Fangxiaoyu","year":"2020","unstructured":"Fangxiaoyu Feng, Yinfei Yang, Daniel Cer, Naveen Arivazhagan, and Wei Wang. 2020. Language-agnostic BERT sentence embedding. arXiv preprint arXiv:2007.01852 (2020)."},{"key":"e_1_3_2_1_15_1","volume-title":"THERADIA WoZ: An Ecological Corpus for Appraisal-based Affect Research in Healthcare. arXiv preprint arXiv:2405.06728","author":"Fournier Hippolyte","year":"2024","unstructured":"Hippolyte Fournier, Sina Alisamir, Safaa Azzakhnini, Hanna Chainay, Olivier Koenig, Isabella Zsoldos, El\u00e9eonore Tr\u00e2n, G\u00e9rard Bailly, Fr\u00e9d\u00e9eric Elisei, B\u00e9atrice Bouchot, 2024. THERADIA WoZ: An Ecological Corpus for Appraisal-based Affect Research in Healthcare. arXiv preprint arXiv:2405.06728 (2024)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2022.3216993"},{"key":"e_1_3_2_1_17_1","volume-title":"Whisper-at: Noise-robust automatic speech recognizers are also strong general audio event taggers. arXiv preprint arXiv:2307.03183","author":"Gong Yuan","year":"2023","unstructured":"Yuan Gong, Sameer Khurana, Leonid Karlinsky, and James Glass. 2023. Whisper-at: Noise-robust automatic speech recognizers are also strong general audio event taggers. arXiv preprint arXiv:2307.03183 (2023)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU57964.2023.10389742"},{"key":"e_1_3_2_1_19_1","volume-title":"Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100","author":"Gulati Anmol","year":"2020","unstructured":"Anmol Gulati, James Qin, Chung-Cheng Chiu, Niki Parmar, Yu Zhang, Jiahui Yu, Wei Han, Shibo Wang, Zhengdong Zhang, Yonghui Wu, 2020. Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_21_1","volume-title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units","author":"Hsu Wei-Ning","year":"2021","unstructured":"Wei-Ning Hsu, Benjamin Bolte, Yao-Hung\u00a0Hubert Tsai, Kushal Lakhotia, Ruslan Salakhutdinov, and Abdelrahman Mohamed. 2021. Hubert: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM transactions on audio, speech, and language processing 29 (2021), 3451\u20133460."},{"key":"e_1_3_2_1_22_1","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev Mustafa Suleyman and Andrew Zisserman. 2017. The Kinetics Human Action Video Dataset. arxiv:1705.06950\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1705.06950"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Adrien Lafore Cl\u00e9ment Pag\u00e9s Leila Moudjari Sebasti\u00e3o Quintas Herv\u00e9 Bredin Thomas Pellegrini Farah Benamara Isabelle Ferran\u00e9 J\u00e9r\u00f4me Bertrand Marie-Fran\u00e7oise Bertrand 2024. IRIT-MFU Multi-modal systems for emotion classification for Odyssey 2024 challenge. In Odyssey 2024: The Speaker and Language Recognition Workshop.","DOI":"10.21437\/odyssey.2024-42"},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of The 12th Language Resources and Evaluation Conference. European Language Resources Association","author":"Le Hang","year":"2020","unstructured":"Hang Le, Lo\u00efc Vial, Jibril Frej, Vincent Segonne, Maximin Coavoux, Benjamin Lecouteux, Alexandre Allauzen, Beno\u00eet Crabb\u00e9, Laurent Besacier, and Didier Schwab. 2020. FlauBERT: Unsupervised Language Model Pre-training for French. In Proceedings of The 12th Language Resources and Evaluation Conference. European Language Resources Association, Marseille, France, 2479\u20132490. https:\/\/www.aclweb.org\/anthology\/2020.lrec-1.302"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.645"},{"key":"e_1_3_2_1_26_1","unstructured":"OpenAI Josh Achiam et al. 2024. GPT-4 Technical Report. arxiv:2303.08774\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054317"},{"key":"e_1_3_2_1_28_1","volume-title":"International conference on machine learning. PMLR, 28492\u201328518","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International conference on machine learning. PMLR, 28492\u201328518."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_1_30_1","volume-title":"EVAC 2024 \u2013 Empathic Virtual Agent Challenge: Appraisal-based Recognition of Affective States. In Proceedings of the 26th International Conference on Multimodal Interaction (ICMI","author":"Ringeval Fabien","year":"2024","unstructured":"Fabien Ringeval, Bj\u00f6rn Schuller, G\u00e9rard Bailly, Safaa Azzakhnini, and Hippolyte Fournier. 2024. EVAC 2024 \u2013 Empathic Virtual Agent Challenge: Appraisal-based Recognition of Affective States. In Proceedings of the 26th International Conference on Multimodal Interaction (ICMI 2024). ACM, San Jos\u00e9, Costa Rica."},{"key":"e_1_3_2_1_31_1","volume-title":"a distilled version of BERT: smaller, faster, cheaper and lighter. ArXiv abs\/1910.01108","author":"Sanh Victor","year":"2019","unstructured":"Victor Sanh, Lysandre Debut, Julien Chaumond, and Thomas Wolf. 2019. DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter. ArXiv abs\/1910.01108 (2019)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-1242"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","unstructured":"Joel Shor and Subhashini Venugopalan. 2022. TRILLsson: Distilling Universal Paralinguistic Speech Representations. https:\/\/doi.org\/10.21437\/interspeech.2022-118","DOI":"10.21437\/interspeech.2022-118"},{"key":"e_1_3_2_1_34_1","volume-title":"Multi-task French speech analysis with deep learning Emotion recognition and speaker diarization models for end-to-end conversational analysis tool. unknown","author":"Sintes Jules","year":"2023","unstructured":"Jules Sintes. 2023. Multi-task French speech analysis with deep learning Emotion recognition and speaker diarization models for end-to-end conversational analysis tool. unknown (2023)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Sergey Ioffe Vincent Vanhoucke and Alex Alemi. 2016. Inception-v4 Inception-ResNet and the Impact of Residual Connections on Learning. arxiv:1602.07261\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1602.07261","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"e_1_3_2_1_37_1","first-page":"2023","article-title":"Silero vad: pre-trained enterprise-grade voice activity detector (vad), number detector and language classifier","volume":"31","author":"Team Silero","year":"2021","unstructured":"Silero Team. 2021. Silero vad: pre-trained enterprise-grade voice activity detector (vad), number detector and language classifier. Retrieved March 31 (2021), 2023.","journal-title":"Retrieved March"},{"key":"e_1_3_2_1_38_1","volume-title":"VoxPopuli: A large-scale multilingual speech corpus for representation learning, semi-supervised learning and interpretation. arXiv preprint arXiv:2101.00390","author":"Wang Changhan","year":"2021","unstructured":"Changhan Wang, Morgane Riviere, Ann Lee, Anne Wu, Chaitanya Talnikar, Daniel Haziza, Mary Williamson, Juan Pino, and Emmanuel Dupoux. 2021. VoxPopuli: A large-scale multilingual speech corpus for representation learning, semi-supervised learning and interpretation. arXiv preprint arXiv:2101.00390 (2021)."},{"key":"e_1_3_2_1_39_1","unstructured":"Liang Wang Nan Yang Xiaolong Huang Linjun Yang Rangan Majumder and Furu Wei. 2024. Improving Text Embeddings with Large Language Models. arxiv:2401.00368\u00a0[cs.CL]"},{"key":"e_1_3_2_1_40_1","volume-title":"EMO-SUPERB: An in-depth look at speech emotion recognition. arXiv preprint arXiv:2402.13018","author":"Wu Haibin","year":"2024","unstructured":"Haibin Wu, Huang-Cheng Chou, Kai-Wei Chang, Lucas Goncalves, Jiawei Du, Jyh-Shing\u00a0Roger Jang, Chi-Chun Lee, and Hung-Yi Lee. 2024. EMO-SUPERB: An in-depth look at speech emotion recognition. arXiv preprint arXiv:2402.13018 (2024)."},{"key":"e_1_3_2_1_41_1","volume-title":"Superb: Speech processing universal performance benchmark. arXiv preprint arXiv:2105.01051","author":"Chi Po-Han","year":"2021","unstructured":"Shu-wen Yang, Po-Han Chi, Yung-Sung Chuang, Cheng-I\u00a0Jeff Lai, Kushal Lakhotia, Yist\u00a0Y Lin, Andy\u00a0T Liu, Jiatong Shi, Xuankai Chang, Guan-Ting Lin, 2021. Superb: Speech processing universal performance benchmark. arXiv preprint arXiv:2105.01051 (2021)."},{"volume-title":"Multimodal speech emotion recognition using audio and text. In 2018 IEEE spoken language technology workshop (SLT)","author":"Yoon Seunghyun","key":"e_1_3_2_1_42_1","unstructured":"Seunghyun Yoon, Seokhyun Byun, and Kyomin Jung. 2018. Multimodal speech emotion recognition using audio and text. In 2018 IEEE spoken language technology workshop (SLT). IEEE, 112\u2013118."}],"event":{"name":"ICMI '24: INTERNATIONAL CONFERENCE ON MULTIMODAL INTERACTION","acronym":"ICMI '24","location":"San Jose Costa Rica"},"container-title":["International Conference on Multimodel Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3678957.3689332","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3678957.3689332","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:10:13Z","timestamp":1750295413000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3678957.3689332"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,4]]},"references-count":42,"alternative-id":["10.1145\/3678957.3689332","10.1145\/3678957"],"URL":"https:\/\/doi.org\/10.1145\/3678957.3689332","relation":{},"subject":[],"published":{"date-parts":[[2024,11,4]]},"assertion":[{"value":"2024-11-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}