{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,23]],"date-time":"2025-10-23T01:09:21Z","timestamp":1761181761862,"version":"build-2065373602"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","funder":[{"name":"Guangdong Province Ordinary Colleges and Universities Young Innovative Talents Project","award":["2023KQNCX036"],"award-info":[{"award-number":["2023KQNCX036"]}]},{"name":"University-Industry Collaborative Education Program of Ministry of Education","award":["241003632084003"],"award-info":[{"award-number":["241003632084003"]}]},{"name":"Guangdong Provincial Higher Education Teaching Research and Reform Project","award":["202430803"],"award-info":[{"award-number":["202430803"]}]},{"name":"Guangdong Higher Vocational Education Teaching Reform Research and Practice Project &#x28;Undergraduate Pilot Program Reform Project for Vocational Colleges&#x29;","award":["2023QNJS02"],"award-info":[{"award-number":["2023QNJS02"]}]},{"name":"Research Fund of Guangdong Polytechnic Normal University","award":["2022SDKYA015"],"award-info":[{"award-number":["2022SDKYA015"]}]},{"name":"Special Fund for Science and Technology Innovation Strategy of Guangdong Province &#x28;Climbing Plan&#x29;","award":["pdjh2024a226"],"award-info":[{"award-number":["pdjh2024a226"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746270.3760223","type":"proceedings-article","created":{"date-parts":[[2025,10,20]],"date-time":"2025-10-20T15:14:09Z","timestamp":1760973249000},"page":"41-50","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["DARE to Disagree: A Multi-Agent Adversarial Debate Framework for Open-Vocabulary Multimodal Emotion Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-4028-448X","authenticated-orcid":false,"given":"Yuesheng","family":"Huang","sequence":"first","affiliation":[{"name":"School of Computer Science, Guangdong Polytechnic Normal University, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1582-419X","authenticated-orcid":false,"given":"Meiqi","family":"Feng","sequence":"additional","affiliation":[{"name":"School of Computer Science, Guangdong Polytechnic Normal University, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3332-208X","authenticated-orcid":false,"given":"Zhenming","family":"He","sequence":"additional","affiliation":[{"name":"School of Computer Science, Guangdong Polytechnic Normal University, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6024-4978","authenticated-orcid":false,"given":"Yueyuan","family":"Peng","sequence":"additional","affiliation":[{"name":"School of Computer Science, Guangdong Polytechnic Normal University, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8586-9535","authenticated-orcid":false,"given":"Jiawen","family":"Li","sequence":"additional","affiliation":[{"name":"School of Computer Science, Guangdong Polytechnic Normal University, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,26]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10115-020-01449-0"},{"key":"e_1_3_2_1_2_1","volume-title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens. arXiv preprint arXiv:2404.03413","author":"Ataallah Kirolos","year":"2024","unstructured":"Kirolos Ataallah, Xiaoqian Shen, Eslam Abdelrahman, Essam Sleiman, Deyao Zhu, Jian Ding, and Mohamed Elhoseiny. 2024. MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens. arXiv preprint arXiv:2404.03413 (2024)."},{"key":"e_1_3_2_1_3_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_4_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems Vol. 33 (2020) 1877-1901."},{"key":"e_1_3_2_1_5_1","volume-title":"IEMOCAP: Interactive emotional dyadic motion capture database. Language resources and evaluation","author":"Busso Carlos","year":"2008","unstructured":"Carlos Busso, Murtaza Bulut, Chi-Chun Lee, Abe Kazemzadeh, Emily Mower, Samuel Kim, Jeannette N Chang, Sungbok Lee, and Shrikanth S Narayanan. 2008. IEMOCAP: Interactive emotional dyadic motion capture database. Language resources and evaluation, Vol. 42 (2008), 335-359."},{"key":"e_1_3_2_1_6_1","volume-title":"Affective Computing","author":"Campeau S","year":"1997","unstructured":"S Campeau, WA Falls, and W Cullinan. [n.d.]. Picard, RW (1997), Affective Computing. Cambridge."},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al., 2024. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198."},{"key":"e_1_3_2_1_8_1","first-page":"110805","article-title":"Emotion-llama: Multimodal emotion recognition and reasoning with instruction tuning","volume":"37","author":"Cheng Zebang","year":"2024","unstructured":"Zebang Cheng, Zhi-Qi Cheng, Jun-Yan He, Kai Wang, Yuxiang Lin, Zheng Lian, Xiaojiang Peng, and Alexander Hauptmann. 2024a. Emotion-llama: Multimodal emotion recognition and reasoning with instruction tuning. Advances in Neural Information Processing Systems, Vol. 37 (2024), 110805-110853.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_9_1","unstructured":"Zesen Cheng Sicong Leng Hang Zhang Yifei Xin Xin Li Guanzheng Chen Yongxin Zhu Wenqi Zhang Ziyang Luo Deli Zhao et al. 2024b. Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms. arXiv preprint arXiv:2406.07476 (2024)."},{"key":"e_1_3_2_1_10_1","article-title":"Natural language processing (almost) from scratch","volume":"12","author":"Collobert Ronan","year":"2011","unstructured":"Ronan Collobert, Jason Weston, L\u00e9on Bottou, Michael Karlen, Koray Kavukcuoglu, and Pavel Kuksa. 2011. Natural language processing (almost) from scratch. Journal of machine learning research, Vol. 12, 7 (2011).","journal-title":"Journal of machine learning research"},{"key":"e_1_3_2_1_11_1","volume-title":"GoEmotions: A dataset of fine-grained emotions. arXiv preprint arXiv:2005.00547","author":"Demszky Dorottya","year":"2020","unstructured":"Dorottya Demszky, Dana Movshovitz-Attias, Jeongwoo Ko, Alan Cowen, Gaurav Nemade, and Sujith Ravi. 2020. GoEmotions: A dataset of fine-grained emotions. arXiv preprint arXiv:2005.00547 (2020)."},{"key":"e_1_3_2_1_12_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Du Yilun","year":"2023","unstructured":"Yilun Du, Shuang Li, Antonio Torralba, Joshua B Tenenbaum, and Igor Mordatch. 2023. Improving factuality and reasoning in language models through multiagent debate. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_13_1","volume-title":"An argument for basic emotions. Cognition & emotion","author":"Ekman Paul","year":"1992","unstructured":"Paul Ekman. 1992. An argument for basic emotions. Cognition & emotion, Vol. 6, 3-4 (1992), 169-200."},{"key":"e_1_3_2_1_14_1","volume-title":"Survey on speech emotion recognition: Features, classification schemes, and databases. Pattern recognition","author":"Ayadi Moataz El","year":"2011","unstructured":"Moataz El Ayadi, Mohamed S Kamel, and Fakhri Karray. 2011. Survey on speech emotion recognition: Features, classification schemes, and databases. Pattern recognition, Vol. 44, 3 (2011), 572-587."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2012.06.016"},{"key":"e_1_3_2_1_16_1","volume-title":"Can Generated Images Serve as a Viable Modality for Text-Centric Multimodal Learning? arXiv preprint arXiv:2506.17623","author":"Huang Yuesheng","year":"2025","unstructured":"Yuesheng Huang, Peng Zhang, Riliang Liu, and Jiaqi Liang. 2025. Can Generated Images Serve as a Viable Modality for Text-Centric Multimodal Learning? arXiv preprint arXiv:2506.17623 (2025)."},{"key":"e_1_3_2_1_17_1","unstructured":"Aaron Hurst Adam Lerer Adam P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et al. 2024. Gpt-4o system card. arXiv preprint arXiv:2410.21276 (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01300"},{"key":"e_1_3_2_1_19_1","volume-title":"Llava-onevision: Easy visual task transfer. arXiv preprint arXiv:2408.03326","author":"Li Bo","year":"2024","unstructured":"Bo Li, Yuanhan Zhang, Dong Guo, Renrui Zhang, Feng Li, Hao Zhang, Kaichen Zhang, Peiyuan Zhang, Yanwei Li, Ziwei Liu, et al., 2024c. Llava-onevision: Easy visual task transfer. arXiv preprint arXiv:2408.03326 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Materials & Continua","volume":"80","author":"Li Jiawen","year":"2024","unstructured":"Jiawen Li, Yuesheng Huang, Yayi Lu, Leijun Wang, Yongqi Ren, and Rongjun Chen. 2024a. Sentiment Analysis Using E-Commerce Review Keyword-Generated Image with a Hybrid Machine Learning-Based Model. Computers, Materials & Continua, Vol. 80, 1 (2024)."},{"key":"e_1_3_2_1_21_1","volume-title":"Videochat: Chat-centric video understanding. arXiv preprint arXiv:2305.06355","author":"Li KunChang","year":"2023","unstructured":"KunChang Li, Yinan He, Yi Wang, Yizhuo Li, Wenhai Wang, Ping Luo, Yali Wang, Limin Wang, and Yu Qiao. 2023. Videochat: Chat-centric video understanding. arXiv preprint arXiv:2305.06355 (2023)."},{"key":"e_1_3_2_1_22_1","volume-title":"Deep facial expression recognition: A survey","author":"Li Shan","year":"2020","unstructured":"Shan Li and Weihong Deng. 2020. Deep facial expression recognition: A survey. IEEE transactions on affective computing, Vol. 13, 3 (2020), 1195-1215."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.277"},{"key":"e_1_3_2_1_24_1","volume-title":"European Conference on Computer Vision. Springer, 323-340","author":"Li Yanwei","year":"2024","unstructured":"Yanwei Li, Chengyao Wang, and Jiaya Jia. 2024b. Llama-vid: An image is worth 2 tokens in large language models. In European Conference on Computer Vision. Springer, 323-340."},{"key":"e_1_3_2_1_25_1","unstructured":"Zheng Lian Haoyu Chen Lan Chen Haiyang Sun Licai Sun Yong Ren Zebang Cheng Bin Liu Rui Liu Xiaojiang Peng et al. 2025a. AffectGPT: A New Dataset Model and Benchmark for Emotion Understanding with Multimodal Large Language Models. arXiv preprint arXiv:2501.16566 (2025)."},{"key":"e_1_3_2_1_26_1","unstructured":"Zheng Lian Rui Liu Kele Xu Bin Liu Xuefei Liu Yazhou Zhang Xin Liu Yong Li Zebang Cheng Haolin Zuo et al. 2025b. Mer 2025: When affective computing meets large language models. arXiv preprint arXiv:2504.19423 (2025)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612836"},{"key":"e_1_3_2_1_28_1","unstructured":"Zheng Lian Haiyang Sun Licai Sun Lan Chen Haoyu Chen Hao Gu Zhuofan Wen Shun Chen Siyuan Zhang Hailiang Yao et al. 2024a. Open-vocabulary Multimodal Emotion Recognition: Dataset Metric and Benchmark. arXiv preprint arXiv:2410.01495 (2024)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689092.3689959"},{"key":"e_1_3_2_1_30_1","volume-title":"Video-llava: Learning united visual representation by alignment before projection. arXiv preprint arXiv:2311.10122","author":"Lin Bin","year":"2023","unstructured":"Bin Lin, Yang Ye, Bin Zhu, Jiaxi Cui, Munan Ning, Peng Jin, and Li Yuan. 2023. Video-llava: Learning united visual representation by alignment before projection. arXiv preprint arXiv:2311.10122 (2023)."},{"key":"e_1_3_2_1_31_1","unstructured":"Haotian Liu Chunyuan Li Yuheng Li Bo Li Yuanhan Zhang Sheng Shen and Yong Jae Lee. 2024. LLaVA-NeXT: Improved reasoning OCR and world knowledge. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/"},{"key":"e_1_3_2_1_32_1","volume-title":"Visual instruction tuning. Advances in neural information processing systems","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual instruction tuning. Advances in neural information processing systems, Vol. 36 (2023), 34892-34916."},{"key":"e_1_3_2_1_33_1","volume-title":"Videogpt: Integrating image and video encoders for enhanced video understanding. arXiv preprint arXiv:2406.09418","author":"Maaz Muhammad","year":"2024","unstructured":"Muhammad Maaz, Hanoona Rasheed, Salman Khan, and Fahad Khan. 2024. Videogpt: Integrating image and video encoders for enhanced video understanding. arXiv preprint arXiv:2406.09418 (2024)."},{"key":"e_1_3_2_1_34_1","first-page":"46534","article-title":"Self-refine: Iterative refinement with self-feedback","volume":"36","author":"Madaan Aman","year":"2023","unstructured":"Aman Madaan, Niket Tandon, Prakhar Gupta, Skyler Hallinan, Luyu Gao, Sarah Wiegreffe, Uri Alon, Nouha Dziri, Shrimai Prabhumoye, Yiming Yang, et al., 2023. Self-refine: Iterative refinement with self-feedback. Advances in Neural Information Processing Systems, Vol. 36 (2023), 46534-46594.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2017.2740923"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1511\/2001.4.344"},{"key":"e_1_3_2_1_37_1","volume-title":"Meld: A multimodal multi-party dataset for emotion recognition in conversations. arXiv preprint arXiv:1810.02508","author":"Poria Soujanya","year":"2018","unstructured":"Soujanya Poria, Devamanyu Hazarika, Navonil Majumder, Gautam Naik, Erik Cambria, and Rada Mihalcea. 2018. Meld: A multimodal multi-party dataset for emotion recognition in conversations. arXiv preprint arXiv:1810.02508 (2018)."},{"key":"e_1_3_2_1_38_1","volume-title":"Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever.","author":"Radford Alec","year":"2022","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2022. Robust Speech Recognition via Large-Scale Weak Supervision. arXiv:2212.04356"},{"key":"e_1_3_2_1_39_1","unstructured":"Alec Radford Karthik Narasimhan Tim Salimans Ilya Sutskever et al. 2018. Improving language understanding by generative pre-training. (2018)."},{"key":"e_1_3_2_1_40_1","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et al. 2019. Language models are unsupervised multitask learners. OpenAI blog Vol. 1 8 (2019) 9."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1037\/h0077714"},{"key":"e_1_3_2_1_42_1","volume-title":"Facial expression recognition based on local binary patterns: A comprehensive study. Image and vision Computing","author":"Shan Caifeng","year":"2009","unstructured":"Caifeng Shan, Shaogang Gong, and Peter W McOwan. 2009. Facial expression recognition based on local binary patterns: A comprehensive study. Image and vision Computing, Vol. 27, 6 (2009), 803-816."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICETET.2010.85"},{"key":"e_1_3_2_1_44_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_45_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al., 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_46_1","volume-title":"Proceedings of the conference. Association for computational linguistics. Meeting","volume":"2019","author":"Hubert Tsai Yao-Hung","year":"2019","unstructured":"Yao-Hung Hubert Tsai, Shaojie Bai, Paul Pu Liang, J Zico Kolter, Louis-Philippe Morency, and Ruslan Salakhutdinov. 2019. Multimodal transformer for unaligned multimodal language sequences. In Proceedings of the conference. Association for computational linguistics. Meeting, Vol. 2019. 6558."},{"key":"e_1_3_2_1_47_1","first-page":"190","article-title":"Segment-based speech emotion recognition using recurrent neural networks. In 2017 seventh international conference on affective computing and intelligent interaction (ACII)","author":"Tzinis Efthymios","year":"2017","unstructured":"Efthymios Tzinis and Alexandras Potamianos. 2017. Segment-based speech emotion recognition using recurrent neural networks. In 2017 seventh international conference on affective computing and intelligent interaction (ACII). IEEE, 190-195.","journal-title":"IEEE"},{"key":"e_1_3_2_1_48_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-024-40231-1"},{"key":"e_1_3_2_1_50_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024a. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_51_1","unstructured":"Michael Wooldridge. 2009. An introduction to multiagent systems. John wiley & sons."},{"key":"e_1_3_2_1_52_1","volume-title":"Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250","author":"Zadeh Amir","year":"2017","unstructured":"Amir Zadeh, Minghai Chen, Soujanya Poria, Erik Cambria, and Louis-Philippe Morency. 2017. Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250 (2017)."},{"key":"e_1_3_2_1_53_1","volume-title":"Proceedings of 5th International Joint Conference on Natural Language Processing. 336-344","author":"Zirn C\u00e4cilia","year":"2011","unstructured":"C\u00e4cilia Zirn, Mathias Niepert, Heiner Stuckenschmidt, and Michael Strube. 2011. Fine-grained sentiment analysis with structural features. In Proceedings of 5th International Joint Conference on Natural Language Processing. 336-344."}],"event":{"name":"MM '25:The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland"},"container-title":["Proceedings of the 3rd International Workshop on Multimodal and Responsible Affective Computing"],"original-title":[],"deposited":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T17:23:17Z","timestamp":1761153797000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746270.3760223"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,26]]},"references-count":53,"alternative-id":["10.1145\/3746270.3760223","10.1145\/3746270"],"URL":"https:\/\/doi.org\/10.1145\/3746270.3760223","relation":{},"subject":[],"published":{"date-parts":[[2025,10,26]]},"assertion":[{"value":"2025-10-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}