{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T06:19:27Z","timestamp":1784787567243,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":65,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62271312, 62132006"],"award-info":[{"award-number":["62271312, 62132006"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"STCSM","award":["22DZ2229005"],"award-info":[{"award-number":["22DZ2229005"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755777","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:55:00Z","timestamp":1761375300000},"page":"7064-7073","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["EEmo-Bench: A Benchmark for Multi-modal Large Language Models on Image Evoked Emotion Assessment"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-5490-1097","authenticated-orcid":false,"given":"Lancheng","family":"Gao","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2623-4756","authenticated-orcid":false,"given":"Ziheng","family":"Jia","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6429-5763","authenticated-orcid":false,"given":"Yunhao","family":"Zeng","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8162-1949","authenticated-orcid":false,"given":"Wei","family":"Sun","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5131-2563","authenticated-orcid":false,"given":"Yiming","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3641-1429","authenticated-orcid":false,"given":"Wei","family":"Zhou","sequence":"additional","affiliation":[{"name":"Cardiff University, Cardiff, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8165-9322","authenticated-orcid":false,"given":"Guangtao","family":"Zhai","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5693-0416","authenticated-orcid":false,"given":"Xiongkuo","family":"Min","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00642"},{"key":"e_1_3_2_2_2_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 11569-11579","author":"Achlioptas Panos","unstructured":"Panos Achlioptas, Maks Ovsjanikov, Kilichbek Haydarov, Mohamed Elhoseiny, and Leonidas J. Guibas. 2021. ArtEmis: Affective Language for Visual Art. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 11569-11579."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2024.3419593"},{"key":"e_1_3_2_2_4_1","volume-title":"Feb 24, 2025.","year":"2025","unstructured":"Anthropic. 2025. Claude 3.7 Sonnet and Claude Code. https:\/\/www.anthropic.com\/blog Announcement on Anthropic blog, Feb 24, 2025."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2308.12966"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2502.13923"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.paid.2010.03.013"},{"key":"e_1_3_2_2_8_1","first-page":"19","article-title":"Methodology for the subjective assessment of the quality of television pictures","volume":"4","author":"RIR","year":"2002","unstructured":"RIR BT. 2002. Methodology for the subjective assessment of the quality of television pictures. International Telecommunication Union, Vol. 4 (2002), 19.","journal-title":"International Telecommunication Union"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2501.17811"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/2024.FINDINGS-ACL.128"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2412.05271"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1702247114"},{"key":"e_1_3_2_2_13_1","volume-title":"The Geneva affective picture database (GAPED): a new 730-picture database focusing on valence and normative significance. Behavior research methods","author":"Dan-Glauser Elise S","year":"2011","unstructured":"Elise S Dan-Glauser and Klaus R Scherer. 2011. The Geneva affective picture database (GAPED): a new 730-picture database focusing on valence and normative significance. Behavior research methods, Vol. 43 (2011), 468-477."},{"key":"e_1_3_2_2_14_1","volume-title":"GoEmotions: A dataset of fine-grained emotions. arXiv preprint arXiv:2005.00547","author":"Demszky Dorottya","year":"2020","unstructured":"Dorottya Demszky, Dana Movshovitz-Attias, Jeongwoo Ko, Alan Cowen, Gaurav Nemade, and Sujith Ravi. 2020. GoEmotions: A dataset of fine-grained emotions. arXiv preprint arXiv:2005.00547 (2020)."},{"key":"e_1_3_2_2_15_1","volume-title":"What emotion categories or dimensions can observers judge from facial behavior? Emotions in the human face","author":"Ekman Paul","year":"1982","unstructured":"Paul Ekman. 1982. What emotion categories or dimensions can observers judge from facial behavior? Emotions in the human face (1982), 39-55."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2306.13394"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680649"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2401.08276"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2011.941851"},{"key":"e_1_3_2_2_20_1","first-page":"38","article-title":"Multimodal sentiment analysis: A survey and comparison. International Journal of Service Science","volume":"10","author":"Kaur Ramandeep","year":"2019","unstructured":"Ramandeep Kaur and Sandeep Kautish. 2019. Multimodal sentiment analysis: A survey and comparison. International Journal of Service Science, Management, Engineering, and Technology (IJSSMET), Vol. 10, 2 (2019), 38-58.","journal-title":"Management, Engineering, and Technology (IJSSMET)"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.285"},{"key":"e_1_3_2_2_22_1","first-page":"39","article-title":"International affective picture system (IAPS): Technical manual and affective ratings","volume":"1","author":"Lang Peter J","year":"1997","unstructured":"Peter J Lang, Margaret M Bradley, Bruce N Cuthbert, et al., 1997. International affective picture system (IAPS): Technical manual and affective ratings. NIMH Center for the Study of Emotion and Attention, Vol. 1, 39-58 (1997), 3.","journal-title":"NIMH Center for the Study of Emotion and Attention"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2408.03326"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2407.07895"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2408.08632"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2403.05525"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2010.5543262"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1873965"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF02686918"},{"key":"e_1_3_2_2_30_1","volume-title":"Society of mind","author":"Minsky Marvin","unstructured":"Marvin Minsky. 1988. Society of mind. Simon and Schuster."},{"key":"e_1_3_2_2_31_1","volume-title":"Proceedings of the eleventh international conference on language resources and evaluation (LREC","author":"Mohammad Saif","year":"2018","unstructured":"Saif Mohammad and Svetlana Kiritchenko. 2018. Wikiart emotions: An annotated dataset of emotions evoked by art. In Proceedings of the eleventh international conference on language resources and evaluation (LREC 2018)."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2017.2740923"},{"key":"e_1_3_2_2_33_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. CoRR Vol. abs\/2303.08774 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2303.08774 arXiv:2303.08774"},{"key":"e_1_3_2_2_34_1","volume-title":"Eq-bench: An emotional intelligence benchmark for large language models. arXiv preprint arXiv:2312.06281","author":"Paech Samuel J","year":"2023","unstructured":"Samuel J Paech. 2023. Eq-bench: An emotional intelligence benchmark for large language models. arXiv preprint arXiv:2312.06281 (2023)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298687"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","unstructured":"Machel Reid Nikolay Savinov Denis Teplyashin Dmitry Lepikhin Timothy P. Lillicrap Jean-Baptiste Alayrac Radu Soricut Angeliki Lazaridou Orhan Firat Julian Schrittwieser Ioannis Antonoglou Rohan Anil Sebastian Borgeaud Andrew M. Dai Katie Millican Ethan Dyer Mia Glaese Thibault Sottiaux Benjamin Lee Fabio Viola Malcolm Reynolds Yuanzhong Xu James Molloy Jilin Chen Michael Isard Paul Barham Tom Hennigan Ross McIlroy Melvin Johnson Johan Schalkwyk Eli Collins Eliza Rutherford Erica Moreira Kareem Ayoub Megha Goel Clemens Meyer Gregory Thornton Zhen Yang Henryk Michalewski Zaheer Abbas Nathan Schucher Ankesh Anand Richard Ives James Keeling Karel Lenc Salem Haykal Siamak Shakeri Pranav Shyam Aakanksha Chowdhery Roman Ring Stephen Spencer Eren Sezener and et al. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. CoRR Vol. abs\/2403.05530 (2024). https:\/\/doi.org\/10.48550\/ARXIV.2403.05530 arXiv:2403.05530","DOI":"10.48550\/ARXIV.2403.05530"},{"key":"e_1_3_2_2_37_1","volume-title":"A Circumplex Model of Affect Journal of Personality and Social Psychology 39. \u00cd6I-I78","author":"Russell JA","year":"1980","unstructured":"JA Russell. 1980. A Circumplex Model of Affect Journal of Personality and Social Psychology 39. \u00cd6I-I78 (1980)."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2006.881959"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/1816041.1816099"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.3389\/frobt.2020.532279"},{"key":"e_1_3_2_2_41_1","unstructured":"OpenGVLab Team. 2024. Internvl2: Better than the best-expanding performance boundaries of open-source multimodal models with the progressive scaling strategy."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2409.12191"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2013.6738665"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDMW.2015.142"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/2024.ACL-LONG.498"},{"key":"e_1_3_2_2_46_1","volume-title":"Norms of valence, arousal, and dominance for 13,915 English lemmas. Behavior research methods","author":"Warriner Amy Beth","year":"2013","unstructured":"Amy Beth Warriner, Victor Kuperman, and Marc Brysbaert. 2013. Norms of valence, arousal, and dominance for 13,915 English lemmas. Behavior research methods, Vol. 45 (2013), 1191-1207."},{"key":"e_1_3_2_2_47_1","volume-title":"Visual chatgpt: Talking, drawing and editing with visual foundation models. arXiv preprint arXiv:2303.04671","author":"Wu Chenfei","year":"2023","unstructured":"Chenfei Wu, Shengming Yin, Weizhen Qi, Xiaodong Wang, Zecheng Tang, and Nan Duan. 2023. Visual chatgpt: Talking, drawing and editing with visual foundation models. arXiv preprint arXiv:2303.04671 (2023)."},{"key":"e_1_3_2_2_48_1","volume-title":"Q-Bench: A Benchmark for General-Purpose Foundation Models on Low-level Vision. In The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Wu Haoning","year":"2024","unstructured":"Haoning Wu, Zicheng Zhang, Erli Zhang, Chaofeng Chen, Liang Liao, Annan Wang, Chunyi Li, Wenxiu Sun, Qiong Yan, Guangtao Zhai, and Weisi Lin. 2024c. Q-Bench: A Benchmark for General-Purpose Foundation Models on Low-level Vision. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=0V5TVt9bk0"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02408"},{"key":"e_1_3_2_2_50_1","volume-title":"Forty-first International Conference on Machine Learning, ICML 2024","author":"Wu Haoning","year":"2024","unstructured":"Haoning Wu, Zicheng Zhang, Weixia Zhang, Chaofeng Chen, Liang Liao, Chunyi Li, Yixuan Gao, Annan Wang, Erli Zhang, Wenxiu Sun, Qiong Yan, Xiongkuo Min, Guangtao Zhai, and Weisi Lin. 2024b. Q-Align: Teaching LMMs for Visual Scoring via Discrete Text-Defined Levels. In Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21-27, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=PHjkVjR78A"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2412.10302"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01864"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2408.04840"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01239"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1609\/AAAI.V29I1.9179"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/2835776.2835779"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/2024.FINDINGS-NAACL.246"},{"key":"e_1_3_2_2_58_1","volume-title":"On the out-of-distribution generalization of multimodal large language models. arXiv preprint arXiv:2402.06599","author":"Zhang Xingxuan","year":"2024","unstructured":"Xingxuan Zhang, Jiansheng Li, Wenjing Chu, Junjia Hai, Renzhe Xu, Yuqing Yang, Shikai Guan, Jiazheng Xu, and Peng Cui. 2024c. On the out-of-distribution generalization of multimodal large language models. arXiv preprint arXiv:2402.06599 (2024)."},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2409.20063"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"crossref","unstructured":"Zicheng Zhang Tengchuan Kou Shushi Wang Chunyi Li Wei Sun Wei Wang Xiaoyu Li Zongyu Wang Xuezhi Cao Xiongkuo Min et al. 2025. Q-Eval-100K: Evaluating Visual Quality and Alignment Level for Text-to-Vision Content. arXiv preprint arXiv:2503.02357 (2025).","DOI":"10.1109\/CVPR52734.2025.00993"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3445770"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3094362"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"crossref","unstructured":"Qing Zhou Carlos Valiente and Nancy Eisenberg. 2003. Empathy and its measurement. (2003).","DOI":"10.1037\/10612-017"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2411.11235"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2404.09619"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755777","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T03:56:50Z","timestamp":1765339010000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755777"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":65,"alternative-id":["10.1145\/3746027.3755777","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755777","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}