{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:15:53Z","timestamp":1765307753988,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001381","name":"National Research Foundation Singapore","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001381","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3762243","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:44:48Z","timestamp":1761371088000},"page":"14323-14325","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["CogMAEC'25: The 1st Workshop on Cognition-oriented Multimodal Affective and Empathetic Computing"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3026-6347","authenticated-orcid":false,"given":"Hao","family":"Fei","sequence":"first","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0513-5540","authenticated-orcid":false,"given":"Bobo","family":"Li","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2274-5719","authenticated-orcid":false,"given":"Meng","family":"Luo","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3162-935X","authenticated-orcid":false,"given":"Qian","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Auckland, Auckland, New Zealand"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9973-3305","authenticated-orcid":false,"given":"Lizi","family":"Liao","sequence":"additional","affiliation":[{"name":"Singapore Management University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1816-1761","authenticated-orcid":false,"given":"Fei","family":"Li","sequence":"additional","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3895-5510","authenticated-orcid":false,"given":"Min","family":"Zhang","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology (Shenzhen), Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6478-8699","authenticated-orcid":false,"given":"Bj\u00f6rn W.","family":"Schuller","sequence":"additional","affiliation":[{"name":"Imperial College London, London, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9636-388X","authenticated-orcid":false,"given":"Mong-Li","family":"Lee","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3030-1280","authenticated-orcid":false,"given":"Erik","family":"Cambria","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","series-title":"In 2010 AAAI fall symposium series.","volume-title":"Senticnet: A publicly available semantic resource for opinion mining","author":"Cambria Erik","year":"2010","unstructured":"Erik Cambria, Robert Speer, Catherine Havasi, and Amir Hussain. 2010. Senticnet: A publicly available semantic resource for opinion mining. In 2010 AAAI fall symposium series."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-76827-9_11"},{"key":"e_1_3_2_1_3_1","volume-title":"LLaVAC: Fine-tuning LLaVA as a Multimodal Sentiment Classifier. arXiv preprint arXiv:2502.02938","author":"Thodsaporn","year":"2025","unstructured":"Thodsaporn Chay-intr, Yujun Chen, Kobkrit Viriyayudhakorn, and Thanaruk Theeramunkong. 2025. LLaVAC: Fine-tuning LLaVA as a Multimodal Sentiment Classifier. arXiv preprint arXiv:2502.02938 (2025)."},{"key":"e_1_3_2_1_4_1","volume-title":"Towards reasoning era: A survey of long chain-of-thought for reasoning large language models. arXiv preprint arXiv:2503.09567","author":"Chen Qiguang","year":"2025","unstructured":"Qiguang Chen, Libo Qin, Jinhao Liu, Dengyun Peng, Jiannan Guan, Peng Wang, Mengkang Hu, Yuhang Zhou, Te Gao, and Wanxiang Che. 2025. Towards reasoning era: A survey of long chain-of-thought for reasoning large language models. arXiv preprint arXiv:2503.09567 (2025)."},{"key":"e_1_3_2_1_5_1","volume-title":"Emotion-LLaMA: Multimodal Emotion Recognition and Reasoning with Instruction Tuning. arXiv preprint arXiv:2406.11161","author":"Cheng Zebang","year":"2024","unstructured":"Zebang Cheng, Zhi-Qi Cheng, Jun-Yan He, Jingdong Sun, Kai Wang, Yuxiang Lin, Zheng Lian, Xiaojiang Peng, and Alexander Hauptmann. 2024. Emotion-LLaMA: Multimodal Emotion Recognition and Reasoning with Instruction Tuning. arXiv preprint arXiv:2406.11161 (2024)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3586075"},{"key":"e_1_3_2_1_7_1","first-page":"1014","article-title":"Llms to the moon? reddit market sentiment analysis with large language models","author":"Deng Xiang","year":"2023","unstructured":"Xiang Deng, Vasilisa Bashlovkina, Feng Han, Simon Baumgartner, and Michael Bendersky. 2023. Llms to the moon? reddit market sentiment analysis with large language models. In Companion Proceedings of WWW. 1014-1019.","journal-title":"Companion Proceedings of WWW."},{"key":"e_1_3_2_1_8_1","first-page":"1459","article-title":"Opensmile: the munich versatile and fast open-source audio feature extractor","author":"Eyben Florian","year":"2010","unstructured":"Florian Eyben, Martin W\u00f6llmer, and Bj\u00f6rn Schuller. 2010. Opensmile: the munich versatile and fast open-source audio feature extractor. In Proceedings of ACM MM. 1459-1462.","journal-title":"Proceedings of ACM MM."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3564281"},{"key":"e_1_3_2_1_10_1","volume-title":"Reasoning implicit sentiment with chain-of-thought prompting. arXiv preprint arXiv:2305.11255","author":"Fei Hao","year":"2023","unstructured":"Hao Fei, Bobo Li, Qian Liu, Lidong Bing, Fei Li, and Tat-Seng Chua. 2023. Reasoning implicit sentiment with chain-of-thought prompting. arXiv preprint arXiv:2305.11255 (2023)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3129483"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00730"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the International Conference on Machine Learning.","author":"Fei Hao","year":"2024","unstructured":"Hao Fei, Shengqiong Wu, Wei Ji, Hanwang Zhang, Meishan Zhang, Mong-Li Lee, and Wynne Hsu. 2024b. Video-of-thought: Step-by-step video reasoning from perception to cognition. In Proceedings of the International Conference on Machine Learning."},{"key":"e_1_3_2_1_14_1","volume-title":"Editing. Proceedings of the Advances in neural information processing systems.","author":"Fei Hao","year":"2024","unstructured":"Hao Fei, Shengqiong Wu, Hanwang Zhang, Tat-Seng Chua, and Shuicheng Yan. 2024c. VITRON: A Unified Pixel-level Vision LLM for Understanding, Generating, Segmenting, Editing. Proceedings of the Advances in neural information processing systems."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3393452"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6271"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the International Conference on Machine Learning.","author":"Fei Hao","year":"2025","unstructured":"Hao Fei, Yuan Zhou, Juncheng Li, Xiangtai Li, Qingshan Xu, Bobo Li, Shengqiong Wu, Yaoting Wang, Junbao Zhou, Jiahao Meng, et al., 2025. On path to multimodal generalist: General-level and general-bench. In Proceedings of the International Conference on Machine Learning."},{"key":"e_1_3_2_1_18_1","first-page":"7037","article-title":"MM-DFN: Multimodal dynamic fusion network for emotion recognition in conversations","author":"Hu Dou","year":"2022","unstructured":"Dou Hu, Xiaolong Hou, Lingwei Wei, Lianxin Jiang, and Yang Mo. 2022. MM-DFN: Multimodal dynamic fusion network for emotion recognition in conversations. In Proceedings of ICASSP. 7037-7041.","journal-title":"Proceedings of ICASSP."},{"key":"e_1_3_2_1_19_1","volume-title":"Recent trends of multimodal affective computing: A survey from NLP perspective. arXiv preprint arXiv:2409.07388","author":"Hu Guimin","year":"2024","unstructured":"Guimin Hu, Yi Xin, Weimin Lyu, Haojian Huang, Chang Sun, Zhihong Zhu, Lin Gui, Ruichu Cai, Erik Cambria, and Hasti Seifi. 2024. Recent trends of multimodal affective computing: A survey from NLP perspective. arXiv preprint arXiv:2409.07388 (2024)."},{"key":"e_1_3_2_1_20_1","unstructured":"Bobo Li Hao Fei Fei Li Tat-seng Chua and Donghong Ji. 2024a. Multimodal Emotion-Cause Pair Extraction with Holistic Interaction and Label Constraint. ACM Trans. Multimedia Comput. Commun. Appl. (2024)."},{"key":"e_1_3_2_1_21_1","volume-title":"Diaasq: A benchmark of conversational aspect-based sentiment quadruple analysis. arXiv preprint arXiv:2211.05705","author":"Li Bobo","year":"2022","unstructured":"Bobo Li, Hao Fei, Fei Li, Yuhan Wu, Jinsong Zhang, Shengqiong Wu, Jingye Li, Yijiang Liu, Lizi Liao, Tat-Seng Chua, et al., 2022. Diaasq: A benchmark of conversational aspect-based sentiment quadruple analysis. arXiv preprint arXiv:2211.05705 (2022)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29807"},{"key":"e_1_3_2_1_23_1","unstructured":"Jian Li Weiheng Lu Hao Fei Meng Luo Ming Dai Min Xia Yizhang Jin Zhenye Gan Ding Qi Chaoyou Fu et al. 2024c. A survey on benchmarks of multimodal large language models. arXiv preprint arXiv:2408.08632 (2024)."},{"key":"e_1_3_2_1_24_1","volume-title":"Vision-language pre-training for multimodal aspect-based sentiment analysis. arXiv preprint arXiv:2204.07955","author":"Ling Yan","year":"2022","unstructured":"Yan Ling, Jianfei Yu, and Rui Xia. 2022. Vision-language pre-training for multimodal aspect-based sentiment analysis. arXiv preprint arXiv:2204.07955 (2022)."},{"key":"e_1_3_2_1_25_1","first-page":"7667","article-title":"Panosent: A panoptic sextuple extraction benchmark for multimodal conversational aspect-based sentiment analysis","author":"Luo Meng","year":"2024","unstructured":"Meng Luo, Hao Fei, Bobo Li, Shengqiong Wu, Qian Liu, Soujanya Poria, Erik Cambria, Mong-Li Lee, and Wynne Hsu. 2024a. Panosent: A panoptic sextuple extraction benchmark for multimodal conversational aspect-based sentiment analysis. In Proceedings of ACM MM. 7667-7676.","journal-title":"Proceedings of ACM MM."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.semeval-1.226"},{"key":"e_1_3_2_1_27_1","volume-title":"Multimodal sentiment analysis using hierarchical fusion with context modeling. Knowledge-based systems","author":"Majumder Navonil","year":"2018","unstructured":"Navonil Majumder, Devamanyu Hazarika, Alexander Gelbukh, Erik Cambria, and Soujanya Poria. 2018. Multimodal sentiment analysis using hierarchical fusion with context modeling. Knowledge-based systems, Vol. 161 (2018), 124-133."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.asej.2014.04.011"},{"key":"e_1_3_2_1_29_1","volume-title":"A review on sentiment analysis and emotion detection from text. Social network analysis and mining","author":"Nandwani Pansy","year":"2021","unstructured":"Pansy Nandwani and Rupali Verma. 2021. A review on sentiment analysis and emotion detection from text. Social network analysis and mining, Vol. 11, 1 (2021), 81."},{"key":"e_1_3_2_1_30_1","first-page":"70","article-title":"Sentiment analysis: Capturing favorability using natural language processing","author":"Nasukawa Tetsuya","year":"2003","unstructured":"Tetsuya Nasukawa and Jeonghee Yi. 2003. Sentiment analysis: Capturing favorability using natural language processing. In Proceedings of the K-CAP. 70-77.","journal-title":"Proceedings of the K-CAP."},{"key":"e_1_3_2_1_31_1","volume-title":"Large language models meet nlp: A survey. arXiv preprint arXiv:2405.12819","author":"Qin Libo","year":"2024","unstructured":"Libo Qin, Qiguang Chen, Xiachong Feng, Yang Wu, Yongheng Zhang, Yinghui Li, Min Li, Wanxiang Che, and Philip S Yu. 2024. Large language models meet nlp: A survey. arXiv preprint arXiv:2405.12819 (2024)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i15.17616"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2017.08.003"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.671"},{"key":"e_1_3_2_1_35_1","volume-title":"Multimodal chain-of-thought reasoning: A comprehensive survey. arXiv preprint arXiv:2503.12605","author":"Wang Yaoting","year":"2025","unstructured":"Yaoting Wang, Shengqiong Wu, Yuecheng Zhang, Shuicheng Yan, Ziwei Liu, Jiebo Luo, and Hao Fei. 2025. Multimodal chain-of-thought reasoning: A comprehensive survey. arXiv preprint arXiv:2503.12605 (2025)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.823"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01321"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.146"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21404"},{"key":"e_1_3_2_1_40_1","volume-title":"Towards Semantic Equivalence of Tokenization in Multimodal LLM. arXiv preprint arXiv:2406.05127","author":"Wu Shengqiong","year":"2024","unstructured":"Shengqiong Wu, Hao Fei, Xiangtai Li, Jiayi Ji, Hanwang Zhang, Tat-Seng Chua, and Shuicheng Yan. 2024a. Towards Semantic Equivalence of Tokenization in Multimodal LLM. arXiv preprint arXiv:2406.05127 (2024)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i8.32913"},{"key":"e_1_3_2_1_42_1","volume-title":"Proceedings of the International Conference on Machine Learning. 53366-53397","author":"Wu Shengqiong","year":"2024","unstructured":"Shengqiong Wu, Hao Fei, Leigang Qu, Wei Ji, and Tat-Seng Chua. 2024b. NExT-GPT: Any-to-Any Multimodal LLM. In Proceedings of the International Conference on Machine Learning. 53366-53397."},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems. 79240-79259","author":"Wu Shengqiong","year":"2023","unstructured":"Shengqiong Wu, Hao Fei, Hanwang Zhang, and Tat-Seng Chua. 2023c. Imagine that! abstract-to-intricate text-to-image synthesis with scene graph hallucination diffusion. In Proceedings of the 37th International Conference on Neural Information Processing Systems. 79240-79259."},{"key":"e_1_3_2_1_44_1","volume-title":"Omni-Emotion: Extending Video MLLM with Detailed Face and Audio Modeling for Multimodal Emotion Analysis. arXiv preprint arXiv:2501.09502","author":"Yang Qize","year":"2025","unstructured":"Qize Yang, Detao Bai, Yi-Xing Peng, and Xihan Wei. 2025. Omni-Emotion: Extending Video MLLM with Detailed Face and Audio Modeling for Multimodal Emotion Analysis. arXiv preprint arXiv:2501.09502 (2025)."},{"key":"e_1_3_2_1_45_1","volume-title":"Towards Multimodal Empathetic Response Generation: A Rich Text-Speech-Vision Avatar-based Benchmark. In THE WEB CONFERENCE","author":"Zhang Han","year":"2025","unstructured":"Han Zhang, Zixiang Meng, Meng Luo, Hong Han, Lizi Liao, Erik Cambria, and Hao Fei. 2025. Towards Multimodal Empathetic Response Generation: A Rich Text-Speech-Vision Avatar-based Benchmark. In THE WEB CONFERENCE 2025."},{"key":"e_1_3_2_1_46_1","volume-title":"Sinno Jialin Pan, and Lidong Bing","author":"Zhang Wenxuan","year":"2023","unstructured":"Wenxuan Zhang, Yue Deng, Bing Liu, Sinno Jialin Pan, and Lidong Bing. 2023. Sentiment analysis in the era of large language models: A reality check. arXiv preprint arXiv:2305.15005 (2023)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3230975"},{"key":"e_1_3_2_1_48_1","volume-title":"AoM: Detecting aspect-oriented information for multimodal aspect-based sentiment analysis. arXiv preprint arXiv:2306.01004","author":"Zhou Ru","year":"2023","unstructured":"Ru Zhou, Wenya Guo, Xumeng Liu, Shenglong Yu, Ying Zhang, and Xiaojie Yuan. 2023. AoM: Detecting aspect-oriented information for multimodal aspect-based sentiment analysis. arXiv preprint arXiv:2306.01004 (2023)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3762243","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:12:52Z","timestamp":1765307572000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3762243"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":48,"alternative-id":["10.1145\/3746027.3762243","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3762243","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}