{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T16:11:33Z","timestamp":1772727093021,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"???????????????????","award":["62006117"],"award-info":[{"award-number":["62006117"]}]},{"name":"???????????????????????","award":["62076133"],"award-info":[{"award-number":["62076133"]}]},{"name":"IaaS??????????????????","award":["62272232"],"award-info":[{"award-number":["62272232"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681601","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"5820-5828","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Observe before Generate: Emotion-Cause aware Video Caption for Multimodal Emotion Cause Generation in Conversations"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8955-9411","authenticated-orcid":false,"given":"Fanfan","family":"Wang","sequence":"first","affiliation":[{"name":"School of Computer Science and Engineering, Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8830-193X","authenticated-orcid":false,"given":"Heqing","family":"Ma","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9825-7853","authenticated-orcid":false,"given":"Xiangqing","family":"Shen","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8380-0609","authenticated-orcid":false,"given":"Jianfei","family":"Yu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0621-1058","authenticated-orcid":false,"given":"Rui","family":"Xia","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3319990"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65--72","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65--72."},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of The 12th Language Resources and Evaluation Conference. 1554--1566","author":"Maria Bostan Laura Ana","year":"2020","unstructured":"Laura Ana Maria Bostan, Evgeny Kim, and Roman Klinger. 2020. GoodNewsEveryone: A Corpus of News Headlines Annotated with Emotions, Semantic Roles, and Reader Perception. In Proceedings of The 12th Language Resources and Evaluation Conference. 1554--1566."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29732"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3132684"},{"key":"e_1_3_2_1_6_1","first-page":"1","article-title":"Scaling instruction-finetuned language models","volume":"25","author":"Chung Hyung Won","year":"2024","unstructured":"Hyung Won Chung, Le Hou, Shayne Longpre, Barret Zoph, Yi Tay, William Fedus, Yunxuan Li, Xuezhi Wang, Mostafa Dehghani, Siddhartha Brahma, et al. 2024. Scaling instruction-finetuned language models. Journal of Machine Learning Research, Vol. 25, 70 (2024), 1--53.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1080\/02699939208411068"},{"key":"e_1_3_2_1_8_1","volume-title":"EmpathyEar: An Open-source Avatar Multimodal Empathetic Chatbot. arXiv preprint arXiv:2406.15177","author":"Fei Hao","year":"2024","unstructured":"Hao Fei, Han Zhang, Bin Wang, Lizi Liao, Qian Liu, and Erik Cambria. 2024. EmpathyEar: An Open-source Avatar Multimodal Empathetic Chatbot. arXiv preprint arXiv:2406.15177 (2024)."},{"key":"e_1_3_2_1_9_1","first-page":"807","article-title":"Improving empathetic response generation by recognizing emotion cause in conversations. In Findings of the association for computational linguistics","volume":"2021","author":"Gao Jun","year":"2021","unstructured":"Jun Gao, Yuhan Liu, Haolin Deng, Wei Wang, Yu Cao, Jiachen Du, and Ruifeng Xu. 2021. Improving empathetic response generation by recognizing emotion cause in conversations. In Findings of the association for computational linguistics: EMNLP 2021. 807--819.","journal-title":"EMNLP"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-18032-8_1"},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the NTCIR-13 Conference.","author":"Gao Qinghong","year":"2017","unstructured":"Qinghong Gao, Jiannan Hu, Ruifeng Xu, Gui Lin, Yulan He, Qin Lu, and Kam-Fai Wong. 2017. Overview of NTCIR-13 ECA task. In Proceedings of the NTCIR-13 Conference."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-18117-2_12"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.344"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Lin Gui Jiannan Hu Yulan He Ruifeng Xu Qin Lu and Jiachen Du. 2017. A question answering approach to emotion cause extraction. In Empirical Methods in Natural Language Processing (EMNLP). 1593--1602.","DOI":"10.18653\/v1\/D17-1167"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Lin Gui Dongyin Wu Ruifeng Xu Qin Lu Yu Zhou et al. 2016. Event-Driven Emotion Cause Extraction with Corpus Construction.. In EMNLP. World Scientific 1639--1649.","DOI":"10.18653\/v1\/D16-1170"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-10-2993-6_8"},{"key":"e_1_3_2_1_17_1","volume-title":"Emotion theory and research: Highlights, unanswered questions, and emerging issues. Annual review of psychology","author":"Izard Carroll E","year":"2009","unstructured":"Carroll E Izard. 2009. Emotion theory and research: Highlights, unanswered questions, and emerging issues. Annual review of psychology, Vol. 60, 1 (2009), 1--25."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the 27th International Conference on Computational Linguistics. 1345--1359","author":"Kim Evgeny","year":"2018","unstructured":"Evgeny Kim and Roman Klinger. 2018. Who feels what and why? annotation of a literature corpus with semantic roles of emotions. In Proceedings of the 27th International Conference on Computational Linguistics. 1345--1359."},{"key":"e_1_3_2_1_19_1","volume-title":"NAACL HLT Workshop on Computational Approaches to Analysis and Generation of Emotion in Text. 45--53","author":"Mei Lee Sophia Yat","year":"2010","unstructured":"Sophia Yat Mei Lee, Ying Chen, and Chu-Ren Huang. 2010. A text-driven rule-based system for emotion cause detection. In NAACL HLT Workshop on Computational Approaches to Analysis and Generation of Emotion in Text. 45--53."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/584"},{"key":"e_1_3_2_1_21_1","volume-title":"ECPEC: emotion-cause pair extraction in conversations","author":"Li Wei","year":"2022","unstructured":"Wei Li, Yang Li, Vlad Pandelea, Mengshi Ge, Luyao Zhu, and Erik Cambria. 2022. ECPEC: emotion-cause pair extraction in conversations. IEEE Transactions on Affective Computing (2022)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463042"},{"key":"e_1_3_2_1_23_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74--81.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74--81."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.chb.2004.02.010"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.737"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12559-021-09925-7"},{"key":"e_1_3_2_1_28_1","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of machine learning research, Vol. 21, 140 (2020), 1--67.","journal-title":"Journal of machine learning research"},{"key":"e_1_3_2_1_29_1","volume-title":"2022 21st IEEE International Conference on Machine Learning and Applications (ICMLA). 140--147","author":"Riyadh Md","unstructured":"Md Riyadh and M. Omair Shafiq. 2022. Towards Emotion Cause Generation in Natural Language Processing using Deep Learning. In 2022 21st IEEE International Conference on Machine Learning and Applications (ICMLA). 140--147."},{"key":"e_1_3_2_1_30_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Yonghui Wu Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2022.3226559"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.semeval-1.277"},{"key":"e_1_3_2_1_34_1","volume-title":"Generative Emotion Cause Triplet Extraction in Conversations with Commonsense Knowledge. In Findings of the Association for Computational Linguistics: EMNLP","author":"Wang Fanfan","year":"2023","unstructured":"Fanfan Wang, Jianfei Yu, and Rui Xia. 2023. Generative Emotion Cause Triplet Extraction in Conversations with Commonsense Knowledge. In Findings of the Association for Computational Linguistics: EMNLP 2023, Houda Bouamor, Juan Pino, and Kalika Bali (Eds.). 3952--3963."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1096"},{"key":"e_1_3_2_1_36_1","volume-title":"RTHN: A RNN-Transformer Hierarchical Network for Emotion Cause Extraction. In International Joint Conference on Artificial Intelligence (IJCAI). 5285--5291","author":"Xia Rui","year":"2019","unstructured":"Rui Xia, Mengran Zhang, and Zixiang Ding. 2019. RTHN: A RNN-Transformer Hierarchical Network for Emotion Cause Extraction. In International Joint Conference on Artificial Intelligence (IJCAI). 5285--5291."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.642"},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the 29th International Conference on Computational Linguistics. 6762--6772","author":"Zhang Duzhen","year":"2022","unstructured":"Duzhen Zhang, Zhen Yang, Fandong Meng, Xiuyi Chen, and Jie Zhou. 2022. TSAM: A Two-Stream Attention Model for Causal Emotion Entailment. In Proceedings of the 29th International Conference on Computational Linguistics. 6762--6772."},{"key":"e_1_3_2_1_39_1","volume-title":"BERTScore: Evaluating Text Generation with BERT. In 8th International Conference on Learning Representations, ICLR 2020","author":"Zhang Tianyi","year":"2020","unstructured":"Tianyi Zhang, Varsha Kishore, Felix Wu, Kilian Q. Weinberger, and Yoav Artzi. 2020. BERTScore: Evaluating Text Generation with BERT. In 8th International Conference on Learning Representations, ICLR 2020, Addis Ababa, Ethiopia, April 26--30, 2020."},{"key":"e_1_3_2_1_40_1","volume-title":"Knowledge-Bridged Causal Interaction Network for Causal Emotion Entailment. In Thirty-Seventh AAAI Conference on Artificial Intelligence, AAAI","author":"Zhao Weixiang","year":"2023","unstructured":"Weixiang Zhao, Yanyan Zhao, Zhuojun Li, and Bing Qin. 2023. Knowledge-Bridged Causal Interaction Network for Causal Emotion Entailment. In Thirty-Seventh AAAI Conference on Artificial Intelligence, AAAI 2023, Brian Williams, Yiling Chen, and Jennifer Neville (Eds.). 14020--14028."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681601","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681601","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:49Z","timestamp":1750295869000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681601"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":40,"alternative-id":["10.1145\/3664647.3681601","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681601","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}