{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,24]],"date-time":"2026-01-24T18:39:34Z","timestamp":1769279974403,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","funder":[{"name":"Key R&D Program of Zhejiang Province","award":["2023C01181"],"award-info":[{"award-number":["2023C01181"]}]},{"name":"Xi'an Science and Technology Plan Key Industrial Chain Technology Research Project","award":["23ZDCYJSGG0007"],"award-info":[{"award-number":["23ZDCYJSGG0007"]}]},{"name":"Xi'an Science and Technology Plan Key Industrial Chain and Core Technology Research Project","award":["23LLRH0022"],"award-info":[{"award-number":["23LLRH0022"]}]},{"name":"Aviation Science Foundation","award":["2023M071070002, 2024M071070001"],"award-info":[{"award-number":["2023M071070002, 2024M071070001"]}]},{"DOI":"10.13039\/501100006374","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62472348"],"award-info":[{"award-number":["62472348"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Innovation Capability Support Plan of Shaanxi","award":["2022PT-33"],"award-info":[{"award-number":["2022PT-33"]}]},{"name":"Qinchuangyuan Construction of Two Chain Integration Important Project","award":["23LLRHZDZX0006"],"award-info":[{"award-number":["23LLRHZDZX0006"]}]},{"name":"Key Research and Development Program of Shaanxi","award":["2023-YBGY-230, 2024GX-YBXM-533"],"award-info":[{"award-number":["2023-YBGY-230, 2024GX-YBXM-533"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3731715.3733453","type":"proceedings-article","created":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T18:29:43Z","timestamp":1750876183000},"page":"1497-1506","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Towards Emotion Analysis in Short-form Videos: A Large-Scale Dataset and Baseline"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6244-0269","authenticated-orcid":false,"given":"Xuecheng","family":"Wu","sequence":"first","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2765-6582","authenticated-orcid":false,"given":"Heli","family":"Sun","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1569-5362","authenticated-orcid":false,"given":"Junxiao","family":"Xue","sequence":"additional","affiliation":[{"name":"Zhejiang Lab, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-9769-2734","authenticated-orcid":false,"given":"Jiayu","family":"Nie","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4051-9566","authenticated-orcid":false,"given":"Xiangyan","family":"Kong","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Harbin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7167-1354","authenticated-orcid":false,"given":"Ruofan","family":"Zhai","sequence":"additional","affiliation":[{"name":"Zhengzhou University, Zhengzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5929-6455","authenticated-orcid":false,"given":"Danlei","family":"Huang","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6463-5158","authenticated-orcid":false,"given":"Liang","family":"He","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Consistency is Key: Disentangling Label Variation in Natural Language Processing with Intra-Annotator Agreement. arXiv preprint arXiv:2301.10684","author":"Abercrombie Gavin","year":"2023","unstructured":"Gavin Abercrombie, Verena Rieser, and Dirk Hovy. 2023. Consistency is Key: Disentangling Label Variation in Natural Language Processing with Intra-Annotator Agreement. arXiv preprint arXiv:2301.10684 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","author":"Ba Jimmy Lei","year":"2016","unstructured":"Jimmy Lei Ba, Jamie Ryan Kiros, and Geoffrey E Hinton. 2016. Layer normalization. arXiv preprint arXiv:1607.06450 (2016)."},{"key":"e_1_3_2_1_3_1","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"2","author":"Bertasius Gedas","year":"2021","unstructured":"Gedas Bertasius, Heng Wang, and Lorenzo Torresani. 2021. Is space-time attention all you need for video understanding?. In ICML, Vol. 2. 4.","journal-title":"ICML"},{"key":"e_1_3_2_1_4_1","volume-title":"IEMOCAP: Interactive emotional dyadic motion capture database. Language resources and evaluation 42","author":"Busso Carlos","year":"2008","unstructured":"Carlos Busso, Murtaza Bulut, Chi-Chun Lee, Abe Kazemzadeh, Emily Mower, Samuel Kim, Jeannette N Chang, Sungbok Lee, and Shrikanth S Narayanan. 2008. IEMOCAP: Interactive emotional dyadic motion capture database. Language resources and evaluation 42 (2008), 335--359."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_6_1","volume-title":"2022 26th International Conference on Pattern Recognition (ICPR). IEEE, 2822--2828","author":"Chumachenko Kateryna","year":"2022","unstructured":"Kateryna Chumachenko, Alexandros Iosifidis, and Moncef Gabbouj. 2022. Selfattention fusion for audiovisual emotion recognition with incomplete data. In 2022 26th International Conference on Pattern Recognition (ICPR). IEEE, 2822--2828."},{"key":"e_1_3_2_1_7_1","volume-title":"Ecapatdnn: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification. arXiv preprint arXiv:2005.07143","author":"Desplanques Brecht","year":"2020","unstructured":"Brecht Desplanques, Jenthe Thienpondt, and Kris Demuynck. 2020. Ecapatdnn: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification. arXiv preprint arXiv:2005.07143 (2020)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/MMUL.2012.26"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"e_1_3_2_1_10_1","volume-title":"LLM Agents in Interaction: Measuring Personality Consistency and Linguistic Alignment in Interacting Populations of Large Language Models. arXiv preprint arXiv:2402.02896","author":"Frisch Ivar","year":"2024","unstructured":"Ivar Frisch and Mario Giulianelli. 2024. LLM Agents in Interaction: Measuring Personality Consistency and Linguistic Alignment in Interacting Populations of Large Language Models. arXiv preprint arXiv:2402.02896 (2024)."},{"key":"e_1_3_2_1_11_1","volume-title":"Res2net: A new multi-scale backbone architecture","author":"Gao Shang-Hua","year":"2019","unstructured":"Shang-Hua Gao, Ming-Ming Cheng, Kai Zhao, Xin-Yu Zhang, Ming-Hsuan Yang, and Philip Torr. 2019. Res2net: A new multi-scale backbone architecture. IEEE transactions on pattern analysis and machine intelligence 43, 2 (2019), 652--662."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR.2006.39"},{"key":"e_1_3_2_1_13_1","volume-title":"Computing inter-rater reliability for observational data: an overview and tutorial. Tutorials in quantitative methods for psychology 8, 1","author":"Hallgren Kevin A","year":"2012","unstructured":"Kevin A Hallgren. 2012. Computing inter-rater reliability for observational data: an overview and tutorial. Tutorials in quantitative methods for psychology 8, 1 (2012), 23."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2004.840618"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00685"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413620"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v28i1.8724"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11945"},{"key":"e_1_3_2_1_20_1","volume-title":"Aff-wild2: Extending the aff-wild database for affect recognition. arXiv preprint arXiv:1811.07770","author":"Kollias Dimitrios","year":"2018","unstructured":"Dimitrios Kollias and Stefanos Zafeiriou. 2018. Aff-wild2: Extending the aff-wild database for affect recognition. arXiv preprint arXiv:1811.07770 (2018)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"e_1_3_2_1_22_1","volume-title":"NEURIPS 2021 Workshop for Data Centric AI.","author":"Lavitas Liliya","year":"2021","unstructured":"Liliya Lavitas, Olivia Redfield, Allen Lee, Daniel Fletcher, Matthias Eck, and Sunil Janardhanan. 2021. Annotation quality framework-accuracy, credibility, and consistency. In NEURIPS 2021 Workshop for Data Centric AI."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.01024"},{"key":"e_1_3_2_1_24_1","volume-title":"Yun Liu, Ming-Ming Cheng, and Juergen Gall.","author":"Li Shijie","year":"2020","unstructured":"Shijie Li, Yazan Abu Farha, Yun Liu, Ming-Ming Cheng, and Juergen Gall. 2020. Ms-tcn: Multi-stage temporal convolutional network for action segmentation. IEEE transactions on pattern analysis and machine intelligence 45, 6 (2020), 6647--6658."},{"key":"e_1_3_2_1_25_1","volume-title":"MER 2024: Semi-Supervised Learning, Noise Robustness, and Open-Vocabulary Multimodal Emotion Recognition. arXiv preprint arXiv:2404","author":"Lian Zheng","year":"2024","unstructured":"Zheng Lian, Haiyang Sun, Licai Sun, Zhuofan Wen, Siyuan Zhang, Shun Chen, Hao Gu, Jinming Zhao, Ziyang Ma, Xie Chen, Jiangyan Yi, Rui Liu, et al. 2024. MER 2024: Semi-Supervised Learning, Noise Robustness, and Open-Vocabulary Multimodal Emotion Recognition. arXiv preprint arXiv:2404.17113 (2024)."},{"key":"e_1_3_2_1_26_1","volume-title":"Explainable multimodal emotion reasoning. arXiv preprint arXiv:2306.15401","author":"Lian Zheng","year":"2023","unstructured":"Zheng Lian, Licai Sun, Mingyu Xu, Haiyang Sun, Ke Xu, Zhuofan Wen, Shun Chen, Bin Liu, and Jianhua Tao. 2023. Explainable multimodal emotion reasoning. arXiv preprint arXiv:2306.15401 (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548407"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"e_1_3_2_1_30_1","unstructured":"John D Mayer Peter Salovey and David R Caruso. 2002. Mayer-Salovey-Caruso emotional intelligence test (MSCEIT) users manual. (2002)."},{"key":"e_1_3_2_1_31_1","first-page":"18571","article-title":"How would the viewer feel? Estimating wellbeing from video scenarios","volume":"35","author":"Mazeika Mantas","year":"2022","unstructured":"Mantas Mazeika, Eric Tang, Andy Zou, Steven Basart, Jun Shern Chan, Dawn Song, David Forsyth, Jacob Steinhardt, and Dan Hendrycks. 2022. How would the viewer feel? Estimating wellbeing from video scenarios. Advances in Neural Information Processing Systems 35 (2022), 18571--18585.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2023.104676"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-020-08836-3"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00806"},{"key":"e_1_3_2_1_35_1","volume-title":"The psychology and biology of emotion","author":"Plutchik Robert","unstructured":"Robert Plutchik. 1994. The psychology and biology of emotion. HarperCollins College Publishers."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1081"},{"key":"e_1_3_2_1_37_1","volume-title":"Meld: A multimodal multi-party dataset for emotion recognition in conversations. arXiv preprint arXiv:1810.02508","author":"Poria Soujanya","year":"2018","unstructured":"Soujanya Poria, Devamanyu Hazarika, Navonil Majumder, Gautam Naik, Erik Cambria, and Rada Mihalcea. 2018. Meld: A multimodal multi-party dataset for emotion recognition in conversations. arXiv preprint arXiv:1810.02508 (2018)."},{"key":"e_1_3_2_1_38_1","volume-title":"The circumplex model of affect: An integrative approach to affective neuroscience, cognitive development, and psychopathology. Development and psychopathology 17, 3","author":"Posner Jonathan","year":"2005","unstructured":"Jonathan Posner, James A Russell, and Bradley S Peterson. 2005. The circumplex model of affect: An integrative approach to affective neuroscience, cognitive development, and psychopathology. Development and psychopathology 17, 3 (2005), 715--734."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TBIOM.2022.3233083"},{"key":"e_1_3_2_1_40_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2011.6116320"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patrec.2021.03.007"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/MC.2018.3620963"},{"key":"e_1_3_2_1_44_1","volume-title":"International Journal of economics, commerce and management 2, 11","author":"Singh Ajay S","year":"2014","unstructured":"Ajay S Singh and Micah B Masuku. 2014. Sampling techniques & determination of sample size in applied statistics research: An overview. International Journal of economics, commerce and management 2, 11 (2014), 1--22."},{"key":"e_1_3_2_1_45_1","volume-title":"Opinion mining in social media: Modeling, simulating, and forecasting political opinions in the web. Government information quarterly 29, 4","author":"Sobkowicz Pawel","year":"2012","unstructured":"Pawel Sobkowicz, Michael Kaschesky, and Guillaume Bouchard. 2012. Opinion mining in social media: Modeling, simulating, and forecasting political opinions in the web. Government information quarterly 29, 4 (2012), 470--479."},{"key":"e_1_3_2_1_46_1","volume-title":"Msaf: Multimodal split attention fusion. arXiv preprint arXiv:2012.07175","author":"Su Lang","year":"2020","unstructured":"Lang Su, Chuqing Hu, Guofa Li, and Dongpu Cao. 2020. Msaf: Multimodal split attention fusion. arXiv preprint arXiv:2012.07175 (2020)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612365"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102382"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747278"},{"key":"e_1_3_2_1_51_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00203"},{"key":"e_1_3_2_1_53_1","volume-title":"CAM: A Fast and Efficient Network For Speaker Verification Using Context-Aware Masking. arXiv preprint arXiv:2303.00332","author":"Wang Hui","year":"2023","unstructured":"Hui Wang, Siqi Zheng, Yafeng Chen, Luyao Cheng, and Qian Chen. 2023. CAM: A Fast and Efficient Network For Speaker Verification Using Context-Aware Masking. arXiv preprint arXiv:2303.00332 (2023)."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00693"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911996.2912006"},{"key":"e_1_3_2_1_56_1","volume-title":"ICASSP 2023--2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Xu Ruize","unstructured":"Ruize Xu, Ruoxuan Feng, Shi-Xiong Zhang, and Di Hu. 2023. MMCosine: Multi-Modal Cosine Loss Towards Balanced Audio-Visual Fine-Grained Learning. In ICASSP 2023--2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1--5."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW56347.2022.00269"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00608"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00422"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.9987"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547869"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1208"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01811"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01219"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i01.5364"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475292"},{"key":"e_1_3_2_1_67_1","volume-title":"Prompting visual-language models for dynamic facial expression recognition. arXiv preprint arXiv:2308.13382","author":"Zhao Zengqun","year":"2023","unstructured":"Zengqun Zhao and Ioannis Patras. 2023. Prompting visual-language models for dynamic facial expression recognition. arXiv preprint arXiv:2308.13382 (2023)."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340555.3355713"}],"event":{"name":"ICMR '25: International Conference on Multimedia Retrieval","location":"Chicago IL USA","acronym":"ICMR '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2025 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731715.3733453","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T04:07:41Z","timestamp":1755749261000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731715.3733453"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":68,"alternative-id":["10.1145\/3731715.3733453","10.1145\/3731715"],"URL":"https:\/\/doi.org\/10.1145\/3731715.3733453","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-06-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}