{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,23]],"date-time":"2025-10-23T01:09:16Z","timestamp":1761181756358,"version":"build-2065373602"},"publisher-location":"New York, NY, USA","reference-count":20,"publisher":"ACM","funder":[{"name":"National Key Research and Development Program of China","award":["2023YFC3306205"],"award-info":[{"award-number":["2023YFC3306205"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746270.3760218","type":"proceedings-article","created":{"date-parts":[[2025,10,20]],"date-time":"2025-10-20T15:14:09Z","timestamp":1760973249000},"page":"23-29","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["ZeroES: Zero-Shot Ensemble for Open-Vocabulary Video Emotion Recognition with Large Multimodal Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6152-3943","authenticated-orcid":false,"given":"Jun","family":"Xie","sequence":"first","affiliation":[{"name":"Lenovo Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1498-1989","authenticated-orcid":false,"given":"Xiaohui","family":"Fan","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1530-7486","authenticated-orcid":false,"given":"Zhenghao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Lenovo Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8278-486X","authenticated-orcid":false,"given":"Feng","family":"Chen","sequence":"additional","affiliation":[{"name":"Lenovo Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3333-5863","authenticated-orcid":false,"given":"Hongzhu","family":"Yi","sequence":"additional","affiliation":[{"name":"Lenovo Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5538-0645","authenticated-orcid":false,"given":"Yingjian","family":"Zhu","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8887-3735","authenticated-orcid":false,"given":"Xiongjun","family":"Guan","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-9111-4817","authenticated-orcid":false,"given":"Xinming","family":"Wang","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6187-5641","authenticated-orcid":false,"given":"Yue","family":"Bi","sequence":"additional","affiliation":[{"name":"Shandong University, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2980-6281","authenticated-orcid":false,"given":"Tao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6088-3517","authenticated-orcid":false,"given":"Zhepeng","family":"Wang","sequence":"additional","affiliation":[{"name":"Lenovo Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,26]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.matpr.2021.07.046"},{"volume-title":"Internet Imaging VI.","author":"Sebe Nicu","key":"e_1_3_2_1_2_1","unstructured":"Nicu Sebe, Ira Cohen, Theo Gevers, and Thomas S Huang. 2005. Multimodal approaches for emotion recognition: a survey. In Internet Imaging VI. Vol. 5670. SPIE, 56--67."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.3390\/app14178071"},{"key":"e_1_3_2_1_4_1","unstructured":"Zheng Lian et al. 2025. Open-vocabulary multimodal emotion recognition: dataset metric and benchmark. (2025). https:\/\/openreview.net\/forum?id=f1uXrAjpOH."},{"key":"e_1_3_2_1_5_1","unstructured":"Zheng Lian et al. 2025. Ov-mer: towards open-vocabulary multimodal emotion recognition. (2025). https : \/ \/ arxiv . org\/ abs \/2410.01495 arXiv: 2410.01495 [cs.HC]."},{"key":"e_1_3_2_1_6_1","unstructured":"Josh Achiam et al. 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774."},{"key":"e_1_3_2_1_7_1","unstructured":"Gemini Team et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689092.3689403"},{"key":"e_1_3_2_1_9_1","unstructured":"Mengying Ge Dongkai Tang and Mingyang Li. 2024. Video emotion open vocabulary recognition based on multimodal large language model. (2024). https:\/\/arxiv.org\/abs\/2408.11286 arXiv: 2408.11286 [cs.CV]."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689092.3689402"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689092.3689404"},{"key":"e_1_3_2_1_12_1","unstructured":"Gheorghe Comanici et al. 2025. Gemini 2.5: pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:2507.06261."},{"key":"e_1_3_2_1_13_1","unstructured":"Jinguo Zhu et al. 2025. Internvl3: exploring advanced training and test-time recipes for open-source multimodal models. (2025). https:\/\/arxiv.org\/abs\/2504.10479 arXiv: 2504.10479 [cs.CV]."},{"key":"e_1_3_2_1_14_1","unstructured":"Zheng Lian et al. 2025. Mer 2025: when affective computing meets large language models. arXiv preprint arXiv:2504.19423."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612836"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689092.3689959"},{"key":"e_1_3_2_1_17_1","unstructured":"Zheng Lian et al. 2025. Affectgpt: a new dataset model and benchmark for emotion understanding with multimodal large language models. (2025). https:\/\/arxiv.org\/abs\/2501.16566 arXiv: 2501.16566 [cs.HC]."},{"key":"e_1_3_2_1_18_1","unstructured":"Zheng Lian Haiyang Sun Licai Sun Jiangyan Yi Bin Liu and Jianhua Tao. 2024. Affectgpt: dataset and framework for explainable multimodal emotion recognition. (2024). https : \/ \/ arxiv . org\/ abs \/2407.07653 arXiv: 2407.07653 [cs.HC]."},{"key":"e_1_3_2_1_19_1","unstructured":"Jin Xu et al. 2025. Qwen2.5-omni technical report. (2025). https:\/\/arxiv.org\/abs\/2503.20215 arXiv: 2503.20215 [cs.CL]."},{"key":"e_1_3_2_1_20_1","unstructured":"Shuai Bai et al. 2025. Qwen2.5-vl technical report. (2025). https:\/\/arxiv.org\/abs\/2502.13923 arXiv: 2502.13923 [cs.CV]."}],"event":{"name":"MM '25:The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland"},"container-title":["Proceedings of the 3rd International Workshop on Multimodal and Responsible Affective Computing"],"original-title":[],"deposited":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T17:22:42Z","timestamp":1761153762000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746270.3760218"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,26]]},"references-count":20,"alternative-id":["10.1145\/3746270.3760218","10.1145\/3746270"],"URL":"https:\/\/doi.org\/10.1145\/3746270.3760218","relation":{},"subject":[],"published":{"date-parts":[[2025,10,26]]},"assertion":[{"value":"2025-10-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}