{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T22:05:51Z","timestamp":1783029951958,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":60,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62306342, 62236004, 62206078, and 62476073"],"award-info":[{"award-number":["62306342, 62236004, 62206078, and 62476073"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Scientific Research Fund of Hunan Provincial Education Department","award":["24B0001"],"award-info":[{"award-number":["24B0001"]}]},{"name":"Excellent Young Scientists Fund in Hunan Province","award":["2024JJ4070"],"award-info":[{"award-number":["2024JJ4070"]}]},{"name":"Science and Technology Innovation Program of Hunan Province","award":["2024RC3024"],"award-info":[{"award-number":["2024RC3024"]}]},{"name":"CCF-Zhipu Large Model Innovation Fund","award":["NO.CCF-Zhipu202406"],"award-info":[{"award-number":["NO.CCF-Zhipu202406"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755837","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:56:43Z","timestamp":1761371803000},"page":"5267-5276","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-2493-2950","authenticated-orcid":false,"given":"Yongheng","family":"Zhang","sequence":"first","affiliation":[{"name":"School of Computer Science and Engineering, Central South University, ChangSha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6255-091X","authenticated-orcid":false,"given":"Xu","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Central South University, ChangSha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1824-4540","authenticated-orcid":false,"given":"Ruihan","family":"Tao","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Central South University, ChangSha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9154-7858","authenticated-orcid":false,"given":"Qiguang","family":"Chen","sequence":"additional","affiliation":[{"name":"Research Center for SCIR, Harbin Institute of Technology, Harbin, Heilongjiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3026-6347","authenticated-orcid":false,"given":"Hao","family":"Fei","sequence":"additional","affiliation":[{"name":"NExT Research Center, National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3907-0335","authenticated-orcid":false,"given":"Wanxiang","family":"Che","sequence":"additional","affiliation":[{"name":"Research Center for SCIR, Harbin Institute of Technology, Harbin, Heilongjiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3619-675X","authenticated-orcid":false,"given":"Libo","family":"Qin","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Central South University, ChangSha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","first-page":"53168","volume-title":"Zhang (Eds.)","volume":"37","author":"Chandrasegaran Keshigeyan","year":"2024","unstructured":"Keshigeyan Chandrasegaran, Agrim Gupta, Lea M. Hadzic, Taran Kota, Jimming He, Cristobal Eyzaguirre, Zane Durante, Manling Li, Jiajun Wu, and Li Fei-Fei. 2024. HourVideo: 1-Hour Video-Language Understanding. In Advances in Neural Information Processing Systems, A. Globerson, L. Mackey, D. Belgrave, A. Fan, U. Paquet, J. Tomczak, and C. Zhang (Eds.), Vol. 37. Curran Associates, Inc., 53168-53197. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2024\/file\/5f2809607f692d79a01c05c43d702883-Paper-Datasets_and_Benchmarks_Track.pdf"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0614"},{"key":"e_1_3_2_1_4_1","volume-title":"Towards reasoning era: A survey of long chain-of-thought for reasoning large language models. arXiv preprint arXiv:2503.09567","author":"Chen Qiguang","year":"2025","unstructured":"Qiguang Chen, Libo Qin, Jinhao Liu, Dengyun Peng, Jiannan Guan, Peng Wang, Mengkang Hu, Yuhang Zhou, Te Gao, and Wanxiang Che. 2025a. Towards reasoning era: A survey of long chain-of-thought for reasoning large language models. arXiv preprint arXiv:2503.09567 (2025)."},{"key":"e_1_3_2_1_5_1","first-page":"54872","article-title":"Unlocking the capabilities of thought: A reasoning boundary framework to quantify and optimize chain-of-thought","volume":"37","author":"Chen Qiguang","year":"2024","unstructured":"Qiguang Chen, Libo Qin, Jiaqi Wang, Jingxuan Zhou, and Wanxiang Che. 2024a. Unlocking the capabilities of thought: A reasoning boundary framework to quantify and optimize chain-of-thought. Advances in Neural Information Processing Systems, Vol. 37 (2024), 54872-54904.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_6_1","volume-title":"M3CoT: A Novel Benchmark for Multi-Domain Multi-step Multi-modal Chain-of-Thought. arXiv preprint arXiv:2405.16473","author":"Chen Qiguang","year":"2024","unstructured":"Qiguang Chen, Libo Qin, Jin Zhang, Zhi Chen, Xiao Xu, and Wanxiang Che. 2024b. M3CoT: A Novel Benchmark for Multi-Domain Multi-step Multi-modal Chain-of-Thought. arXiv preprint arXiv:2405.16473 (2024)."},{"key":"e_1_3_2_1_7_1","unstructured":"Qiguang Chen Mingda Yang Libo Qin Jinhao Liu Zheng Yan Jiannan Guan Dengyun Peng Yiyan Ji Hanjing Li Mengkang Hu et al. 2025b. AI4Research: A Survey of Artificial Intelligence for Scientific Research. arXiv preprint arXiv:2507.01903 (2025)."},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al., 2024d. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198."},{"key":"e_1_3_2_1_9_1","volume-title":"Videgothink: Assessing egocentric video understanding capabilities for embodied ai. arXiv preprint arXiv:2410.11623","author":"Cheng Sijie","year":"2024","unstructured":"Sijie Cheng, Kechen Fang, Yangyang Yu, Sicheng Zhou, Bohao Li, Ye Tian, Tingguang Li, Lei Han, and Yang Liu. 2024. Videgothink: Assessing egocentric video understanding capabilities for embodied ai. arXiv preprint arXiv:2410.11623 (2024)."},{"key":"e_1_3_2_1_10_1","volume-title":"Zhi Chen, Wanxiang Che, et al.","author":"Cheng Zihui","year":"2025","unstructured":"Zihui Cheng, Qiguang Chen, Xiao Xu, Jiaqi Wang, Weiyun Wang, Hao Fei, Yidong Wang, Alex Jinpeng Wang, Zhi Chen, Wanxiang Che, et al., 2025a. Visual thoughts: A unified perspective of understanding multimodal chain-of-thought. arXiv preprint arXiv:2505.15510 (2025)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i22.34538"},{"key":"e_1_3_2_1_12_1","volume-title":"Sheng Zheng, Jingyao Zheng, Lik-Hang Lee, Tae-Ho Kim, Choong Seon Hong, and Chaoning Zhang.","author":"Cho Joseph","year":"2024","unstructured":"Joseph Cho, Fachrina Dewi Puspitasari, Sheng Zheng, Jingyao Zheng, Lik-Hang Lee, Tae-Ho Kim, Choong Seon Hong, and Chaoning Zhang. 2024. Sora as an agi world model? a complete survey on text-to-video generation. arXiv preprint arXiv:2403.05131 (2024)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACVW60836.2024.00106"},{"key":"e_1_3_2_1_14_1","volume-title":"Dysen-VDM: Empowering Dynamics-Aware Text-to-Video Diffusion with LLMs. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE Computer Society, 7641-7653","author":"Fei Hao","year":"2024","unstructured":"Hao Fei, Shengqiong Wu, Wei Ji, Hanwang Zhang, and Tat-Seng Chua. 2024a. Dysen-VDM: Empowering Dynamics-Aware Text-to-Video Diffusion with LLMs. In 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE Computer Society, 7641-7653."},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning. 13109-13125","author":"Fei Hao","year":"2024","unstructured":"Hao Fei, Shengqiong Wu, Wei Ji, Hanwang Zhang, Meishan Zhang, Mong Li Lee, and Wynne Hsu. 2024b. Video-of-thought: step-by-step video reasoning from perception to cognition. In Proceedings of the 41st International Conference on Machine Learning. 13109-13125."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3393452"},{"key":"e_1_3_2_1_17_1","volume-title":"Video-r1: Reinforcing video reasoning in mllms. arXiv preprint arXiv:2503.21776","author":"Feng Kaituo","year":"2025","unstructured":"Kaituo Feng, Kaixiong Gong, Bohao Li, Zonghao Guo, Yibing Wang, Tianshuo Peng, Junfei Wu, Xiaoying Zhang, Benyou Wang, and Xiangyu Yue. 2025. Video-r1: Reinforcing video reasoning in mllms. arXiv preprint arXiv:2503.21776 (2025)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02438"},{"key":"e_1_3_2_1_19_1","volume-title":"Let's think frame by frame with vip: A video infilling and prediction dataset for evaluating video chain-of-thought. arXiv preprint arXiv:2305.13903","author":"Himakunthala Vaishnavi","year":"2023","unstructured":"Vaishnavi Himakunthala, Andy Ouyang, Daniel Rose, Ryan He, Alex Mei, Yujie Lu, Chinmay Sonar, Michael Saxon, and William Yang Wang. 2023. Let's think frame by frame with vip: A video infilling and prediction dataset for evaluating video chain-of-thought. arXiv preprint arXiv:2305.13903 (2023)."},{"key":"e_1_3_2_1_20_1","volume-title":"CoS: Chain-of-Shot Prompting for Long Video Understanding. arXiv preprint arXiv:2502.06428","author":"Hu Jian","year":"2025","unstructured":"Jian Hu, Zixu Cheng, Chenyang Si, Wei Li, and Shaogang Gong. 2025. CoS: Chain-of-Shot Prompting for Long Video Understanding. arXiv preprint arXiv:2502.06428 (2025)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612088"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCC.2009.2023380"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3672758.3672824"},{"key":"e_1_3_2_1_24_1","volume-title":"Question-Free Fine-Tuning for Adaptive Reasoning. arXiv preprint arXiv:2506.12860","author":"Liu Wanlong","year":"2025","unstructured":"Wanlong Liu, Junxiao Xu, Fei Yu, Yukang Lin, Ke Ji, Wenyu Chen, Yan Xu, Yasheng Wang, Lifeng Shang, and Benyou Wang. 2025. QFFT, Question-Free Fine-Tuning for Adaptive Reasoning. arXiv preprint arXiv:2506.12860 (2025)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"e_1_3_2_1_26_1","volume-title":"Umap: Uniform manifold approximation and projection for dimension reduction. arXiv preprint arXiv:1802.03426","author":"McInnes Leland","year":"2018","unstructured":"Leland McInnes, John Healy, and James Melville. 2018. Umap: Uniform manifold approximation and projection for dimension reduction. arXiv preprint arXiv:1802.03426 (2018)."},{"key":"e_1_3_2_1_27_1","volume-title":"An In-Depth Exploration. In The Thirty-eighth Annual Conference on Neural Information Processing Systems.","author":"Qin Libo","unstructured":"Libo Qin, Qiguang Chen, Hao Fei, Zhi Chen, Min Li, and Wanxiang Che. [n.d.]. What Factors Affect Multi-Modal In-Context Learning? An In-Depth Exploration. In The Thirty-eighth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_1_28_1","volume-title":"Large language models meet nlp: A survey. arXiv preprint arXiv:2405.12819","author":"Qin Libo","year":"2024","unstructured":"Libo Qin, Qiguang Chen, Xiachong Feng, Yang Wu, Yongheng Zhang, Yinghui Li, Min Li, Wanxiang Che, and Philip S Yu. 2024. Large language models meet nlp: A survey. arXiv preprint arXiv:2405.12819 (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"Patterns","volume":"6","author":"Qin Libo","year":"2025","unstructured":"Libo Qin, Qiguang Chen, Yuhang Zhou, Zhi Chen, Yinghui Li, Lizi Liao, Min Li, Wanxiang Che, and Philip S Yu. 2025. A survey of multilingual large language models. Patterns, Vol. 6, 1 (2025)."},{"key":"e_1_3_2_1_30_1","volume-title":"International conference on machine learning. PmLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PmLR, 8748-8763."},{"key":"e_1_3_2_1_31_1","unstructured":"Yunlong Tang Jing Bi Siting Xu Luchuan Song Susan Liang Teng Wang Daoan Zhang Jie An Jingyang Lin Rongyi Zhu et al. 2023. Video understanding with large language models: A survey. arXiv preprint arXiv:2312.17432 (2023)."},{"key":"e_1_3_2_1_32_1","volume-title":"Ryan Burnell, Libin Bai, Anmol Gulati, Garrett Tanzer, Damien Vincent, Zhufeng Pan, Shibo Wang, et al.","author":"Team Gemini","year":"2024","unstructured":"Gemini Team, Petko Georgiev, Ving Ian Lei, Ryan Burnell, Libin Bai, Anmol Gulati, Garrett Tanzer, Damien Vincent, Zhufeng Pan, Shibo Wang, et al., 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv preprint arXiv:2403.05530 (2024)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00375"},{"key":"e_1_3_2_1_34_1","volume-title":"Roy Ka-Wei Lee, and Ee-Peng Lim","author":"Wang Lei","year":"2023","unstructured":"Lei Wang, Wanyu Xu, Yihuai Lan, Zhiqiang Hu, Yunshi Lan, Roy Ka-Wei Lee, and Ee-Peng Lim. 2023a. Plan-and-solve prompting: Improving zero-shot chain-of-thought reasoning by large language models. arXiv preprint arXiv:2305.04091 (2023)."},{"key":"e_1_3_2_1_35_1","volume-title":"S3 agent: Unlocking the power of VLLM for zero-shot multi-modal sarcasm detection. ACM Transactions on Multimedia Computing, Communications and Applications","author":"Wang Peng","year":"2024","unstructured":"Peng Wang, Yongheng Zhang, Hao Fei, Qiguang Chen, Yukai Wang, Jiasheng Si, Wenpeng Lu, Min Li, and Libo Qin. 2024b. S3 agent: Unlocking the power of VLLM for zero-shot multi-modal sarcasm detection. ACM Transactions on Multimedia Computing, Communications and Applications (2024)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Shuai Wang Ivona Najdenkoska Hongyi Zhu Stevan Rudinac Monika Kackovic Nachoem Wijnberg and Marcel Worring. 2025a. ArtRAG: Retrieval-Augmented Generation with Structured Context for Visual Art Understanding. arXiv:2505.06020 [cs.AI] https:\/\/arxiv.org\/abs\/2505.06020","DOI":"10.1145\/3746027.3755673"},{"key":"e_1_3_2_1_37_1","volume-title":"Aakanksha Chowdhery, and Denny Zhou.","author":"Wang Xuezhi","year":"2022","unstructured":"Xuezhi Wang, Jason Wei, Dale Schuurmans, Quoc Le, Ed Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou. 2022. Self-consistency improves chain of thought reasoning in language models. arXiv preprint arXiv:2203.11171 (2022)."},{"key":"e_1_3_2_1_38_1","volume-title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey. arXiv preprint arXiv:2503.12605","author":"Wang Yaoting","year":"2025","unstructured":"Yaoting Wang, Shengqiong Wu, Yuecheng Zhang, William Wang, Ziwei Liu, Jiebo Luo, and Hao Fei. 2025b. Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey. arXiv preprint arXiv:2503.12605 (2025)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.alvr-1.8"},{"key":"e_1_3_2_1_40_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al., 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, Vol. 35 (2022), 24824-24837."},{"key":"e_1_3_2_1_41_1","volume-title":"European Conference on Computer Vision. Springer, 453-470","author":"Weng Yuetian","year":"2024","unstructured":"Yuetian Weng, Mingfei Han, Haoyu He, Xiaojun Chang, and Bohan Zhuang. 2024. Longvlm: Efficient long video understanding via large language models. In European Conference on Computer Vision. Springer, 453-470."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00192"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681489"},{"key":"e_1_3_2_1_44_1","volume-title":"The role of chain-of-thought in complex vision-language reasoning task. arXiv preprint arXiv:2311.09193","author":"Wu Yifan","year":"2023","unstructured":"Yifan Wu, Pengchuan Zhang, Wenhan Xiong, Barlas Oguz, James C Gee, and Yixin Nie. 2023. The role of chain-of-thought in complex vision-language reasoning task. arXiv preprint arXiv:2311.09193 (2023)."},{"key":"e_1_3_2_1_45_1","volume-title":"Beyond chain-of-thought: A survey of chain-of-x paradigms for llms. arXiv preprint arXiv:2404.15676","author":"Xia Yu","year":"2024","unstructured":"Yu Xia, Rui Wang, Xu Liu, Mingyan Li, Tong Yu, Xiang Chen, Julian McAuley, and Shuai Li. 2024. Beyond chain-of-thought: A survey of chain-of-x paradigms for llms. arXiv preprint arXiv:2404.15676 (2024)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3614425"},{"key":"e_1_3_2_1_47_1","volume-title":"ActiveRAG: Autonomously Knowledge Assimilation and Accommodation through Retrieval-Augmented Agents. arXiv preprint arXiv:2402.13547","author":"Xu Zhipeng","year":"2024","unstructured":"Zhipeng Xu, Zhenghao Liu, Yukun Yan, Shuo Wang, Shi Yu, Zheni Zeng, Chaojun Xiao, Zhiyuan Liu, Ge Yu, and Chenyan Xiong. 2024. ActiveRAG: Autonomously Knowledge Assimilation and Accommodation through Retrieval-Augmented Agents. arXiv preprint arXiv:2402.13547 (2024)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01032"},{"key":"e_1_3_2_1_49_1","unstructured":"An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei et al. 2024. Qwen2. 5 technical report. arXiv preprint arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_50_1","volume-title":"Tree of thoughts: Deliberate problem solving with large language models. Advances in neural information processing systems","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Dian Yu, Jeffrey Zhao, Izhak Shafran, Tom Griffiths, Yuan Cao, and Karthik Narasimhan. 2023. Tree of thoughts: Deliberate problem solving with large language models. Advances in neural information processing systems, Vol. 36 (2023), 11809-11822."},{"key":"e_1_3_2_1_51_1","volume-title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi. arXiv preprint arXiv:2404.16006","author":"Ying Kaining","year":"2024","unstructured":"Kaining Ying, Fanqing Meng, Jin Wang, Zhiqian Li, Han Lin, Yue Yang, Hao Zhang, Wenbo Zhang, Yuqi Lin, Shuo Liu, et al., 2024. Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi. arXiv preprint arXiv:2404.16006 (2024)."},{"key":"e_1_3_2_1_52_1","first-page":"1","article-title":"Find Details in Long Videos: Tower-of-Thoughts and Self-Retrieval Augmented Generation for Video Understanding. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Yue Tong","year":"2025","unstructured":"Tong Yue, Mingrui Xiao, Dafeng Zhang, Xin Liu, Yali Li, and Shengjin Wang. 2025. Find Details in Long Videos: Tower-of-Thoughts and Self-Retrieval Augmented Generation for Video Understanding. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1-5.","journal-title":"IEEE"},{"key":"e_1_3_2_1_53_1","volume-title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding. arXiv preprint arXiv:2501.13106","author":"Zhang Boqiang","year":"2025","unstructured":"Boqiang Zhang, Kehan Li, Zesen Cheng, Zhiqiang Hu, Yuqian Yuan, Guanzheng Chen, Sicong Leng, Yuming Jiang, Hang Zhang, Xin Li, Peng Jin, Wenqi Zhang, Fan Wang, Lidong Bing, and Deli Zhao. 2025a. VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding. arXiv preprint arXiv:2501.13106 (2025). https:\/\/arxiv.org\/abs\/2501.13106"},{"key":"e_1_3_2_1_54_1","first-page":"9191","article-title":"AutoCAP","volume":"2024","author":"Zhang Yongheng","year":"2024","unstructured":"Yongheng Zhang, Qiguang Chen, Min Li, Wanxiang Che, and Libo Qin. 2024a. AutoCAP: Towards Automatic Cross-lingual Alignment Planning for Zero-shot Chain-of-Thought. In Findings of the Association for Computational Linguistics ACL 2024. 9191-9200.","journal-title":"Towards Automatic Cross-lingual Alignment Planning for Zero-shot Chain-of-Thought. In Findings of the Association for Computational Linguistics ACL"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.388"},{"key":"e_1_3_2_1_56_1","volume-title":"CCHall: A Novel Benchmark for Joint Cross-Lingual and Cross-Modal Hallucinations Detection in Large Language Models. arXiv preprint arXiv:2505.19108","author":"Zhang Yongheng","year":"2025","unstructured":"Yongheng Zhang, Xu Liu, Ruoxi Zhou, Qiguang Chen, Hao Fei, Wenpeng Lu, and Libo Qin. 2025b. CCHall: A Novel Benchmark for Joint Cross-Lingual and Cross-Modal Hallucinations Detection in Large Language Models. arXiv preprint arXiv:2505.19108 (2025)."},{"key":"e_1_3_2_1_57_1","unstructured":"Wayne Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong et al. 2023b. A survey of large language models. arXiv preprint arXiv:2303.18223 Vol. 1 2 (2023)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00637"},{"key":"e_1_3_2_1_59_1","volume-title":"Mlvu: A comprehensive benchmark for multi-task long video understanding. arXiv preprint arXiv:2406.04264","author":"Zhou Junjie","year":"2024","unstructured":"Junjie Zhou, Yan Shu, Bo Zhao, Boya Wu, Shitao Xiao, Xi Yang, Yongping Xiong, Bo Zhang, Tiejun Huang, and Zheng Liu. 2024a. Mlvu: A comprehensive benchmark for multi-task long video understanding. arXiv preprint arXiv:2406.04264 (2024)."},{"key":"e_1_3_2_1_60_1","volume-title":"A survey on generative ai and llm for video generation, understanding, and streaming. arXiv preprint arXiv:2404.16038","author":"Zhou Pengyuan","year":"2024","unstructured":"Pengyuan Zhou, Lin Wang, Zhi Liu, Yanbin Hao, Pan Hui, Sasu Tarkoma, and Jussi Kangasharju. 2024b. A survey on generative ai and llm for video generation, understanding, and streaming. arXiv preprint arXiv:2404.16038 (2024)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755837","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:03:28Z","timestamp":1765339408000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755837"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":60,"alternative-id":["10.1145\/3746027.3755837","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755837","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}