{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,10]],"date-time":"2026-04-10T17:13:47Z","timestamp":1775841227985,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":90,"publisher":"ACM","funder":[{"name":"the National Natural Science Foundation of China","award":["62176029"],"award-info":[{"award-number":["62176029"]}]},{"name":"the National Natural Science Foundation of China","award":["62506050"],"award-info":[{"award-number":["62506050"]}]},{"name":"the National Natural Science Foundation of China","award":["62301538"],"award-info":[{"award-number":["62301538"]}]},{"name":"China Postdoctoral Science Foundation Funded Project","award":["2024M763867"],"award-info":[{"award-number":["2024M763867"]}]},{"name":"Chongqing Higher Education Teaching Reform Research Project","award":["242009"],"award-info":[{"award-number":["242009"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792726","type":"proceedings-article","created":{"date-parts":[[2026,4,9]],"date-time":"2026-04-09T21:54:39Z","timestamp":1775771679000},"page":"2501-2512","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["CFVBench: A Comprehensive Video Benchmark for Fine-grained Multimodal Retrieval-Augmented Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5830-0802","authenticated-orcid":false,"given":"Kaiwen","family":"Wei","sequence":"first","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0233-7410","authenticated-orcid":false,"given":"Xiao","family":"Liu","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0263-5037","authenticated-orcid":false,"given":"Jie","family":"Zhang","sequence":"additional","affiliation":[{"name":"Independent Researcher, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3598-4841","authenticated-orcid":false,"given":"Zijian","family":"Wang","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4864-2795","authenticated-orcid":false,"given":"Ruida","family":"Liu","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5000-7609","authenticated-orcid":false,"given":"Yuming","family":"Yang","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2846-6749","authenticated-orcid":false,"given":"Xin","family":"Xiao","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7256-4357","authenticated-orcid":false,"given":"Xiao","family":"Sun","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9810-5684","authenticated-orcid":false,"given":"Haoyang","family":"Zeng","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2676-9223","authenticated-orcid":false,"given":"Changzai","family":"Pan","sequence":"additional","affiliation":[{"name":"Independent Researcher, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7466-0234","authenticated-orcid":false,"given":"Yidan","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of the Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5169-4634","authenticated-orcid":false,"given":"Jiang","family":"Zhong","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0371-5584","authenticated-orcid":false,"given":"Peijin","family":"Wang","sequence":"additional","affiliation":[{"name":"University of the Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4017-8885","authenticated-orcid":false,"given":"Yingchao","family":"Feng","sequence":"additional","affiliation":[{"name":"University of the Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Musiclm: Generating music from text. arXiv preprint arXiv:2301.11325","author":"Agostinelli Andrea","year":"2023","unstructured":"Andrea Agostinelli, Timo I Denk, Zal\u00e1n Borsos, Jesse Engel, Mauro Verzetti, Antoine Caillon, Qingqing Huang, Aren Jansen, Adam Roberts, Marco Tagliasacchi, et al. 2023. Musiclm: Generating music from text. arXiv preprint arXiv:2301.11325 (2023)."},{"key":"e_1_3_2_1_2_1","unstructured":"Anthropic. 2025. Introducing Claude 4. https:\/\/www.anthropic.com\/news\/claude-4"},{"key":"e_1_3_2_1_3_1","unstructured":"Lei Bai Zhongrui Cai Maosong Cao Weihan Cao Chiyu Chen Haojiong Chen Kai Chen Pengcheng Chen Ying Chen Yongkang Chen et al. 2025. Intern-s1: A scientific multimodal foundation model. arXiv preprint arXiv:2508.15763 (2025)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0614"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3185"},{"key":"e_1_3_2_1_7_1","volume-title":"Can pre-trained vision and language models answer visual information-seeking questions? arXiv preprint arXiv:2302.11713","author":"Chen Yang","year":"2023","unstructured":"Yang Chen, Hexiang Hu, Yi Luan, Haitian Sun, Soravit Changpinyo, Alan Ritter, and Ming-Wei Chang. 2023. Can pre-trained vision and language models answer visual information-seeking questions? arXiv preprint arXiv:2302.11713 (2023)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"e_1_3_2_1_10_1","volume-title":"MAGNET: A Multi-agent Framework for Finding Audio-Visual Needles by Reasoning over Multi-Video Haystacks. arXiv preprint arXiv:2506.07016","author":"Chowdhury Sanjoy","year":"2025","unstructured":"Sanjoy Chowdhury, Mohamed Elmoghany, Yohan Abeysinghe, Junjie Fei, Sayan Nag, Salman Khan, Mohamed Elhoseiny, and Dinesh Manocha. 2025. MAGNET: A Multi-agent Framework for Finding Audio-Visual Needles by Reasoning over Multi-Video Haystacks. arXiv preprint arXiv:2506.07016 (2025)."},{"key":"e_1_3_2_1_11_1","unstructured":"XTuner Contributors. 2023. XTuner: A Toolkit for Efficiently Fine-tuning LLM. https:\/\/github.com\/InternLM\/xtuner."},{"key":"e_1_3_2_1_12_1","unstructured":"DeepSeek-AI Daya Guo and etc. Dejian Yang. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. arXiv:2501.12948 [cs.CL] https:\/\/arxiv.org\/abs\/2501.12948"},{"key":"e_1_3_2_1_13_1","volume-title":"European Conference on Computer Vision. Springer, 75-92","author":"Fan Yue","year":"2024","unstructured":"Yue Fan, Xiaojian Ma, Rujie Wu, Yuntao Du, Jiaqi Li, Zhi Gao, and Qing Li. 2024. Videoagent: A memory-augmented multimodal agent for video understanding. In European Conference on Computer Vision. Springer, 75-92."},{"key":"e_1_3_2_1_14_1","first-page":"89098","article-title":"Mmbench-video: A long-form multi-shot benchmark for holistic video understanding","volume":"37","author":"Fang Xinyu","year":"2024","unstructured":"Xinyu Fang, Kangrui Mao, Haodong Duan, Xiangyu Zhao, Yining Li, Dahua Lin, and Kai Chen. 2024. Mmbench-video: A long-form multi-shot benchmark for holistic video understanding. Advances in Neural Information Processing Systems 37 (2024), 89098-89124.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_15_1","volume-title":"The equivalence of weighted kappa and the intraclass correlation coefficient as measures of reliability. Educational and psychological measurement 33, 3","author":"Fleiss Joseph L","year":"1973","unstructured":"Joseph L Fleiss and Jacob Cohen. 1973. The equivalence of weighted kappa and the intraclass correlation coefficient as measures of reliability. Educational and psychological measurement 33, 3 (1973), 613-619."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02245"},{"key":"e_1_3_2_1_17_1","unstructured":"Sanchit Gandhi Patrick von Platen and Alexander M Rush. 2023. Distil-Whisper: Robust knowledge distillation via large-scale pseudo labelling."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6713"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02197"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"e_1_3_2_1_21_1","unstructured":"Team GLM : Aohan Zeng and et.al. Bin Xu. 2024. ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools. arXiv:2406.12793 [cs.CL] https:\/\/arxiv.org\/abs\/2406.12793"},{"key":"e_1_3_2_1_22_1","unstructured":"Google. 2025. Continuing to bring you our latest models with an improved Gemini 2.5 Flash and Flash-Lite release. https:\/\/developers.googleblog.com\/en\/continuing-to-bring-you-our-latest-models-with-an-improved-gemini-2-5-flash-and-flash-lite-release\/"},{"key":"e_1_3_2_1_23_1","volume-title":"Mrag-bench: Vision-centric evaluation for retrievalaugmented multimodal models. arXiv preprint arXiv:2410.08182","author":"Hu Wenbo","year":"2024","unstructured":"Wenbo Hu, Jia-Chen Gu, Zi-Yi Dou, Mohsen Fayyaz, Pan Lu, Kai-Wei Chang, and Nanyun Peng. 2024. Mrag-bench: Vision-centric evaluation for retrievalaugmented multimodal models. arXiv preprint arXiv:2410.08182 (2024)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i4.32427"},{"key":"e_1_3_2_1_25_1","volume-title":"Likert scale: Explored and explained. British journal of applied science & technology 7, 4","author":"Joshi Ankur","year":"2015","unstructured":"Ankur Joshi, Saket Kale, Satish Chandel, and D Kumar Pal. 2015. Likert scale: Explored and explained. British journal of applied science & technology 7, 4 (2015), 396."},{"key":"e_1_3_2_1_26_1","volume-title":"An empirical comparison of video frame sampling methods for multi-modal rag retrieval. arXiv preprint arXiv:2408.03340","author":"Kandhare Mahesh","year":"2024","unstructured":"Mahesh Kandhare and Thibault Gisselbrecht. 2024. An empirical comparison of video frame sampling methods for multi-modal rag retrieval. arXiv preprint arXiv:2408.03340 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"Enhancing llm factual accuracy with rag to counter hallucinations: A case study on domain-specific queries in private knowledge-bases. arXiv preprint arXiv:2403.10446","author":"Li Jiarui","year":"2024","unstructured":"Jiarui Li, Ye Yuan, and Zehua Zhang. 2024. Enhancing llm factual accuracy with rag to counter hallucinations: A case study on domain-specific queries in private knowledge-bases. arXiv preprint arXiv:2403.10446 (2024)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02095"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730077"},{"key":"e_1_3_2_1_30_1","unstructured":"Yangning Li Yinghui Li Xinyu Wang Yong Jiang Zhen Zhang Xinran Zheng Hui Wang Hai-Tao Zheng Philip S Yu Fei Huang et al. 2024. Benchmarking multimodal retrieval augmented generation with dynamic vqa dataset and selfadaptive planning agent. arXiv preprint arXiv:2411.02937 (2024)."},{"key":"e_1_3_2_1_31_1","volume-title":"European Conference on Computer Vision. Springer, 323-340","author":"Li Yanwei","year":"2024","unstructured":"Yanwei Li, Chengyao Wang, and Jiaya Jia. 2024. Llama-vid: An image is worth 2 tokens in large language models. In European Conference on Computer Vision. Springer, 323-340."},{"key":"e_1_3_2_1_32_1","volume-title":"Omnibench: Towards the future of universal omni-language models. arXiv preprint arXiv:2409.15272","author":"Li Yizhi","year":"2024","unstructured":"Yizhi Li, Ge Zhang, Yinghao Ma, Ruibin Yuan, Kang Zhu, Hangyu Guo, Yiming Liang, Jiaheng Liu, Zekun Wang, Jian Yang, et al. 2024. Omnibench: Towards the future of universal omni-language models. arXiv preprint arXiv:2409.15272 (2024)."},{"key":"e_1_3_2_1_33_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81."},{"key":"e_1_3_2_1_34_1","volume-title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want. arXiv preprint arXiv:2403.20271","author":"Lin Weifeng","year":"2024","unstructured":"Weifeng Lin, XinyuWei, Ruichuan An, Peng Gao, Bocheng Zou, Yulin Luo, Siyuan Huang, Shanghang Zhang, and Hongsheng Li. 2024. Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want. arXiv preprint arXiv:2403.20271 (2024)."},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. 4478-4487","author":"Liu Xiulong","year":"2024","unstructured":"Xiulong Liu, Zhikang Dong, and Peng Zhang. 2024. Tackling data bias in musicavqa: Crafting a balanced dataset for unbiased question-answering. In Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. 4478-4487."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746786"},{"key":"e_1_3_2_1_37_1","volume-title":"Video-rag: Visually-aligned retrieval-augmented long video comprehension. arXiv preprint arXiv:2411.13093","author":"Luo Yongdong","year":"2024","unstructured":"Yongdong Luo, Xiawu Zheng, Xiao Yang, Guilin Li, Haojia Lin, Jinfa Huang, Jiayi Ji, Fei Chao, Jiebo Luo, and Rongrong Ji. 2024. Video-rag: Visually-aligned retrieval-augmented long video comprehension. arXiv preprint arXiv:2411.13093 (2024)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437963.3441777"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01764"},{"key":"e_1_3_2_1_40_1","first-page":"46212","article-title":"Egoschema: A diagnostic benchmark for very long-form video language understanding","volume":"36","author":"Mangalam Karttikeya","year":"2023","unstructured":"Karttikeya Mangalam, Raiymbek Akshulakov, and Jitendra Malik. 2023. Egoschema: A diagnostic benchmark for very long-form video language understanding. Advances in Neural Information Processing Systems 36 (2023), 46212-46244.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_41_1","volume-title":"Multi-RAG: A Multimodal Retrieval-Augmented Generation System for Adaptive Video Understanding. arXiv preprint arXiv:2505.23990","author":"Mao Mingyang","year":"2025","unstructured":"Mingyang Mao, Mariela M Perez-Cabarcas, Utteja Kallakuri, Nicholas R Waytowich, Xiaomin Lin, and Tinoosh Mohsenin. 2025. Multi-RAG: A Multimodal Retrieval-Augmented Generation System for Adaptive Video Understanding. arXiv preprint arXiv:2505.23990 (2025)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00331"},{"key":"e_1_3_2_1_43_1","volume-title":"A survey of multimodal retrieval-augmented generation. arXiv preprint arXiv:2504.08748","author":"Mei Lang","year":"2025","unstructured":"Lang Mei, Siyu Mo, Zhihan Yang, and Chong Chen. 2025. A survey of multimodal retrieval-augmented generation. arXiv preprint arXiv:2504.08748 (2025)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3419446"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00289"},{"key":"e_1_3_2_1_46_1","volume-title":"Video-bench: A comprehensive benchmark and toolkit for evaluating video-based large language models. arXiv preprint arXiv:2311.16103","author":"Ning Munan","year":"2023","unstructured":"Munan Ning, Bin Zhu, Yujia Xie, Bin Lin, Jiaxi Cui, Lu Yuan, Dongdong Chen, and Li Yuan. 2023. Video-bench: A comprehensive benchmark and toolkit for evaluating video-based large language models. arXiv preprint arXiv:2311.16103 (2023)."},{"key":"e_1_3_2_1_47_1","volume-title":"Nomic Embed: Training a Reproducible Long Context Text Embedder. arXiv:2402.01613 [cs.CL]","author":"Nussbaum Zach","year":"2024","unstructured":"Zach Nussbaum, John X. Morris, Brandon Duderstadt, and Andriy Mulyar. 2024. Nomic Embed: Training a Reproducible Long Context Text Embedder. arXiv:2402.01613 [cs.CL]"},{"key":"e_1_3_2_1_48_1","unstructured":"OpenAI. 2025. Introducing GPT-5. https:\/\/openai.com\/index\/introducing-gpt-5\/"},{"key":"e_1_3_2_1_49_1","volume-title":"Evaluation: from precision, recall and F-measure to ROC, informedness, markedness and correlation. arXiv preprint arXiv:2010.16061","author":"Powers David MW","year":"2020","unstructured":"David MW Powers. 2020. Evaluation: from precision, recall and F-measure to ROC, informedness, markedness and correlation. arXiv preprint arXiv:2010.16061 (2020)."},{"key":"e_1_3_2_1_50_1","volume-title":"Tiger: Unifying text-to-image generation and retrieval with large multimodal models. arXiv preprint arXiv:2406.05814","author":"Qu Leigang","year":"2024","unstructured":"Leigang Qu, Haochuan Li, Tan Wang, Wenjie Wang, Yongqi Li, Liqiang Nie, and Tat-Seng Chua. 2024. Tiger: Unifying text-to-image generation and retrieval with large multimodal models. arXiv preprint arXiv:2406.05814 (2024)."},{"key":"e_1_3_2_1_51_1","volume-title":"Khyathi Raghavi Chandu, et al","author":"Rastogi Abhinav","year":"2025","unstructured":"Abhinav Rastogi, Albert Q Jiang, Andy Lo, Gabrielle Berrada, Guillaume Lample, Jason Rute, Joep Barmentlo, Karmesh Yadav, Kartik Khandelwal, Khyathi Raghavi Chandu, et al. 2025. Magistral. arXiv preprint arXiv:2506.10910 (2025)."},{"key":"e_1_3_2_1_52_1","volume-title":"Cinepile: A long video question answering dataset and benchmark. arXiv preprint arXiv:2405.08813","author":"Rawal Ruchit","year":"2024","unstructured":"Ruchit Rawal, Khalid Saifullah, Miquel Farr\u00e9, Ronen Basri, David Jacobs, Gowthami Somepalli, and Tom Goldstein. 2024. Cinepile: A long video question answering dataset and benchmark. arXiv preprint arXiv:2405.08813 (2024)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_1_54_1","volume-title":"Videorag: Retrieval-augmented generation with extreme long-context videos. arXiv preprint arXiv:2502.01549","author":"Ren Xubin","year":"2025","unstructured":"Xubin Ren, Lingrui Xu, Long Xia, ShuaiqiangWang, Dawei Yin, and Chao Huang. 2025. Videorag: Retrieval-augmented generation with extreme long-context videos. arXiv preprint arXiv:2502.01549 (2025)."},{"key":"e_1_3_2_1_55_1","volume-title":"VideoRAG: Retrieval-Augmented Generation with Extreme Long-Context Videos. arXiv preprint arXiv:2502.01549","author":"Ren Xubin","year":"2025","unstructured":"Xubin Ren, Lingrui Xu, Long Xia, ShuaiqiangWang, Dawei Yin, and Chao Huang. 2025. VideoRAG: Retrieval-Augmented Generation with Extreme Long-Context Videos. arXiv preprint arXiv:2502.01549 (2025)."},{"key":"e_1_3_2_1_56_1","volume-title":"How2: a large-scale dataset for multimodal language understanding. arXiv preprint arXiv:1811.00347","author":"Sanabria Ramon","year":"2018","unstructured":"Ramon Sanabria, Ozan Caglayan, Shruti Palaskar, Desmond Elliott, Lo\u00efc Barrault, Lucia Specia, and Florian Metze. 2018. How2: a large-scale dataset for multimodal language understanding. arXiv preprint arXiv:1811.00347 (2018)."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20074-8_9"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018876"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01725"},{"key":"e_1_3_2_1_60_1","first-page":"46345","article-title":"Learning to tokenize for generative retrieval","volume":"36","author":"Sun Weiwei","year":"2023","unstructured":"Weiwei Sun, Lingyong Yan, Zheng Chen, ShuaiqiangWang, Haichao Zhu, Pengjie Ren, Zhumin Chen, Dawei Yin, Maarten Rijke, and Zhaochun Ren. 2023. Learning to tokenize for generative retrieval. Advances in Neural Information Processing Systems 36 (2023), 46345-46361.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_61_1","volume-title":"Joon Son Chung, and Tae-Hyun Oh","author":"Sung-Bin Kim","year":"2024","unstructured":"Kim Sung-Bin, Oh Hyun-Bin, JungMok Lee, Arda Senocak, Joon Son Chung, and Tae-Hyun Oh. 2024. Avhbench: A cross-modal hallucination benchmark for audio-visual large language models. arXiv preprint arXiv:2410.18325 (2024)."},{"key":"e_1_3_2_1_62_1","volume-title":"Multimodalqa: Complex question answering over text, tables and images. arXiv preprint arXiv:2104.06039","author":"Talmor Alon","year":"2021","unstructured":"Alon Talmor, Ori Yoran, Amnon Catav, Dan Lahav, Yizhong Wang, Akari Asai, Gabriel Ilharco, Hannaneh Hajishirzi, and Jonathan Berant. 2021. Multimodalqa: Complex question answering over text, tables and images. arXiv preprint arXiv:2104.06039 (2021)."},{"key":"e_1_3_2_1_63_1","unstructured":"Gemma Team. 2025. Gemma 3. https:\/\/google\/Gemma3Report"},{"key":"e_1_3_2_1_64_1","unstructured":"Qwen Team. 2025. Qwen2.5-VL. https:\/\/qwenlm.github.io\/blog\/qwen2.5-vl\/"},{"key":"e_1_3_2_1_65_1","unstructured":"Qwen Team. 2025. Qwen3 Technical Report. arXiv:2505.09388 [cs.CL] https:\/\/arxiv.org\/abs\/2505.09388"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICECA55336.2022.10009215"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01271"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01566"},{"key":"e_1_3_2_1_69_1","volume-title":"Explicit knowledge-based reasoning for visual question answering. arXiv preprint arXiv:1511.02570","author":"Wang Peng","year":"2015","unstructured":"Peng Wang, Qi Wu, Chunhua Shen, Anton van den Hengel, and Anthony Dick. 2015. Explicit knowledge-based reasoning for visual question answering. arXiv preprint arXiv:1511.02570 (2015)."},{"key":"e_1_3_2_1_70_1","unstructured":"Weiyun Wang Zhangwei Gao Lixin Gu Hengjun Pu Long Cui Xingguang Wei Zhaoyang Liu Linglin Jing Shenglong Ye Jie Shao et al. 2025. InternVL3.5: Advancing Open-Source Multimodal Models in Versatility Reasoning and Efficiency. arXiv preprint arXiv:2508.18265 (2025)."},{"key":"e_1_3_2_1_71_1","volume-title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation. arXiv preprint arXiv:2307.06942","author":"Wang Yi","year":"2023","unstructured":"Yi Wang, Yinan He, Yizhuo Li, Kunchang Li, Jiashuo Yu, Xin Ma, Xinyuan Chen, Yaohui Wang, Ping Luo, Ziwei Liu, Yali Wang, Limin Wang, and Yu Qiao. 2023. InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation. arXiv preprint arXiv:2307.06942 (2023)."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645359"},{"key":"e_1_3_2_1_73_1","first-page":"28828","article-title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","volume":"37","author":"Wu Haoning","year":"2024","unstructured":"Haoning Wu, Dongxu Li, Bei Chen, and Junnan Li. 2024. Longvideobench: A benchmark for long-context interleaved video-language understanding. Advances in Neural Information Processing Systems 37 (2024), 28828-28857.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548291"},{"key":"e_1_3_2_1_76_1","volume-title":"Benchmarking Multimodal RAG through a Chart-based Document Question-Answering Generation Framework. arXiv preprint arXiv:2502.14864","author":"Yang Yuming","year":"2025","unstructured":"Yuming Yang, Jiang Zhong, Li Jin, Jingwang Huang, Jingpeng Gao, Qing Liu, Yang Bai, Jingyuan Zhang, Rui Jiang, and Kaiwen Wei. 2025. Benchmarking Multimodal RAG through a Chart-based Document Question-Answering Generation Framework. arXiv preprint arXiv:2502.14864 (2025)."},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"crossref","unstructured":"Yuan Yao Tianyu Yu Ao Zhang Chongyi Wang Junbo Cui Hongji Zhu Tianchi Cai Haoyu Li Weilin Zhao Zhihui He et al. 2025. MiniCPM-V: A GPT-4V Level MLLM on Your Phone. Nat Commun 16 5509 (2025) (2025).","DOI":"10.1038\/s41467-025-61040-5"},{"key":"e_1_3_2_1_78_1","volume-title":"European Conference on Computer Vision. Springer, 146- 164","author":"Ye Qilang","year":"2024","unstructured":"Qilang Ye, Zitong Yu, Rui Shao, Xinyu Xie, Philip Torr, and Xiaochun Cao. 2024. Cat: Enhancing multimodal large language model to answer questions in dynamic audio-visual scenarios. In European Conference on Computer Vision. Springer, 146- 164."},{"key":"e_1_3_2_1_79_1","volume-title":"MRAMG-Bench: A BeyondText Benchmark for Multimodal Retrieval-Augmented Multimodal Generation. arXiv e-prints","author":"Yu Qinhan","year":"2025","unstructured":"Qinhan Yu, Zhiyou Xiao, Binghui Li, Zhengren Wang, Chong Chen, and Wentao Zhang. 2025. MRAMG-Bench: A BeyondText Benchmark for Multimodal Retrieval-Augmented Multimodal Generation. arXiv e-prints (2025), arXiv-2502."},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645477"},{"key":"e_1_3_2_1_81_1","volume-title":"Ziyang Wang, Shoubin Yu, Mohit Bansal, and Gedas Bertasius.","author":"Zhang Ce","year":"2023","unstructured":"Ce Zhang, Taixi Lu, Md Mohaiminul Islam, Ziyang Wang, Shoubin Yu, Mohit Bansal, and Gedas Bertasius. 2023. A simple llm framework for long-range video question-answering. arXiv preprint arXiv:2312.17235 (2023)."},{"key":"e_1_3_2_1_82_1","volume-title":"Adversarial retriever-ranker for dense text retrieval. arXiv preprint arXiv:2110.03611","author":"Zhang Hang","year":"2021","unstructured":"Hang Zhang, Yeyun Gong, Yelong Shen, Jiancheng Lv, Nan Duan, and Weizhu Chen. 2021. Adversarial retriever-ranker for dense text retrieval. arXiv preprint arXiv:2110.03611 (2021)."},{"key":"e_1_3_2_1_83_1","volume-title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams. arXiv preprint arXiv:2506.23825","author":"Zhang Haoji","year":"2025","unstructured":"Haoji Zhang, Yiqin Wang, Yansong Tang, Yong Liu, Jiashi Feng, and Xiaojie Jin. 2025. Flash-VStream: Efficient Real-Time Understanding for Long Video Streams. arXiv preprint arXiv:2506.23825 (2025)."},{"key":"e_1_3_2_1_84_1","volume-title":"Long context transfer from language to vision. arXiv preprint arXiv:2406.16852","author":"Zhang Peiyuan","year":"2024","unstructured":"Peiyuan Zhang, Kaichen Zhang, Bo Li, Guangtao Zeng, Jingkang Yang, Yuanhan Zhang, Ziyue Wang, Haoran Tan, Chunyuan Li, and Ziwei Liu. 2024. Long context transfer from language to vision. arXiv preprint arXiv:2406.16852 (2024)."},{"key":"e_1_3_2_1_85_1","volume-title":"Deep Video Discovery: Agentic Search with Tool Use for Long-form Video Understanding. arXiv preprint arXiv:2505.18079","author":"Zhang Xiaoyi","year":"2025","unstructured":"Xiaoyi Zhang, Zhaoyang Jia, Zongyu Guo, Jiahao Li, Bin Li, Houqiang Li, and Yan Lu. 2025. Deep Video Discovery: Agentic Search with Tool Use for Long-form Video Understanding. arXiv preprint arXiv:2505.18079 (2025)."},{"key":"e_1_3_2_1_86_1","unstructured":"Yidan Zhang Ting Zhang Dong Chen Yujing Wang Qi Chen Xing Xie Hao Sun Weiwei Deng Qi Zhang Fan Yang et al. 2023. IRGen: Generative Modeling for Image Retrieval. CoRR abs\/2303.10126 (2023)."},{"key":"e_1_3_2_1_87_1","volume-title":"Chengwei Qin, Bosheng Ding, Xiaobao Guo, Minzhi Li, Xingxuan Li, et al.","author":"Zhao Ruochen","year":"2023","unstructured":"Ruochen Zhao, Hailin Chen,WeishiWang, Fangkai Jiao, Xuan Long Do, Chengwei Qin, Bosheng Ding, Xiaobao Guo, Minzhi Li, Xingxuan Li, et al. 2023. Retrieving multimodal information for augmented generation: A survey. arXiv preprint arXiv:2303.10868 (2023)."},{"key":"e_1_3_2_1_88_1","volume-title":"Retrieval augmented generation (rag) and beyond: A comprehensive survey on how to make your llms use external data more wisely. arXiv preprint arXiv:2409.14924","author":"Zhao Siyun","year":"2024","unstructured":"Siyun Zhao, Yuqing Yang, Zilong Wang, Zhiyuan He, Luna K Qiu, and Lili Qiu. 2024. Retrieval augmented generation (rag) and beyond: A comprehensive survey on how to make your llms use external data more wisely. arXiv preprint arXiv:2409.14924 (2024)."},{"key":"e_1_3_2_1_89_1","volume-title":"Zhifeng Li, Wei Liu, and Li Yuan.","author":"Zhu Bin","year":"2023","unstructured":"Bin Zhu, Bin Lin, Munan Ning, Yang Yan, Jiaxi Cui, Wang HongFa, Yatian Pang, Wenhao Jiang, Junwu Zhang, Zongwei Li, Cai Wan Zhang, Zhifeng Li, Wei Liu, and Li Yuan. 2023. LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment. arXiv:2310.01852 [cs.CV]"},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.418"}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"deposited":{"date-parts":[[2026,4,10]],"date-time":"2026-04-10T16:29:42Z","timestamp":1775838582000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792726"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":90,"alternative-id":["10.1145\/3774904.3792726","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792726","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}