{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:17:04Z","timestamp":1784179024505,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":70,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62461146205"],"award-info":[{"award-number":["62461146205"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Natural Science Foundation of China","award":["62206042"],"award-info":[{"award-number":["62206042"]}]},{"name":"Fundamental Research Funds for the Central Universities","award":["N25ZLL045"],"award-info":[{"award-number":["N25ZLL045"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755625","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:27:39Z","timestamp":1761377259000},"page":"4817-4826","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Benchmarking Retrieval-Augmented Generation in Multi-Modal Contexts"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0083-3224","authenticated-orcid":false,"given":"Zhenghao","family":"Liu","sequence":"first","affiliation":[{"name":"Northeastern University, China, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3777-7643","authenticated-orcid":false,"given":"Xingsheng","family":"Zhu","sequence":"additional","affiliation":[{"name":"Northeastern University, China, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4234-9345","authenticated-orcid":false,"given":"Tianshuo","family":"Zhou","sequence":"additional","affiliation":[{"name":"Northeastern University, China, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6982-8059","authenticated-orcid":false,"given":"Xinyi","family":"Zhang","sequence":"additional","affiliation":[{"name":"Northeastern University, China, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2710-1613","authenticated-orcid":false,"given":"Xiaoyuan","family":"Yi","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2421-7098","authenticated-orcid":false,"given":"Yukun","family":"Yan","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3171-8889","authenticated-orcid":false,"given":"Ge","family":"Yu","sequence":"additional","affiliation":[{"name":"Northeastern University, China, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6011-6115","authenticated-orcid":false,"given":"Maosong","family":"Sun","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. ArXiv preprint (2023). https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"e_1_3_2_1_2_1","unstructured":"Armen Aghajanyan Bernie Huang Candace Ross Vladimir Karpukhin Hu Xu Naman Goyal Dmytro Okhonko Mandar Joshi Gargi Ghosh Mike Lewis et al. 2022. Cm3: A causal masked multimodal model of the internet. ArXiv preprint (2022). https:\/\/arxiv.org\/abs\/2201.07520"},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of NeurIPS. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/960a172bc7fbf0177ccccbb411a7d800-Abstract-Conference.html","author":"Alayrac Jean-Baptiste","year":"2022","unstructured":"Jean-Baptiste Alayrac, Jeff Donahue, Pauline Luc, Antoine Miech, Iain Barr, Yana Hasson, Karel Lenc, Arthur Mensch, Katherine Millican, Malcolm Reynolds, Roman Ring, Eliza Rutherford, Serkan Cabi, Tengda Han, Zhitao Gong, Sina Samangooei, Marianne Monteiro, Jacob L. Menick, Sebastian Borgeaud, Andy Brock, Aida Nematzadeh, Sahand Sharifzadeh, Mikolaj Binkowski, Ricardo Barreira, Oriol Vinyals, Andrew Zisserman, and Kar\u00e9n Simonyan. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. In Proceedings of NeurIPS. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/960a172bc7fbf0177ccccbb411a7d800-Abstract-Conference.html"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of ICLR. https:\/\/openreview.net\/forum?id=hSyW5go0v8","author":"Asai Akari","year":"2024","unstructured":"Akari Asai, Zeqiu Wu, Yizhong Wang, Avirup Sil, and Hannaneh Hajishirzi. 2024a. Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection. In Proceedings of ICLR. https:\/\/openreview.net\/forum?id=hSyW5go0v8"},{"key":"e_1_3_2_1_5_1","volume-title":"Luke Zettlemoyer, Hannaneh Hajishirzi, and Wen-tau Yih.","author":"Asai Akari","year":"2024","unstructured":"Akari Asai, Zexuan Zhong, Danqi Chen, Pang Wei Koh, Luke Zettlemoyer, Hannaneh Hajishirzi, and Wen-tau Yih. 2024b. Reliable, adaptable, and attributable language models with retrieval. ArXiv preprint (2024). https:\/\/arxiv.org\/abs\/2403.03187"},{"key":"e_1_3_2_1_6_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00188"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01600"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.375"},{"key":"e_1_3_2_1_10_1","first-page":"1178","article-title":"MORE","volume":"2024","author":"Cui Wanqing","year":"2024","unstructured":"Wanqing Cui, Keping Bi, Jiafeng Guo, and Xueqi Cheng. 2024. MORE: Multi-mOdal REtrieval Augmented Generative Commonsense Reasoning. In Findings of the Association for Computational Linguistics ACL 2024. 1178-1192.","journal-title":"Multi-mOdal REtrieval Augmented Generative Commonsense Reasoning. In Findings of the Association for Computational Linguistics ACL"},{"key":"e_1_3_2_1_11_1","volume-title":"RA-BLIP: Multimodal Adaptive Retrieval-Augmented Bootstrapping Language-Image Pre-training. arXiv preprint arXiv:2410.14154","author":"Ding Muhe","year":"2024","unstructured":"Muhe Ding, Yang Ma, Pengda Qin, Jianlong Wu, Yuhong Li, and Liqiang Nie. 2024. RA-BLIP: Multimodal Adaptive Retrieval-Augmented Bootstrapping Language-Image Pre-training. arXiv preprint arXiv:2410.14154 (2024)."},{"key":"e_1_3_2_1_12_1","first-page":"2843","article-title":"Unsupervised Corpus Aware Language Model Pre-training for Dense Passage Retrieval","author":"Gao Luyu","year":"2022","unstructured":"Luyu Gao and Jamie Callan. 2022. Unsupervised Corpus Aware Language Model Pre-training for Dense Passage Retrieval. In Proceedings of ACL. 2843-2853. https:\/\/aclanthology.org\/2022.acl-long.203\/","journal-title":"Proceedings of ACL."},{"key":"e_1_3_2_1_13_1","first-page":"6626","article-title":"GANs Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium","author":"Heusel Martin","year":"2017","unstructured":"Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. GANs Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium. In Proceedings of NeurIPS. 6626-6637. https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/8a1d694707eb0fefe65871369074926d-Abstract.html","journal-title":"Proceedings of NeurIPS."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of ICLR. https:\/\/openreview.net\/forum?id=nZeVKeeFYf9","author":"Hu Edward J.","year":"2022","unstructured":"Edward J. Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In Proceedings of ICLR. https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"e_1_3_2_1_15_1","volume-title":"MRAG-Bench: Vision-Centric Evaluation for Retrieval-Augmented Multimodal Models. ArXiv preprint","author":"Hu Wenbo","year":"2024","unstructured":"Wenbo Hu, Jia-Chen Gu, Zi-Yi Dou, Mohsen Fayyaz, Pan Lu, Kai-Wei Chang, and Nanyun Peng. 2024. MRAG-Bench: Vision-Centric Evaluation for Retrieval-Augmented Multimodal Models. ArXiv preprint (2024). https:\/\/arxiv.org\/abs\/2410.08182"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02238"},{"key":"e_1_3_2_1_17_1","unstructured":"Lei Huang Weijiang Yu Weitao Ma Weihong Zhong Zhangyin Feng Haotian Wang Qianglong Chen Weihua Peng Xiaocheng Feng Bing Qin et al. 2023. A survey on hallucination in large language models: Principles taxonomy challenges and open questions. ArXiv preprint (2023). https:\/\/arxiv.org\/abs\/2311.05232"},{"key":"e_1_3_2_1_18_1","unstructured":"Aaron Hurst Adam Lerer Adam P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et al. 2024. Gpt-4o system card. arXiv preprint arXiv:2410.21276 (2024)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3571730"},{"key":"e_1_3_2_1_20_1","first-page":"535","article-title":"Billion-scale similarity search with GPUs","volume":"3","author":"Johnson Jeff","year":"2019","unstructured":"Jeff Johnson, Matthijs Douze, and Herv\u00e9 J\u00e9gou. 2019. Billion-scale similarity search with GPUs. IEEE Transactions on Big Data, 3 (2019), 535-547. https:\/\/ieeexplore.ieee.org\/document\/8733051","journal-title":"IEEE Transactions on Big Data"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.215"},{"key":"e_1_3_2_1_22_1","first-page":"6769","article-title":"Dense Passage Retrieval for Open-Domain Question Answering","author":"Karpukhin Vladimir","year":"2020","unstructured":"Vladimir Karpukhin, Barlas Oguz, Sewon Min, Patrick Lewis, Ledell Wu, Sergey Edunov, Danqi Chen, and Wen-tau Yih. 2020. Dense Passage Retrieval for Open-Domain Question Answering. In Proceedings of EMNLP. 6769-6781. https:\/\/aclanthology.org\/2020.emnlp-main.550\/","journal-title":"Proceedings of EMNLP."},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of NeurIPS. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/6b493230205f780e1bc26945df7481e5-Abstract.html","author":"Lewis Patrick S. H.","year":"2020","unstructured":"Patrick S. H. Lewis, Ethan Perez, Aleksandra Piktus, Fabio Petroni, Vladimir Karpukhin, Naman Goyal, Heinrich K\u00fcttler, Mike Lewis, Wen-tau Yih, Tim Rockt\u00e4schel, Sebastian Riedel, and Douwe Kiela. 2020. Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks. In Proceedings of NeurIPS. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/6b493230205f780e1bc26945df7481e5-Abstract.html"},{"key":"e_1_3_2_1_24_1","volume-title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models. arXiv preprint arXiv:2407.07895","author":"Li Feng","year":"2024","unstructured":"Feng Li, Renrui Zhang, Hao Zhang, Yuanhan Zhang, Bo Li, Wei Li, Zejun Ma, and Chunyuan Li. 2024. Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models. arXiv preprint arXiv:2407.07895 (2024)."},{"key":"e_1_3_2_1_25_1","first-page":"19730","article-title":"BLIP-2","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven C. H. Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In Proceedings of ICML. 19730-19742. https:\/\/proceedings.mlr.press\/v202\/li23q.html","journal-title":"In Proceedings of ICML."},{"key":"e_1_3_2_1_26_1","first-page":"12888","article-title":"BLIP","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven C. H. Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In Proceedings of ICML. 12888-12900. https:\/\/proceedings.mlr.press\/v162\/li22n.html","journal-title":"In Proceedings of ICML."},{"key":"e_1_3_2_1_27_1","first-page":"74","article-title":"ROUGE","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. ROUGE: A Package for Automatic Evaluation of Summaries. In Text Summarization Branches Out. 74-81. https:\/\/aclanthology.org\/W04-1013\/","journal-title":"A Package for Automatic Evaluation of Summaries. In Text Summarization Branches Out."},{"key":"e_1_3_2_1_28_1","volume-title":"Microsoft coco: Common objects in context","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft coco: Common objects in context. In Proceedings of ECCV. Springer, 740-755. https:\/\/link.springer.com\/chapter\/10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of ICLR. https:\/\/openreview.net\/forum?id=22OTbutug9","author":"Lin Xi Victoria","year":"2024","unstructured":"Xi Victoria Lin, Xilun Chen, Mingda Chen, Weijia Shi, Maria Lomeli, Richard James, Pedro Rodriguez, Jacob Kahn, Gergely Szilvasy, Mike Lewis, Luke Zettlemoyer, and Wen-tau Yih. 2024. RA-DIT: Retrieval-Augmented Dual Instruction Tuning. In Proceedings of ICLR. https:\/\/openreview.net\/forum?id=22OTbutug9"},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of NeurIPS. http:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/6dcf277ea32ce3288914faf369fe6de0-Abstract-Conference.html","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023a. Visual Instruction Tuning. In Proceedings of NeurIPS. http:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/6dcf277ea32ce3288914faf369fe6de0-Abstract-Conference.html"},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of ICLR. https:\/\/openreview.net\/pdf?id=PQOlkgsBsik","author":"Liu Zhenghao","year":"2023","unstructured":"Zhenghao Liu, Chenyan Xiong, Yuanhuiyi Lv, Zhiyuan Liu, and Ge Yu. 2023b. Universal Vision-Language Dense Retrieval: Learning A Unified Representation Space for Multi-Modal Retrieval. In Proceedings of ICLR. https:\/\/openreview.net\/pdf?id=PQOlkgsBsik"},{"key":"e_1_3_2_1_32_1","unstructured":"Haoyu Lu Wen Liu Bo Zhang Bingxuan Wang Kai Dong Bo Liu Jingxiang Sun Tongzheng Ren Zhuoshu Li Hao Yang Yaofeng Sun Chengqi Deng Hanwei Xu Zhenda Xie and Chong Ruan. 2024. DeepSeek-VL: Towards Real-World Vision-Language Understanding. https:\/\/arxiv.org\/abs\/2403.05525"},{"key":"e_1_3_2_1_33_1","first-page":"3195","article-title":"OK-VQA: A Visual Question Answering Benchmark Requiring External Knowledge","author":"Marino Kenneth","year":"2019","unstructured":"Kenneth Marino, Mohammad Rastegari, Ali Farhadi, and Roozbeh Mottaghi. 2019. OK-VQA: A Visual Question Answering Benchmark Requiring External Knowledge. In Proceedings of CVPR. 3195-3204. http:\/\/openaccess.thecvf.com\/content_CVPR_2019\/html\/Marino_OK-VQA_A_Visual_Question_Answering_Benchmark_Requiring_External_Knowledge_CVPR_2019_paper.html","journal-title":"Proceedings of CVPR."},{"key":"e_1_3_2_1_34_1","volume-title":"FACTIFY: A Multi-Modal Fact Verification Dataset.. In DE-FACTIFY@ AAAI. https:\/\/ceur-ws.org\/Vol-3199\/paper18.pdf","author":"Mishra Shreyash","year":"2022","unstructured":"Shreyash Mishra, S Suryavardan, Amrit Bhaskar, Parul Chopra, Aishwarya N Reganti, Parth Patwa, Amitava Das, Tanmoy Chakraborty, Amit P Sheth, Asif Ekbal, et al., 2022. FACTIFY: A Multi-Modal Fact Verification Dataset.. In DE-FACTIFY@ AAAI. https:\/\/ceur-ws.org\/Vol-3199\/paper18.pdf"},{"key":"e_1_3_2_1_35_1","volume-title":"Sgpt: Gpt sentence embeddings for semantic search. ArXiv preprint","author":"Muennighoff Niklas","year":"2022","unstructured":"Niklas Muennighoff. 2022. Sgpt: Gpt sentence embeddings for semantic search. ArXiv preprint (2022). https:\/\/arxiv.org\/abs\/2202.08904"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2303.08774"},{"key":"e_1_3_2_1_37_1","first-page":"2523","article-title":"KILT: a Benchmark for Knowledge Intensive Language Tasks","author":"Petroni Fabio","year":"2021","unstructured":"Fabio Petroni, Aleksandra Piktus, Angela Fan, Patrick Lewis, Majid Yazdani, Nicola De Cao, James Thorne, Yacine Jernite, Vladimir Karpukhin, Jean Maillard, Vassilis Plachouras, Tim Rockt\u00e4schel, and Sebastian Riedel. 2021. KILT: a Benchmark for Knowledge Intensive Language Tasks. In Proceedings of NAACL-HLT. 2523-2544. https:\/\/aclanthology.org\/2021.naacl-main.200\/","journal-title":"Proceedings of NAACL-HLT."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00605"},{"key":"e_1_3_2_1_39_1","first-page":"2825","article-title":"RocketQAv2","author":"Ren Ruiyang","year":"2021","unstructured":"Ruiyang Ren, Yingqi Qu, Jing Liu, Wayne Xin Zhao, QiaoQiao She, Hua Wu, Haifeng Wang, and Ji-Rong Wen. 2021. RocketQAv2: A Joint Training Method for Dense Passage Retrieval and Passage Re-ranking. In Proceedings of EMNLP. 2825-2835. https:\/\/aclanthology.org\/2021.emnlp-main.224\/","journal-title":"In Proceedings of EMNLP."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","unstructured":"Stephen Robertson Hugo Zaragoza et al. 2009. The probabilistic relevance framework: BM25 and beyond. Foundations and Trends\u00ae in Information Retrieval 4 (2009) 333-389. https:\/\/doi.org\/10.1561\/1500000019","DOI":"10.1561\/1500000019"},{"key":"e_1_3_2_1_41_1","volume-title":"Laion-400m: Open dataset of clip-filtered 400 million image-text pairs. ArXiv preprint","author":"Schuhmann Christoph","year":"2021","unstructured":"Christoph Schuhmann, Richard Vencu, Romain Beaumont, Robert Kaczmarczyk, Clayton Mullis, Aarush Katta, Theo Coombes, Jenia Jitsev, and Aran Komatsuzaki. 2021. Laion-400m: Open dataset of clip-filtered 400 million image-text pairs. ArXiv preprint (2021). https:\/\/arxiv.org\/abs\/2111.02114"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018876"},{"key":"e_1_3_2_1_43_1","volume-title":"UniRAG: Universal Retrieval Augmentation for Multi-Modal Large Language Models. ArXiv preprint","author":"Sharifymoghaddam Sahel","year":"2024","unstructured":"Sahel Sharifymoghaddam, Shivani Upadhyay, Wenhu Chen, and Jimmy Lin. 2024. UniRAG: Universal Retrieval Augmentation for Multi-Modal Large Language Models. ArXiv preprint (2024). https:\/\/arxiv.org\/abs\/2405.10311"},{"key":"e_1_3_2_1_44_1","first-page":"8371","article-title":"REPLUG","author":"Shi Weijia","year":"2024","unstructured":"Weijia Shi, Sewon Min, Michihiro Yasunaga, Minjoon Seo, Richard James, Mike Lewis, Luke Zettlemoyer, and Wen-tau Yih. 2024. REPLUG: Retrieval-Augmented Black-Box Language Models. In Proceedings of NAACL-HLT. 8371-8384. https:\/\/aclanthology.org\/2024.naacl-long.463\/","journal-title":"Retrieval-Augmented Black-Box Language Models. In Proceedings of NAACL-HLT."},{"key":"e_1_3_2_1_45_1","first-page":"3784","article-title":"Retrieval Augmentation Reduces Hallucination in Conversation","author":"Shuster Kurt","year":"2021","unstructured":"Kurt Shuster, Spencer Poff, Moya Chen, Douwe Kiela, and Jason Weston. 2021. Retrieval Augmentation Reduces Hallucination in Conversation. In Proceedings of EMNLP Findings. 3784-3803. https:\/\/aclanthology.org\/2021.findings-emnlp.320\/","journal-title":"Proceedings of EMNLP Findings."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01365"},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of ICLR. https:\/\/openreview.net\/forum?id=mL8Q9OOamV","author":"Sun Quan","year":"2024","unstructured":"Quan Sun, Qiying Yu, Yufeng Cui, Fan Zhang, Xiaosong Zhang, Yueze Wang, Hongcheng Gao, Jingjing Liu, Tiejun Huang, and Xinlong Wang. 2024b. Emu: Generative Pretraining in Multimodality. In Proceedings of ICLR. https:\/\/openreview.net\/forum?id=mL8Q9OOamV"},{"key":"e_1_3_2_1_48_1","first-page":"2189","article-title":"Multimodal misinformation detection using large vision-language models","author":"Tahmasebi Sahar","year":"2024","unstructured":"Sahar Tahmasebi, Eric M\u00fcller-Budack, and Ralph Ewerth. 2024. Multimodal misinformation detection using large vision-language models. In Proceedings of CIKM. 2189-2199. https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3627673.3679826","journal-title":"Proceedings of CIKM."},{"key":"e_1_3_2_1_49_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. ArXiv preprint (2023). https:\/\/arxiv.org\/abs\/2312.11805"},{"key":"e_1_3_2_1_50_1","first-page":"809","article-title":"FEVER: a Large-scale Dataset for Fact Extraction and VERification","author":"Thorne James","year":"2018","unstructured":"James Thorne, Andreas Vlachos, Christos Christodoulopoulos, and Arpit Mittal. 2018. FEVER: a Large-scale Dataset for Fact Extraction and VERification. In Proceedings of NAACL-HLT. 809-819. https:\/\/aclanthology.org\/N18-1074\/","journal-title":"Proceedings of NAACL-HLT."},{"key":"e_1_3_2_1_51_1","volume-title":"Llama: Open and efficient foundation language models. ArXiv preprint","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al., 2023. Llama: Open and efficient foundation language models. ArXiv preprint (2023). https:\/\/arxiv.org\/abs\/2302.13971"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_1_53_1","volume-title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. ArXiv preprint","author":"Wang Peng","year":"2024","unstructured":"Peng Wang, Shuai Bai, Sinan Tan, Shijie Wang, Zhihao Fan, Jinze Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Yang Fan, Kai Dang, Mengfei Du, Xuancheng Ren, Rui Men, Dayiheng Liu, Chang Zhou, Jingren Zhou, and Junyang Lin. 2024. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. ArXiv preprint (2024). https:\/\/arxiv.org\/abs\/2409.12191"},{"key":"e_1_3_2_1_54_1","unstructured":"Jason Wei Yi Tay Rishi Bommasani Colin Raffel Barret Zoph Sebastian Borgeaud Dani Yogatama Maarten Bosma Denny Zhou Donald Metzler et al. 2022. Emergent Abilities of Large Language Models. Transactions on Machine Learning Research (2022). https:\/\/openreview.net\/forum?id=yzkSU5zdwD"},{"key":"e_1_3_2_1_55_1","volume-title":"Proceedings of ICLR. https:\/\/openreview.net\/forum?id=zeFrfgyZln","author":"Xiong Lee","year":"2021","unstructured":"Lee Xiong, Chenyan Xiong, Ye Li, Kwok-Fung Tang, Jialin Liu, Paul N. Bennett, Junaid Ahmed, and Arnold Overwijk. 2021b. Approximate Nearest Neighbor Negative Contrastive Learning for Dense Text Retrieval. In Proceedings of ICLR. https:\/\/openreview.net\/forum?id=zeFrfgyZln"},{"key":"e_1_3_2_1_56_1","volume-title":"Proceedings of ICLR. https:\/\/openreview.net\/forum?id=EMHoBG0avc1","author":"Xiong Wenhan","year":"2021","unstructured":"Wenhan Xiong, Xiang Lorraine Li, Srini Iyer, Jingfei Du, Patrick S. H. Lewis, William Yang Wang, Yashar Mehdad, Scott Yih, Sebastian Riedel, Douwe Kiela, and Barlas Oguz. 2021a. Answering Complex Open-Domain Questions with Multi-Hop Dense Retrieval. In Proceedings of ICLR. https:\/\/openreview.net\/forum?id=EMHoBG0avc1"},{"key":"e_1_3_2_1_57_1","volume-title":"Corrective retrieval augmented generation. ArXiv preprint","author":"Yan Shi-Qi","year":"2024","unstructured":"Shi-Qi Yan, Jia-Chen Gu, Yun Zhu, and Zhen-Hua Ling. 2024. Corrective retrieval augmented generation. ArXiv preprint (2024). https:\/\/arxiv.org\/abs\/2401.15884"},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of ICLR. https:\/\/openreview.net\/pdf?id=WE_vluYUL-X","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik R. Narasimhan, and Yuan Cao. 2023. ReAct: Synergizing Reasoning and Acting in Language Models. In Proceedings of ICLR. https:\/\/openreview.net\/pdf?id=WE_vluYUL-X"},{"key":"e_1_3_2_1_59_1","unstructured":"Yuan Yao Tianyu Yu Ao Zhang Chongyi Wang Junbo Cui Hongji Zhu Tianchi Cai Haoyu Li Weilin Zhao Zhihui He et al. 2024. MiniCPM-V: A GPT-4V Level MLLM on Your Phone. ArXiv preprint (2024). https:\/\/arxiv.org\/abs\/2408.01800"},{"key":"e_1_3_2_1_60_1","first-page":"39755","article-title":"Retrieval-Augmented Multimodal Language Modeling","author":"Yasunaga Michihiro","year":"2023","unstructured":"Michihiro Yasunaga, Armen Aghajanyan, Weijia Shi, Richard James, Jure Leskovec, Percy Liang, Mike Lewis, Luke Zettlemoyer, and Wen-Tau Yih. 2023. Retrieval-Augmented Multimodal Language Modeling. In Proceedings of ICML. 39755-39769. https:\/\/proceedings.mlr.press\/v202\/yasunaga23a.html","journal-title":"Proceedings of ICML."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00166"},{"key":"e_1_3_2_1_62_1","unstructured":"Lili Yu Bowen Shi Ramakanth Pasunuru Benjamin Muller Olga Golovneva Tianlu Wang Arun Babu Binh Tang Brian Karrer Shelly Sheynin et al. 2023a. Scaling autoregressive multi-modal models: Pretraining and instruction tuning. ArXiv preprint (2023). https:\/\/arxiv.org\/abs\/2309.02591"},{"key":"e_1_3_2_1_63_1","volume-title":"Visrag: Vision-based retrieval-augmented generation on multi-modality documents. arXiv preprint arXiv:2410.10594","author":"Yu Shi","year":"2024","unstructured":"Shi Yu, Chaoyue Tang, Bokai Xu, Junbo Cui, Junhao Ran, Yukun Yan, Zhenghao Liu, Shuo Wang, Xu Han, Zhiyuan Liu, et al., 2024. Visrag: Vision-based retrieval-augmented generation on multi-modality documents. arXiv preprint arXiv:2410.10594 (2024)."},{"key":"e_1_3_2_1_64_1","first-page":"2421","article-title":"Augmentation-Adapted Retriever Improves Generalization of Language Models as Generic Plug-In","author":"Yu Zichun","year":"2023","unstructured":"Zichun Yu, Chenyan Xiong, Shi Yu, and Zhiyuan Liu. 2023b. Augmentation-Adapted Retriever Improves Generalization of Language Models as Generic Plug-In. In Proceedings of ACL. 2421-2436. https:\/\/aclanthology.org\/2023.acl-long.136\/","journal-title":"Proceedings of ACL."},{"key":"e_1_3_2_1_65_1","volume-title":"BERTScore: Evaluating Text Generation with BERT. In Proceedings of ICLR.","author":"Zhang Tianyi","unstructured":"Tianyi Zhang, Varsha Kishore, Felix Wu, Kilian Q Weinberger, and Yoav Artzi. [n.d.]. BERTScore: Evaluating Text Generation with BERT. In Proceedings of ICLR."},{"key":"e_1_3_2_1_66_1","unstructured":"Wayne Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong et al. 2023. A survey of large language models. ArXiv preprint (2023). https:\/\/arxiv.org\/abs\/2303.18223"},{"key":"e_1_3_2_1_67_1","first-page":"400","article-title":"LlamaFactory: Unified Efficient Fine-Tuning of 100 Language Models","author":"Zheng Yaowei","year":"2024","unstructured":"Yaowei Zheng, Richong Zhang, Junhao Zhang, Yanhan Ye, and Zheyan Luo. 2024. LlamaFactory: Unified Efficient Fine-Tuning of 100 Language Models. In Proceedings of ACL. 400-410. https:\/\/aclanthology.org\/2024.acl-demos.38\/","journal-title":"Proceedings of ACL."},{"key":"e_1_3_2_1_68_1","first-page":"3185","article-title":"VISTA","author":"Zhou Junjie","year":"2024","unstructured":"Junjie Zhou, Zheng Liu, Shitao Xiao, Bo Zhao, and Yongping Xiong. 2024a. VISTA: Visualized Text Embedding For Universal Multi-Modal Retrieval. In Proceedings of ACL. 3185-3200. https:\/\/aclanthology.org\/2024.acl-long.175\/","journal-title":"Visualized Text Embedding For Universal Multi-Modal Retrieval. In Proceedings of ACL."},{"key":"e_1_3_2_1_69_1","first-page":"14608","article-title":"MARVEL: Unlocking the Multi-Modal Capability of Dense Retrieval via Visual Module Plugin","author":"Zhou Tianshuo","year":"2024","unstructured":"Tianshuo Zhou, Sen Mei, Xinze Li, Zhenghao Liu, Chenyan Xiong, Zhiyuan Liu, Yu Gu, and Ge Yu. 2024b. MARVEL: Unlocking the Multi-Modal Capability of Dense Retrieval via Visual Module Plugin. In Proceedings of ACL. 14608-14624. https:\/\/aclanthology.org\/2024.acl-long.783\/","journal-title":"Proceedings of ACL."},{"key":"e_1_3_2_1_70_1","unstructured":"Jinguo Zhu Weiyun Wang Zhe Chen Zhaoyang Liu Shenglong Ye Lixin Gu Hao Tian Yuchen Duan Weijie Su Jie Shao et al. 2025. Internvl3: Exploring advanced training and test-time recipes for open-source multimodal models. arXiv preprint arXiv:2504.10479 (2025)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755625","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:58:41Z","timestamp":1765342721000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755625"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":70,"alternative-id":["10.1145\/3746027.3755625","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755625","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}