{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:54:32Z","timestamp":1784138072853,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":59,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754761","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:27:39Z","timestamp":1761377259000},"page":"2781-2790","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":29,"title":["HM-RAG: Hierarchical Multi-Agent Multimodal Retrieval Augmented Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-6925-1833","authenticated-orcid":false,"given":"Pei","family":"Liu","sequence":"first","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3906-8629","authenticated-orcid":false,"given":"Xin","family":"Liu","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1754-1770","authenticated-orcid":false,"given":"Ruoyu","family":"Yao","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9131-3245","authenticated-orcid":false,"given":"Junming","family":"Liu","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5823-7340","authenticated-orcid":false,"given":"Siyuan","family":"Meng","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2398-1409","authenticated-orcid":false,"given":"Ding","family":"Wang","sequence":"additional","affiliation":[{"name":"Shanghai Artificial Intelligence Laboratory, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9405-8232","authenticated-orcid":false,"given":"Jun","family":"Ma","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia Leoni Aleman Diogo Almeida Janko Altenschmidt Sam Altman Shyamal Anadkat et al. 2023. GPT-4 Technical Report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1609\/icwsm.v12i1.14983"},{"key":"e_1_3_2_1_3_1","unstructured":"Abhijit Anand Vinay Setty Avishek Anand et al. 2023. Context Aware Query Rewriting for Text Rankers using LLM. arXiv preprint arXiv:2308.16753 (2023)."},{"key":"e_1_3_2_1_4_1","volume-title":"Self-rag: Learning to retrieve, generate, and critique through self-reflection. arXiv preprint arXiv:2310.11511","author":"Asai Akari","year":"2023","unstructured":"Akari Asai, Zeqiu Wu, Yizhong Wang, Avirup Sil, and Hannaneh Hajishirzi. 2023. Self-rag: Learning to retrieve, generate, and critique through self-reflection. arXiv preprint arXiv:2310.11511 (2023)."},{"key":"e_1_3_2_1_5_1","volume-title":"RAG Beyond Text: Enhancing Image Retrieval in RAG Systems. In 2024 International Conference on Electrical, Computer and Energy Technologies (ICECET. IEEE, 1-6.","author":"Bag Sukanya","year":"2024","unstructured":"Sukanya Bag, Ayushman Gupta, Rajat Kaushik, and Chirag Jain. 2024. RAG Beyond Text: Enhancing Image Retrieval in RAG Systems. In 2024 International Conference on Electrical, Computer and Energy Technologies (ICECET. IEEE, 1-6."},{"key":"e_1_3_2_1_6_1","volume-title":"Visual RAG: Expanding MLLM Visual Knowledge without Fine-tuning. arXiv preprint arXiv:2501.10834","author":"Bonomo Mirco","year":"2025","unstructured":"Mirco Bonomo and Simone Bianco. 2025. Visual RAG: Expanding MLLM Visual Knowledge without Fine-tuning. arXiv preprint arXiv:2501.10834 (2025)."},{"key":"e_1_3_2_1_7_1","volume-title":"MLLM Is a Strong Reranker: Advancing Multimodal Retrieval-augmented Generation via Knowledge-enhanced Reranking and Noise-injected Training. arXiv preprint arXiv:2407.21439","author":"Chen Zhanpeng","year":"2024","unstructured":"Zhanpeng Chen, Chengjin Xu, Yiyan Qi, and Jian Guo. 2024. MLLM Is a Strong Reranker: Advancing Multimodal Retrieval-augmented Generation via Knowledge-enhanced Reranking and Noise-injected Training. arXiv preprint arXiv:2407.21439 (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"Intelligent Agents: Definitions, Methods, and Prospects. arXiv preprint arXiv:2401.03428","author":"Cheng Yuheng","year":"2024","unstructured":"Yuheng Cheng, Ceyao Zhang, Zhengwen Zhang, Xiangrui Meng, Sirui Hong, Wenhao Li, Zihao Wang, Zekai Wang, Feng Yin, Junhua Zhao, et al., 2024. Exploring Large Language Model based Intelligent Agents: Definitions, Methods, and Prospects. arXiv preprint arXiv:2401.03428 (2024)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Yuxin Dong Shuo Wang Hongye Zheng Jiajing Chen Zhenhong Zhang and Chihang Wang. 2024. Advanced RAG Models with Graph Structures: Optimizing Complex Knowledge Reasoning and Text Generation. In 2024 5th International Symposium on Computer Engineering and Intelligent Communications (ISCEIC). IEEE 626-630.","DOI":"10.1109\/ISCEIC63613.2024.10810209"},{"key":"e_1_3_2_1_10_1","volume-title":"Words: Transformers for Image Recognition at Scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, et al., 2020. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_11_1","volume-title":"Leandro Youiti Silva Okimoto, Leonardo Yuto Suzuki Camelo, Hendrio Luis de Souza Bragan\u00e7a, Rubens Fernandes, Andre Printes, F\u00e1bio Cardoso, Raimundo Gomes, and Israel Gondres Torn\u00e9.","author":"Aquino Gustavo","year":"2025","unstructured":"Gustavo de Aquino e Aquino, N\u00e1dila da Silva de Azevedo, Leandro Youiti Silva Okimoto, Leonardo Yuto Suzuki Camelo, Hendrio Luis de Souza Bragan\u00e7a, Rubens Fernandes, Andre Printes, F\u00e1bio Cardoso, Raimundo Gomes, and Israel Gondres Torn\u00e9. 2025. From RAG to Multi-Agent Systems: A Survey of Modern Approaches in LLM Development. (2025)."},{"key":"e_1_3_2_1_12_1","volume-title":"Robert Osazuwa Ness, and Jonathan Larson","author":"Edge Darren","year":"2024","unstructured":"Darren Edge, Ha Trinh, Newman Cheng, Joshua Bradley, Alex Chao, Apurva Mody, Steven Truitt, Dasha Metropolitansky, Robert Osazuwa Ness, and Jonathan Larson. 2024. From Local to Global: A GraphRAG Approach to Query-Focused Summarization. arXiv preprint arXiv:2404.16130 (2024)."},{"key":"e_1_3_2_1_13_1","volume-title":"ColPali: Efficient Document Retrieval with Vision Language Models. In The Thirteenth International Conference on Learning Representations.","author":"Faysse Manuel","year":"2024","unstructured":"Manuel Faysse, Hugues Sibille, Tony Wu, Bilel Omrani, Gautier Viaud, C\u00e9line Hudelot, and Pierre Colombo. 2024. ColPali: Efficient Document Retrieval with Vision Language Models. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_14_1","volume-title":"Rishabh Ranjan, Joshua Robinson, Rex Ying, Jiaxuan You, and Jure Leskovec.","author":"Fey Matthias","year":"2023","unstructured":"Matthias Fey, Weihua Hu, Kexin Huang, Jan Eric Lenssen, Rishabh Ranjan, Joshua Robinson, Rex Ying, Jiaxuan You, and Jure Leskovec. 2023. Relational Deep Learning: Graph Representation Learning on Relational Databases. arXiv preprint arXiv:2312.04615 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"Retrieval-Augmented Generation for Large Language Models: A Survey. arXiv preprint arXiv:2312.10997","author":"Gao Yunfan","year":"2023","unstructured":"Yunfan Gao, Yun Xiong, Xinyu Gao, Kangxiang Jia, Jinliu Pan, Yuxi Bi, Yi Dai, Jiawei Sun, Haofen Wang, and Haofen Wang. 2023. Retrieval-Augmented Generation for Large Language Models: A Survey. arXiv preprint arXiv:2312.10997, Vol. 2 (2023)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Jeanie Genesis and Frazier Keane. 2025. Integrating Knowledge Retrieval with Generation: A Comprehensive Survey of RAG Models in NLP. (2025).","DOI":"10.20944\/preprints202504.0351.v1"},{"key":"e_1_3_2_1_17_1","volume-title":"Rada Mihalcea, and Soujanya Poria.","author":"Ghosal Deepanway","year":"2023","unstructured":"Deepanway Ghosal, Navonil Majumder, Roy Ka-Wei Lee, Rada Mihalcea, and Soujanya Poria. 2023. Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts. arXiv preprint arXiv:2310.20159 (2023)."},{"key":"e_1_3_2_1_18_1","volume-title":"LightRAG: Simple and Fast Retrieval-Augmented Generation. arXiv preprint arXiv:2410.05779","author":"Guo Zirui","year":"2024","unstructured":"Zirui Guo, Lianghao Xia, Yanhua Yu, Tu Ao, and Chao Huang. 2024. LightRAG: Simple and Fast Retrieval-Augmented Generation. arXiv preprint arXiv:2410.05779 (2024)."},{"key":"e_1_3_2_1_19_1","volume-title":"A Comprehensive Survey of Retrieval-Augmented Generation (RAG): Evolution, Current Landscape and Future Directions. arXiv preprint arXiv:2410.12837","author":"Gupta Shailja","year":"2024","unstructured":"Shailja Gupta, Rajesh Ranjan, and Surya Narayan Singh. 2024. A Comprehensive Survey of Retrieval-Augmented Generation (RAG): Evolution, Current Landscape and Future Directions. arXiv preprint arXiv:2410.12837 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Retrieval Augmented Language Model Pre-Training. In International Conference on Machine Learning. PMLR, 3929-3938","author":"Guu Kelvin","year":"2020","unstructured":"Kelvin Guu, Kenton Lee, Zora Tung, Panupong Pasupat, and Mingwei Chang. 2020. Retrieval Augmented Language Model Pre-Training. In International Conference on Machine Learning. PMLR, 3929-3938."},{"key":"e_1_3_2_1_21_1","volume-title":"MDocAgent: A Multi-Modal Multi-Agent Framework for Document Understanding. arXiv preprint arXiv:2503.13964","author":"Han Siwei","year":"2025","unstructured":"Siwei Han, Peng Xia, Ruiyi Zhang, Tong Sun, Yun Li, Hongtu Zhu, and Huaxiu Yao. 2025. MDocAgent: A Multi-Modal Multi-Agent Framework for Document Understanding. arXiv preprint arXiv:2503.13964 (2025)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_23_1","volume-title":"SCITUNE: Aligning Large Language Models with Scientific Multimodal Instructions. arXiv preprint arXiv:2307.01139","author":"Horawalavithana Sameera","year":"2023","unstructured":"Sameera Horawalavithana, Sai Munikoti, Ian Stewart, and Henry Kvinge. 2023. SCITUNE: Aligning Large Language Models with Scientific Multimodal Instructions. arXiv preprint arXiv:2307.01139 (2023)."},{"key":"e_1_3_2_1_24_1","volume-title":"Tool Documentation Enables Zero-Shot Tool-Usage with Large Language Models. arXiv preprint arXiv:2308.00675","author":"Hsieh Cheng-Yu","year":"2023","unstructured":"Cheng-Yu Hsieh, Si-An Chen, Chun-Liang Li, Yasuhisa Fujii, Alexander Ratner, Chen-Yu Lee, Ranjay Krishna, and Tomas Pfister. 2023. Tool Documentation Enables Zero-Shot Tool-Usage with Large Language Models. arXiv preprint arXiv:2308.00675 (2023)."},{"key":"e_1_3_2_1_25_1","unstructured":"Anwen Hu Haiyang Xu Jiabo Ye Ming Yan Liang Zhang Bo Zhang Chen Li Ji Zhang Qin Jin Fei Huang et al. 2024. mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding. arXiv preprint arXiv:2403.12895 (2024)."},{"key":"e_1_3_2_1_26_1","unstructured":"Aaron Hurst Adam Lerer Adam P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et al. 2024. GPT-4o System Card. arXiv preprint arXiv:2410.21276 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"Few-shot Learning with Retrieval Augmented Language Models. arXiv preprint arXiv:2208.03299","author":"Izacard Gautier","year":"2022","unstructured":"Gautier Izacard, Patrick Lewis, Maria Lomeli, Lucas Hosseini, Fabio Petroni, Timo Schick, Jane Dwivedi-Yu, Armand Joulin, Sebastian Riedel, and Edouard Grave. 2022. Few-shot Learning with Retrieval Augmented Language Models. arXiv preprint arXiv:2208.03299, Vol. 1, 2 (2022), 4."},{"key":"e_1_3_2_1_28_1","first-page":"99","article-title":"A Graph-Agent-Based Approach to Enhancing Knowledge-Based QA with Advanced RAG","volume":"25","author":"Jeong Cheonsu","year":"2024","unstructured":"Cheonsu Jeong. 2024a. A Graph-Agent-Based Approach to Enhancing Knowledge-Based QA with Advanced RAG. Knowledge Management Research, Vol. 25, 3 (2024), 99-119.","journal-title":"Knowledge Management Research"},{"key":"e_1_3_2_1_29_1","volume-title":"A Study on the Implementation Method of an Agent-Based Advanced RAG System Using Graph. arXiv preprint arXiv:2407.19994","author":"Jeong Cheonsu","year":"2024","unstructured":"Cheonsu Jeong. 2024b. A Study on the Implementation Method of an Agent-Based Advanced RAG System Using Graph. arXiv preprint arXiv:2407.19994 (2024)."},{"key":"e_1_3_2_1_30_1","volume-title":"Active Retrieval Augmented Generation. arXiv preprint arXiv:2305.06983","author":"Jiang Zhengbao","year":"2023","unstructured":"Zhengbao Jiang, Frank F Xu, Luyu Gao, Zhiqing Sun, Qian Liu, Jane Dwivedi-Yu, Yiming Yang, Jamie Callan, and Graham Neubig. 2023. Active Retrieval Augmented Generation. arXiv preprint arXiv:2305.06983 (2023)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401075"},{"key":"e_1_3_2_1_32_1","volume-title":"PaperQA: Retrieval-Augmented Generative Agent for Scientific Research. arXiv preprint arXiv:2312.07559","author":"L\u00e1la Jakub","year":"2023","unstructured":"Jakub L\u00e1la, Odhran O'Donoghue, Aleksandar Shtedritski, Sam Cox, Samuel G Rodriques, and Andrew D White. 2023. PaperQA: Retrieval-Augmented Generative Agent for Scientific Research. arXiv preprint arXiv:2312.07559 (2023)."},{"key":"e_1_3_2_1_33_1","first-page":"9459","article-title":". Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks","volume":"33","author":"Lewis Patrick","year":"2020","unstructured":"Patrick Lewis, Ethan Perez, Aleksandra Piktus, Fabio Petroni, Vladimir Karpukhin, Naman Goyal, Heinrich K\u00fcttler, Mike Lewis, Wen-tau Yih, Tim Rockt\u00e4schel, et al., 2020. Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks. Advances in Neural Information Processing Systems, Vol. 33 (2020), 9459-9474.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_34_1","volume-title":"International Conference on Machine Learning. PMLR","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In International Conference on Machine Learning. PMLR, 19730-19742."},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics. 2814-2833","author":"Li Weijie","year":"2025","unstructured":"Weijie Li, Jin Wang, Liang-Chih Yu, and Xuejie Zhang. 2025. Topology-of-Question-Decomposition: Enhancing Large Language Models with Information Retrieval for Knowledge-Intensive Tasks. In Proceedings of the 31st International Conference on Computational Linguistics. 2814-2833."},{"key":"e_1_3_2_1_36_1","first-page":"34892","article-title":"Visual Instruction Tuning","volume":"36","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual Instruction Tuning. Advances in Neural Information Processing Systems, Vol. 36 (2023), 34892-34916.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_37_1","volume-title":"Aligning Vision to Language: Text-Free Multimodal Knowledge Graph Construction for Enhanced LLMs Reasoning. arXiv preprint arXiv:2503.12972","author":"Liu Junming","year":"2025","unstructured":"Junming Liu, Siyuan Meng, Yanting Gao, Song Mao, Pinlong Cai, Guohang Yan, Yirong Chen, Zilin Bian, Botian Shi, and Ding Wang. 2025a. Aligning Vision to Language: Text-Free Multimodal Knowledge Graph Construction for Enhanced LLMs Reasoning. arXiv preprint arXiv:2503.12972 (2025)."},{"key":"e_1_3_2_1_38_1","volume-title":"SiQA: A Large Multi-Modal Question Answering Model for Structured Images Based on RAG. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1-5.","author":"Liu Jiawang","year":"2025","unstructured":"Jiawang Liu, Ye Tao, Fei Wang, Hui Li, and Xiugong Qin. 2025b. SiQA: A Large Multi-Modal Question Answering Model for Structured Images Based on RAG. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1-5."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_40_1","first-page":"2507","article-title":"Learn to Explain: Multimodal Reasoning via Thought Chains for Science Question Answering","volume":"35","author":"Lu Pan","year":"2022","unstructured":"Pan Lu, Swaroop Mishra, Tanglin Xia, Liang Qiu, Kai-Wei Chang, Song-Chun Zhu, Oyvind Tafjord, Peter Clark, and Ashwin Kalyan. 2022. Learn to Explain: Multimodal Reasoning via Thought Chains for Science Question Answering. Advances in Neural Information Processing Systems, Vol. 35 (2022), 2507-2521.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01480"},{"key":"e_1_3_2_1_42_1","volume-title":"GNN-RAG: Graph Neural Retrieval for Large Language Model Reasoning. arXiv preprint arXiv:2405.20139","author":"Mavromatis Costas","year":"2024","unstructured":"Costas Mavromatis and George Karypis. 2024. GNN-RAG: Graph Neural Retrieval for Large Language Model Reasoning. arXiv preprint arXiv:2405.20139 (2024)."},{"key":"e_1_3_2_1_43_1","volume-title":"Shi Qiu, Muhammad Saqib, Saeed Anwar, Muhammad Usman, Naveed Akhtar, Nick Barnes, and Ajmal Mian.","author":"Naveed Humza","year":"2023","unstructured":"Humza Naveed, Asad Ullah Khan, Shi Qiu, Muhammad Saqib, Saeed Anwar, Muhammad Usman, Naveed Akhtar, Nick Barnes, and Ajmal Mian. 2023. A Comprehensive Overview of Large Language Models. arXiv preprint arXiv:2307.06435 (2023)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/AIxSET62544.2024.00030"},{"key":"e_1_3_2_1_45_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. PmLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. PmLR, 8748-8763."},{"key":"e_1_3_2_1_46_1","volume-title":"Beyond Text: Optimizing RAG with Multimodal Inputs for Industrial Applications. arXiv preprint arXiv:2410.21943","author":"Riedler Monica","year":"2024","unstructured":"Monica Riedler and Stefan Langer. 2024. Beyond Text: Optimizing RAG with Multimodal Inputs for Industrial Applications. arXiv preprint arXiv:2410.21943 (2024)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1017\/nlp.2024.53"},{"key":"e_1_3_2_1_48_1","first-page":"68539","article-title":"Toolformer: Language models can teach themselves to use tools","volume":"36","author":"Schick Timo","year":"2023","unstructured":"Timo Schick, Jane Dwivedi-Yu, Roberto Dess`i, Roberta Raileanu, Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom. 2023. Toolformer: Language models can teach themselves to use tools. Advances in Neural Information Processing Systems, Vol. 36 (2023), 68539-68551.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_49_1","volume-title":"DRAGIN: Dynamic Retrieval Augmented Generation based on the Real-time Information Needs of Large Language Models. arXiv preprint arXiv:2403.10081","author":"Su Weihang","year":"2024","unstructured":"Weihang Su, Yichen Tang, Qingyao Ai, Zhijing Wu, and Yiqun Liu. 2024. DRAGIN: Dynamic Retrieval Augmented Generation based on the Real-time Information Needs of Large Language Models. arXiv preprint arXiv:2403.10081 (2024)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1186\/s13326-024-00320-3"},{"key":"e_1_3_2_1_51_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_52_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_53_1","volume-title":"Medical Graph RAG: Towards Safe Medical Large Language Model via Graph Retrieval-Augmented Generation. arXiv preprint arXiv:2408.04187","author":"Wu Junde","year":"2024","unstructured":"Junde Wu, Jiayuan Zhu, Yunli Qi, Jingkun Chen, Min Xu, Filippo Menolascina, and Vicente Grau. 2024. Medical Graph RAG: Towards Safe Medical Large Language Model via Graph Retrieval-Augmented Generation. arXiv preprint arXiv:2408.04187 (2024)."},{"key":"e_1_3_2_1_54_1","volume-title":"MMed-RAG: Versatile Multimodal RAG System for Medical Vision Language Models. arXiv preprint arXiv:2410.13085","author":"Xia Peng","year":"2024","unstructured":"Peng Xia, Kangyu Zhu, Haoran Li, Tianze Wang, Weijia Shi, Sheng Wang, Linjun Zhang, James Zou, and Huaxiu Yao. 2024. MMed-RAG: Versatile Multimodal RAG System for Medical Vision Language Models. arXiv preprint arXiv:2410.13085 (2024)."},{"key":"e_1_3_2_1_55_1","unstructured":"An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei et al. 2024. Qwen2.5 Technical Report. arXiv preprint arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_56_1","volume-title":"MM-BigBench: Evaluating Multimodal Models on Multimodal Content Comprehension Tasks. arXiv preprint arXiv:2310.09036","author":"Yang Xiaocui","year":"2023","unstructured":"Xiaocui Yang, Wenfang Wu, Shi Feng, Ming Wang, Daling Wang, Yang Li, Qi Sun, Yifei Zhang, Xiaoming Fu, and Soujanya Poria. 2023. MM-BigBench: Evaluating Multimodal Models on Multimodal Content Comprehension Tasks. arXiv preprint arXiv:2310.09036 (2023)."},{"key":"e_1_3_2_1_57_1","volume-title":"First Conference on Language Modeling.","author":"Zhang Tianjun","year":"2024","unstructured":"Tianjun Zhang, Shishir G Patil, Naman Jain, Sheng Shen, Matei Zaharia, Ion Stoica, and Joseph E Gonzalez. 2024. RAFT: Adapting Language Model to Domain Specific RAG. In First Conference on Language Modeling."},{"key":"e_1_3_2_1_58_1","first-page":"5168","article-title":"DDCoT: Duty-Distinct Chain-of-Thought Prompting for Multimodal Reasoning in Language Models","volume":"36","author":"Zheng Ge","year":"2023","unstructured":"Ge Zheng, Bin Yang, Jiajin Tang, Hong-Yu Zhou, and Sibei Yang. 2023. DDCoT: Duty-Distinct Chain-of-Thought Prompting for Multimodal Reasoning in Language Models. Advances in Neural Information Processing Systems, Vol. 36 (2023), 5168-5191.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599563"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754761","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:03:17Z","timestamp":1765342997000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754761"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":59,"alternative-id":["10.1145\/3746027.3754761","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754761","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}