{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T08:13:49Z","timestamp":1765008829232,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476132,62476134"],"award-info":[{"award-number":["62476132,62476134"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,9]]},"DOI":"10.1145\/3743093.3771062","type":"proceedings-article","created":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T08:06:16Z","timestamp":1765008376000},"page":"1-8","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Global Question-Aware Multimodal Retrieval-Augmented Generation for Multimedia Multi-Hop Question Answering"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-6697-0432","authenticated-orcid":false,"given":"Zhixiao","family":"Shen","sequence":"first","affiliation":[{"name":"Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8380-0609","authenticated-orcid":false,"given":"Jianfei","family":"Yu","sequence":"additional","affiliation":[{"name":"Nanjing University of Science and Technology, Nanjing, China and University of Chinese Academy of Sciences, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5612-7818","authenticated-orcid":false,"given":"Wenya","family":"Wang","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0621-1058","authenticated-orcid":false,"given":"Rui","family":"Xia","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China and University of Chinese Academy of Sciences, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,6]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Jinze Bai Shuai Bai Shusheng Yang Shijie Wang Sinan Tan Peng Wang Junyang Lin Chang Zhou and Jingren Zhou. 2023. Qwen-vl: A versatile vision-language model for understanding localization text reading and beyond. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.12966 1 2 (2023) 3."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.375"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","unstructured":"Muhe Ding Yang Ma Pengda Qin Jianlong Wu Yuhong Li and Liqiang Nie. 2024. RA-BLIP: Multimodal Adaptive Retrieval-Augmented Bootstrapping Language-Image Pre-training. CoRR abs\/2410.14154 (2024). arXiv:https:\/\/arXiv.org\/abs\/2410.1415410.48550\/ARXIV.2410.14154","DOI":"10.48550\/ARXIV.2410.14154"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","unstructured":"Darren Edge Ha Trinh Newman Cheng Joshua Bradley Alex Chao Apurva Mody Steven Truitt and Jonathan Larson. 2024. From Local to Global: A Graph RAG Approach to Query-Focused Summarization. CoRR abs\/2404.16130 (2024). arXiv:https:\/\/arXiv.org\/abs\/2404.1613010.48550\/ARXIV.2404.16130","DOI":"10.48550\/ARXIV.2404.16130"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","unstructured":"Ting Jiang Minghui Song Zihan Zhang Haizhen Huang Weiwei Deng Feng Sun Qi Zhang Deqing Wang and Fuzhen Zhuang. 2024. E5-V: Universal Embeddings with Multimodal Large Language Models. CoRR abs\/2407.12580 (2024). arXiv:https:\/\/arXiv.org\/abs\/2407.1258010.48550\/ARXIV.2407.12580","DOI":"10.48550\/ARXIV.2407.12580"},{"key":"e_1_3_3_1_7_2","unstructured":"Zaid Khan Vijay\u00a0Kumar BG Samuel Schulter Manmohan Chandraker and Yun Fu. 2024. Exploring question decomposition for zero-shot VQA. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3595916.3626389"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3595916.3626391"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/2022.ACL-LONG.290"},{"key":"e_1_3_3_1_11_2","volume-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025","author":"Li Yangning","year":"2025","unstructured":"Yangning Li, Yinghui Li, Xinyu Wang, Yong Jiang, Zhen Zhang, Xinran Zheng, Hui Wang, Hai-Tao Zheng, Fei Huang, Jingren Zhou, and Philip\u00a0S. Yu. 2025. Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent. In The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025. OpenReview.net. https:\/\/openreview.net\/forum?id=VvDEuyVXkG"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3696409.3700223"},{"key":"e_1_3_3_1_13_2","unstructured":"Haotian Liu Chunyuan Li Yuheng Li Bo Li Yuanhan Zhang Sheng Shen and Yong\u00a0Jae Lee. 2024. LLaVA-NeXT: Improved reasoning OCR and world knowledge. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/"},{"key":"e_1_3_3_1_14_2","unstructured":"Zhenghao Liu Chenyan Xiong Yuanhuiyi Lv Zhiyuan Liu and Ge Yu. 2022. Universal vision-language dense retrieval: Learning a unified representation space for multi-modal retrieval. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.00179 (2022)."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3595916.3626424"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3613848"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21370"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Stephen Robertson Hugo Zaragoza et\u00a0al. 2009. The probabilistic relevance framework: BM25 and beyond. Foundations and Trends\u00ae in Information Retrieval 3 4 (2009) 333\u2013389.","DOI":"10.1561\/1500000019"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","unstructured":"Sahel Sharifymoghaddam Shivani Upadhyay Wenhu Chen and Jimmy Lin. 2024. UniRAG: Universal Retrieval Augmentation for Multi-Modal Large Language Models. CoRR abs\/2405.10311 (2024). arXiv:https:\/\/arXiv.org\/abs\/2405.1031110.48550\/ARXIV.2405.10311","DOI":"10.48550\/ARXIV.2405.10311"},{"key":"e_1_3_3_1_20_2","volume-title":"International Conference on Learning Representations","author":"Talmor Alon","year":"2021","unstructured":"Alon Talmor, Ori Yoran, Amnon Catav, Dan Lahav, Yizhong Wang, Akari Asai, Gabriel Ilharco, Hannaneh Hajishirzi, and Jonathan Berant. 2021. MultiModalQA: complex question answering over text, tables and images. In International Conference on Learning Representations."},{"key":"e_1_3_3_1_21_2","unstructured":"Gemini Team Petko Georgiev Ving\u00a0Ian Lei Ryan Burnell Libin Bai Anmol Gulati Garrett Tanzer Damien Vincent Zhufeng Pan Shibo Wang et\u00a0al. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.05530 (2024)."},{"key":"e_1_3_3_1_22_2","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et\u00a0al. 2024. Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.12191 (2024)."},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73021-4_23"},{"key":"e_1_3_3_1_24_2","unstructured":"Shitao Xiao Zheng Liu Peitian Zhang and N Muennighof. 2023. C-pack: packaged resources to advance general Chinese embedding. 2023. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.07597 (2023)."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3595916.3626459"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3595916.3626394"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611964"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/2023.FINDINGS-ACL.292"},{"key":"e_1_3_3_1_29_2","volume-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025","author":"Yu Shi","year":"2025","unstructured":"Shi Yu, Chaoyue Tang, Bokai Xu, Junbo Cui, Junhao Ran, Yukun Yan, Zhenghao Liu, Shuo Wang, Xu Han, Zhiyuan Liu, and Maosong Sun. 2025. VisRAG: Vision-based Retrieval-augmented Generation on Multi-modality Documents. In The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025. OpenReview.net. https:\/\/openreview.net\/forum?id=zG459X3Xge"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/2024.ACL-LONG.175"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.783"}],"event":{"name":"MMAsia '25: ACM Multimedia Asia","location":"Kuala Lumpur Malaysia","acronym":"MMAsia '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 7th ACM International Conference on Multimedia in Asia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3743093.3771062","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T08:09:09Z","timestamp":1765008549000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3743093.3771062"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,6]]},"references-count":30,"alternative-id":["10.1145\/3743093.3771062","10.1145\/3743093"],"URL":"https:\/\/doi.org\/10.1145\/3743093.3771062","relation":{},"subject":[],"published":{"date-parts":[[2025,12,6]]},"assertion":[{"value":"2025-12-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}