{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:08:43Z","timestamp":1784138923384,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":96,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Natural Science Foundation of China","award":["62576082"],"award-info":[{"award-number":["62576082"]}]},{"name":"National Natural Science Foundation of China","award":["U23B2019"],"award-info":[{"award-number":["U23B2019"]}]},{"name":"National Natural Science Foundation of China","award":["62461146205"],"award-info":[{"award-number":["62461146205"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809602","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:28:19Z","timestamp":1783693699000},"page":"2162-2173","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["<scp>ReAlign:<\/scp>\n                    Optimizing the Visual Document Retriever with Reasoning-Guided Fine-Grained Alignment"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-8144-2413","authenticated-orcid":false,"given":"Hao","family":"Yang","sequence":"first","affiliation":[{"name":"Northeastern University, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7594-3132","authenticated-orcid":false,"given":"Yifan","family":"Ji","sequence":"additional","affiliation":[{"name":"Northeastern University, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3671-5978","authenticated-orcid":false,"given":"Zhipeng","family":"Xu","sequence":"additional","affiliation":[{"name":"Northeastern University, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0083-3224","authenticated-orcid":false,"given":"Zhenghao","family":"Liu","sequence":"additional","affiliation":[{"name":"Northeastern University, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2421-7098","authenticated-orcid":false,"given":"Yukun","family":"Yan","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0807-6889","authenticated-orcid":false,"given":"Zulong","family":"Chen","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5408-3145","authenticated-orcid":false,"given":"Shuo","family":"Wang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7422-6254","authenticated-orcid":false,"given":"Yu","family":"Gu","sequence":"additional","affiliation":[{"name":"Northeastern University, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3171-8889","authenticated-orcid":false,"given":"Ge","family":"Yu","sequence":"additional","affiliation":[{"name":"Northeastern University, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Technical Report: A Highly Capable Language Model Locally on Your Phone. arXiv preprint arXiv:2404.14219","author":"Abdin Marah","year":"2024","unstructured":"Marah Abdin, Jyoti Aneja, Hany Awadalla, et al., 2024. Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone. arXiv preprint arXiv:2404.14219 (2024)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/s13735-016-0110-y"},{"key":"e_1_3_2_1_3_1","first-page":"3500","article-title":"A brief review of document image retrieval methods: Recent advances","author":"Alaei Fahimeh","year":"2016","unstructured":"Fahimeh Alaei, Alireza Alaei, Michael Blumenstein, and Umapada Pal. 2016a. A brief review of document image retrieval methods: Recent advances. In Proceedings of IJCNN. 3500-3507.","journal-title":"Proceedings of IJCNN."},{"key":"e_1_3_2_1_4_1","volume-title":"Document Image Retrieval Based on Texture Features: A Recognition-Free Approach. In 2016 International Conference on Digital Image Computing: Techniques and Applications (DICTA). 1-7.","author":"Alaei Fahimeh","year":"2016","unstructured":"Fahimeh Alaei, Alireza Alaei, Umapada Pal, and Michael Blumenstein. 2016b. Document Image Retrieval Based on Texture Features: A Recognition-Free Approach. In 2016 International Conference on Digital Image Computing: Techniques and Applications (DICTA). 1-7."},{"key":"e_1_3_2_1_5_1","first-page":"993","article-title":"DocFormer","author":"Appalaraju Srikar","year":"2021","unstructured":"Srikar Appalaraju, Bhavan Jasani, Bhargava Urala Kota, Yusheng Xie, and R Manmatha. 2021. DocFormer: End-to-End Transformer for Document Understanding. In Proceedings of ICCV. 993-1003.","journal-title":"End-to-End Transformer for Document Understanding. In Proceedings of ICCV."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i2.27828"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10115-006-0014-x"},{"key":"e_1_3_2_1_8_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2.5-VL Technical Report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_9_1","first-page":"1436","article-title":"GlobalDoc","author":"Bakkali Souhail","year":"2025","unstructured":"Souhail Bakkali, Sanket Biswas, Zuheng Ming, Micka\u00ebl Coustaty, Mar\u00e7al Rusi nol, Oriol Ramos Terrades, and Josep Llad\u00f3s. 2025. GlobalDoc: A Cross-Modal Vision-Language Framework for Real-World Document Image Retrieval and Classification. In Proceedings of WACV. 1436-1446.","journal-title":"In Proceedings of WACV."},{"key":"e_1_3_2_1_10_1","first-page":"102","article-title":"Assessing the Impact of OCR Errors in Information Retrieval","author":"Bazzo Guilherme Torresan","year":"2020","unstructured":"Guilherme Torresan Bazzo, Gustavo Acauan Lorentz, Danny Suarez Vargas, and Viviane P Moreira. 2020. Assessing the Impact of OCR Errors in Information Retrieval. In Proceedings of ECIR. 102-109.","journal-title":"Proceedings of ECIR."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485127"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01788"},{"key":"e_1_3_2_1_13_1","unstructured":"Cheng Cui Ting Sun Manhui Lin Tingquan Gao Yubo Zhang Jiaxuan Liu Xueqing Wang Zelun Zhang Changda Zhou Hongen Liu et al. 2025b. PaddleOCR 3.0 Technical Report. arXiv preprint arXiv:2507.05595 (2025)."},{"key":"e_1_3_2_1_14_1","volume-title":"Attention Grounded Enhancement for Visual Document Retrieval. arXiv preprint arXiv:2511.13415","author":"Cui Wanqing","year":"2025","unstructured":"Wanqing Cui, Wei Huang, Yazhi Guo, Yibo Hu, Meiguang Jin, Junfeng Ma, and Keping Bi. 2025a. Attention Grounded Enhancement for Visual Document Retrieval. arXiv preprint arXiv:2511.13415 (2025)."},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of ICLR.","author":"Dao Tri","year":"2024","unstructured":"Tri Dao. 2024. FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning. In Proceedings of ICLR."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1006\/cviu.1998.0692"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of ICLR.","author":"Faysse Manuel","year":"2025","unstructured":"Manuel Faysse, Hugues Sibille, Tony Wu, Bilel Omrani, Gautier Viaud, C\u00e9line Hudelot, and Pierre Colombo. 2025. ColPali: Efficient Document Retrieval with Vision Language Models. In Proceedings of ICLR."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2012.6467497"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.02.023"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02767"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of ICLR.","author":"Hu Edward J","year":"2022","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, Weizhu Chen, et al., 2022. LoRA: Low-Rank Adaptation of Large Language Models. In Proceedings of ICLR."},{"key":"e_1_3_2_1_22_1","volume-title":"Unsupervised dense information retrieval with contrastive learning. Transactions on Machine Learning Research (TMLR)","author":"Izacard Gautier","year":"2021","unstructured":"Gautier Izacard, Mathilde Caron, Lucas Hosseini, Sebastian Riedel, Piotr Bojanowski, Armand Joulin, and Edouard Grave. 2021. Unsupervised dense information retrieval with contrastive learning. Transactions on Machine Learning Research (TMLR) (2021)."},{"key":"e_1_3_2_1_23_1","first-page":"292","article-title":"Learning Refined Document Representations for Dense Retrieval via Deliberate Thinking","author":"Ji Yifan","year":"2025","unstructured":"Yifan Ji, Zhipeng Xu, Zhenghao Liu, Yukun Yan, Shi Yu, Yishan Li, Zhiyuan Liu, Yu Gu, Ge Yu, and Maosong Sun. 2025. Learning Refined Document Representations for Dense Retrieval via Deliberate Thinking. In Proceedings of SIGIR-AP. 292-302.","journal-title":"Proceedings of SIGIR-AP."},{"key":"e_1_3_2_1_24_1","volume-title":"E5-V: Universal Embeddings with Multimodal Large Language Models. arXiv preprint arXiv:2407.12580","author":"Jiang Ting","year":"2024","unstructured":"Ting Jiang, Minghui Song, Zihan Zhang, Haizhen Huang, Weiwei Deng, Feng Sun, Qi Zhang, Deqing Wang, and Fuzhen Zhuang. 2024. E5-V: Universal Embeddings with Multimodal Large Language Models. arXiv preprint arXiv:2407.12580 (2024)."},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of ICLR.","author":"Jiang Ziyan","year":"2025","unstructured":"Ziyan Jiang, Rui Meng, Xinyi Yang, Semih Yavuz, Yingbo Zhou, and Wenhu Chen. 2025. VLM2Vec: Training Vision-Language Models for Massive Multimodal Embedding Tasks. In Proceedings of ICLR."},{"key":"e_1_3_2_1_26_1","first-page":"9339","article-title":"Your Large Vision-Language Model Only Needs A Few Attention Heads For Visual Grounding","author":"Kang Seil","year":"2025","unstructured":"Seil Kang, Jinyeong Kim, Junhyeok Kim, and Seong Jae Hwang. 2025. Your Large Vision-Language Model Only Needs A Few Attention Heads For Visual Grounding. In Proceedings of CVPR. 9339-9350.","journal-title":"Proceedings of CVPR."},{"key":"e_1_3_2_1_27_1","first-page":"6769","article-title":"Dense Passage Retrieval for Open-Domain Question Answering","author":"Karpukhin Vladimir","year":"2020","unstructured":"Vladimir Karpukhin, Barlas Oguz, Sewon Min, Patrick SH Lewis, Ledell Wu, Sergey Edunov, Danqi Chen, and Wen-tau Yih. 2020. Dense Passage Retrieval for Open-Domain Question Answering. In Proceedings of EMNLP. 6769-6781.","journal-title":"Proceedings of EMNLP."},{"key":"e_1_3_2_1_28_1","volume-title":"Recent Advances, Challenges, and Future Trends. ACM Transactions on Information Systems","author":"Ke Wenjun","year":"2025","unstructured":"Wenjun Ke, Yifan Zheng, Yining Li, Hengyuan Xu, Dong Nie, Peng Wang, and Yao He. 2025. Large Language Models in Document Intelligence: A Comprehensive Survey, Recent Advances, Challenges, and Future Trends. ACM Transactions on Information Systems (2025), 1-64."},{"key":"e_1_3_2_1_29_1","volume-title":"International Journal of Software Engineering and Its Applications","author":"Keyvanpour Mohammadreza","year":"2013","unstructured":"Mohammadreza Keyvanpour and Reza Tavoli. 2013. Document image retrieval: Algorithms, analysis and promising directions. International Journal of Software Engineering and Its Applications (2013), 93-106."},{"key":"e_1_3_2_1_30_1","first-page":"498","article-title":"OCR-Free Document Understanding Transformer","author":"Kim Geewook","year":"2022","unstructured":"Geewook Kim, Teakgyu Hong, Moonbin Yim, JeongYeon Nam, Jinyoung Park, Jinyeong Yim, Wonseok Hwang, Sangdoo Yun, Dongyoon Han, and Seunghyun Park. 2022. OCR-Free Document Understanding Transformer. In Proceedings of ECCV. 498-517.","journal-title":"Proceedings of ECCV."},{"key":"e_1_3_2_1_31_1","first-page":"3148","article-title":"good4cir","author":"Kolouju Pranavi","year":"2025","unstructured":"Pranavi Kolouju, Eric Xing, Robert Pless, Nathan Jacobs, and Abby Stylianou. 2025. good4cir: Generating Detailed Synthetic Captions for Composed Image Retrieval. In Proceedings of CVPR. 3148-3157.","journal-title":"Generating Detailed Synthetic Captions for Composed Image Retrieval. In Proceedings of CVPR."},{"key":"e_1_3_2_1_32_1","first-page":"8285","article-title":"Open-WikiTable : Dataset for Open Domain Question Answering with Complex Reasoning over Table","author":"Kweon Sunjun","year":"2023","unstructured":"Sunjun Kweon, Yeonsu Kwon, Seonhee Cho, Yohan Jo, and Edward Choi. 2023. Open-WikiTable : Dataset for Open Domain Question Answering with Complex Reasoning over Table. In Findings of ACL. 8285-8297.","journal-title":"Findings of ACL."},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of ICLR.","author":"Lee Chankyu","year":"2025","unstructured":"Chankyu Lee, Rajarshi Roy, Mengyao Xu, Jonathan Raiman, Mohammad Shoeybi, Bryan Catanzaro, and Wei Ping. 2025. NV-Embed: Improved Techniques for Training LLMs as Generalist Embedding Models. In Proceedings of ICLR."},{"key":"e_1_3_2_1_34_1","first-page":"3490","article-title":"Llama2Vec","author":"Li Chaofan","year":"2024","unstructured":"Chaofan Li, Zheng Liu, Shitao Xiao, Yingxia Shao, and Defu Lian. 2024b. Llama2Vec: Unsupervised Adaptation of Large Language Models for Dense Retrieval. In Proceedings of ACL. 3490-3500.","journal-title":"In Proceedings of ACL."},{"key":"e_1_3_2_1_35_1","first-page":"9098","article-title":"DyFo","author":"Li Geng","year":"2025","unstructured":"Geng Li, Jinglin Xu, Yunzhen Zhao, and Yuxin Peng. 2025b. DyFo: A Training-Free Dynamic Focus Visual Search for Enhancing LMMs in Fine-Grained Visual Understanding. In Proceedings of CVPR. 9098-9108.","journal-title":"In Proceedings of CVPR."},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of ICML.","author":"Li Jinhao","year":"2024","unstructured":"Jinhao Li, Haopeng Li, Sarah Erfani, Lei Feng, James Bailey, and Feng Liu. 2024a. Visual-Text Cross Alignment: Refining the Similarity Score in Vision-Language Models. In Proceedings of ICML."},{"key":"e_1_3_2_1_37_1","first-page":"14369","article-title":"Multimodal ArXiv","author":"Li Lei","year":"2024","unstructured":"Lei Li, Yuqi Wang, Runxin Xu, Peiyi Wang, Xiachong Feng, Lingpeng Kong, and Qi Liu. 2024c. Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models. In Proceedings of ACL. 14369-14387.","journal-title":"In Proceedings of ACL."},{"key":"e_1_3_2_1_38_1","volume-title":"Towards Visual Text Grounding of Multimodal Large Language Model. arXiv preprint arXiv:2504.04974","author":"Li Ming","year":"2025","unstructured":"Ming Li, Ruiyi Zhang, Jian Chen, Chenguang Wang, Jiuxiang Gu, Yufan Zhou, Franck Dernoncourt, Wanrong Zhu, Tianyi Zhou, and Tong Sun. 2025c. Towards Visual Text Grounding of Multimodal Large Language Model. arXiv preprint arXiv:2504.04974 (2025)."},{"key":"e_1_3_2_1_39_1","first-page":"287","article-title":"More Robust Dense Retrieval with Contrastive Dual Learning","author":"Li Yizhi","year":"2021","unstructured":"Yizhi Li, Zhenghao Liu, Chenyan Xiong, and Zhiyuan Liu. 2021a. More Robust Dense Retrieval with Contrastive Dual Learning. In Proceedings of SIGIR. 287-296.","journal-title":"Proceedings of SIGIR."},{"key":"e_1_3_2_1_40_1","volume-title":"RegionRAG: Region-level Retrieval-Augmented Generation for Visual Document Understanding. arXiv preprint arXiv:2510.27261","author":"Li Yinglu","year":"2025","unstructured":"Yinglu Li, Zhiying Lu, Zhihang Liu, Chuanbin Liu, and Hongtao Xie. 2025a. RegionRAG: Region-level Retrieval-Augmented Generation for Visual Document Understanding. arXiv preprint arXiv:2510.27261 (2025)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475345"},{"key":"e_1_3_2_1_42_1","first-page":"2356","article-title":"Pyserini","author":"Lin Jimmy","year":"2021","unstructured":"Jimmy Lin, Xueguang Ma, Sheng-Chieh Lin, Jheng-Hong Yang, Ronak Pradeep, and Rodrigo Nogueira. 2021. Pyserini: A Python Toolkit for Reproducible Information Retrieval Research with Sparse and Dense Representations. In Proceedings of SIGIR. 2356-2362.","journal-title":"In Proceedings of SIGIR."},{"key":"e_1_3_2_1_43_1","volume-title":"Look as You Think: Unifying Reasoning and Visual Evidence Attribution for Verifiable Document RAG via Reinforcement Learning. arXiv preprint arXiv:2511.12003","author":"Liu Shuochen","year":"2025","unstructured":"Shuochen Liu, Pengfei Luo, Chao Zhang, Yuhao Chen, Haotian Zhang, Qi Liu, Xin Kou, Tong Xu, and Enhong Chen. 2025. Look as You Think: Unifying Reasoning and Visual Evidence Attribution for Verifiable Document RAG via Reinforcement Learning. arXiv preprint arXiv:2511.12003 (2025)."},{"key":"e_1_3_2_1_44_1","volume-title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document. arXiv preprint arXiv:2403.04473","author":"Liu Yuliang","year":"2024","unstructured":"Yuliang Liu, Biao Yang, Qiang Liu, Zhang Li, Zhiyin Ma, Shuo Zhang, and Xiang Bai. 2024. TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document. arXiv preprint arXiv:2403.04473 (2024)."},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of ICLR.","author":"Liu Zhenghao","year":"2023","unstructured":"Zhenghao Liu, Chenyan Xiong, Yuanhuiyi Lv, Zhiyuan Liu, and Ge Yu. 2023. Universal Vision-Language Dense Retrieval: Learning A Unified Representation Space for Multi-Modal Retrieval. In Proceedings of ICLR."},{"key":"e_1_3_2_1_46_1","volume-title":"Multimodal Reference Visual Grounding. arXiv preprint arXiv:2504.02876","author":"Lu Yangxiao","year":"2025","unstructured":"Yangxiao Lu, Ruosen Li, Liqiang Jing, Jikai Wang, Xinya Du, Yunhui Guo, Nicholas Ruozzi, and Yu Xiang. 2025. Multimodal Reference Visual Grounding. arXiv preprint arXiv:2504.02876 (2025)."},{"key":"e_1_3_2_1_47_1","first-page":"6492","article-title":"Unifying Multimodal Retrieval via Document Screenshot Embedding","author":"Ma Xueguang","year":"2024","unstructured":"Xueguang Ma, Sheng-Chieh Lin, Minghan Li, Wenhu Chen, and Jimmy Lin. 2024. Unifying Multimodal Retrieval via Document Screenshot Embedding. In Proceedings of EMNLP. 6492-6505.","journal-title":"Proceedings of EMNLP."},{"key":"e_1_3_2_1_48_1","volume-title":"ViDoRe Benchmark V2: Raising the Bar for Visual Retrieval. arXiv preprint arXiv:2505.17166","author":"Mac\u00e9 Quentin","year":"2025","unstructured":"Quentin Mac\u00e9, Ant\u00f3nio Loison, and Manuel Faysse. 2025. ViDoRe Benchmark V2: Raising the Bar for Visual Retrieval. arXiv preprint arXiv:2505.17166 (2025)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-22913-8_9"},{"key":"e_1_3_2_1_50_1","first-page":"2263","article-title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","author":"Masry Ahmed","year":"2022","unstructured":"Ahmed Masry, Xuan Long Do, Jia Qing Tan, Shafiq Joty, and Enamul Hoque. 2022. ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning. In Findings of ACL. 2263-2279.","journal-title":"Findings of ACL."},{"key":"e_1_3_2_1_51_1","first-page":"1697","article-title":"InfographicVQA","author":"Mathew Minesh","year":"2022","unstructured":"Minesh Mathew, Viraj Bagal, Rub\u00e8n Tito, Dimosthenis Karatzas, Ernest Valveny, and CV Jawahar. 2022. InfographicVQA. In Proceedings of WACV. 1697-1706.","journal-title":"Proceedings of WACV."},{"key":"e_1_3_2_1_52_1","first-page":"2200","article-title":"DocVQA","author":"Mathew Minesh","year":"2021","unstructured":"Minesh Mathew, Dimosthenis Karatzas, and CV Jawahar. 2021. DocVQA: A Dataset for VQA on Document Images. In Proceedings of WACV. 2200-2209.","journal-title":"A Dataset for VQA on Document Images. In Proceedings of WACV."},{"key":"e_1_3_2_1_53_1","volume-title":"Statistical learning for OCR error correction. Information Processing & Management","author":"Mei Jie","year":"2018","unstructured":"Jie Mei, Aminul Islam, Abidalrahman Moh'd, Yajing Wu, and Evangelos Milios. 2018. Statistical learning for OCR error correction. Information Processing & Management (2018), 874-887."},{"key":"e_1_3_2_1_54_1","first-page":"1527","article-title":"PlotQA: Reasoning over Scientific Plots","author":"Methani Nitesh","year":"2020","unstructured":"Nitesh Methani, Pritha Ganguly, Mitesh M Khapra, and Pratyush Kumar. 2020. PlotQA: Reasoning over Scientific Plots. In Proceedings of WACV. 1527-1536.","journal-title":"Proceedings of WACV."},{"key":"e_1_3_2_1_55_1","first-page":"30807","article-title":"SERVAL: Surprisingly Effective Zero-Shot Visual Document Retrieval Powered by Large Vision and Language Models","author":"Nguyen Thong","year":"2025","unstructured":"Thong Nguyen, Yibin Lei, Jia-Huei Ju, and Andrew Yates. 2025. SERVAL: Surprisingly Effective Zero-Shot Visual Document Retrieval Powered by Large Vision and Language Models. In Proceedings of EMNLP. 30807-30822.","journal-title":"Proceedings of EMNLP."},{"key":"e_1_3_2_1_56_1","volume-title":"Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748","author":"van den Oord Aaron","year":"2018","unstructured":"Aaron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)."},{"key":"e_1_3_2_1_57_1","first-page":"3744","author":"Peng Qiming","year":"2022","unstructured":"Qiming Peng, Yinxu Pan, Wenjin Wang, Bin Luo, Zhenyu Zhang, Zhengjie Huang, Yuhui Cao, Weichong Yin, Yongfeng Chen, Yin Zhang, et al., 2022. ERNIE-Layout: Layout Knowledge Enhanced Pre-training for Visually-rich Document Understanding. In Findings of EMNLP. 3744-3756.","journal-title":"In Findings of EMNLP."},{"key":"e_1_3_2_1_58_1","volume-title":"Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer. In International Conference on Document Analysis and Recognition. 732-747","author":"Jurkiewicz Dawid","year":"2021","unstructured":"Rafa? Powalski, ?ukasz Borchmann, Dawid Jurkiewicz, Tomasz Dwojak, Micha? Pietruszka, and Gabriela Pa?ka. 2021. Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer. In International Conference on Document Analysis and Recognition. 732-747."},{"key":"e_1_3_2_1_59_1","first-page":"8748","volume-title":"Proceedings of ICML","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of ICML, Vol. 139. 8748-8763."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1561\/1500000019"},{"key":"e_1_3_2_1_61_1","first-page":"3419","article-title":"Towards Debiasing Fact Verification Models","author":"Schuster Tal","year":"2019","unstructured":"Tal Schuster, Darsh Shah, Yun Jie Serene Yeo, Daniel Roberto Filizzola Ortiz, Enrico Santus, and Regina Barzilay. 2019. Towards Debiasing Fact Verification Models. In Proceedings of EMNLP-IJCNLP. 3419-3425.","journal-title":"Proceedings of EMNLP-IJCNLP."},{"key":"e_1_3_2_1_62_1","first-page":"6602","article-title":"ZoomEye: Enhancing Multimodal LLMs with Human-Like Zooming Capabilities through Tree-Based Image Exploration","author":"Shen Haozhan","year":"2025","unstructured":"Haozhan Shen, Kangjia Zhao, Tiancheng Zhao, Ruochen Xu, Zilun Zhang, Mingwei Zhu, and Jianwei Yin. 2025. ZoomEye: Enhancing Multimodal LLMs with Human-Like Zooming Capabilities through Tree-Based Image Exploration. In Proceedings of EMNLP. 6602-6618.","journal-title":"Proceedings of EMNLP."},{"key":"e_1_3_2_1_63_1","first-page":"4613","article-title":"Where To Look","author":"Shih Kevin J","year":"2016","unstructured":"Kevin J Shih, Saurabh Singh, and Derek Hoiem. 2016. Where To Look: Focus Regions for Visual Question Answering. In Proceedings of CVPR. 4613-4621.","journal-title":"Focus Regions for Visual Question Answering. In Proceedings of CVPR."},{"key":"e_1_3_2_1_64_1","first-page":"1423","article-title":"REVISE: A Framework for Revising OCRed text in Practical Information Systems with Data Contamination Strategy","author":"Shim Gyuho","year":"2025","unstructured":"Gyuho Shim, Seongtae Hong, and Heui-Seok Lim. 2025. REVISE: A Framework for Revising OCRed text in Practical Information Systems with Data Contamination Strategy. In Proceedings of ACL. 1423-1434.","journal-title":"Proceedings of ACL."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2025.104368"},{"key":"e_1_3_2_1_66_1","first-page":"23935","article-title":"Unveil","author":"Sun Hao","year":"2025","unstructured":"Hao Sun, Yingyan Hou, Jiayan Guo, Bo Wang, Chunyu Yang, Jinsong Ni, and Yan Zhang. 2025. Unveil: Unified Visual-Textual Integration and Distillation for Multi-modal Document Retrieval. In Proceedings of ACL. 23935-23945.","journal-title":"In Proceedings of ACL."},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2011.213"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02312"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i11.26598"},{"key":"e_1_3_2_1_70_1","first-page":"13878","article-title":"VisualMRC","author":"Tanaka Ryota","year":"2021","unstructured":"Ryota Tanaka, Kyosuke Nishida, and Sen Yoshida. 2021. VisualMRC: Machine Reading Comprehension on Document Images. In Proceedings of AAAI. 13878-13888.","journal-title":"Machine Reading Comprehension on Document Images. In Proceedings of AAAI."},{"key":"e_1_3_2_1_71_1","volume-title":"ModernVBERT: Towards Smaller Visual Document Retrievers. arXiv preprint arXiv:2510.01149","author":"Teiletche Paul","year":"2025","unstructured":"Paul Teiletche, Quentin Mac\u00e9, Max Conti, Antonio Loison, Gautier Viaud, Pierre Colombo, and Manuel Faysse. 2025. ModernVBERT: Towards Smaller Visual Document Retrievers. arXiv preprint arXiv:2510.01149 (2025)."},{"key":"e_1_3_2_1_72_1","volume-title":"HKRAG: Holistic Knowledge Retrieval-Augmented Generation Over Visually-Rich Documents. arXiv preprint arXiv:2511.20227","author":"Tong Anyang","year":"2025","unstructured":"Anyang Tong, Xiang Niu, ZhiPing Liu, Chang Tian, Yanyan Wei, Zenglin Shi, and Meng Wang. 2025. HKRAG: Holistic Knowledge Retrieval-Augmented Generation Over Visually-Rich Documents. arXiv preprint arXiv:2511.20227 (2025)."},{"key":"e_1_3_2_1_73_1","volume-title":"Ibrahim Alabdulmohsin, Nikhil Parthasarathy, Talfan Evans, Lucas Beyer, Ye Xia, Basil Mustafa, et al.","author":"Tschannen Michael","year":"2025","unstructured":"Michael Tschannen, Alexey Gritsenko, Xiao Wang, Muhammad Ferjad Naeem, Ibrahim Alabdulmohsin, Nikhil Parthasarathy, Talfan Evans, Lucas Beyer, Ye Xia, Basil Mustafa, et al., 2025. SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features. arXiv preprint arXiv:2502.14786 (2025)."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01789"},{"key":"e_1_3_2_1_75_1","volume-title":"Pixel Reasoner: Incentivizing Pixel-Space Reasoning with Curiosity-Driven Reinforcement Learning. arXiv preprint arXiv:2505.15966","author":"Wang Haozhe","year":"2025","unstructured":"Haozhe Wang, Alex Su, Weiming Ren, Fangzhen Lin, and Wenhu Chen. 2025c. Pixel Reasoner: Incentivizing Pixel-Space Reasoning with Curiosity-Driven Reinforcement Learning. arXiv preprint arXiv:2505.15966 (2025)."},{"key":"e_1_3_2_1_76_1","first-page":"7747","article-title":"LiLT: A Simple yet Effective Language-Independent Layout Transformer for Structured Document Understanding","author":"Wang Jiapeng","year":"2022","unstructured":"Jiapeng Wang, Lianwen Jin, and Kai Ding. 2022. LiLT: A Simple yet Effective Language-Independent Layout Transformer for Structured Document Understanding. In Proceedings of ACL. 7747-7757.","journal-title":"Proceedings of ACL."},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.642"},{"key":"e_1_3_2_1_78_1","first-page":"9113","article-title":"ViDoRAG: Visual Document Retrieval-Augmented Generation via Dynamic Iterative Reasoning Agents","author":"Wang Qiuchen","year":"2025","unstructured":"Qiuchen Wang, Ruixue Ding, Zehui Chen, Weiqi Wu, Shihang Wang, Pengjun Xie, and Feng Zhao. 2025a. ViDoRAG: Visual Document Retrieval-Augmented Generation via Dynamic Iterative Reasoning Agents. In Proceedings of EMNLP. 9113-9134.","journal-title":"Proceedings of EMNLP."},{"key":"e_1_3_2_1_79_1","volume-title":"VRAG-RL: Empower Vision-Perception-Based RAG for Visually Rich Information Understanding via Iterative Reasoning with Reinforcement Learning. arXiv preprint arXiv:2505.22019","author":"Wang Qiuchen","year":"2025","unstructured":"Qiuchen Wang, Ruixue Ding, Yu Zeng, Zehui Chen, Lin Chen, Shihang Wang, Pengjun Xie, Fei Huang, and Feng Zhao. 2025b. VRAG-RL: Empower Vision-Perception-Based RAG for Visually Rich Information Understanding via Iterative Reasoning with Reinforcement Learning. arXiv preprint arXiv:2505.22019 (2025)."},{"key":"e_1_3_2_1_80_1","first-page":"9929","article-title":"Understanding contrastive representation learning through alignment and uniformity on the hypersphere","author":"Wang Tongzhou","year":"2020","unstructured":"Tongzhou Wang and Phillip Isola. 2020. Understanding contrastive representation learning through alignment and uniformity on the hypersphere. In Proceedings of ICML. 9929-9939.","journal-title":"Proceedings of ICML."},{"key":"e_1_3_2_1_81_1","unstructured":"Jason Wei Yi Tay Rishi Bommasani Colin Raffel Barret Zoph Sebastian Borgeaud Dani Yogatama Maarten Bosma Denny Zhou Donald Metzler et al. 2022. Emergent Abilities of Large Language Models. Transactions on Machine Learning Research (2022)."},{"key":"e_1_3_2_1_82_1","first-page":"447","article-title":"Visual Matching is Enough for Scene Text Retrieval","author":"Wen Lilong","year":"2023","unstructured":"Lilong Wen, Yingrong Wang, Dongxiang Zhang, and Gang Chen. 2023. Visual Matching is Enough for Scene Text Retrieval. In Proceedings of WSDM. 447-455.","journal-title":"Proceedings of WSDM."},{"key":"e_1_3_2_1_83_1","first-page":"641","article-title":"C-Pack","author":"Xiao Shitao","year":"2024","unstructured":"Shitao Xiao, Zheng Liu, Peitian Zhang, Niklas Muennighoff, Defu Lian, and Jian-Yun Nie. 2024. C-Pack: Packed Resources For General Chinese Embeddings. In Proceedings of SIGIR. 641-649.","journal-title":"Packed Resources For General Chinese Embeddings. In Proceedings of SIGIR."},{"key":"e_1_3_2_1_84_1","volume-title":"Proceedings of ICLR.","author":"Xiong Lee","year":"2021","unstructured":"Lee Xiong, Chenyan Xiong, Ye Li, Kwok-Fung Tang, Jialin Liu, Paul Bennett, Junaid Ahmed, and Arnold Overwijk. 2021. Approximate Nearest Neighbor Negative Contrastive Learning for Dense Text Retrieval. In Proceedings of ICLR."},{"key":"e_1_3_2_1_85_1","first-page":"1192","article-title":"LayoutLM","author":"Xu Yiheng","year":"2020","unstructured":"Yiheng Xu, Minghao Li, Lei Cui, Shaohan Huang, Furu Wei, and Ming Zhou. 2020. LayoutLM: Pre-training of Text and Layout for Document Image Understanding. In Proceedings of SIGKDD. 1192-1200.","journal-title":"In Proceedings of SIGKDD."},{"key":"e_1_3_2_1_86_1","volume-title":"Proceedings of ICLR.","author":"Yu Shi","year":"2025","unstructured":"Shi Yu, Chaoyue Tang, Bokai Xu, Junbo Cui, Junhao Ran, Yukun Yan, Zhenghao Liu, Shuo Wang, Xu Han, Zhiyuan Liu, et al., 2025. VisRAG: Vision-based Retrieval-augmented Generation on Multi-modality Documents. In Proceedings of ICLR."},{"key":"e_1_3_2_1_87_1","volume-title":"TextHawk: Exploring Efficient Fine-Grained Perception of Multimodal Large Language Models. arXiv preprint arXiv:2404.09204","author":"Yu Ya-Qi","year":"2024","unstructured":"Ya-Qi Yu, Minghui Liao, Jihao Wu, Yongxin Liao, Xiaoyu Zheng, and Wei Zeng. 2024. TextHawk: Exploring Efficient Fine-Grained Perception of Multimodal Large Language Models. arXiv preprint arXiv:2404.09204 (2024)."},{"key":"e_1_3_2_1_88_1","first-page":"3104","article-title":"VILE","author":"Yuan Huaying","year":"2023","unstructured":"Huaying Yuan, Zhicheng Dou, Yujia Zhou, Yu Guo, and Ji-Rong Wen. 2023. VILE: Block-Aware Visual Enhanced Document Retrieval. In Proceedings of CIKM. 3104-3113.","journal-title":"Block-Aware Visual Enhanced Document Retrieval. In Proceedings of CIKM."},{"key":"e_1_3_2_1_89_1","volume-title":"A Document Image Retrieval System. Engineering Applications of Artificial Intelligence","author":"Zagoris Konstantinos","year":"2010","unstructured":"Konstantinos Zagoris, Kavallieratou Ergina, and Nikos Papamarkos. 2010. A Document Image Retrieval System. Engineering Applications of Artificial Intelligence (2010), 872-879."},{"key":"e_1_3_2_1_90_1","unstructured":"Aohan Zeng Xin Lv Qinkai Zheng Zhenyu Hou Bin Chen Chengxing Xie Cunxiang Wang Da Yin Hao Zeng Jiajie Zhang et al. 2025. GLM-4.5: Agentic Reasoning and Coding (ARC) Foundation Models. arXiv preprint arXiv:2508.06471 (2025)."},{"key":"e_1_3_2_1_91_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.culher.2019.05.018"},{"key":"e_1_3_2_1_92_1","first-page":"17443","article-title":"OCR Hinders RAG","author":"Zhang Junyuan","year":"2025","unstructured":"Junyuan Zhang, Qintong Zhang, Bin Wang, Linke Ouyang, Zichen Wen, Ying Li, Ka-Ho Chow, Conghui He, and Wentao Zhang. 2025c. OCR Hinders RAG: Evaluating the Cascading Impact of OCR on Retrieval-Augmented Generation. In Proceedings of ICCV. 17443-17453.","journal-title":"In Proceedings of ICCV."},{"key":"e_1_3_2_1_93_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i11.26634"},{"key":"e_1_3_2_1_94_1","unstructured":"Xintong Zhang Zhi Gao Bofei Zhang Pengxiang Li Xiaowen Zhang Yang Liu Tao Yuan Yuwei Wu Yunde Jia Song-Chun Zhu et al. 2025a. Chain-of-Focus: Adaptive Visual Search and Zooming for Multimodal Reasoning via RL. arXiv preprint arXiv:2505.15436 (2025)."},{"key":"e_1_3_2_1_95_1","volume-title":"Embedding: Advancing Text Embedding and Reranking Through Foundation Models. arXiv preprint arXiv:2506.05176","author":"Zhang Yanzhao","year":"2025","unstructured":"Yanzhao Zhang, Mingxin Li, Dingkun Long, Xin Zhang, Huan Lin, Baosong Yang, Pengjun Xie, An Yang, Dayiheng Liu, Junyang Lin, et al., 2025b. Qwen3 Embedding: Advancing Text Embedding and Reranking Through Foundation Models. arXiv preprint arXiv:2506.05176 (2025)."},{"key":"e_1_3_2_1_96_1","unstructured":"Wayne Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong et al. 2023. A Survey of Large Language Models. arXiv preprint arXiv:2303.18223 (2023)."}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:24:03Z","timestamp":1784136243000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809602"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":96,"alternative-id":["10.1145\/3805712.3809602","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809602","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}