{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:05:41Z","timestamp":1784138741010,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"&#x5c;&quot;Pioneer&#x5c;&quot; and &#x5c;&quot;Leading Goose&#x5c;&quot; R&#x5c;&#x5c;&amp;D Program of Zhejiang","award":["2025C02032"],"award-info":[{"award-number":["2025C02032"]}]},{"name":"National Natural Science Founda-tion of China","award":["92570101"],"award-info":[{"award-number":["92570101"]}]},{"name":"Fundamental Research Funds forthe Central Universities","award":["226-202500080"],"award-info":[{"award-number":["226-202500080"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809603","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:28:19Z","timestamp":1783693699000},"page":"1789-1799","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["RegionSLM: Region-aware Question Answering on Document Screenshots"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1297-768X","authenticated-orcid":false,"given":"Chao","family":"Wang","sequence":"first","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia and University of Technology Sydney, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9572-2345","authenticated-orcid":false,"given":"Hehe","family":"Fan","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9767-1021","authenticated-orcid":false,"given":"Huichen","family":"Yang","sequence":"additional","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4927-3937","authenticated-orcid":false,"given":"Sarvnaz","family":"Karimi","sequence":"additional","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia and Monash University, Melbourne, VIC, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4149-839X","authenticated-orcid":false,"given":"Lina","family":"Yao","sequence":"additional","affiliation":[{"name":"University of New South Wales, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0512-880X","authenticated-orcid":false,"given":"Yi","family":"Yang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"993","article-title":"Docformer: End-to-end transformer for document understanding","author":"Appalaraju Srikar","year":"2021","unstructured":"Srikar Appalaraju, Bhavan Jasani, Bhargava Urala Kota, Yusheng Xie, and R Manmatha. 2021. Docformer: End-to-end transformer for document understanding. In ICCV. 993-1003.","journal-title":"ICCV."},{"key":"e_1_3_2_1_2_1","first-page":"3058","article-title":"ScreenAI: a vision-language model for UI and infographics understanding","author":"Baechler Gilles","year":"2024","unstructured":"Gilles Baechler, Srinivas Sunkara, Maria Wang, Fedir Zubach, Hassan Mansoor, Vincent Etter, Victor C\u0103rbune, Jason Lin, Jindong Chen, and Abhanshu Sharma. 2024. ScreenAI: a vision-language model for UI and infographics understanding. In IJCAI. 3058-3068.","journal-title":"IJCAI."},{"key":"e_1_3_2_1_3_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_4_1","volume-title":"Due: End-to-end document understanding benchmark. In NeurIPS Datasets and Benchmarks Track.","author":"Borchmann \u0141ukasz","year":"2021","unstructured":"\u0141ukasz Borchmann, Micha\u0142 Pietruszka, Tomasz Stanislawek, Dawid Jurkiewicz, Micha\u0142 Turski, Karolina Szyndler, and Filip Grali\u0144ski. 2021. Due: End-to-end document understanding benchmark. In NeurIPS Datasets and Benchmarks Track."},{"key":"e_1_3_2_1_5_1","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In NAACL-HLT. 4171-4186.","journal-title":"NAACL-HLT."},{"key":"e_1_3_2_1_6_1","volume-title":"Colpali: Efficient document retrieval with vision language models. In ICLR.","author":"Faysse Manuel","year":"2025","unstructured":"Manuel Faysse, Hugues Sibille, Tony Wu, Bilel Omrani, Gautier Viaud, C\u00e9line Hudelot, and Pierre Colombo. 2025. Colpali: Efficient document retrieval with vision language models. In ICLR."},{"key":"e_1_3_2_1_7_1","volume-title":"Improving language understanding from screenshots. arXiv preprint arXiv:2402.14073","author":"Gao Tianyu","year":"2024","unstructured":"Tianyu Gao, Zirui Wang, Adithya Bhaskar, and Danqi Chen. 2024. Improving language understanding from screenshots. arXiv preprint arXiv:2402.14073 (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"Lambert: Layout-aware language modeling for information extraction","author":"Garncarek \u0141ukasz","year":"2021","unstructured":"\u0141ukasz Garncarek, Rafa\u0142 Powalski, Tomasz Stanis\u0142awek, Bartosz Topolski, Piotr Halama, Micha\u0142 Turski, and Filip Grali\u0144ski. 2021. Lambert: Layout-aware language modeling for information extraction. In ICDAR. Springer, 532-547."},{"key":"e_1_3_2_1_9_1","first-page":"39","article-title":"Unidoc: Unified pretraining framework for document understanding","volume":"34","author":"Gu Jiuxiang","year":"2021","unstructured":"Jiuxiang Gu, Jason Kuen, Vlad I Morariu, Handong Zhao, Rajiv Jain, Nikolaos Barmpalios, Ani Nenkova, and Tong Sun. 2021. Unidoc: Unified pretraining framework for document understanding. NeurIPS, Vol. 34 (2021), 39-50.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_10_1","first-page":"4583","article-title":"XYLayoutLM: Towards layout-aware multimodal networks for visually-rich document understanding","author":"Gu Zhangxuan","year":"2022","unstructured":"Zhangxuan Gu, Changhua Meng, Ke Wang, Jun Lan, Weiqiang Wang, Ming Gu, and Liqing Zhang. 2022. XYLayoutLM: Towards layout-aware multimodal networks for visually-rich document understanding. In CVPR. 4583-4592.","journal-title":"CVPR."},{"key":"e_1_3_2_1_11_1","first-page":"991","article-title":"Evaluation of deep convolutional nets for document image classification and retrieval","author":"Harley Adam W","year":"2015","unstructured":"Adam W Harley, Alex Ufkes, and Konstantinos G Derpanis. 2015. Evaluation of deep convolutional nets for document image classification and retrieval. In ICDAR. IEEE, 991-995.","journal-title":"ICDAR. IEEE"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21322"},{"key":"e_1_3_2_1_13_1","first-page":"9427","article-title":"Screenqa: Large-scale question-answer pairs over mobile app screenshots","author":"Hsiao Yu-Chung","year":"2025","unstructured":"Yu-Chung Hsiao, Fedir Zubach, Gilles Baechler, Srinivas Sunkara, Victor C\u0103rbune, Jason Lin, Maria Wang, Yun Zhu, and Jindong Chen. 2025. Screenqa: Large-scale question-answer pairs over mobile app screenshots. In NAACL-HLT. 9427-9452.","journal-title":"NAACL-HLT."},{"key":"e_1_3_2_1_14_1","first-page":"3096","article-title":"mplug-docowl 1.5: Unified structure learning for OCR-free document understanding","author":"Hu Anwen","year":"2024","unstructured":"Anwen Hu, Haiyang Xu, Jiabo Ye, Ming Yan, Liang Zhang, Bo Zhang, Ji Zhang, Qin Jin, Fei Huang, and Jingren Zhou. 2024. mplug-docowl 1.5: Unified structure learning for OCR-free document understanding. In Findings of EMNLP. 3096-3120.","journal-title":"Findings of EMNLP."},{"key":"e_1_3_2_1_15_1","first-page":"4083","article-title":"Layoutlmv3: Pre-training for document ai with unified text and image masking","author":"Huang Yupan","year":"2022","unstructured":"Yupan Huang, Tengchao Lv, Lei Cui, Yutong Lu, and Furu Wei. 2022. Layoutlmv3: Pre-training for document ai with unified text and image masking. In ACM MM. 4083-4091.","journal-title":"ACM MM."},{"key":"e_1_3_2_1_16_1","volume-title":"ICDAR2019 competition on scanned receipt OCR and information extraction. In ICDAR. IEEE, 1516-1520","author":"Huang Zheng","year":"2019","unstructured":"Zheng Huang, Kai Chen, Jianhua He, Xiang Bai, Dimosthenis Karatzas, Shijian Lu, and CV Jawahar. 2019. ICDAR2019 competition on scanned receipt OCR and information extraction. In ICDAR. IEEE, 1516-1520."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDARW.2019.10029"},{"key":"e_1_3_2_1_18_1","volume-title":"ICLR workshops.","author":"Kahou Samira Ebrahimi","year":"2018","unstructured":"Samira Ebrahimi Kahou, Vincent Michalski, Adam Atkinson, \u00c1kos K\u00e1d\u00e1r, Adam Trischler, and Yoshua Bengio. 2018. ''FigureQA'': An annotated figure dataset for visual reasoning. In ICLR workshops."},{"key":"e_1_3_2_1_19_1","first-page":"8580","article-title":"AxCell: Automatic Extraction of Results from Machine Learning Papers","author":"Kardas Marcin","year":"2020","unstructured":"Marcin Kardas, Piotr Czapla, Pontus Stenetorp, Sebastian Ruder, Sebastian Riedel, Ross Taylor, and Robert Stojnic. 2020. AxCell: Automatic Extraction of Results from Machine Learning Papers. In EMNLP. 8580-8594.","journal-title":"EMNLP."},{"key":"e_1_3_2_1_20_1","first-page":"498","article-title":"OCR-free document understanding transformer","author":"Kim Geewook","year":"2022","unstructured":"Geewook Kim, Teakgyu Hong, Moonbin Yim, JeongYeon Nam, Jinyoung Park, Jinyeong Yim, Wonseok Hwang, Sangdoo Yun, Dongyoon Han, and Seunghyun Park. 2022. OCR-free document understanding transformer. In ECCV. 498-517.","journal-title":"ECCV."},{"key":"e_1_3_2_1_21_1","first-page":"3735","article-title":"Formnet: Structural encoding beyond sequential modeling in form document information extraction","author":"Lee Chen-Yu","year":"2022","unstructured":"Chen-Yu Lee, Chun-Liang Li, Timothy Dozat, Vincent Perot, Guolong Su, Nan Hua, Joshua Ainslie, Renshen Wang, Yasuhisa Fujii, and Tomas Pfister. 2022. Formnet: Structural encoding beyond sequential modeling in form document information extraction. In ACL. 3735-3754.","journal-title":"ACL."},{"key":"e_1_3_2_1_22_1","first-page":"18893","article-title":"Pix2struct: Screenshot parsing as pretraining for visual language understanding","author":"Lee Kenton","year":"2023","unstructured":"Kenton Lee, Mandar Joshi, Iulia Raluca Turc, Hexiang Hu, Fangyu Liu, Julian Martin Eisenschlos, Urvashi Khandelwal, Peter Shaw, Ming-Wei Chang, and Kristina Toutanova. 2023. Pix2struct: Screenshot parsing as pretraining for visual language understanding. In ICLR. 18893-18912.","journal-title":"ICLR."},{"key":"e_1_3_2_1_23_1","first-page":"6309","article-title":"StructuralLM: Structural pre-training for form understanding","author":"Li Chenliang","year":"2021","unstructured":"Chenliang Li, Bin Bi, Ming Yan, Wei Wang, Songfang Huang, Fei Huang, and Luo Si. 2021a. StructuralLM: Structural pre-training for form understanding. In ACL-IJCNLP. 6309-6318.","journal-title":"ACL-IJCNLP."},{"key":"e_1_3_2_1_24_1","first-page":"3530","article-title":"Dit: Self-supervised pre-training for document image transformer","author":"Li Junlong","year":"2022","unstructured":"Junlong Li, Yiheng Xu, Tengchao Lv, Lei Cui, Cha Zhang, and Furu Wei. 2022. Dit: Self-supervised pre-training for document image transformer. In ACM MM. 3530-3539.","journal-title":"ACM MM."},{"key":"e_1_3_2_1_25_1","first-page":"5652","article-title":"Selfdoc: Self-supervised document representation learning","author":"Li Peizhao","year":"2021","unstructured":"Peizhao Li, Jiuxiang Gu, Jason Kuen, Vlad I Morariu, Handong Zhao, Rajiv Jain, Varun Manjunatha, and Hongfu Liu. 2021b. Selfdoc: Self-supervised document representation learning. In CVPR. 5652-5660.","journal-title":"CVPR."},{"key":"e_1_3_2_1_26_1","first-page":"1912","article-title":"Structext: Structured text understanding with multi-modal transformers","author":"Li Yulin","year":"2021","unstructured":"Yulin Li, Yuxi Qian, Yuechen Yu, Xiameng Qin, Chengquan Zhang, Yan Liu, Kun Yao, Junyu Han, Jingtuo Liu, and Errui Ding. 2021c. Structext: Structured text understanding with multi-modal transformers. In ACM MM. 1912-1920.","journal-title":"ACM MM."},{"key":"e_1_3_2_1_27_1","first-page":"34892","article-title":"Visual instruction tuning","volume":"36","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual instruction tuning. NeurIPS, Vol. 36 (2023), 34892-34916.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_28_1","first-page":"15630","article-title":"LayoutLLM","author":"Luo Chuwei","year":"2024","unstructured":"Chuwei Luo, Yufan Shen, Zhaoqing Zhu, Qi Zheng, Zhi Yu, and Cong Yao. 2024. LayoutLLM: Layout Instruction Tuning with Large Language Models for Document Understanding. In CVPR. 15630-15640.","journal-title":"In CVPR."},{"key":"e_1_3_2_1_29_1","first-page":"2263","article-title":"Chartqa: A benchmark for question answering about charts with visual and logical reasoning","author":"Masry Ahmed","year":"2022","unstructured":"Ahmed Masry, Xuan Long Do, Jia Qing Tan, Shafiq Joty, and Enamul Hoque. 2022. Chartqa: A benchmark for question answering about charts with visual and logical reasoning. In Findings ACL. 2263-2279.","journal-title":"Findings ACL."},{"key":"e_1_3_2_1_30_1","first-page":"1697","article-title":"Infographicvqa","author":"Mathew Minesh","year":"2022","unstructured":"Minesh Mathew, Viraj Bagal, Rub\u00e8n Tito, Dimosthenis Karatzas, Ernest Valveny, and CV Jawahar. 2022. Infographicvqa. In WACV. 1697-1706.","journal-title":"WACV."},{"key":"e_1_3_2_1_31_1","first-page":"2200","article-title":"Docvqa: A dataset for vqa on document images","author":"Mathew Minesh","year":"2021","unstructured":"Minesh Mathew, Dimosthenis Karatzas, and CV Jawahar. 2021. Docvqa: A dataset for vqa on document images. In WACV. 2200-2209.","journal-title":"WACV."},{"key":"e_1_3_2_1_32_1","first-page":"947","article-title":"OCR-VQA'': Visual question answering by reading text in images","author":"Mishra Anand","year":"2019","unstructured":"Anand Mishra, Shashank Shekhar, Ajeet Kumar Singh, and Anirban Chakraborty. 2019. ''OCR-VQA'': Visual question answering by reading text in images. In ICDAR. 947-952.","journal-title":"ICDAR."},{"key":"e_1_3_2_1_33_1","volume-title":"Workshop on Document Intelligence at NeurIPS.","author":"Park Seunghyun","year":"2019","unstructured":"Seunghyun Park, Seung Shin, Bado Lee, Junyeop Lee, Jaeheung Surh, Minjoon Seo, and Hwalsuk Lee. 2019. Cord: a consolidated receipt dataset for post-ocr parsing. In Workshop on Document Intelligence at NeurIPS."},{"key":"e_1_3_2_1_34_1","first-page":"1470","article-title":"Compositional semantic parsing on semi-structured tables","author":"Pasupat Panupong","year":"2015","unstructured":"Panupong Pasupat and Percy Liang. 2015. Compositional semantic parsing on semi-structured tables. In ACL-IJCNLP. 1470-1480.","journal-title":"ACL-IJCNLP."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-86331-9_47"},{"key":"e_1_3_2_1_36_1","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of Machine Learning Research, Vol. 21, 140 (2020), 1-67.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_37_1","first-page":"1","article-title":"Language Modeling with Pixels","author":"Rust Phillip","year":"2023","unstructured":"Phillip Rust, Jonas F Lotz, Emanuele Bugliarello, Elizabeth Salesky, Miryam de Lhoneux, and Desmond Elliott. 2023. Language Modeling with Pixels. In ICLR. 1-32.","journal-title":"ICLR."},{"key":"e_1_3_2_1_38_1","volume-title":"Textcaps: a dataset for image captioning with reading comprehension","author":"Sidorov Oleksii","unstructured":"Oleksii Sidorov, Ronghang Hu, Marcus Rohrbach, and Amanpreet Singh. 2020. Textcaps: a dataset for image captioning with reading comprehension. In ECCV. Springer, 742-758."},{"key":"e_1_3_2_1_39_1","volume-title":"Meet Shah, Yu Jiang, Xinlei Chen, Dhruv Batra, Devi Parikh, and Marcus Rohrbach.","author":"Singh Amanpreet","year":"2019","unstructured":"Amanpreet Singh, Vivek Natarajan, Meet Shah, Yu Jiang, Xinlei Chen, Dhruv Batra, Devi Parikh, and Marcus Rohrbach. 2019. Towards VQA models that can read. In CVPR. 8317-8326."},{"key":"e_1_3_2_1_40_1","first-page":"56549","article-title":"DocVXQA","author":"Souibgui Mohamed Ali","year":"2025","unstructured":"Mohamed Ali Souibgui, Changkyu Choi, Andrey Barsky, Kangsoo Jung, Ernest Valveny, and Dimosthenis Karatzas. 2025. DocVXQA: Context-Aware Visual Explanations for Document Question Answering. In ICLR. 56549-56569.","journal-title":"Context-Aware Visual Explanations for Document Question Answering. In ICLR."},{"key":"e_1_3_2_1_41_1","volume-title":"Kleister: key information extraction datasets involving long documents with complex layouts","author":"Stanis\u0142awek Tomasz","unstructured":"Tomasz Stanis\u0142awek, Filip Grali\u0144ski, Anna Wr\u00f3blewska, Dawid Lipi\u0144ski, Agnieszka Kaliska, Paulina Rosalska, Bartosz Topolski, and Przemys\u0142aw Biecek. 2021. Kleister: key information extraction datasets involving long documents with complex layouts. In ICDAR. Springer, 564-579."},{"key":"e_1_3_2_1_42_1","volume-title":"Deepform: Understand structured documents at scale. Weights & Biases report","author":"Svetlichnaya Stacey","year":"2020","unstructured":"Stacey Svetlichnaya. 2020. Deepform: Understand structured documents at scale. Weights & Biases report (2020)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i15.17635"},{"key":"e_1_3_2_1_44_1","first-page":"19254","article-title":"Unifying vision, text, and layout for universal document processing","author":"Tang Zineng","year":"2023","unstructured":"Zineng Tang, Ziyi Yang, Guoxin Wang, Yuwei Fang, Yang Liu, Chenguang Zhu, Michael Zeng, Cha Zhang, and Mohit Bansal. 2023. Unifying vision, text, and layout for universal document processing. In CVPR. 19254-19264.","journal-title":"CVPR."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Chao Wang Hehe Fan Ruijie Quan Lina Yao and Yi Yang. 2025a. ProtChatGPT: Towards Understanding Proteins with Hybrid Representation and Large Language Models. In SIGIR. 1076\u20131086.","DOI":"10.1145\/3726302.3730064"},{"key":"e_1_3_2_1_46_1","first-page":"23539","article-title":"Adapting text-to-image generation with feature difference instruction for generic image restoration","author":"Wang Chao","year":"2025","unstructured":"Chao Wang, Hehe Fan, Huichen Yang, Sarvnaz Karimi, Lina Yao, and Yi Yang. 2025b. Adapting text-to-image generation with feature difference instruction for generic image restoration. In CVPR. 23539-23550.","journal-title":"CVPR."},{"key":"e_1_3_2_1_47_1","first-page":"7747","article-title":"Lilt: A simple yet effective language-independent layout transformer for structured document understanding","author":"Wang Jiapeng","year":"2022","unstructured":"Jiapeng Wang, Lianwen Jin, and Kai Ding. 2022. Lilt: A simple yet effective language-independent layout transformer for structured document understanding. In ACL. 7747-7757.","journal-title":"ACL."},{"key":"e_1_3_2_1_48_1","volume-title":"Hongmin Wang and William Yang Wang","author":"Yunkai Zhang Hong Wang Jianshu Chen","year":"2020","unstructured":"Jianshu Chen Yunkai Zhang Hong Wang Shiyang Li Xiyou Zhou Wenhu Chen, Hongmin Wang and William Yang Wang. 2020. TabFact: A Large-scale Dataset for Table-based Fact Verification. In ICLR. Addis Ababa, Ethiopia."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v40i13.38097"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.370"},{"key":"e_1_3_2_1_51_1","first-page":"1192","article-title":"LayoutLM","author":"Xu Yiheng","year":"2020","unstructured":"Yiheng Xu, Minghao Li, Lei Cui, Shaohan Huang, Furu Yu, Yijuan Chen, and Ming Zhou. 2020. LayoutLM: Pre-training of Text and Layout for Document Image Understanding. In SIGKDD. 1192-1200.","journal-title":"In SIGKDD."},{"key":"e_1_3_2_1_52_1","volume-title":"Layoutxlm: Multimodal pre-training for multilingual visually-rich document understanding. arXiv preprint arXiv:2104.08836","author":"Xu Yiheng","year":"2021","unstructured":"Yiheng Xu, Tengchao Lv, Lei Cui, Guoxin Wang, Yijuan Lu, Dinei Florencio, Cha Zhang, and Furu Wei. 2021b. Layoutxlm: Multimodal pre-training for multilingual visually-rich document understanding. arXiv preprint arXiv:2104.08836 (2021)."},{"key":"e_1_3_2_1_53_1","first-page":"2579","article-title":"Layoutlmv2: Multi-modal pre-training for visually-rich document understanding","author":"Xu Yang","year":"2021","unstructured":"Yang Xu, Yiheng Xu, Tengchao Lv, Lei Cui, Furu Wei, Guoxin Wang, Yijuan Lu, Dinei Florencio, Cha Zhang, Wanxiang Che, et al., 2021c. Layoutlmv2: Multi-modal pre-training for visually-rich document understanding. In ACL-IJCNLP. 2579-2591.","journal-title":"ACL-IJCNLP."},{"key":"e_1_3_2_1_54_1","first-page":"837","article-title":"Docthinker: Explainable multimodal large language models with rule-based reinforcement learning for document understanding","author":"Yu Wenwen","year":"2025","unstructured":"Wenwen Yu, Zhibo Yang, Yuliang Liu, and Xiang Bai. 2025. Docthinker: Explainable multimodal large language models with rule-based reinforcement learning for document understanding. In ICCV. 837-847.","journal-title":"ICCV."},{"key":"e_1_3_2_1_55_1","volume-title":"LLaVAR: Enhanced visual instruction tuning for text-rich image understanding. arXiv preprint arXiv:2306.17107","author":"Zhang Yanzhe","year":"2023","unstructured":"Yanzhe Zhang, Ruiyi Zhang, Jiuxiang Gu, Yufan Zhou, Nedim Lipka, Diyi Yang, and Tong Sun. 2023. LLaVAR: Enhanced visual instruction tuning for text-rich image understanding. arXiv preprint arXiv:2306.17107 (2023)."}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:21:40Z","timestamp":1784136100000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809603"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":55,"alternative-id":["10.1145\/3805712.3809603","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809603","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}