{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:58:13Z","timestamp":1785488293363,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774556","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["FormLens: From Ink to Insight with Adapting Vision-Language Models for Handwritten Form Digitization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-8452-9043","authenticated-orcid":false,"given":"Shaon","family":"Bhattacharyya","sequence":"first","affiliation":[{"name":"CVIT, International Institute of Information Technology Hyderabad, Hyderabad, Telengana, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4808-8860","authenticated-orcid":false,"given":"Ajoy","family":"Mondal","sequence":"additional","affiliation":[{"name":"CVIT, International Institute of Information Technology Hyderabad, Hyderabad, Telengana, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6767-7057","authenticated-orcid":false,"given":"C. V.","family":"Jawahar","sequence":"additional","affiliation":[{"name":"CVIT, International Institute of Information Technology Hyderabad, Hyderabad, Telengana, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_3_2_2","unstructured":"Amazon Web Services. [n.d.]. Amazon Textract: Automatically extract text handwriting and data from documents. https:\/\/aws.amazon.com\/textract\/."},{"key":"e_1_3_3_3_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00103"},{"key":"e_1_3_3_3_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00959"},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"crossref","unstructured":"Youngmin Baek Bado Lee Dongyoon Han Sangdoo Yun and Hwalsuk Lee. 2019. Character region awareness for text detection. (2019) 9365\u20139374.","DOI":"10.1109\/CVPR.2019.00959"},{"key":"e_1_3_3_3_6_2","unstructured":"Jinze Bai and others.2023. Qwen Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.16609 (2023)."},{"key":"e_1_3_3_3_7_2","unstructured":"Lukas Blecher Guillem Cucurull Thomas Scialom and Robert Stojnic. 2023. Nougat: Neural optical understanding for academic documents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.13418 (2023)."},{"key":"e_1_3_3_3_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219861"},{"key":"e_1_3_3_3_9_2","doi-asserted-by":"crossref","unstructured":"Roberto Brunelli. 2008. Template matching techniques in computer vision. Theory and Practice (2008) 25\u201328.","DOI":"10.1002\/9780470744055"},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"crossref","unstructured":"Denis Coquenet Cl\u00e9ment Chatelain and Thierry Paquet. 2022. End-to-end handwritten paragraph text recognition using a vertical attention network. IEEE 45 (2022) 508\u2013524.","DOI":"10.1109\/TPAMI.2022.3144899"},{"key":"e_1_3_3_3_11_2","unstructured":"Cheng et\u00a0al. Cui. 2025. PaddleOCR 3.0 Technical Report."},{"key":"e_1_3_3_3_12_2","doi-asserted-by":"crossref","unstructured":"Dan Deng Haifeng Liu Xuelong Li and Deng Cai. 2018. Pixellink: Detecting scene text via instance segmentation. (2018).","DOI":"10.1609\/aaai.v32i1.12269"},{"key":"e_1_3_3_3_13_2","unstructured":"Daniel\u00a0Hernandez Diaz Siyang Qin Reeve Ingle Yasuhisa Fujii and Alessandro Bissacco. 2021. Rethinking text line recognition models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2104.07787 (2021)."},{"key":"e_1_3_3_3_14_2","first-page":"15641","volume-title":"CVPR","author":"al. Wan et","year":"2024","unstructured":"Wan et al.2024. Omniparser: A unified framework for text spotting key information extraction and table recognition. In CVPR. 15641\u201315653."},{"key":"e_1_3_3_3_15_2","unstructured":"Google Cloud. [n.d.]. Form Parser | Document AI. https:\/\/cloud.google.com\/document-ai\/docs\/form-parser. Accessed on [Current Date]."},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"crossref","unstructured":"Samuel Grieggs Bingyu Shen Greta Rauch Pei Li Jiaqi Ma David Chiang Brian Price and Walter\u00a0J Scheirer. 2021. Measuring human perception to improve handwritten document transcription. IEEE 44 (2021) 6594\u20136601.","DOI":"10.1109\/TPAMI.2021.3092688"},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"crossref","unstructured":"Felix Hertlein Alexander Naumann and Patrick Philipp. 2023. Inv3D: a high-resolution 3D invoice dataset for template-guided single-image document unwarping. International Journal on Document Analysis and Recognition (IJDAR) (2023).","DOI":"10.1007\/s10032-023-00434-x"},{"key":"e_1_3_3_3_18_2","unstructured":"Hu and others.2022. Lora: Low-rank adaptation of large language models. ICLR 1 (2022)."},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548112"},{"key":"e_1_3_3_3_20_2","unstructured":"JaidedAI. Year. EasyOCR. https:\/\/github.com\/JaidedAI\/EasyOCR."},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"crossref","unstructured":"Anil\u00a0K Jain Robert P.\u00a0W. Duin and Jianchang Mao. 2000. Statistical pattern recognition: A review. IEEE Transactions on pattern analysis and machine intelligence 22 (2000) 4\u201337.","DOI":"10.1109\/34.824819"},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDARW.2019.10029"},{"key":"e_1_3_3_3_23_2","unstructured":"Mahsa Kholghi Xiaoxiao Shao Subhojeet Ghosh et\u00a0al. 2021. FormX: A comprehensive benchmark for visual information extraction from forms. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2106.11363 (2021)."},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19815-1_29"},{"key":"e_1_3_3_3_25_2","unstructured":"Geewook Kim Teakgyu Hong Moonbin Yim Jinyoung Park Jinyeong Yim Wonseok Hwang Sangdoo Yun Dongyoon Han and Seunghyun Park. 2021. Donut: Document understanding transformer without ocr. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2111.15664 (2021)."},{"key":"e_1_3_3_3_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3478328"},{"key":"e_1_3_3_3_27_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-86337-1_27"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i11.26538"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548038"},{"key":"e_1_3_3_3_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00098"},{"key":"e_1_3_3_3_31_2","doi-asserted-by":"crossref","unstructured":"Rujiao Long Hangdi Xing Zhibo Yang Qi Zheng Zhi Yu Fei Huang and Cong Yao. 2025. LORE++: Logical location regression network for table structure recognition with pre-training. PR (2025).","DOI":"10.1016\/j.patcog.2024.110816"},{"key":"e_1_3_3_3_32_2","unstructured":"Shangbang Long Jiaqiang Ruan Wenjie Zhang Xin He Wenhao Wu and Cong Yao. 2018. Textsnake: A flexible representation for detecting text of arbitrary shapes. (2018) 20\u201336."},{"key":"e_1_3_3_3_33_2","unstructured":"Microsoft Azure. [n.d.]. Azure Form Recognizer documentation. https:\/\/azure.microsoft.com\/en-us\/products\/ai-services\/ai-document-intelligence\/. Accessed on [Current Date]."},{"key":"e_1_3_3_3_34_2","unstructured":"Mindee. 2021. docTR: Document Text Recognition. https:\/\/github.com\/mindee\/doctr."},{"key":"e_1_3_3_3_35_2","unstructured":"docTR Mindee et\u00a0al. 2022. docTR: a Document Text Recognition toolbox. GitHub repository (2022). https:\/\/github.com\/mindee\/doctr"},{"key":"e_1_3_3_3_36_2","doi-asserted-by":"publisher","unstructured":"Isaac Sanchez and Enrique Vidal. 2016. Bentham Dataset R0. 10.5281\/zenodo.44519","DOI":"10.5281\/zenodo.44519"},{"key":"e_1_3_3_3_37_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-86549-8_9"},{"key":"e_1_3_3_3_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2007.4376991"},{"key":"e_1_3_3_3_39_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00459"},{"key":"e_1_3_3_3_40_2","unstructured":"Nishant Subramani Alexandre Matton Malcolm Greaves and Adrian Lam. 2020. A survey of deep learning approaches for ocr and document understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2011.13534 (2020)."},{"key":"e_1_3_3_3_41_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01845"},{"key":"e_1_3_3_3_42_2","unstructured":"Haoran Wei and others.2024. General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model. ArXiv (2024)."},{"key":"e_1_3_3_3_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403172"},{"key":"e_1_3_3_3_44_2","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403172"},{"key":"e_1_3_3_3_45_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.253"},{"key":"e_1_3_3_3_46_2","unstructured":"Yang Xu Yiheng Xu Tengchao Lv Lei Cui Furu Wei Guoxin Wang Yijuan Lu Dinei Florencio Cha Zhang Wanxiang Che et\u00a0al. 2020. Layoutlmv2: Multi-modal pre-training for visually-rich document understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2012.14740."},{"key":"e_1_3_3_3_47_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.462"},{"key":"e_1_3_3_3_48_2","unstructured":"Susan Zhang and others.2022. OPT: Open Pre-trained Transformer Language Models. ArXiv (2022)."},{"key":"e_1_3_3_3_49_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00074"},{"key":"e_1_3_3_3_50_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.283"},{"key":"e_1_3_3_3_51_2","doi-asserted-by":"crossref","unstructured":"Yi Zhu Zhou Yanpeng Chunwei Wang Yang Cao Jianhua Han Lu Hou and Hang Xu. 2024. Unit: Unifying image and text recognition in one vision encoder. Advances in Neural Information Processing Systems 37 (2024) 122185\u2013122205.","DOI":"10.52202\/079017-3883"}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774556","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:08:21Z","timestamp":1785485301000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774556"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":50,"alternative-id":["10.1145\/3774521.3774556","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774556","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}