{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T19:07:27Z","timestamp":1782932847073,"version":"3.54.5"},"reference-count":28,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,7,29]],"date-time":"2025-07-29T00:00:00Z","timestamp":1753747200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,29]],"date-time":"2025-07-29T00:00:00Z","timestamp":1753747200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["IJDAR"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s10032-025-00548-4","type":"journal-article","created":{"date-parts":[[2025,7,29]],"date-time":"2025-07-29T14:03:30Z","timestamp":1753797810000},"page":"289-305","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["How well do MLLMs understand handwritten legal documents? A novel dataset for benchmarking"],"prefix":"10.1007","volume":"29","author":[{"given":"Sagar","family":"Chakraborty","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gaurav","family":"Harit","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Saptarshi","family":"Ghosh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,29]]},"reference":[{"key":"548_CR1","unstructured":"Bai, J., Bai, S., Yang, S., et\u00a0al.: Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond. arXiv:2308.12966 (2023)"},{"key":"548_CR2","doi-asserted-by":"crossref","unstructured":"Chakraborty, S., Harit, G., Ghosh, S.: Transdocanalyser: A framework for semi-structured offline handwritten documents analysis with an application to legal domain. In: Fink, G.A., Jain, R., Kise, K., et al. (eds.) Document Analysis and Recognition - ICDAR 2023, pp. 45\u201362. Springer Nature Switzerland, Cham (2023)","DOI":"10.1007\/978-3-031-41676-7_3"},{"key":"548_CR3","doi-asserted-by":"crossref","unstructured":"Dai, W., Li, J., Li, D., et\u00a0al.: InstructBLIP: Towards general-purpose vision-language models with instruction tuning. In: Thirty-seventh Conference on Neural Information Processing Systems, https:\/\/openreview.net\/forum?id=vvoWPYqZJA (2023)","DOI":"10.52202\/075280-2142"},{"key":"548_CR4","doi-asserted-by":"crossref","unstructured":"Deng, J., Heybati, K., Shammas-Toma, M.: When vision meets reality: Exploring the clinical applicability of gpt-4 with vision (2024)","DOI":"10.1016\/j.clinimag.2024.110101"},{"key":"548_CR5","doi-asserted-by":"publisher","unstructured":"Diem, M., Fiel, S., Kleber, F., et\u00a0al.: Icfhr 2014 competition on handwritten digit string recognition in challenging datasets (hdsrc 2014). In: 2014 14th International Conference on Frontiers in Handwriting Recognition, pp 779\u2013784, (2014). https:\/\/doi.org\/10.1109\/ICFHR.2014.136","DOI":"10.1109\/ICFHR.2014.136"},{"key":"548_CR6","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast r-cnn. In: Proceedings of IEEE International Conference on Computer Vision (ICCV), pp 1440\u20131448 (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"548_CR7","unstructured":"Han, X., You, Q., Liu, Y., et\u00a0al.: Infimm-eval: Complex open-ended reasoning evaluation for multi-modal large language models (2023). arXiv:2311.11567"},{"key":"548_CR8","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., et\u00a0al.: Mask r-cnn. In: Proceedings of IEEE International Conference on Computer Vision (ICCV), pp 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"548_CR9","unstructured":"Hsu, B., Dai, Y., Kothapalli, V., et\u00a0al.: Liger kernel: Efficient triton kernels for llm training. (2024) https:\/\/api.semanticscholar.org\/CorpusID:273351123"},{"key":"548_CR10","unstructured":"Hu, EJ., yelong shen, Wallis, P., et\u00a0al.: LoRA: Low-rank adaptation of large language models. In: International Conference on Learning Representations, (2022) https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"548_CR11","doi-asserted-by":"crossref","unstructured":"Huang, Z., Chen, K., He, J., et\u00a0al.: Competition on scanned receipt ocr and information extraction. In: Proceedings of International Conference on Document Analysis and Recognition (ICDAR), pp 1516\u20131520 (2019)","DOI":"10.1109\/ICDAR.2019.00244"},{"key":"548_CR12","doi-asserted-by":"crossref","unstructured":"Jaume, G., Kemal\u00a0Ekenel, H., Thiran, JP.: Funsd: A dataset for form understanding in noisy scanned documents. In: 2019 International Conference on Document Analysis and Recognition Workshops (ICDARW), pp 1\u20136 (2019)","DOI":"10.1109\/ICDARW.2019.10029"},{"key":"548_CR13","unstructured":"Li, J., Li, D., Savarese, S., et\u00a0al.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of the 40th International Conference on Machine Learning. JMLR.org, ICML\u201923 (2023a)"},{"key":"548_CR14","doi-asserted-by":"crossref","unstructured":"Li, M., Lv, T., Chen, J., et\u00a0al.: TrOCR: Transformer-based Optical Character Recognition with Pre-trained Models. In: Proceedings of AAAI (2023b)","DOI":"10.1609\/aaai.v37i11.26538"},{"key":"548_CR15","doi-asserted-by":"crossref","unstructured":"Li, Z., Yang, B., Liu, Q., et\u00a0al.: Monkey: Image resolution and text label are important things for large multi-modal models (2024). arXiv:2311.06607","DOI":"10.1109\/CVPR52733.2024.02527"},{"key":"548_CR16","unstructured":"Liu, H., Li, C., Li, Y., et\u00a0al.: Llava-next: Improved reasoning, ocr, and world knowledge (2024a). https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/"},{"key":"548_CR17","doi-asserted-by":"crossref","unstructured":"Liu, Y., Duan, H., Zhang, Y., et\u00a0al.: MMBench: Is your multi-modal model an all-around player? (2024b) https:\/\/openreview.net\/forum?id=BfMQIJ0nLc","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"548_CR18","doi-asserted-by":"crossref","unstructured":"Liu, Y., Li, Z., Yang, B., et\u00a0al.: On the hidden mystery of ocr in large multimodal models. (2024c). arXiv:2305.07895","DOI":"10.1007\/s11432-024-4235-6"},{"key":"548_CR19","unstructured":"Lu, P., Bansal, H., Xia, T., et\u00a0al.: Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts. In: The Twelfth International Conference on Learning Representations, (2024). https:\/\/openreview.net\/forum?id=KUNzEQMWU7"},{"key":"548_CR20","unstructured":"Lyu, C., Zhang, W., Huang, H., et\u00a0al.: Rtmdet: An empirical study of designing real-time object detectors. arXiv:2212.07784. (2022) https:\/\/api.semanticscholar.org\/CorpusID:254685870"},{"key":"548_CR21","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1007\/s100320200071","volume":"5","author":"UV Marti","year":"2002","unstructured":"Marti, U.V., Bunke, H.: The iam-database: An english sentence database for offline handwriting recognition. Int. J. Doc. Anal. Recogn. 5, 39\u201346 (2002)","journal-title":"Int. J. Doc. Anal. Recogn."},{"key":"548_CR22","doi-asserted-by":"crossref","unstructured":"Palm, RB., Winther, O., Laws, F.: Cloudscan - a configuration-free invoice analysis system using recurrent neural networks. In: Proceedings of IAPR International Conference on Document Analysis and Recognition (ICDAR), pp 406\u2013413 (2017)","DOI":"10.1109\/ICDAR.2017.74"},{"key":"548_CR23","unstructured":"Ren, S., He, K., Girshick, R., et\u00a0al.: Faster r-cnn: Towards real-time object detection with region proposal networks. In: Proceedings of the International Conference on Neural Information Processing Systems - Volume 1. MIT Press, p 91\u201399 (2015)"},{"key":"548_CR24","doi-asserted-by":"crossref","unstructured":"\u0160imsa, \u0160., \u0160ulc, M., U\u0159i\u010d\u00e1\u0159, M., et\u00a0al.: DocILE Benchmark for Document Information Localization and Extraction. In: Proc. International Conference on Document Analysis and Recognition (ICDAR), pp 147\u2013166 (2023)","DOI":"10.1007\/978-3-031-41679-8_9"},{"key":"548_CR25","unstructured":"Team, G., Anil, R., et\u00a0al.: SB Gemini: A family of highly capable multimodal models. arXiv:2312.11805 (2024)"},{"key":"548_CR26","unstructured":"University, DI. Doctors prescriptions handwriting dataset. https:\/\/universe.roboflow.com\/daffodil-international-university-s5vpr\/doctors-prescriptions-handwriting, visited on 2024-05-24 (2023)"},{"key":"548_CR27","doi-asserted-by":"publisher","unstructured":"Wu, J., Gan, W., Chen, Z., et\u00a0al.: Multimodal large language models: A survey. In: 2023 IEEE International Conference on Big Data (BigData), pp 2247\u20132256, (2023). https:\/\/doi.org\/10.1109\/BigData59044.2023.10386743","DOI":"10.1109\/BigData59044.2023.10386743"},{"key":"548_CR28","unstructured":"Yu, W., Yang, Z., Li, L., et\u00a0al.: Mm-vet: Evaluating large multimodal models for integrated capabilities (2023). arXiv:2308.02490"}],"container-title":["International Journal on Document Analysis and Recognition (IJDAR)"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-025-00548-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10032-025-00548-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-025-00548-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T07:08:09Z","timestamp":1781939289000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10032-025-00548-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,29]]},"references-count":28,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["548"],"URL":"https:\/\/doi.org\/10.1007\/s10032-025-00548-4","relation":{},"ISSN":["1433-2833","1433-2825"],"issn-type":[{"value":"1433-2833","type":"print"},{"value":"1433-2825","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,29]]},"assertion":[{"value":"16 November 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 June 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 July 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 July 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interests"}}]}}