{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T16:10:51Z","timestamp":1780071051909,"version":"3.54.0"},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T00:00:00Z","timestamp":1780012800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T00:00:00Z","timestamp":1780012800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["New Gener. Comput."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1007\/s00354-026-00325-9","type":"journal-article","created":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T15:46:54Z","timestamp":1780069614000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["IndicCharGrid: A Character Grid-Based Representation for Spatially-Aware Document Understanding in Indian Language Documents"],"prefix":"10.1007","volume":"44","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-6344-8581","authenticated-orcid":false,"given":"Akkshita","family":"Trivedi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sandeep","family":"Khanna","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Santanu","family":"Chaudhury","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gaurav","family":"Harit","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,29]]},"reference":[{"key":"325_CR1","volume-title":"The World\u2019s Writing Systems","year":"1996","unstructured":"Bright, W., Daniels, P.T. (eds.): The World\u2019s Writing Systems. Oxford University Press, New York (1996)"},{"issue":"12","key":"325_CR2","first-page":"3947","volume":"47","author":"N Roy","year":"2014","unstructured":"Roy, N., Bhattacharya, U., Parui, S.K.: A segmentation based approach to complex indic character recognition. Pattern Recogn. 47(12), 3947\u20133960 (2014)","journal-title":"Pattern Recogn."},{"issue":"9","key":"325_CR3","doi-asserted-by":"publisher","first-page":"1887","DOI":"10.1016\/j.patcog.2004.02.003","volume":"37","author":"U Pal","year":"2004","unstructured":"Pal, U., Chaudhuri, B.B.: Indian script character recognition: a survey. Pattern Recogn. 37(9), 1887\u20131899 (2004)","journal-title":"Pattern Recogn."},{"key":"325_CR4","doi-asserted-by":"crossref","unstructured":"Jawahar, C.V., Kumar, M.P., Babu, R.V.: A bilingual ocr for Hindi and Telugu documents. In: Proceedings of the Eighth International Conference on Document Analysis and Recognition (ICDAR), pp. 408\u2013412. IEEE (2005)","DOI":"10.1109\/ICDAR.2003.1227699"},{"key":"325_CR5","unstructured":"Bansal, A., Krishnan, P.: Challenges in deep ocr for indic scripts. In: Proceedings of the 13th IAPR International Workshop on Document Analysis Systems (DAS), pp. 25\u201332. IEEE (2018)"},{"key":"325_CR6","first-page":"123","volume":"23","author":"R Sarkar","year":"2020","unstructured":"Sarkar, R., et al.: Text recognition in low resource indic scripts using deep convolutional architectures. Int. J. Doc. Anal. Recogn. 23, 123\u2013140 (2020)","journal-title":"Int. J. Doc. Anal. Recogn."},{"key":"325_CR7","first-page":"116","volume":"118","author":"R Mandal","year":"2019","unstructured":"Mandal, R., et al.: Detection and recognition of text in Indian languages in complex document layouts. Pattern Recogn. Lett. 118, 116\u2013123 (2019)","journal-title":"Pattern Recogn. Lett."},{"issue":"3","key":"325_CR8","first-page":"945","volume":"24","author":"V Murdock","year":"2021","unstructured":"Murdock, V., et al.: Layout aware ocr for heterogeneous multilingual documents. Pattern Anal. Appl. 24(3), 945\u2013958 (2021)","journal-title":"Pattern Anal. Appl."},{"key":"325_CR9","volume":"102","author":"S Bakkali","year":"2020","unstructured":"Bakkali, S., et al.: A survey on document layout analysis. Pattern Recogn. 102, 107248 (2020)","journal-title":"Pattern Recogn."},{"key":"325_CR10","unstructured":"Sarkhel, R., Natarajan, M., Jawahar, C.V.: Document structure extraction in complex indian language documents. In: Proceedings of the Tenth Indian Conference on Computer Vision, Graphics and Image Processing (ICVGIP), pp. 1\u201310. ACM (2018)"},{"issue":"2","key":"325_CR11","first-page":"743","volume":"26","author":"R Kumar","year":"2023","unstructured":"Kumar, R., et al.: Visual text analysis in indic documents using multimodal feature integration. Pattern Anal. Appl. 26(2), 743\u2013759 (2023)","journal-title":"Pattern Anal. Appl."},{"key":"325_CR12","doi-asserted-by":"crossref","unstructured":"Katti, A.R., Reisswig, C., Guder, C., Brarda, S., Bickel, S., H\u00f6hne, J., Faddoul, J.B.: Chargrid: towards understanding 2d documents. arXiv preprint arXiv:1809.08799 (2018)","DOI":"10.18653\/v1\/D18-1476"},{"key":"325_CR13","doi-asserted-by":"crossref","unstructured":"Redmon, J.: You only look once: unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (2016)","DOI":"10.1109\/CVPR.2016.91"},{"key":"325_CR14","doi-asserted-by":"crossref","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C.-Y., Berg, A.C.: Ssd: single shot multibox detector. In: Computer Vision\u2014ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part I 14, pp. 21\u201337 (2016). Springer","DOI":"10.1007\/978-3-319-46448-0_2"},{"issue":"6","key":"325_CR15","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"325_CR16","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"325_CR17","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., Darrell, T.: Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3431\u20133440 (2015)","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"325_CR18","doi-asserted-by":"crossref","unstructured":"Baek, Y., Lee, B., Han, D., Yun, S., Lee, H.: Character region awareness for text detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9365\u20139374 (2019)","DOI":"10.1109\/CVPR.2019.00959"},{"key":"325_CR19","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X., Hou, W., Lu, T., Yu, G., Shao, S.: Shape robust text detection with progressive scale expansion network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9336\u20139345 (2019)","DOI":"10.1109\/CVPR.2019.00956"},{"key":"325_CR20","doi-asserted-by":"crossref","unstructured":"Liao, M., Wan, Z., Yao, C., Chen, K., Bai, X.: Real-time scene text detection with differentiable binarization. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11474\u201311481 (2020)","DOI":"10.1609\/aaai.v34i07.6812"},{"key":"325_CR21","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"325_CR22","doi-asserted-by":"crossref","unstructured":"Chaurasia, A., Culurciello, E.: Linknet: exploiting encoder representations for efficient semantic segmentation. In: 2017 IEEE Visual Communications and Image Processing (VCIP), pp. 1\u20134 (2017)","DOI":"10.1109\/VCIP.2017.8305148"},{"key":"325_CR23","doi-asserted-by":"crossref","unstructured":"Kudale, D., Kasuba, B.V., Subramanian, V., Chaudhuri, P., Ramakrishnan, G.: Textron: weakly supervised multilingual text detection through data programming. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2871\u20132880 (2024)","DOI":"10.1109\/WACV57701.2024.00285"},{"key":"325_CR24","doi-asserted-by":"crossref","unstructured":"Kanezaki, A.: Unsupervised image segmentation by backpropagation. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1543\u20131547. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8462533"},{"key":"325_CR25","doi-asserted-by":"crossref","unstructured":"Xu, Y., Li, M., Cui, L., Huang, S., Wei, F., Zhou, M.: Layoutlm: pre-training of text and layout for document image understanding. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, pp. 1192\u20131200 (2020)","DOI":"10.1145\/3394486.3403172"},{"key":"325_CR26","doi-asserted-by":"crossref","unstructured":"Garncarek, \u0141., Powalski, R., Stanis\u0142awek, T., Topolski, B., Halama, P., Turski, M., Grali\u0144ski, F.: Lambert: layout-aware language modeling for information extraction. In: Document Analysis and Recognition\u2013ICDAR 2021: 16th International Conference, Lausanne, Switzerland, September 5\u201310, 2021, Proceedings, Part I, pp. 532\u2013547 (2021). Springer","DOI":"10.1007\/978-3-030-86549-8_34"},{"key":"325_CR27","doi-asserted-by":"crossref","unstructured":"Xu, Y., Xu, Y., Lv, T., Cui, L., Wei, F., Wang, G., Lu, Y., Florencio, D., Zhang, C., Che, W., Zhang, M., Zhou, L.: Layoutlmv2: multi-modal pre-training for visually-rich document understanding. In: ACL-IJCNLP 2021 (2021)","DOI":"10.18653\/v1\/2021.acl-long.201"},{"key":"325_CR28","doi-asserted-by":"crossref","unstructured":"Huang, Y., Lv, T., Cui, L., Lu, Y., Wei, F.: Layoutlmv3: pre-training for document ai with unified text and image masking. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4083\u20134091 (2022)","DOI":"10.1145\/3503161.3548112"},{"key":"325_CR29","unstructured":"Zhao, X., Niu, E., Wu, Z., Wang, X.: Cutie: learning to understand documents with convolutional universal text information extractor. arXiv preprint arXiv:1903.12363 (2019)"},{"key":"325_CR30","unstructured":"Denk, T.I., Reisswig, C.: BERTgrid: contextualized embedding for 2d document representation and understanding. In: Workshop on Document Intelligence at NeurIPS 2019 (2019)"},{"key":"325_CR31","unstructured":"Devlin, J., Chang, M., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. arXiv:1810.04805 (2018)"},{"key":"325_CR32","doi-asserted-by":"crossref","unstructured":"Kim, G., Hong, T., Yim, M., Nam, J., Park, J., Yim, J., Hwang, W., Yun, S., Han, D., Park, S.: Ocr-free document understanding transformer. In: European Conference on Computer Vision, pp. 498\u2013517 (2022). Springer","DOI":"10.1007\/978-3-031-19815-1_29"},{"key":"325_CR33","doi-asserted-by":"crossref","unstructured":"Bhattacharyya, S., Ghosh, S., Deb, P., Mondal, A., Jawahar, C.: Adapting vision-language models for Hindi ocr. In: International Conference on Document Analysis and Recognition, pp. 577\u2013594. Springer (2025)","DOI":"10.1007\/978-3-032-04624-6_34"},{"key":"325_CR34","unstructured":"Reisswig, C., Katti, A.R., Spinaci, M., H\u00f6hne, J.: Chargrid-ocr: end-to-end trainable optical character recognition through semantic segmentation and object detection. In: Workshop on Document Intelligence at NeurIPS 2019 (2019)"},{"key":"325_CR35","unstructured":"Velayuthan, M., Sarveswaran, K.: Egalitarian language representation in language models: it all begins with tokenizers. In: Proceedings of the 31st International Conference on Computational Linguistics, pp. 5987\u20135996 (2025)"},{"key":"325_CR36","doi-asserted-by":"crossref","unstructured":"Kolavi, A., Samarth, P., Jain, V.: Nayana ocr: a scalable framework for document ocr in low-resource languages. In: Proceedings of the 1st Workshop on Language Models for Underserved Communities (LM4UC 2025), pp. 86\u2013103 (2025)","DOI":"10.18653\/v1\/2025.lm4uc-1.11"},{"key":"325_CR37","unstructured":"Du, Y., Li, C., Guo, R., Yin, X., Liu, W., Zhou, J., Bai, Y., Yu, Z., Yang, Y., Dang, Q., Wang, H.: PP-OCR: a practical ultra lightweight OCR system. CoRR. arXiv:2009.09941 (2020)"},{"key":"325_CR38","unstructured":"Li, M., Lv, T., Cui, L., Lu, Y., Flor\u00eancio, D.A.F., Zhang, C., Li, Z., Wei, F.: Trocr: transformer-based optical character recognition with pre-trained models. CoRR. arXiv:2109.10282 (2021)"},{"key":"325_CR39","doi-asserted-by":"crossref","unstructured":"Smith, R.: An overview of the tesseract ocr engine. In: Ninth International Conference on Document Analysis and Recognition (ICDAR 2007), vol. 2, pp. 629\u2013633. IEEE (2007)","DOI":"10.1109\/ICDAR.2007.4376991"}],"container-title":["New Generation Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00354-026-00325-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00354-026-00325-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00354-026-00325-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T15:47:01Z","timestamp":1780069621000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00354-026-00325-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,29]]},"references-count":39,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,8]]}},"alternative-id":["325"],"URL":"https:\/\/doi.org\/10.1007\/s00354-026-00325-9","relation":{},"ISSN":["0288-3635","1882-7055"],"issn-type":[{"value":"0288-3635","type":"print"},{"value":"1882-7055","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,29]]},"assertion":[{"value":"4 February 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 May 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"20"}}