{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T11:02:37Z","timestamp":1784372557993,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":25,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819235193","type":"print"},{"value":"9789819235209","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3520-9_46","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T10:03:52Z","timestamp":1784369032000},"page":"581-591","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MGL-LMM: Generative OCR with Semantic Inpainting for Degraded Mongolian Lead-Type Printed Documents"],"prefix":"10.1007","author":[{"given":"Zicheng","family":"Luo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Min","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Na","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bao","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingdaoerji","family":"Ren","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nier","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"issue":"11","key":"46_CR1","doi-asserted-by":"publisher","first-page":"2298","DOI":"10.1109\/TPAMI.2016.2646371","volume":"39","author":"B Shi","year":"2017","unstructured":"Shi, B., Bai, X., Yao, C.: An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE Trans. Pattern Anal. Mach. Intell. 39(11), 2298\u20132304 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"46_CR2","unstructured":"Li, M., et al.: TrOCR: transformer-based optical character recognition with pre-trained models. In: Proceedings of the AAAI, pp. 13058\u201313066 (2023)"},{"key":"46_CR3","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. In: ICLR (2021)"},{"key":"46_CR4","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Louradour, J., Collobert, R., Weston, J.: Curriculum learning. In: Proceedings of the ICML, pp. 41\u201348 (2009)","DOI":"10.1145\/1553374.1553380"},{"key":"46_CR5","unstructured":"Yang, A., et al.: Qwen2 technical report. arXiv preprint arXiv:2407.10671 (2024)"},{"key":"46_CR6","doi-asserted-by":"publisher","unstructured":"Kim, G., et al.: OCR-free document understanding transformer. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022, ECCV 2022. LNCS, vol. 13688, pp. 498\u2013517. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19815-1_29","DOI":"10.1007\/978-3-031-19815-1_29"},{"key":"46_CR7","unstructured":"Liu, Y., et al.: TextMonkey: an OCR-free large multimodal model for understanding document. arXiv preprint arXiv:2403.04473 (2024)"},{"key":"46_CR8","doi-asserted-by":"crossref","unstructured":"Borisyuk, F., Gordo, A., Sivakumar, V.: Rosetta: large scale system for text detection and recognition in images. In: Proceedings of the KDD, pp. 71\u201379 (2018)","DOI":"10.1145\/3219819.3219861"},{"issue":"1","key":"46_CR9","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1007\/s11263-020-01369-0","volume":"129","author":"S Long","year":"2021","unstructured":"Long, S., He, X., Yao, C.: Scene text detection and recognition: the deep learning era. Int. J. Comput. Vis. 129(1), 161\u2013184 (2021)","journal-title":"Int. J. Comput. Vis."},{"key":"46_CR10","doi-asserted-by":"crossref","unstructured":"Talebi, H., Milanfar, P.: Learning to resize images for computer vision tasks. In: Proceedings of the ICCV, pp. 11284\u201311294 (2021)","DOI":"10.1109\/ICCV48922.2021.00055"},{"key":"46_CR11","doi-asserted-by":"crossref","unstructured":"Fang, S., Xie, H., Wang, Y., et al.: Read like humans: autonomous, bidirectional and iterative language modeling for scene text recognition. In: Proceedings of the CVPR, pp. 7073\u20137082 (2021)","DOI":"10.1109\/CVPR46437.2021.00702"},{"key":"46_CR12","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Wu, Q., et al.: Visual instruction tuning. In: NeurIPS, vol. 36, pp. 34892\u201334916 (2023)","DOI":"10.52202\/075280-1516"},{"key":"46_CR13","doi-asserted-by":"crossref","unstructured":"Wei, H.R., Kong, L.Y., Chen, J.Y., et al.: Vary: scaling up the vision vocabulary for large vision-language model. arXiv preprint arXiv:2312.06109 (2023)","DOI":"10.1007\/978-3-031-73235-5_23"},{"key":"46_CR14","unstructured":"Li, J., Li, D., Savarese, S., et al.: BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of the ICML, pp. 19730\u201319742 (2023)"},{"key":"46_CR15","unstructured":"Blecher, L., Cucurull, G., Scialom, T., et al.: Nougat: neural optical understanding for academic documents. arXiv preprint arXiv:2308.13418 (2023)"},{"key":"46_CR16","unstructured":"Touvron, H., Cord, M., Douze, M., et al.: Training data-efficient image transformers & distillation through attention. In: Proceedings of the ICML, pp. 10347\u201310357 (2021)"},{"key":"46_CR17","doi-asserted-by":"publisher","first-page":"8490","DOI":"10.1109\/ACCESS.2023.3238315","volume":"11","author":"F Alrasheedi","year":"2023","unstructured":"Alrasheedi, F., Zhong, X., Huang, P.C.: Padding module: learning the padding in deep neural networks. IEEE Access 11, 8490\u20138497 (2023)","journal-title":"IEEE Access"},{"key":"46_CR18","doi-asserted-by":"crossref","unstructured":"Alayrac, J.B., Donahue, J., Luc, P., et al.: Flamingo: a visual language model for few-shot learning. In: NeurIPS, vol. 35, pp. 23716\u201323736 (2022)","DOI":"10.52202\/068431-1723"},{"key":"46_CR19","unstructured":"Liu, Z., He, Y., Wang, W., et al.: InternGPT: solving vision-centric tasks by interacting with ChatGPT beyond language. arXiv preprint arXiv:2305.05662 (2023)"},{"issue":"9","key":"46_CR20","doi-asserted-by":"crossref","first-page":"4555","DOI":"10.1109\/TPAMI.2021.3072422","volume":"44","author":"X Wang","year":"2022","unstructured":"Wang, X., Chen, Y., Zhu, W.: A survey on curriculum learning. IEEE Trans. Pattern Anal. Mach. Intell. 44(9), 4555\u20134576 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"46_CR21","unstructured":"Lee, K., Joshi, M., Turc, I., et al.: Pix2struct: screenshot parsing as pretraining for visual language understanding. In: Proceedings of the ICML, pp. 18843\u201318859 (2023)"},{"key":"46_CR22","doi-asserted-by":"crossref","unstructured":"Zeng, H., Cao, J., Zhang, K., et al.: Unmixing diffusion for self-supervised hyper-spectral image denoising. In: Proceedings of the CVPR, pp. 25290\u201325301 (2024)","DOI":"10.1109\/CVPR52733.2024.02628"},{"key":"46_CR23","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: ICLR (2019)"},{"key":"46_CR24","unstructured":"Du, Y., Li, C., Guo, R., et al.: PP-OCR: a practical ultra lightweight OCR system. arXiv preprint arXiv:2009.09941 (2020)"},{"key":"46_CR25","doi-asserted-by":"publisher","unstructured":"Bautista, D., Atienza, R.: Scene text recognition with permuted autoregressive sequence models. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022, ECCV 2022. LNCS, vol. 13688, pp. 178\u2013196. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19815-1_11","DOI":"10.1007\/978-3-031-19815-1_11"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3520-9_46","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T10:03:54Z","timestamp":1784369034000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3520-9_46"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819235193","9789819235209"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3520-9_46","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}