{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T08:10:27Z","timestamp":1783930227831,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":29,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234318","type":"print"},{"value":"9789819234325","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T00:00:00Z","timestamp":1783987200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T00:00:00Z","timestamp":1783987200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3432-5_41","type":"book-chapter","created":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T07:41:15Z","timestamp":1783928475000},"page":"506-516","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["OmniOCR: Generalist OCR for Ethnic Minority Languages"],"prefix":"10.1007","author":[{"given":"Bonan","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zeyu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bingbing","family":"Meng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Han","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hanshuo","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chengping","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daji","family":"Ergu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,14]]},"reference":[{"key":"41_CR1","doi-asserted-by":"publisher","unstructured":"Peng, L., et al.: Multilingual document recognition research and its application in China. In: Second International Conference on Document Image Analysis for Libraries (DIAL 2006), pp. 126\u2013132. IEEE, Lyon, France (2006). https:\/\/doi.org\/10.1109\/DIAL.2006.27","DOI":"10.1109\/DIAL.2006.27"},{"key":"41_CR2","doi-asserted-by":"crossref","unstructured":"Drup, N., Zhao, D.C., Ren, P., Sanglangjie, D., Liu, F., Bawangdui, B.: Study on printed Tibetan character recognition. In: 2010 International Conference on Artificial Intelligence and Computational Intelligence, vol. 1, pp. 280\u2013285. IEEE (2010)","DOI":"10.1109\/AICI.2010.66"},{"key":"41_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2021\/5520338","volume":"2021","author":"D Zhang","year":"2021","unstructured":"Zhang, D., et al.: OCR with the deep CNN model for ligature script-based languages like manchu. Sci. Program. 2021, 1\u20139 (2021). https:\/\/doi.org\/10.1155\/2021\/5520338","journal-title":"Sci. Program."},{"key":"41_CR4","doi-asserted-by":"publisher","unstructured":"Zheng, R., et al.: Segmentation-free multi-font printed manchu word recognition using deep convolutional features and data augmentation. In: 2018 11th International Congress on Image and Signal Processing, BioMedical Engineering and Informatics (CISP-BMEI), pp. 1\u20136. IEEE, Beijing, China (2018). https:\/\/doi.org\/10.1109\/CISP-BMEI.2018.8633208","DOI":"10.1109\/CISP-BMEI.2018.8633208"},{"key":"41_CR5","doi-asserted-by":"publisher","unstructured":"Zhang, H. et al.: Segmentation-free printed traditional mongolian OCR using sequence to sequence with attention model. In: 2017 14th IAPR International Conference on Document Analysis and Recognition (ICDAR), pp. 585\u2013590. IEEE, Kyoto (2017). https:\/\/doi.org\/10.1109\/ICDAR.2017.101","DOI":"10.1109\/ICDAR.2017.101"},{"key":"41_CR6","unstructured":"Chung, Y.H.M., Choi, D.: Finetuning vision-language models as OCR systems for low-resource languages: a case study of Manchu. arXiv preprint arXiv:2507.06761 (2025)"},{"issue":"11","key":"41_CR7","doi-asserted-by":"publisher","first-page":"13094","DOI":"10.1609\/aaai.v37i11.26538","volume":"37","author":"M Li","year":"2023","unstructured":"Li, M., et al.: TrOCR: transformer-based optical character recognition with pre-trained models. AAAI. 37(11), 13094\u201313102 (2023). https:\/\/doi.org\/10.1609\/aaai.v37i11.26538","journal-title":"AAAI."},{"key":"41_CR8","unstructured":"Greif, G., Griesshaber, N., Greif, R.: Multimodal LLMS for OCR, OCR post-correction, and named entity recognition in historical documents. arXiv preprint arXiv:2504.00414 (2025)"},{"key":"41_CR9","unstructured":"Bai, et al.: RolmOCR: a vision-language foundation model for robust multilingual OCR (2025). https:\/\/github.com\/QwenLM\/Qwen2.5-VL"},{"key":"41_CR10","doi-asserted-by":"publisher","unstructured":"Liu, Y., et al.: OCRBench: on the hidden mystery of OCR in large multimodal models. Sci. China Inf. Sci. 67, 12, 220102 (2024). https:\/\/doi.org\/10.1007\/s11432-024-4235-6","DOI":"10.1007\/s11432-024-4235-6"},{"key":"41_CR11","doi-asserted-by":"crossref","unstructured":"Yang, Z.B., et al.: Cc-ocr: a comprehensive and challenging OCR benchmark for evaluating large multimodal models in literacy. arXiv preprint arXiv:2412.02210 (2024)","DOI":"10.1109\/ICCV51701.2025.02019"},{"key":"41_CR12","unstructured":"He, H.B., et al.: Reasoning-OCR: can large multimodal models solve complex logical reasoning problems from OCR Cues? arXiv preprint arXiv:2505.12766 (2025)"},{"key":"41_CR13","unstructured":"Sohail, M.A., Masood, S., Iqbal, H.: Deciphering the underserved: benchmarking LLM OCR for low-resource scripts. arXiv preprint arXiv:2412.16119 (2024)"},{"key":"41_CR14","unstructured":"Chen, S., et al.: Ocean-OCR: Towards general OCR application via a vision-language model. arXiv preprint arXiv:2501.15558 (2025)"},{"key":"41_CR15","doi-asserted-by":"publisher","unstructured":"Sun, P. et al.: Yi characters recognition based on tesseract-OCR. In: 2019 IEEE 3rd Advanced Information Management, Communicates, Electronic and Automation Control Conference (IMCEC), pp. 102\u2013106. IEEE, Chongqing, China (2019). https:\/\/doi.org\/10.1109\/IMCEC46724.2019.8983913","DOI":"10.1109\/IMCEC46724.2019.8983913"},{"key":"41_CR16","unstructured":"Yuan, M.Q., Cairang, X.M., Tang, J.A., et al.: TibetanMNIST tibetan handwritten digit dataset. Heywhale (2018). https:\/\/www.heywhale.com\/mw\/dataset\/5bfe734a954d6e0010683839"},{"key":"41_CR17","unstructured":"Yang, X.Z.: Sui-AIResearch: AI-driven research platform for endangered Sui script protection and digital humanities. https:\/\/github.com\/eastmountyxz\/Sui-AIResearch"},{"key":"41_CR18","doi-asserted-by":"publisher","unstructured":"Liu, X., et al.: Ancient Yi script handwriting sample repository. Sci Data. 11, 1, 1183 (2024). https:\/\/doi.org\/10.1038\/s41597-024-03918-5","DOI":"10.1038\/s41597-024-03918-5"},{"key":"41_CR19","doi-asserted-by":"crossref","unstructured":"Luo, Y.L., Sun, Y.W., Bi, X.J.: Multiple attentional aggregation network for handwritten Dongba character recognition. Expert Syst. Appl. 213, 118865 (2023)","DOI":"10.1016\/j.eswa.2022.118865"},{"key":"41_CR20","unstructured":"Lu, H.D., Zhao, C.Y., Xue, J., Yao, L., Moore, K., Gong, D.: Adaptive rank, reduced forgetting: Knowledge retention in continual learning vision-language models with dynamic rank-selective lora. arXiv preprint arXiv:2412.01004 (2024)"},{"key":"41_CR21","unstructured":"Team, K., et al.: Kimi-vl technical report. arXiv preprint arXiv:2504.07491 (2025)"},{"key":"41_CR22","unstructured":"Moonshot AI: Moonshot V1 (2025). https:\/\/moonshot.cn\/product"},{"key":"41_CR23","unstructured":"Mistral AI: Pixtral Large (2025). https:\/\/mistral.ai\/news\/pixtral-large\/"},{"key":"41_CR24","unstructured":"ByteDance:Doubao-1.5-Vision-Pro (2025). https:\/\/www.volcengine.com\/product\/doubao"},{"key":"41_CR25","unstructured":"Zhipu AI: GLM-4v-Plus (2025). https:\/\/chatglm.cn"},{"key":"41_CR26","unstructured":"Wu, Z.Y., et al.: Deepseek-vl2: mixture-of-experts vision-language models for advanced multimodal understanding. arXiv preprint arXiv:2412.10302 (2024)"},{"key":"41_CR27","unstructured":"Qwen Team, Alibaba Cloud: Qwen-VL-Max (2025). https:\/\/tongyi.aliyun.com"},{"key":"41_CR28","unstructured":"Qwen Team, Alibaba Cloud: Qwen-VL-OCR (2025). https:\/\/tongyi.aliyun.com"},{"key":"41_CR29","unstructured":"Zhu, J.G., et al.: Internvl3: exploring advanced training and test-time recipes for open-source multimodal models. arXiv preprint arXiv:2504.10479 (2025)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3432-5_41","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T07:41:17Z","timestamp":1783928477000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3432-5_41"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,14]]},"ISBN":["9789819234318","9789819234325"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3432-5_41","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,14]]},"assertion":[{"value":"14 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}