{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T11:15:57Z","timestamp":1783595757747,"version":"3.55.0"},"reference-count":32,"publisher":"China Science Publishing & Media Ltd.","issue":"2","content-domain":{"domain":["engine.scichina.com"],"crossmark-restriction":false},"short-container-title":["DI"],"published-print":{"date-parts":[[2026,6,1]]},"DOI":"10.3724\/2096-7004.di.2025.0140","type":"journal-article","created":{"date-parts":[[2025,9,22]],"date-time":"2025-09-22T02:29:13Z","timestamp":1758508153000},"page":"20250140","update-policy":"https:\/\/doi.org\/10.1360\/scp-crossmark-policy-page","source":"Crossref","is-referenced-by-count":0,"title":["TibOCR-Bench: A Comprehensive Benchmark and Training Pipeline for Tibetan Multimodal OCR"],"prefix":"10.3724","volume":"8","author":[{"given":"LAMA","family":"Jie","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Manla","family":"Cairang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kuntharrgyal","family":"Khysru","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hua Guo Cai","family":"Rang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiao","family":"Jiahui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yue","family":"Yingkai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tan","family":"Qian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"2026","published-online":{"date-parts":[[2026,7,9]]},"reference":[{"key":"null","unstructured":"Rowinski Z. and Keutzer K., \u201cNamsel: An optical character recognition system for Tibetan text,\u201d Himalayan Linguistics, vol. 15, no. 1, 2016."},{"key":"null","unstructured":"Drup N., Zhao D., Ren P., and Wang G., \u201cStudy on printed Tibetan character recognition,\u201d in Proc. Int. Conf. Artif. Intell. Comput. Intell., Sanya, China, vol. 1, 2010, pp. 280\u2013285."},{"key":"null","unstructured":"Li Y., Li L., Zhang Z., Li S., and Zhang T., \u201cTSTN: Tibetan spatiotemporal online handwritten character recognition with data augmentation,\u201d Authorea Preprints, 2025."},{"key":"null","unstructured":"Sabbagh C., \u201cEnhanced HTR accuracy for Tibetan historical texts\u2014Optimising image pre-processing for improved transcription quality,\u201d Revue d\u2019Etudes Tibetaines, no. 74, pp. 82\u2013128, 2025."},{"key":"null","unstructured":"Wang X. and Wang W., \u201cAn unsupervised character recognition method for Tibetan historical document images based on deep learning,\u201d Appl. Sci., vol. 14, no. 5, p. 2142, 2024."},{"key":"null","unstructured":"Wu Y., Wang S., Yang H., Dong L., Qiao Y., and Lin L., \u201cAn early evaluation of GPT-4V(ision),\u201d arXiv preprint, arXiv:2310.16534, 2023."},{"key":"null","unstructured":"Gemini Team et al., \u201cGemini: A family of highly capable multimodal models,\u201d arXiv preprint, arXiv:2312.11805, 2023."},{"key":"null","unstructured":"Chen Z. et al., \u201cInternVL: Scaling up vision foundation models and aligning for generic visual-linguistic tasks,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR), Seattle, WA, USA, 2024, pp. 24185\u201324198."},{"key":"null","unstructured":"Achiam J. et al., \u201cGPT-4 technical report,\u201d arXiv preprint, arXiv:2303.08774, 2023."},{"key":"null","unstructured":"Chen Z., Wang W., Tian H., Ye S., Gao Z., and Kit E., \u201cHow far are we to GPT-4V? Closing the gap to commercial multimodal models with open-source suites,\u201d Sci. China Inf. Sci., vol. 67, no. 12, p. 220101, Dec. 2024."},{"key":"null","unstructured":"Dai W. et al., \u201cInstructBLIP: Towards general-purpose vision-language models with instruction tuning,\u201d in Adv. Neural Inf. Process. Syst., vol. 36, 2023, pp. 49250\u201349267."},{"key":"null","unstructured":"Bai J. et al., \u201cQwen technical report,\u201d arXiv preprint, arXiv:2309.16609, 2023."},{"key":"null","unstructured":"Wei H., Kong L., Chen J., and Sun F., \u201cVary: Scaling up the vision vocabulary for large vision-language model,\u201d in Proc. Eur. Conf. Comput. Vis. (ECCV), Cham: Springer Nature Switzerland, 2024, pp. 408\u2013424."},{"key":"null","unstructured":"Gongqu Z., Luo P., Jiayang D., and Zhang D., \u201cA Tibetan ancient Uchen text line dataset for OCR,\u201d in Proc. Int. Conf. Image Process., Comput. Vis. Mach. Learn., 2024, pp. 796\u2013799."},{"key":"null","unstructured":"Zhong X., Tang J., and Yepes A. J., \u201cPubLayNet: Largest dataset ever for document layout analysis,\u201d in Proc. Int. Conf. Doc. Anal. Recognit. (ICDAR), 2019, pp. 1015\u20131022."},{"key":"null","unstructured":"Singh S. P. and Markovitch S., Eds., Proc. 31st AAAI Conf. Artif. Intell. (AAAI-17), San Francisco, CA, USA, Feb. 4\u20139, 2017. AAAI Press, 2017."},{"key":"null","unstructured":"Graves A., Fern\u00e1ndez S., Gomez F., and Schmidhuber J., \u201cConnectionist temporal classification: Labelling unsegmented sequence data with recurrent neural networks,\u201d in Proc. 23rd Int. Conf. Mach. Learn. (ICML), 2006, pp. 369\u2013376."},{"key":"null","unstructured":"Gao L. et al., \u201cICDAR 2019 competition on table detection and recognition (CTDaR),\u201d in Proc. Int. Conf. Doc. Anal. Recognit. (ICDAR), 2019, pp. 1510\u20131515."},{"key":"null","unstructured":"Veit A., Matera T., Neumann L., Matas J., and Belongie S., \u201cCOCO-Text: Dataset and benchmark for text detection and recognition in natural images,\u201d arXiv preprint, arXiv:1601.07140, 2016."},{"key":"null","unstructured":"Liu H., Li C., Wu Q., and Lee Y. J., \u201cVisual instruction tuning,\u201d in Adv. Neural Inf. Process. Syst., 2023, pp. 34892\u201334916."},{"key":"null","unstructured":"Liu C., Wei H., Chen J., and Sun F., \u201cFocus anywhere for fine-grained multi-page document understanding,\u201d arXiv preprint, arXiv:2405.14295, 2024."},{"key":"null","unstructured":"Wang P. et al., \u201cQwen2-VL: Enhancing vision-language model\u2019s perception of the world at any resolution,\u201d arXiv preprint, arXiv:2409.12191, 2024."},{"key":"null","unstructured":"Gupta D., Lenka P., Ekbal A., and Bhattacharyya P., \u201cA unified framework for multilingual and code-mixed visual question answering,\u201d in Proc. 1st Conf. Asia-Pacific Chapter Assoc. Comput. Linguistics (AACL) 10th Int. Joint Conf. Nat. Lang. Process. (IJCNLP), 2020, pp. 900\u2013913."},{"key":"null","unstructured":"Achiam J. et al., \u201cGPT-4 technical report,\u201d arXiv preprint, arXiv:2303.08774, 2023."},{"key":"null","unstructured":"Anthropic, \u201cThe Claude 3 model family: Opus, sonnet, haiku,\u201d Claude-3 Model Card, 2024."},{"key":"null","unstructured":"Liu Y. et al., \u201cTextMonkey: An OCR-free large multimodal model for understanding document,\u201d arXiv preprint, arXiv:2403.04473, 2024."},{"key":"null","unstructured":"Wei H., Liu C., Chen J., and Sun F., \u201cGeneral OCR theory: Towards OCR-2.0 via a unified end-to-end model,\u201d 2024."},{"key":"null","unstructured":"Farid H., \u201cAn overview of perceptual hashing,\u201d J. Online Trust Saf., vol. 1, no. 1, Jul. 2021."},{"key":"null","unstructured":"Bakurov I., Buzzelli M., Schettini R., Zontone M., and de Sousa A. R. B., \u201cStructural similarity index (SSIM) revisited: A data-driven approach,\u201d Expert Syst. Appl., vol. 189, p. 116087, Mar. 2022."},{"key":"null","unstructured":"Yim M., Kim Y., Cho H. C., and Park S., \u201cSynthTIGER: Synthetic text image generator towards better text recognition models,\u201d in Proc. Int. Conf. Doc. Anal. Recognit. (ICDAR), Cham: Springer International Publishing, 2021, pp. 109\u2013124."},{"key":"null","unstructured":"Russell B. C., Torralba A., Murphy K. P., and Freeman W. T., \u201cLabelMe: A database and web-based tool for image annotation,\u201d Int. J. Comput. Vis., vol. 77, no. 1\u20133, pp. 157\u2013173, May 2008."},{"key":"null","unstructured":"Liu A. et al., \u201cDeepSeek-V3 technical report,\u201d arXiv preprint, arXiv:2412.19437, 2024."}],"container-title":["Data Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/757AF9D8942F45C1A7CF158839CEDD35","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0140","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/757AF9D8942F45C1A7CF158839CEDD35","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T10:15:47Z","timestamp":1783592147000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0140"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,1]]},"references-count":32,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2026,7,9]]},"published-print":{"date-parts":[[2026,6,1]]}},"URL":"https:\/\/doi.org\/10.3724\/2096-7004.di.2025.0140","relation":{},"ISSN":["2096-7004"],"issn-type":[{"value":"2096-7004","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6,1]]}}}