{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T23:55:11Z","timestamp":1767311711920,"version":"3.48.0"},"publisher-location":"Cham","reference-count":47,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032093707","type":"print"},{"value":"9783032093714","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-09371-4_21","type":"book-chapter","created":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T23:52:24Z","timestamp":1767311544000},"page":"347-365","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Cross-Lingual Learning for\u00a0Low-Resource Khmer Scene Text Detection and\u00a0Recognition"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-4684-0416","authenticated-orcid":false,"given":"Vannkin","family":"Nom","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9878-4015","authenticated-orcid":false,"given":"Saly","family":"Keo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9383-3842","authenticated-orcid":false,"given":"Souhail","family":"Bakkali","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9658-0833","authenticated-orcid":false,"given":"Muhammad Muzzamil","family":"Luqman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0123-439X","authenticated-orcid":false,"given":"Micka\u00ebl","family":"Coustaty","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5666-475X","authenticated-orcid":false,"given":"Jean-Marc","family":"Ogier","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,2]]},"reference":[{"key":"21_CR1","doi-asserted-by":"crossref","unstructured":"Baek, J., et al.: What is wrong with scene text recognition model comparisons? dataset and model analysis. In: Proceedings of the IEEE\/CVF International Conference On Computer Vision, pp. 4715\u20134723 (2019)","DOI":"10.1109\/ICCV.2019.00481"},{"key":"21_CR2","doi-asserted-by":"crossref","unstructured":"Baek, Y., Lee, B., Han, D., Yun, S., Lee, H.: Character region awareness for text detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9365\u20139374 (2019)","DOI":"10.1109\/CVPR.2019.00959"},{"key":"21_CR3","unstructured":"Bhunia, A., Roy, S., Bhowmick, A., Pal, U., Roy, K., Song, Y.-Z., Bhattacharya, U.: IndOCR: a new benchmark for Indian Script OCR. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022)"},{"key":"21_CR4","doi-asserted-by":"crossref","unstructured":"Born, S., Valy, D., Kong, P.: Encoder-decoder language model for Khmer handwritten text recognition in historical documents. In: 2022 14th International Conference on Software, Knowledge, Information Management and Applications (SKIMA), pp. 234\u2013238. IEEE (2022)","DOI":"10.1109\/SKIMA57145.2022.10029532"},{"key":"21_CR5","doi-asserted-by":"publisher","first-page":"128044","DOI":"10.1109\/ACCESS.2023.3332361","volume":"11","author":"R Buoy","year":"2023","unstructured":"Buoy, R., Iwamura, M., Srun, S., Kise, K.: Toward a low-resource non-latin-complete baseline: an exploration of khmer optical character recognition. IEEE Access 11, 128044\u2013128060 (2023)","journal-title":"IEEE Access"},{"key":"21_CR6","doi-asserted-by":"crossref","unstructured":"Bu\u0161ta, M., Patel, Y., Matas, J.: E2e-MLT-an unconstrained end-to-end method for multi-language scene text. In: Computer Vision\u2013ACCV 2018 Workshops: 14th Asian Conference on Computer Vision, Perth, Australia, December 2\u20136, 2018, Revised Selected Papers 14, pp. 127\u2013143. Springer International Publishing (2019)","DOI":"10.1007\/978-3-030-21074-8_11"},{"key":"21_CR7","doi-asserted-by":"crossref","unstructured":"Ch\u2019ng, C.K., Chan, C.S.: Total-text: a comprehensive dataset for scene text detection and recognition. In: 2017 14th IAPR International Conference On Document Analysis And Recognition (ICDAR), vol. 1, pp. 935\u2013942. IEEE (2017)","DOI":"10.1109\/ICDAR.2017.157"},{"issue":"5","key":"21_CR8","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3406095","volume":"53","author":"R Dabre","year":"2020","unstructured":"Dabre, R., Chu, C., Kunchukuttan, A.: A survey of multilingual neural machine translation. ACM Comput. Surv. (CSUR) 53(5), 1\u201338 (2020)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"21_CR9","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference Of The North American Chapter Of The Association For Computational Linguistics: Human Language Technologies, volume 1 (long and short papers), pp. 4171\u20134186 (2019)","DOI":"10.18653\/v1\/N19-1423"},{"issue":"6","key":"21_CR10","doi-asserted-by":"publisher","first-page":"3290","DOI":"10.3390\/s23063290","volume":"23","author":"K Eltouny","year":"2023","unstructured":"Eltouny, K., Gomaa, M., Liang, X.: Unsupervised learning methods for data-driven vibration-based structural health monitoring: a review. Sensors 23(6), 3290 (2023)","journal-title":"Sensors"},{"key":"21_CR11","doi-asserted-by":"crossref","unstructured":"Gilani, A., Qasim, S.R., Malik, I., Shafait, F.: Table detection using deep learning. In: 2017 14th IAPR International Conference on Document Analysis and Recognition (ICDAR), vol. 1, pp. 771\u2013776. IEEE (2017)","DOI":"10.1109\/ICDAR.2017.131"},{"key":"21_CR12","doi-asserted-by":"crossref","unstructured":"Ghiffari, F.A.A., Alfina, I., Azizah, K.: Cross-lingual transfer learning for Javanese dependency parsing. arXiv preprint arXiv:2401.12072 (2024)","DOI":"10.18653\/v1\/2023.ijcnlp-srw.1"},{"key":"21_CR13","unstructured":"Horton, J., Sok, M., Durdin, M., Ty, R.: Spoof-vulnerable rendering in Khmer Unicode implementations. In: Proceedings of the Sixth Asian Conference on Information Systems, pp. 177\u2013180 (2017)"},{"key":"21_CR14","doi-asserted-by":"crossref","unstructured":"Hedderich, M.A., Lange, L., Adel, H., Str\u00f6tgen, J., Klakow, D.: A survey on recent approaches for natural language processing in low-resource scenarios. arXiv preprint arXiv:2010.12309 (2020)","DOI":"10.18653\/v1\/2021.naacl-main.201"},{"key":"21_CR15","doi-asserted-by":"crossref","unstructured":"Jatowt, A., Coustaty, M., Nguyen, N.V., Doucet, A.: Deep statistical analysis of OCR errors for effective post-OCR processing. In: 2019 ACM\/IEEE Joint Conference on Digital Libraries (JCDL), pp. 29\u201338. IEEE (2019)","DOI":"10.1109\/JCDL.2019.00015"},{"issue":"6","key":"21_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3464378","volume":"20","author":"H Kaing","year":"2021","unstructured":"Kaing, H., et al.: Towards tokenization and part-of-speech tagging for khmer: data and discussion. Trans. Asian Low-Resource Language Inf. Process. 20(6), 1\u201316 (2021)","journal-title":"Trans. Asian Low-Resource Language Inf. Process."},{"key":"21_CR17","doi-asserted-by":"crossref","unstructured":"Krishnan, P., Dutta, K., Jawahar, C.V.: Word spotting and recognition using deep embedding. In: 2018 13th IAPR International Workshop on Document Analysis Systems (DAS), pp. 1\u20136. IEEE (2018)","DOI":"10.1109\/DAS.2018.70"},{"key":"21_CR18","doi-asserted-by":"crossref","unstructured":"Liao, M., et al.: Scene text recognition from two-dimensional perspective. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, no. 01, pp. 8714\u20138721 (2019)","DOI":"10.1609\/aaai.v33i01.33018714"},{"key":"21_CR19","doi-asserted-by":"crossref","unstructured":"Liao, M., Wan, Z., Yao, C., Chen, K., Bai, X.: Real-time scene text detection with differentiable binarization. In: Proceedings of the AAAI Conference On Artificial Intelligence, vol. 34, no. 07, pp. 11474\u201311481 (2020)","DOI":"10.1609\/aaai.v34i07.6812"},{"key":"21_CR20","doi-asserted-by":"crossref","unstructured":"Li, M., et al.: TROCR: transformer-based optical character recognition with pre-trained models. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, no. 11, pp. 13094\u201313102 (2023)","DOI":"10.1609\/aaai.v37i11.26538"},{"issue":"1","key":"21_CR21","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1007\/s11263-020-01369-0","volume":"129","author":"S Long","year":"2021","unstructured":"Long, S., He, X., Yao, C.: Scene text detection and recognition: the deep learning era. Int. J. Comput. Vision 129(1), 161\u2013184 (2021)","journal-title":"Int. J. Comput. Vision"},{"key":"21_CR22","doi-asserted-by":"crossref","unstructured":"Lyu, P., Yao, C., Wu, W., Yan, S., Bai, X.: Multi-oriented scene text detection via corner localization and region segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7553\u20137563 (2018)","DOI":"10.1109\/CVPR.2018.00788"},{"key":"21_CR23","doi-asserted-by":"crossref","unstructured":"Mathew, M., Karatzas, D., Jawahar, C.V.: Docvqa: a dataset for VQA on document images. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2200\u20132209 (2021)","DOI":"10.1109\/WACV48630.2021.00225"},{"key":"21_CR24","doi-asserted-by":"crossref","unstructured":"Muller, B., Anastasopoulos, A., Sagot, B., Seddah, D.: When being unseen from mBERT is just the beginning: handling new languages with multilingual language models. arXiv preprint arXiv:2010.12858 (2020)","DOI":"10.18653\/v1\/2021.naacl-main.38"},{"key":"21_CR25","doi-asserted-by":"crossref","unstructured":"Nayef, N., et al.: Icdar2017 robust reading challenge on multi-lingual scene text detection and script identification-RRC-MLT. In 2017 14th IAPR International Conference on Document Analysis and Recognition (ICDAR), vol. 1, pp. 1454\u20131459. IEEE (2017)","DOI":"10.1109\/ICDAR.2017.237"},{"key":"21_CR26","doi-asserted-by":"crossref","unstructured":"Nayef, N., et al.: ICDAR2019 robust reading challenge on multi-lingual scene text detection and recognition\u2014RRC-MLT-2019. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 1582\u20131587. IEEE (2019)","DOI":"10.1109\/ICDAR.2019.00254"},{"key":"21_CR27","doi-asserted-by":"crossref","unstructured":"Nom, V., Bakkali, S., Luqman, M.M., Coustaty, M., Ogier, J.M.: KhmerST: A low-resource khmer scene text detection and recognition benchmark. In: Proceedings of the Asian Conference on Computer Vision, pp. 1777\u20131792 (2024)","DOI":"10.1007\/978-981-96-0917-8_22"},{"key":"21_CR28","doi-asserted-by":"crossref","unstructured":"Nom, V., et al.: WildKhmerST: a comprehensive dataset and benchmark for Khmer scene text detection and recognition in the wild. In: Proceedings of the International Conference on Document Analysis and Recognition (ICDAR) (2025)","DOI":"10.1007\/978-3-032-04630-7_20"},{"key":"21_CR29","doi-asserted-by":"crossref","unstructured":"Pires, T., Schlinger, E., Garrette, D.: How multilingual is multilingual BERT?. arXiv preprint arXiv:1906.01502 (2019)","DOI":"10.18653\/v1\/P19-1493"},{"key":"21_CR30","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1007\/s10032-006-0027-8","volume":"9","author":"TM Rath","year":"2007","unstructured":"Rath, T.M., Manmatha, R.: Word spotting for historical documents. IJDAR 9, 139\u2013152 (2007)","journal-title":"IJDAR"},{"key":"21_CR31","doi-asserted-by":"crossref","unstructured":"Ruder, S., Peters, M. E., Swayamdipta, S., Wolf, T.: Transfer learning in natural language processing. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Tutorials, pp. 15\u201318 (2019)","DOI":"10.18653\/v1\/N19-5004"},{"key":"21_CR32","doi-asserted-by":"publisher","first-page":"569","DOI":"10.1613\/jair.1.11640","volume":"65","author":"S Ruder","year":"2019","unstructured":"Ruder, S., Vuli\u0107, I., S\u00f8gaard, A.: A survey of cross-lingual word embedding models. J. Artif. Intell. Res. 65, 569\u2013631 (2019)","journal-title":"J. Artif. Intell. Res."},{"key":"21_CR33","doi-asserted-by":"crossref","unstructured":"Schreiber, S., Agne, S., Wolf, I., Dengel, A., Ahmed, S.: Deepdesrt: deep learning for detection and structure recognition of tables in document images. In: 2017 14th IAPR International Conference on Document Analysis and Recognition (ICDAR), vol. 1, pp. 1162\u20131167. IEEE (2017)","DOI":"10.1109\/ICDAR.2017.192"},{"key":"21_CR34","doi-asserted-by":"crossref","unstructured":"Schuster, S., Gupta, S., Shah, R., Lewis, M.: Cross-lingual transfer learning for multilingual task oriented dialog. arXiv preprint arXiv:1810.13327 (2018)","DOI":"10.18653\/v1\/N19-1380"},{"issue":"11","key":"21_CR35","doi-asserted-by":"publisher","first-page":"2298","DOI":"10.1109\/TPAMI.2016.2646371","volume":"39","author":"B Shi","year":"2016","unstructured":"Shi, B., Bai, X., Yao, C.: An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE Trans. Patt. Anal. Mach. Intell. 39(11), 2298\u20132304 (2016)","journal-title":"IEEE Trans. Patt. Anal. Mach. Intell."},{"key":"21_CR36","doi-asserted-by":"crossref","unstructured":"Singh, A., Pang, G., Toh, M., Huang, J., Galuba, W., Hassner, T.: Textocr: towards large-scale end-to-end reasoning for arbitrary-shaped scene text. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8802\u20138812 (2021)","DOI":"10.1109\/CVPR46437.2021.00869"},{"key":"21_CR37","doi-asserted-by":"crossref","unstructured":"Sok, P., Taing, N.: Support vector machine (SVM) based classifier for khmer printed character-set recognition. In: Signal and Information Processing Association Annual Summit and Conference (APSIPA), 2014 Asia-Pacific, pp. 1\u20139. IEEE (2014)","DOI":"10.1109\/APSIPA.2014.7041823"},{"key":"21_CR38","doi-asserted-by":"crossref","unstructured":"Valy, D., Verleysen, M., Chhun, S.: Data augmentation and text recognition on Khmer historical manuscripts. In: 2020 17th International Conference on Frontiers in Handwriting Recognition (ICFHR), pp. 73\u201378. IEEE (2020)","DOI":"10.1109\/ICFHR2020.2020.00024"},{"key":"21_CR39","unstructured":"Wang, K., Babenko, B., Belongie, S.: End-to-end scene text recognition. In: 2011 International Conference on Computer Vision, pp. 1457\u20131464. IEEE (2011)"},{"key":"21_CR40","doi-asserted-by":"crossref","unstructured":"Wang, T., et al.: Decoupled attention network for text recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, no. 07, pp. 12216\u201312224 (2020)","DOI":"10.1609\/aaai.v34i07.6903"},{"key":"21_CR41","unstructured":"Wang, F., Xie, H., Zha, Z., Lu, S.: Scene Text recognition with transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)"},{"key":"21_CR42","doi-asserted-by":"publisher","first-page":"91661","DOI":"10.1109\/ACCESS.2020.2994287","volume":"8","author":"W Weihong","year":"2020","unstructured":"Weihong, W., Jiaoyang, T.: Research on license plate recognition algorithms based on deep learning in complex environment. IEEE Access 8, 91661\u201391675 (2020)","journal-title":"IEEE Access"},{"key":"21_CR43","doi-asserted-by":"crossref","unstructured":"Wu, S., Dredze, M.: Beto, bentz, becas: the surprising cross-lingual effectiveness of BERT. arXiv preprint arXiv:1904.09077 (2019)","DOI":"10.18653\/v1\/D19-1077"},{"issue":"7","key":"21_CR44","doi-asserted-by":"publisher","first-page":"1480","DOI":"10.1109\/TPAMI.2014.2366765","volume":"37","author":"Q Ye","year":"2014","unstructured":"Ye, Q., Doermann, D.: Text detection and recognition in imagery: a survey. IEEE Trans. Pattern Anal. Mach. Intell. 37(7), 1480\u20131500 (2014)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"21_CR45","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1007\/s11390-019-1923-y","volume":"34","author":"TL Yuan","year":"2019","unstructured":"Yuan, T.L., Zhu, Z., Xu, K., Li, C.J., Mu, T.J., Hu, S.M.: A large chinese text dataset in the wild. J. Comput. Sci. Technol. 34, 509\u2013521 (2019)","journal-title":"J. Comput. Sci. Technol."},{"key":"21_CR46","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1007\/s11390-019-1923-y","volume":"34","author":"TL Yuan","year":"2019","unstructured":"Yuan, T.L., Zhu, Z., Xu, K., Li, C.J., Mu, T.J., Hu, S.M.: A large chinese text dataset in the wild. J. Comput. Sci. Technol. 34, 509\u2013521 (2019)","journal-title":"J. Comput. Sci. Technol."},{"key":"21_CR47","doi-asserted-by":"crossref","unstructured":"Zhou, X., et al.: East: an efficient and accurate scene text detector. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5551\u20135560 (2017)","DOI":"10.1109\/CVPR.2017.283"}],"container-title":["Lecture Notes in Computer Science","Document Analysis and Recognition \u2013 ICDAR 2025 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-09371-4_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T23:52:28Z","timestamp":1767311548000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-09371-4_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032093707","9783032093714"],"references-count":47,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-09371-4_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 January 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICDAR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Document Analysis and Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Wuhan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icdar2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/iapr.org\/icdar2025","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}