{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T22:00:03Z","timestamp":1787695203477,"version":"build-2784847793"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T00:00:00Z","timestamp":1757289600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T00:00:00Z","timestamp":1757289600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"JST CRONOS"},{"name":"JSPS Kakenhi","award":["22H00540"],"award-info":[{"award-number":["22H00540"]}]},{"name":"JSPS Kakenhi","award":["22H00540"],"award-info":[{"award-number":["22H00540"]}]},{"name":"JSPS Kakenhi","award":["22H00540"],"award-info":[{"award-number":["22H00540"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["IJDAR"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s10032-025-00554-6","type":"journal-article","created":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T11:45:32Z","timestamp":1757331932000},"page":"241-266","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Addressing the attention drift problem for khmer long textline recognition"],"prefix":"10.1007","volume":"29","author":[{"given":"Rina","family":"Buoy","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sovisal","family":"Chenda","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nguonly","family":"Taing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Marry","family":"Kong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Masakazu","family":"Iwamura","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Koichi","family":"Kise","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,8]]},"reference":[{"key":"554_CR1","doi-asserted-by":"crossref","unstructured":"Buoy, R., Iwamura, M., Srun, S., Kise, K.: Toward a low-resource non-latin-complete baseline: An exploration of khmer optical character recognition. IEEE Access 11, 128044\u2013128060 (2023)","DOI":"10.1109\/ACCESS.2023.3332361"},{"key":"554_CR2","unstructured":"Kaing, H.: Towards Morphological And Syntactic Analyses For The Khmer Language. Ph.D. thesis, Nara Institute of Science and Technology (2022)"},{"key":"554_CR3","doi-asserted-by":"crossref","unstructured":"Valy, D., Verleysen, M., Chhun, S.: Data augmentation and text recognition on khmer historical manuscripts. Proceedings of 17th International Conference on Frontiers in Handwriting Recognition (ICFHR), 73\u201378 (IEEE, 2020)","DOI":"10.1109\/ICFHR2020.2020.00024"},{"key":"554_CR4","doi-asserted-by":"publisher","first-page":"12216","DOI":"10.1609\/aaai.v34i07.6903","volume":"34","author":"T Wang","year":"2020","unstructured":"Wang, T., et al.: Decoupled attention network for text recognition. Proceedings of the AAAI Conference on Artificial Intelligence 34, 12216\u201312224 (2020)","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"554_CR5","unstructured":"Lugosch, L.: Sequence-to-sequence learning with transducers (2020). https:\/\/lorenlugosch.github.io\/posts\/2020\/11\/transducer\/"},{"key":"554_CR6","doi-asserted-by":"crossref","unstructured":"Wan, Z., He, M., Chen, H., Bai, X., Yao, C.: Textscanner: Reading characters in order for robust scene text recognition. 34, 12120\u201312127 (2020). https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/6891","DOI":"10.1609\/aaai.v34i07.6891"},{"key":"554_CR7","doi-asserted-by":"crossref","unstructured":"Qiao, Z., et\u00a0al.: Pimnet: A parallel, iterative and mimicking network for scene text recognition. Proceedings of the 29th ACM International Conference on Multimedia (2021)","DOI":"10.1145\/3474085.3475238"},{"key":"554_CR8","unstructured":"Chan, W., Saharia, C., Hinton, G., Norouzi, M., Jaitly, N.: Imputer: Sequence modelling via imputation and dynamic programming. International Conference on Machine Learning, 1403\u20131413 (PMLR, 2020)"},{"key":"554_CR9","doi-asserted-by":"crossref","unstructured":"Cheng, Z., et\u00a0al.: Focusing attention: Towards accurate text recognition in natural images. Proceedings of the IEEE international conference on computer vision, 5076\u20135084 (2017)","DOI":"10.1109\/ICCV.2017.543"},{"key":"554_CR10","unstructured":"Buoy, R., Iwamura, M., Srun, S., Kise, K.: Parstr: partially autoregressive scene text recognition. International Journal on Document Analysis and Recognition (IJDAR) 1\u201314 (2024)"},{"key":"554_CR11","unstructured":"Vaswani, A., et\u00a0al.: Attention is all you need. Proceedings of Advances in Neural Information Processing Systems, Vol.\u00a030 (Curran Associates, Inc., 2017). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2017\/file\/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf"},{"key":"554_CR12","doi-asserted-by":"crossref","unstructured":"Atienza, R.: Vision transformer for fast and efficient scene text recognition. Proceedings of International Conference on Document Analysis and Recognition, 319\u2013334 (Springer, 2021)","DOI":"10.1007\/978-3-030-86549-8_21"},{"key":"554_CR13","doi-asserted-by":"crossref","unstructured":"Baek, J., et\u00a0al.: What is wrong with scene text recognition model comparisons? dataset and model analysis. Proceedings of the IEEE\/CVF international conference on computer vision, 4715\u20134723 (2019)","DOI":"10.1109\/ICCV.2019.00481"},{"key":"554_CR14","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. Proceedings of the 23rd international conference on Machine learning, 369\u2013376 (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"554_CR15","doi-asserted-by":"crossref","unstructured":"Fang, S., Xie, H., Wang, Y., Mao, Z., Zhang, Y.: Read like humans: Autonomous, bidirectional and iterative language modeling for scene text recognition. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 7098\u20137107 (2021)","DOI":"10.1109\/CVPR46437.2021.00702"},{"key":"554_CR16","doi-asserted-by":"publisher","first-page":"2298","DOI":"10.1109\/TPAMI.2016.2646371","volume":"39","author":"B Shi","year":"2017","unstructured":"Shi, B., Bai, X., Yao, C.: An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE Trans. Pattern Anal. Mach. Intell. 39, 2298\u20132304 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"554_CR17","doi-asserted-by":"publisher","first-page":"2035","DOI":"10.1109\/TPAMI.2018.2848939","volume":"41","author":"B Shi","year":"2019","unstructured":"Shi, B., et al.: Aster: An attentional scene text recognizer with flexible rectification. IEEE Trans. Pattern Anal. Mach. Intell. 41, 2035\u20132048 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"554_CR18","doi-asserted-by":"crossref","unstructured":"Zhan, F., Lu, S.: Esir: End-to-end scene text recognition via iterative image rectification. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 2059\u20132068 (2019)","DOI":"10.1109\/CVPR.2019.00216"},{"key":"554_CR19","doi-asserted-by":"crossref","unstructured":"Bautista, D., Atienza, R.: Scene text recognition with permuted autoregressive sequence models. Proceedings of European conference on computer vision, 178\u2013196 (Springer, 2022)","DOI":"10.1007\/978-3-031-19815-1_11"},{"key":"554_CR20","doi-asserted-by":"crossref","unstructured":"Yang, M., et\u00a0al.: Reading and writing: Discriminative and generative modeling for self-supervised text recognition. Proceedings of the 30th ACM International Conference on Multimedia (2022)","DOI":"10.1145\/3503161.3547784"},{"key":"554_CR21","doi-asserted-by":"crossref","unstructured":"Jiang, Q., Wang, J., Peng, D., Liu, C., Jin, L.: Revisiting scene text recognition: A data perspective. Proceedings of the IEEE\/CVF international conference on computer vision, 20543\u201320554 (2023)","DOI":"10.1109\/ICCV51070.2023.01878"},{"key":"554_CR22","unstructured":"Hernandez\u00a0Diaz, D., Qin, S., Ingle, R., Fujii, Y., Bissacco, A.: Rethinking text line recognition models. arXiv e-prints arXiv\u20132104 (2021)"},{"key":"554_CR23","doi-asserted-by":"crossref","unstructured":"Sok, P., Taing, N.: Support vector machine (svm) based classifier for khmer printed character-set recognition. Proceedings of Signal and Information Processing Association Annual Summit and Conference (APSIPA), 2014 Asia-Pacific, 1\u20139 (IEEE, 2014)","DOI":"10.1109\/APSIPA.2014.7041823"},{"key":"554_CR24","doi-asserted-by":"crossref","unstructured":"Valy, D., Verleysen, M., Chhun, S.: Text recognition on khmer historical documents using glyph class map generation with encoder-decoder model. Proceedings of the 8th International Conference on Pattern Recognition Applications and Methods (2019)","DOI":"10.5220\/0007555507490756"},{"key":"554_CR25","doi-asserted-by":"publisher","first-page":"3","DOI":"10.46223\/HCMCOUJS.tech.en.12.1.2217.2022","volume":"12","author":"R Buoy","year":"2022","unstructured":"Buoy, R., Taing, N., Chenda, S., Kor, S.: Khmer printed character recognition using attention-based seq2seq network. Ho Chi Minh City Open University Journal Of Science - Engineering And Technology 12, 3\u201316 (2022)","journal-title":"Ho Chi Minh City Open University Journal Of Science - Engineering And Technology"},{"key":"554_CR26","doi-asserted-by":"crossref","unstructured":"Buoy, R., Iwamura, M., Srun, S., Kise, K.: Language-aware non-autoregressive khmer textline recognition. Pattern Recognition and Artificial Intelligence, 339\u2013353 (Springer Nature Singapore, Singapore, 2025)","DOI":"10.1007\/978-981-97-8702-9_23"},{"key":"554_CR27","unstructured":"Tesseract-Ocr. Tesseract-ocr\/tesseract: Tesseract open source ocr engine (main repository). https:\/\/github.com\/tesseract-ocr\/tesseract"},{"key":"554_CR28","unstructured":"Surya-Ocr. Surya. https:\/\/github.com\/VikParuchuri\/surya"},{"key":"554_CR29","doi-asserted-by":"crossref","unstructured":"Lee, J., et\u00a0al.: On recognizing texts of arbitrary shapes with 2d self-attention. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, 546\u2013547 (2020)","DOI":"10.1109\/CVPRW50498.2020.00281"},{"key":"554_CR30","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. Proceedings of the IEEE conference on computer vision and pattern recognition, 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"554_CR31","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: BERT: Pre-training of deep bidirectional transformers for language understanding. Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), 4171\u20134186 (Association for Computational Linguistics, Minneapolis, Minnesota, 2019). https:\/\/aclanthology.org\/N19-1423"},{"key":"554_CR32","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"554_CR33","unstructured":"Lamb, A.M., et\u00a0al.: Professor forcing: A new algorithm for training recurrent networks. Advances in neural information processing systems 29 (2016)"},{"key":"554_CR34","doi-asserted-by":"crossref","unstructured":"Nom, V., Bakkali, S., Luqman, M.M., Coustaty, M., Ogier, J.-M.: Khmerst: A low-resource khmer scene text detection and recognition benchmark (2024). https:\/\/arxiv.org\/abs\/2410.18277","DOI":"10.1007\/978-981-96-0917-8_22"},{"key":"554_CR35","unstructured":"Seanghay. Seanghay\/synthkhmer-10k\u00b7 datasets at hugging face. https:\/\/huggingface.co\/datasets\/seanghay\/SynthKhmer-10k"},{"key":"554_CR36","unstructured":"EKYCSolutions. Ekycsolutions\/khmer-ocr-benchmark-dataset: A standardized benchmark dataset for khmer optical character recognition (ocr) engine. https:\/\/github.com\/EKYCSolutions\/khmer-ocr-benchmark-dataset"},{"key":"554_CR37","unstructured":"Chakravuth, K.: Khmer annotation (2023). http:\/\/www.kaggle.com\/datasets\/keatchakravuth\/khmer-annotation"},{"key":"554_CR38","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1007\/s100320200071","volume":"5","author":"U-V Marti","year":"2002","unstructured":"Marti, U.-V., Bunke, H.: The iam-database: an english sentence database for offline handwriting recognition. Int. J. Doc. Anal. Recogn. 5, 39\u201346 (2002)","journal-title":"Int. J. Doc. Anal. Recogn."},{"key":"554_CR39","doi-asserted-by":"crossref","unstructured":"Grosicki, E., Abed, H.-M.: Icdar 2009 handwriting recognition competition. Proceedings of the 10th International Conference on Document Analysis and Recognition (ICDAR), 1398\u20131402 (IEEE, 2009)","DOI":"10.1109\/ICDAR.2009.184"},{"key":"554_CR40","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"554_CR41","doi-asserted-by":"crossref","unstructured":"Touvron, H., Cord, M., J\u00e9gou, H.: Deit iii: Revenge of the vit. European conference on computer vision, 516\u2013533 (Springer, 2022)","DOI":"10.1007\/978-3-031-20053-3_30"},{"key":"554_CR42","doi-asserted-by":"crossref","unstructured":"Yousef, M., Bishop, T.E.: Origaminet: weakly-supervised, segmentation-free, one-step, full page text recognition by learning to unfold. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 14710\u201314719 (2020)","DOI":"10.1109\/CVPR42600.2020.01472"},{"key":"554_CR43","doi-asserted-by":"publisher","first-page":"10563","DOI":"10.1007\/s00521-021-05813-1","volume":"33","author":"J Poulos","year":"2021","unstructured":"Poulos, J., Valle, R.: Character-based handwritten text transcription with attention networks. Neural Comput. Appl. 33, 10563\u201310573 (2021)","journal-title":"Neural Comput. Appl."},{"key":"554_CR44","doi-asserted-by":"crossref","unstructured":"Coquenet, D., Chatelain, C., Paquet, T.: Span: A simple predict & align network for handwritten paragraph recognition. Document Analysis and Recognition \u2013 ICDAR 2021, 70\u201384 (Springer International Publishing, Cham, 2021)","DOI":"10.1007\/978-3-030-86334-0_5"},{"key":"554_CR45","doi-asserted-by":"publisher","first-page":"13094","DOI":"10.1609\/aaai.v37i11.26538","volume":"37","author":"M Li","year":"2023","unstructured":"Li, M., et al.: Trocr: Transformer-based optical character recognition with pre-trained models. Proceedings of the AAAI conference on artificial intelligence 37, 13094\u201313102 (2023)","journal-title":"Proceedings of the AAAI conference on artificial intelligence"},{"key":"554_CR46","doi-asserted-by":"publisher","first-page":"8227","DOI":"10.1109\/TPAMI.2023.3235826","volume":"45","author":"D Coquenet","year":"2023","unstructured":"Coquenet, D., Chatelain, C., Paquet, T.: Dan: a segmentation-free document attention network for handwritten document recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45, 8227\u20138243 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"554_CR47","unstructured":"Li, Y., Chen, D., Tang, T., Shen, X.: Htr-vt: Handwritten text recognition with vision transformer. Pattern Recogn. 158, 110967 (2025)"}],"container-title":["International Journal on Document Analysis and Recognition (IJDAR)"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-025-00554-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10032-025-00554-6","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-025-00554-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T07:08:02Z","timestamp":1781939282000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10032-025-00554-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,8]]},"references-count":47,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["554"],"URL":"https:\/\/doi.org\/10.1007\/s10032-025-00554-6","relation":{},"ISSN":["1433-2833","1433-2825"],"issn-type":[{"value":"1433-2833","type":"print"},{"value":"1433-2825","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,8]]},"assertion":[{"value":"16 November 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 June 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 August 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 September 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Koichi Kise is the editor-in-chief of the International Journal on Document Analysis and Recognition (IJDAR).","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}