{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T03:57:10Z","timestamp":1783828630505,"version":"3.55.0"},"reference-count":113,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2025,5,14]],"date-time":"2025-05-14T00:00:00Z","timestamp":1747180800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,14]],"date-time":"2025-05-14T00:00:00Z","timestamp":1747180800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62376266"],"award-info":[{"award-number":["62376266"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100018527","name":"Key Research Program of Frontier Science, Chinese Academy of Sciences","doi-asserted-by":"publisher","award":["ZDBS-LY-7024"],"award-info":[{"award-number":["ZDBS-LY-7024"]}],"id":[{"id":"10.13039\/501100018527","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s11263-025-02443-1","type":"journal-article","created":{"date-parts":[[2025,5,14]],"date-time":"2025-05-14T14:50:37Z","timestamp":1747234237000},"page":"5589-5609","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["IPAD: Iterative, Parallel, and Diffusion-Based Network for Scene Text Recognition"],"prefix":"10.1007","volume":"133","author":[{"given":"Xiaomeng","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhi","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4188-9953","authenticated-orcid":false,"given":"Yu","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,14]]},"reference":[{"key":"2443_CR1","doi-asserted-by":"crossref","unstructured":"Atienza, R. (2021). Vision transformer for fast and efficient scene text recognition. In: International Conference on Document Analysis and Recognition, pp. 319\u2013334.","DOI":"10.1007\/978-3-030-86549-8_21"},{"key":"2443_CR2","first-page":"17981","volume":"34","author":"J Austin","year":"2021","unstructured":"Austin, J., Johnson, D. D., Ho, J., Tarlow, D., & Van Den Berg, R. (2021). Structured denoising diffusion models in discrete state-spaces. Advances in Neural Information Processing Systems, 34, 17981\u201317993.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2443_CR3","unstructured":"Ba, J.L., Kiros, J.R., & Hinton, G.E. (2016). Layer normalization. arXiv preprint arXiv:1607.06450."},{"key":"2443_CR4","doi-asserted-by":"crossref","unstructured":"Baek, J., Kim, G., Lee, J., Park, S., Han, D., Yun, S., Oh, S. J., & Lee, H. (2019). What is wrong with scene text recognition model comparisons? Dataset and model analysis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4715\u20134723.","DOI":"10.1109\/ICCV.2019.00481"},{"key":"2443_CR5","doi-asserted-by":"crossref","unstructured":"Bautista, D., & Atienza, R. (2022). Scene text recognition with permuted autoregressive sequence models. In: Proceedings of the European Conference on Computer Vision, pp. 178\u2013196.","DOI":"10.1007\/978-3-031-19815-1_11"},{"key":"2443_CR6","unstructured":"Chan, W., Saharia, C., Hinton, G., Norouzi, M., & Jaitly, N. (2020). Imputer: Sequence modelling via imputation and dynamic programming. In: International Conference on Machine Learning, pp. 1403\u20131413. PMLR."},{"key":"2443_CR7","doi-asserted-by":"crossref","unstructured":"Chao, L., Chen, J., & Chu, W. (2020). Variational connectionist temporal classification. In: Proceedings of the European Conference on Computer Vision, pp. 460\u2013476.","DOI":"10.1007\/978-3-030-58604-1_28"},{"key":"2443_CR8","doi-asserted-by":"crossref","unstructured":"Chen, J., Li, B., & Xue, X. (2021). Scene text telescope: Text-focused scene image super-resolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12026\u201312035.","DOI":"10.1109\/CVPR46437.2021.01185"},{"key":"2443_CR9","unstructured":"Chen, J., Yu, H., Ma, J., Guan, M., Xu, X., Wang, X., Qu, S., Li, B., & Xue, X. (2021). Benchmarking Chinese text recognition: Datasets, baselines, and an empirical study. arXiv preprint arXiv:2112.15093."},{"key":"2443_CR10","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Bai, F., Xu, Y., Zheng, G., Pu, S., & Zhou, S. (2017). Focusing attention: Towards accurate text recognition in natural images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5076\u20135084.","DOI":"10.1109\/ICCV.2017.543"},{"key":"2443_CR11","doi-asserted-by":"crossref","unstructured":"Chi, E.A., Salazar, J., & Kirchhoff, K. (2021). Align-refine: Non-autoregressive speech recognition via iterative realignment. In: Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 1920\u20131927.","DOI":"10.18653\/v1\/2021.naacl-main.154"},{"key":"2443_CR12","doi-asserted-by":"crossref","unstructured":"Chng, C.K., Liu, Y., Sun, Y., Ng, C.C., Luo, C., Ni, Z., Fang, C., Zhang, S., Han, J., Ding, E., et al. (2019). ICDAR2019 robust reading challenge on arbitrary-shaped text (RRC-ArT). In: International Conference on Document Analysis and Recognition, pp. 1571\u20131576.","DOI":"10.1109\/ICDAR.2019.00252"},{"issue":"9","key":"2443_CR13","doi-asserted-by":"publisher","first-page":"10850","DOI":"10.1109\/TPAMI.2023.3261988","volume":"45","author":"F-A Croitoru","year":"2023","unstructured":"Croitoru, F.-A., Hondru, V., Ionescu, R. T., & Shah, M. (2023). Diffusion models in vision: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(9), 10850\u201310869.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2443_CR14","doi-asserted-by":"crossref","unstructured":"Da, C., Wang, P., & Yao, C. (2022). Levenshtein OCR. In: Proceedings of the European Conference on Computer Vision, pp. 322\u2013338.","DOI":"10.1007\/978-3-031-19815-1_19"},{"key":"2443_CR15","first-page":"8780","volume":"34","author":"P Dhariwal","year":"2021","unstructured":"Dhariwal, P., & Nichol, A. (2021). Diffusion models beat GANs on image synthesis. Advances in Neural Information Processing Systems, 34, 8780\u20138794.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2443_CR16","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., & Gelly, S., et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations."},{"key":"2443_CR17","doi-asserted-by":"crossref","unstructured":"Du, Y., Chen, Z., Jia, C., Yin, X., Zheng, T., Li, C., Du, Y., & Jiang, Y. -G. (2022). SVTR: Scene text recognition with a single visual model. In: Proceedings of the 31st International Joint Conference on Artificial Intelligence, pp. 884\u2013890.","DOI":"10.24963\/ijcai.2022\/124"},{"key":"2443_CR18","unstructured":"Esser, P., Rombach, R., Blattmann, A., & Ommer, B. (2021). ImageBART: Bidirectional context with multinomial diffusion for autoregressive image synthesis. In: Ranzato, M., Beygelzimer, A., Dauphin, Y., Liang, P.S., Vaughan, J.W. (eds.) Advances in Neural Information Processing Systems, vol. 34, pp. 3518\u20133532."},{"key":"2443_CR19","doi-asserted-by":"crossref","unstructured":"Fang, S., Xie, H., Wang, Y., Mao, Z., & Zhang, Y. (2021). Read like humans: Autonomous, bidirectional and iterative language modeling for scene text recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7098\u20137107.","DOI":"10.1109\/CVPR46437.2021.00702"},{"key":"2443_CR20","doi-asserted-by":"crossref","unstructured":"Fang, S., Xie, H., Zha, Z.-J., Sun, N., Tan, J., & Zhang, Y. (2018). Attention and language ensemble for scene text recognition with convolutional sequence modeling. In: Proceedings of the 26th ACM International Conference on Multimedia, pp. 248\u2013256.","DOI":"10.1145\/3240508.3240571"},{"key":"2443_CR21","doi-asserted-by":"crossref","unstructured":"Feng, W., He, W., Yin, F., Zhang, X.-Y., & Liu, C. -L. (2019). Textdragon: An end-to-end framework for arbitrary shaped text spotting. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9076\u20139085.","DOI":"10.1109\/ICCV.2019.00917"},{"key":"2443_CR22","doi-asserted-by":"crossref","unstructured":"Ghazvininejad, M., Levy, O., Liu, Y., & Zettlemoyer, L. (2019). Mask-predict: Parallel decoding of conditional masked language models. arXiv preprint arXiv:1904.09324.","DOI":"10.18653\/v1\/D19-1633"},{"key":"2443_CR23","unstructured":"Goldberg, Y., & Elhadad, M. (2010). An efficient algorithm for easy-first non-directional dependency parsing. In: Human Language Technologies: The 2010 Annual Conference of the North American Chapter of the Association for Computational Linguistics, pp. 742\u2013750."},{"key":"2443_CR24","unstructured":"Gong, S., Li, M., Feng, J., Wu, Z., & Kong, L. (2023). DiffuSeq: Sequence to sequence text generation with diffusion models. In: International Conference on Learning Representations, pp. 1\u201320."},{"key":"2443_CR25","unstructured":"Gu, J., Bradbury, J., Xiong, C., Li, V.O., & Socher, R. (2017). Non-autoregressive neural machine translation. arXiv preprint arXiv:1711.02281."},{"key":"2443_CR26","doi-asserted-by":"crossref","unstructured":"Gu, S., Chen, D., Bao, J., Wen, F., Zhang, B., Chen, D., Yuan, L., & Guo, B. (2022). Vector quantized diffusion model for text-to-image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10696\u201310706.","DOI":"10.1109\/CVPR52688.2022.01043"},{"key":"2443_CR27","unstructured":"Guan, T., Gu, C., Tu, J., Yang, X., & Feng, Q. (2022). A glyph-driven topology enhancement network for scene text recognition. arXiv preprint arXiv:2203.03382."},{"key":"2443_CR28","doi-asserted-by":"crossref","unstructured":"Guo, L., Liu, J., Zhu, X., He, X., Jiang, J., & Lu, H. (2020). Non-autoregressive image captioning with counterfactuals-critical multi-agent learning. arXiv preprint arXiv:2005.04690.","DOI":"10.24963\/ijcai.2020\/107"},{"key":"2443_CR29","doi-asserted-by":"crossref","unstructured":"Gupta, A., Vedaldi, A., & Zisserman, A. (2016). Synthetic data for text localisation in natural images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2315\u20132324.","DOI":"10.1109\/CVPR.2016.254"},{"key":"2443_CR30","doi-asserted-by":"crossref","unstructured":"He, P., Huang, W., Qiao, Y., Loy, C., & Tang, X. (2016). Reading scene text in deep convolutional sequences. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 30.","DOI":"10.1609\/aaai.v30i1.10465"},{"key":"2443_CR31","doi-asserted-by":"crossref","unstructured":"He, M., Liu, Y., Yang, Z., Zhang, S., Luo, C., Gao, F., Zheng, Q., Wang, Y., Zhang, X., & Jin, L. (2018). ICPR2018 contest on robust reading for multi-type web images. In: International Conference on Pattern Recognition, pp. 7\u201312.","DOI":"10.1109\/ICPR.2018.8546143"},{"key":"2443_CR32","doi-asserted-by":"crossref","unstructured":"He, Z., Sun, T., Tang, Q., Wang, K., Huang, X., & Qiu, X. (2023). DiffusionBERT: Improving generative masked language models with diffusion models. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, pp. 4521\u20134534.","DOI":"10.18653\/v1\/2023.acl-long.248"},{"key":"2443_CR33","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2443_CR34","unstructured":"Ho, J., Chan, W., Saharia, C., Whang, J., Gao, R., Gritsenko, A., Kingma, D.P., Poole, B., Norouzi, M., & Fleet, D.J., et al. (2022). Imagen video: High definition video generation with diffusion models. arXiv preprint arXiv:2210.02303."},{"key":"2443_CR35","unstructured":"Ho, J., Salimans, T., Gritsenko, A., Chan, W., Norouzi, M., & Fleet, D.J. (2022). Video diffusion models. arXiv preprint arXiv:2204.03458 ."},{"key":"2443_CR36","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models. Advances in Neural Information Processing Systems, 33, 6840\u20136851.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2443_CR37","unstructured":"Hoogeboom, E., Nielsen, D., Jaini, P., Forr\u00e9, P., & Welling, M. (2021). Argmax flows and multinomial diffusion: Learning categorical distributions. In: Ranzato, M., Beygelzimer, A., Dauphin, Y., Liang, P.S., Vaughan, J.W. (eds.) Advances in Neural Information Processing Systems, vol. 34, pp. 12454\u201312465."},{"key":"2443_CR38","doi-asserted-by":"crossref","unstructured":"Hu, W., Cai, X., Hou, J., Yi, S., & Lin, Z. (2020). GTC: Guided training of CTC towards efficient and accurate scene text recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11005\u201311012.","DOI":"10.1609\/aaai.v34i07.6735"},{"key":"2443_CR39","unstructured":"Izmailov, P., Podoprikhin, D., Garipov, T., Vetrov, D.P., & Wilson, A.G.(2018). Averaging weights leads to wider optima and better generalization. In: Conference on Uncertainty in Artificial Intelligence."},{"issue":"1","key":"2443_CR40","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11263-015-0823-z","volume":"116","author":"M Jaderberg","year":"2016","unstructured":"Jaderberg, M., Simonyan, K., Vedaldi, A., & Zisserman, A. (2016). Reading text in the wild with convolutional neural networks. International Journal of Computer Vision, 116(1), 1\u201320.","journal-title":"International Journal of Computer Vision"},{"key":"2443_CR41","doi-asserted-by":"crossref","unstructured":"Jiang, Q., Wang, J., Peng, D., Liu, C., & Jin, L. (2023). Revisiting scene text recognition: A data perspective. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 20543\u201320554.","DOI":"10.1109\/ICCV51070.2023.01878"},{"key":"2443_CR42","doi-asserted-by":"crossref","unstructured":"Karatzas, D., Gomez-Bigorda, L., Nicolaou, A., Ghosh, S., Bagdanov, A., Iwamura, M., Matas, J., Neumann, L., Chandrasekhar, V.R., & Lu, S., et al. (2015). ICDAR 2015 competition on robust reading. In: International Conference on Document Analysis and Recognition, pp. 1156\u20131160.","DOI":"10.1109\/ICDAR.2015.7333942"},{"key":"2443_CR43","doi-asserted-by":"crossref","unstructured":"Karatzas, D., Shafait, F., Uchida, S., Iwamura, M., Bigorda, L.G., Mestre, S.R., Mas, J., Mota, D.F., Almazan, J.A., & De\u00a0Las\u00a0Heras, L.P. (2013). ICDAR 2013 robust reading competition. In: International Conference on Document Analysis and Recognition, pp. 1484\u20131493.","DOI":"10.1109\/ICDAR.2013.221"},{"key":"2443_CR44","unstructured":"Kenton, J.D.M.-W.C., & Toutanova, L.K. (2019). BERT: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of Annual Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 4171\u20134186."},{"key":"2443_CR45","unstructured":"Kingma, D.P., & Ba, J. (2015). Adam: A method for stochastic optimization. In: International Conference on Learning Representations, pp. 1\u201315."},{"key":"2443_CR46","unstructured":"Kong, Z., Ping, W., Huang, J., Zhao, K., & Catanzaro, B. (2021). Diffwave: A versatile diffusion model for audio synthesis. In: International Conference on Learning Representations, pp. 1\u201317."},{"key":"2443_CR47","doi-asserted-by":"crossref","unstructured":"Lee, C.-Y., & Osindero, S. (2016). Recursive recurrent nets with attention modeling for OCR in the wild. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2231\u20132239.","DOI":"10.1109\/CVPR.2016.245"},{"key":"2443_CR48","doi-asserted-by":"crossref","unstructured":"Li, H., Wang, P., Shen, C., & Zhang, G. (2019). Show, attend and read: A simple and strong baseline for irregular text recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 8610\u20138617.","DOI":"10.1609\/aaai.v33i01.33018610"},{"key":"2443_CR49","doi-asserted-by":"crossref","unstructured":"Liao, M., Zhang, J., Wan, Z., Xie, F., Liang, J., Lyu, P., Yao, C., & Bai, X. (2019). Scene text recognition from two-dimensional perspective. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 8714\u20138721.","DOI":"10.1609\/aaai.v33i01.33018714"},{"key":"2443_CR50","first-page":"4328","volume":"35","author":"X Li","year":"2022","unstructured":"Li, X., Thickstun, J., Gulrajani, I., Liang, P. S., & Hashimoto, T. B. (2022). Diffusion-lm improves controllable text generation. Advances in Neural Information Processing Systems, 35, 4328\u20134343.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2443_CR51","doi-asserted-by":"crossref","unstructured":"Litman, R., Anschel, O., Tsiper, S., Litman, R., Mazor, S., & Manmatha, R. (2020). SCATTER: Selective context attentional scene text recognizer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11962\u201311972.","DOI":"10.1109\/CVPR42600.2020.01198"},{"key":"2443_CR52","doi-asserted-by":"publisher","first-page":"109","DOI":"10.1016\/j.patcog.2019.01.020","volume":"90","author":"C Luo","year":"2019","unstructured":"Luo, C., Jin, L., & Sun, Z. (2019). MORAN: A multi-object rectified attention network for scene text recognition. Pattern Recognition, 90, 109\u2013118.","journal-title":"Pattern Recognition"},{"key":"2443_CR53","doi-asserted-by":"publisher","first-page":"107980","DOI":"10.1016\/j.patcog.2021.107980","volume":"117","author":"N Lu","year":"2021","unstructured":"Lu, N., Yu, W., Qi, X., Chen, Y., Gong, P., Xiao, R., & Bai, X. (2021). MASTER: Multi-aspect non-local network for scene text recognition. Pattern Recognition, 117, 107980.","journal-title":"Pattern Recognition"},{"key":"2443_CR54","unstructured":"Lyu, P., Zhang, C., Liu, S., Qiao, M., Xu, Y., Wu, L., Yao, K., Han, J., Ding, E., & Wang, J. (2022). MaskOCR: Text recognition with masked encoder-decoder pretraining. arXiv preprintarXiv:2206.00311."},{"key":"2443_CR55","unstructured":"Merity, S., Xiong, C., Bradbury, J., & Socher, R. (2017). Pointer sentinel mixture models. In: International Conference on Learning Representations"},{"key":"2443_CR56","doi-asserted-by":"crossref","unstructured":"Mishra, A., Alahari, K., & Jawahar, C. (2012). Scene text recognition using higher order language priors. In: British Machine Vision Conference.","DOI":"10.5244\/C.26.127"},{"key":"2443_CR57","doi-asserted-by":"crossref","unstructured":"Mishra, A., Alahari, K., & Jawahar, C. (2012). Top-down and bottom-up cues for scene text recognition. In: 2012 IEEE Conference on Computer Vision and Pattern Recognition, pp. 2687\u20132694. IEEE.","DOI":"10.1109\/CVPR.2012.6247990"},{"key":"2443_CR58","doi-asserted-by":"crossref","unstructured":"Na, B., Kim, Y., & Park, S. (2022). Multi-modal text recognition networks: Interactive enhancements between visual and semantic features. In: Proceedings of the European Conference on Computer Vision, pp. 446\u2013463.","DOI":"10.1007\/978-3-031-19815-1_26"},{"key":"2443_CR59","doi-asserted-by":"crossref","unstructured":"Neumann, L., & Matas, J. (2012). Real-time scene text localization and recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3538\u20133545.","DOI":"10.1109\/CVPR.2012.6248097"},{"key":"2443_CR60","unstructured":"Nichol, A.Q., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., McGrew, B., Sutskever, I., & Chen, M. (2022). GLIDE: towards photorealistic image generation and editing with text-guided diffusion models. In: International Conference on Machine Learning, vol. 162, pp. 16784\u201316804."},{"key":"2443_CR61","doi-asserted-by":"crossref","unstructured":"Novikova, T., Barinova, O., Kohli, P., & Lempitsky, V. (2012). Large-lexicon attribute-consistent text recognition in natural images. In: Computer Vision\u2013ECCV 2012: 12th European Conference on Computer Vision, Florence, Italy, October 7-13, 2012, Proceedings, Part VI 12, pp. 752\u2013765. Springer.","DOI":"10.1007\/978-3-642-33783-3_54"},{"key":"2443_CR62","doi-asserted-by":"crossref","unstructured":"Phan, T.Q., Shivakumara, P., Tian, S., & Tan, C.L. (2013). Recognizing text with perspective distortion in natural scenes. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 569\u2013576.","DOI":"10.1109\/ICCV.2013.76"},{"key":"2443_CR63","doi-asserted-by":"crossref","unstructured":"Qian, L., Zhou, H., Bao, Y., Wang, M., Qiu, L., Zhang, W., Yu, Y., & Li, L. (2021). Glancing transformer for non-autoregressive neural machine translation. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics, pp. 1993\u20132003.","DOI":"10.18653\/v1\/2021.acl-long.155"},{"key":"2443_CR64","doi-asserted-by":"crossref","unstructured":"Qiao, Z., Qin, X., Zhou, Y., Yang, F., & Wang, W. (2021). Gaussian constrained attention network for scene text recognition. In: 2020 25th International Conference on Pattern Recognition (ICPR), pp. 3328\u20133335. IEEE","DOI":"10.1109\/ICPR48806.2021.9412806"},{"key":"2443_CR65","doi-asserted-by":"crossref","unstructured":"Qiao, Z., Zhou, Y., Wei, J., Wang, W., Zhang, Y., Jiang, N., Wang, H., & Wang, W. (2021). PIMNet: A parallel, iterative and mimicking network for scene text recognition. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 2046\u20132055.","DOI":"10.1145\/3474085.3475238"},{"key":"2443_CR66","doi-asserted-by":"crossref","unstructured":"Qiao, Z., Zhou, Y., Yang, D., Zhou, Y., & Wang, W. (2020). SEED: Semantics enhanced encoder-decoder framework for scene text recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13528\u201313537.","DOI":"10.1109\/CVPR42600.2020.01354"},{"key":"2443_CR67","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., & Chen, M. (2022). Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125."},{"key":"2443_CR68","unstructured":"Razavi, A., Oord, A., & Vinyals, O. (2019). Generating diverse high-fidelity images with VQ-VAE-2. In: Neural Information Processing Systems, pp. 14837\u201314847."},{"issue":"18","key":"2443_CR69","doi-asserted-by":"publisher","first-page":"8027","DOI":"10.1016\/j.eswa.2014.07.008","volume":"41","author":"A Risnumawan","year":"2014","unstructured":"Risnumawan, A., Shivakumara, P., Chan, C. S., & Tan, C. L. (2014). A robust arbitrary text detection system for natural scene images. Expert Systems with Applications, 41(18), 8027\u20138048.","journal-title":"Expert Systems with Applications"},{"key":"2443_CR70","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2443_CR71","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E. L., Ghasemipour, K., Gontijo Lopes, R., Karagol Ayan, B., Salimans, T., et al. (2022). Photorealistic text-to-image diffusion models with deep language understanding. Advances in Neural Information Processing Systems, 35, 36479\u201336494.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2443_CR72","doi-asserted-by":"crossref","unstructured":"Shi, B., Wang, X., Lyu, P., Yao, C., & Bai, X. (2016). Robust scene text recognition with automatic rectification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4168\u20134176.","DOI":"10.1109\/CVPR.2016.452"},{"key":"2443_CR73","doi-asserted-by":"crossref","unstructured":"Shi, B., Yao, C., Liao, M., Yang, M., Xu, P., Cui, L., Belongie, S., Lu, S., & Bai, X. (2017). ICDAR2017 competition on reading Chinese text in the wild (RCTW-17). In: International Conference on Document Analysis and Recognition, vol. 1, pp. 1429\u20131434.","DOI":"10.1109\/ICDAR.2017.233"},{"issue":"11","key":"2443_CR74","doi-asserted-by":"publisher","first-page":"2298","DOI":"10.1109\/TPAMI.2016.2646371","volume":"39","author":"B Shi","year":"2016","unstructured":"Shi, B., Bai, X., & Yao, C. (2016). An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 39(11), 2298\u20132304.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"9","key":"2443_CR75","doi-asserted-by":"publisher","first-page":"2035","DOI":"10.1109\/TPAMI.2018.2848939","volume":"41","author":"B Shi","year":"2018","unstructured":"Shi, B., Yang, M., Wang, X., Lyu, P., Yao, C., & Bai, X. (2018). ASTER: An attentional scene text recognizer with flexible rectification. IEEE Transactions on Pattern Analysis and Machine Intelligence, 41(9), 2035\u20132048.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2443_CR76","doi-asserted-by":"crossref","unstructured":"Smith, L.N., & Topin, N. (2019). Super-convergence: Very fast training of neural networks using large learning rates. In: Artificial Intelligence and Machine Learning for Multi-Domain Operations Applications, vol. 11006, pp. 369\u2013386.","DOI":"10.1117\/12.2520589"},{"key":"2443_CR77","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., & Ganguli, S. (2015). Deep unsupervised learning using nonequilibrium thermodynamics. In: International Conference on Machine Learning, pp. 2256\u20132265."},{"key":"2443_CR78","unstructured":"Song, J., Meng, C., & Ermon, S. (2021). Denoising diffusion implicit models. In: International Conference on Learning Representations, pp. 1\u201320."},{"key":"2443_CR79","doi-asserted-by":"publisher","first-page":"397","DOI":"10.1016\/j.patcog.2016.10.016","volume":"63","author":"B Su","year":"2017","unstructured":"Su, B., & Lu, S. (2017). Accurate recognition of words in scenes without character segmentation using recurrent neural network. Pattern Recognition, 63, 397\u2013405.","journal-title":"Pattern Recognition"},{"key":"2443_CR80","doi-asserted-by":"crossref","unstructured":"Sun, Y., Ni, Z., Chng, C.-K., Liu, Y., Luo, C., Ng, C.C., Han, J., Ding, E., Liu, J., & Karatzas, D., et al. (2019). ICDAR 2019 competition on large-scale street view text with partial labeling \u2013 RRC-LSVT. In: International Conference on Document Analysis and Recognition, pp. 1557\u20131562.","DOI":"10.1109\/ICDAR.2019.00250"},{"key":"2443_CR81","doi-asserted-by":"crossref","unstructured":"Tan, Y.L., Kong, A.W.-K., & Kim, J.-J. (2022). Pure transformer with integrated experts for scene text recognition. In: European Conference on Computer Vision, pp. 481\u2013497.","DOI":"10.1007\/978-3-031-19815-1_28"},{"key":"2443_CR82","doi-asserted-by":"crossref","unstructured":"Tian, Z., Yi, J., Tao, J., Bai, Y., Zhang, S., & Wen, Z. (2020). Spike-triggered non-autoregressive transformer for end-to-end speech recognition. arXiv preprint arXiv:2005.07903.","DOI":"10.21437\/Interspeech.2020-2086"},{"key":"2443_CR83","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. Advances in Neural Information Processing Systems30."},{"key":"2443_CR84","unstructured":"Veit, A., Matera, T., Neumann, L., Matas, J., & Belongie, S. (2016). COCO-Text: Dataset and benchmark for text detection and recognition in natural images. arXiv preprint arXiv:1601.07140."},{"key":"2443_CR85","doi-asserted-by":"crossref","unstructured":"Wan, Z., He, M., Chen, H., Bai, X., & Yao, C. (2020). Textscanner: Reading characters in order for robust scene text recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 12120\u201312127.","DOI":"10.1609\/aaai.v34i07.6891"},{"key":"2443_CR86","doi-asserted-by":"crossref","unstructured":"Wang, K., & Belongie, S. (2010). Word spotting in the wild. In: Computer Vision\u2013ECCV 2010: 11th European Conference on Computer Vision, Heraklion, Crete, Greece, September 5-11, 2010, Proceedings, Part I 11, pp. 591\u2013604. Springer.","DOI":"10.1007\/978-3-642-15549-9_43"},{"key":"2443_CR87","unstructured":"Wang, J., & Hu, X. (2017). Gated recurrent convolution neural network for ocr. Advances in Neural Information Processing Systems 30."},{"key":"2443_CR88","unstructured":"Wang, K., Babenko, B., & Belongie, S. (2011). End-to-end scene text recognition. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1457\u20131464."},{"key":"2443_CR89","doi-asserted-by":"crossref","unstructured":"Wang, P., Da, C., & Yao, C. (2022). Multi-granularity prediction for scene text recognition. In: Proceedings of the European Conference on Computer Vision, pp. 339\u2013355.","DOI":"10.1007\/978-3-031-19815-1_20"},{"key":"2443_CR90","unstructured":"Wang, T., Wu, D.J., Coates, A., & Ng, A. Y. (2012). End-to-end text recognition with convolutional neural networks. In: Proceedings of the 21st International Conference on Pattern Recognition (ICPR2012), pp. 3304\u20133308. IEEE."},{"key":"2443_CR91","doi-asserted-by":"crossref","unstructured":"Wang, Y., Xie, H., Fang, S., Wang, J., Zhu, S., & Zhang, Y. (2021). From two to one: A new scene text recognizer with visual language modeling network. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14194\u201314203.","DOI":"10.1109\/ICCV48922.2021.01393"},{"key":"2443_CR92","doi-asserted-by":"crossref","unstructured":"Wang, W., Zhou, Y., Lv, J., Wu, D., Zhao, G., Jiang, N., & Wang, W. (2022). TPSNet: Reverse thinking of thin plate splines for arbitrary shape scene text representation. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 5014\u20135025.","DOI":"10.1145\/3503161.3547882"},{"key":"2443_CR93","doi-asserted-by":"crossref","unstructured":"Wang, T., Zhu, Y., Jin, L., Luo, C., Chen, X., Wu, Y., Wang, Q., & Cai, M. (2020). Decoupled attention network for text recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 12216\u201312224.","DOI":"10.1609\/aaai.v34i07.6903"},{"key":"2443_CR94","doi-asserted-by":"crossref","unstructured":"Xie, X., Fu, L., Zhang, Z., Wang, Z., & Bai, X. (2022). Toward understanding WordArt: Corner-guided transformer for scene text recognition. In: Proceedings of the European Conference on Computer Vision, pp. 303\u2013321.","DOI":"10.1007\/978-3-031-19815-1_18"},{"key":"2443_CR95","doi-asserted-by":"crossref","unstructured":"Xu, J., Wang, X., Cheng, W., Cao, Y.-P., Shan, Y., Qie, X., & Gao, S. (2023). Dream3d: Zero-shot text-to-3d synthesis using 3d shape prior and text-to-image diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20908\u201320918.","DOI":"10.1109\/CVPR52729.2023.02003"},{"key":"2443_CR96","doi-asserted-by":"crossref","unstructured":"Yang, M., Guan, Y., Liao, M., He, X., Bian, K., Bai, S., Yao, C., & Bai, X. (2019). Symmetry-constrained rectification network for scene text recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9147\u20139156.","DOI":"10.1109\/ICCV.2019.00924"},{"key":"2443_CR97","doi-asserted-by":"crossref","unstructured":"Yang, X., Qiao, Z., Wei, J., Yang, D., & Zhou, Y. (2024). Masked and permuted implicit context learning for scene text recognition. IEEE Signal Processing Letters.","DOI":"10.1109\/LSP.2024.3381893"},{"key":"2443_CR98","unstructured":"Yang, L., Zhang, Z., Song, Y., Hong, S., Xu, R., Zhao, Y., Shao, Y., Zhang, W., Cui, B., & Yang, M.-H. (2022). Diffusion models: A comprehensive survey of methods and applications. arXiv preprint arXiv:2209.00796."},{"key":"2443_CR99","first-page":"3","volume":"1","author":"X Yang","year":"2017","unstructured":"Yang, X., He, D., Zhou, Z., Kifer, D., & Giles, C. L. (2017). Learning to read irregular text with attention mechanisms. Proceedings of the 27th International Joint Conferences on Artificial Intelligence, 1, 3.","journal-title":"Proceedings of the 27th International Joint Conferences on Artificial Intelligence"},{"key":"2443_CR100","doi-asserted-by":"crossref","unstructured":"Yao, C., Bai, X., Shi, B., & Liu, W. (2014). Strokelets: A learned multi-scale representation for scene text recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4042\u20134049.","DOI":"10.1109\/CVPR.2014.515"},{"key":"2443_CR101","doi-asserted-by":"crossref","unstructured":"Yu, D., Li, X., Zhang, C., Liu, T., Han, J., Liu, J., & Ding, E. (2020). Towards accurate scene text recognition with semantic reasoning networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12113\u201312122.","DOI":"10.1109\/CVPR42600.2020.01213"},{"key":"2443_CR102","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1007\/s11390-019-1923-y","volume":"34","author":"T-L Yuan","year":"2019","unstructured":"Yuan, T.-L., Zhu, Z., Xu, K., Li, C.-J., Mu, T.-J., & Hu, S.-M. (2019). A large Chinese text dataset in the wild. Journal of Computer Science and Technology, 34, 509\u2013521.","journal-title":"Journal of Computer Science and Technology"},{"key":"2443_CR103","doi-asserted-by":"crossref","unstructured":"Yue, X., Kuang, Z., Lin, C., Sun, H., & Zhang, W. (2020). RobustScanner: Dynamically enhancing positional clues for robust text recognition. In: Proceedings of the European Conference on Computer Vision, pp. 135\u2013151.","DOI":"10.1007\/978-3-030-58529-7_9"},{"key":"2443_CR104","doi-asserted-by":"crossref","unstructured":"Zeng, G., Zhang, Y., Zhou, Y., Yang, X., Jiang, N., Zhao, G., Wang, W., & Yin, X.-C. (2023). Beyond OCR+VQA: Towards end-to-end reading and reasoning for robust and accurate textvqa. Pattern Recognition, 138, Article 109337.","DOI":"10.1016\/j.patcog.2023.109337"},{"key":"2443_CR105","doi-asserted-by":"crossref","unstructured":"Zhan, F., & Lu, S. (2019). ESIR: End-to-end scene text recognition via iterative image rectification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2059\u20132068.","DOI":"10.1109\/CVPR.2019.00216"},{"key":"2443_CR106","unstructured":"Zhang, Y., Gueguen, L., Zharkov, I., Zhang, P., Seifert, K., & Kadlec, B. (2017). Uber-Text: A large-scale dataset for optical character recognition from street-level imagery. In: SUNw: Scene Understanding Workshop-CVPR, p. 5."},{"key":"2443_CR107","unstructured":"Zhang, Q., Tao, M., & Chen, Y. (2023) gDDIM: Generalized denoising diffusion implicit models. In: International Conference on Learning Representations, pp. 1\u201331."},{"key":"2443_CR108","doi-asserted-by":"crossref","unstructured":"Zhang, R., Zhou, Y., Jiang, Q., Song, Q., Li, N., Zhou, K., Wang, L., Wang, D., Liao, M., & Yang, M., et al. (2019). ICDAR 2019 robust reading challenge on reading Chinese text on signboard. In: International Conference on Document Analysis and Recognition, pp. 1577\u20131581.","DOI":"10.1109\/ICDAR.2019.00253"},{"key":"2443_CR109","doi-asserted-by":"publisher","first-page":"107559","DOI":"10.1016\/j.patcog.2020.107559","volume":"108","author":"H Zhang","year":"2020","unstructured":"Zhang, H., Liang, L., & Jin, L. (2020). SCUT-HCCDoc: A new benchmark dataset of handwritten Chinese text in unconstrained camera-captured documents. Pattern Recognition, 108, 107559.","journal-title":"Pattern Recognition"},{"key":"2443_CR110","doi-asserted-by":"crossref","unstructured":"Zheng, Y., Qin, W., Wijaya, D., & Betke, M. (2020). Lal: Linguistically aware learning for scene text recognition. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4051\u20134059.","DOI":"10.1145\/3394171.3413913"},{"key":"2443_CR111","doi-asserted-by":"crossref","unstructured":"Zhong, D., Lyu, S., Shivakumara, P., Yin, B., Wu, J., Pal, U., & Lu, Y. (2022). SGBANet: Semantic GAN and balanced attention network for arbitrarily oriented scene text recognition. In: Proceedings of the European Conference on Computer Vision, pp. 464\u2013480.","DOI":"10.1007\/978-3-031-19815-1_27"},{"key":"2443_CR112","doi-asserted-by":"crossref","unstructured":"Zhou, X., Yao, C., Wen, H., Wang, Y., Zhou, S., He, W., & Liang, J. (2017). EAST: An efficient and accurate scene text detector. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5551\u20135560.","DOI":"10.1109\/CVPR.2017.283"},{"key":"2443_CR113","unstructured":"Zhu, Z., Wei, Y., Wang, J., Gan, Z., Zhang, Z., Wang, L., Hua, G., Wang, L., Liu, Z., & Hu, H. (2022). Exploring discrete diffusion models for image captioning. arXiv preprint arXiv:2211.11694."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02443-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02443-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02443-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T14:52:44Z","timestamp":1757170364000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02443-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,14]]},"references-count":113,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["2443"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02443-1","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,14]]},"assertion":[{"value":"23 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 March 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}