{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,14]],"date-time":"2026-04-14T23:02:25Z","timestamp":1776207745086,"version":"3.50.1"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,6,22]],"date-time":"2023-06-22T00:00:00Z","timestamp":1687392000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,6,22]],"date-time":"2023-06-22T00:00:00Z","timestamp":1687392000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["IJDAR"],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1007\/s10032-023-00444-9","type":"journal-article","created":{"date-parts":[[2023,6,22]],"date-time":"2023-06-22T09:10:33Z","timestamp":1687425033000},"page":"45-56","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Chinese text recognition enhanced by glyph and character semantic information"],"prefix":"10.1007","volume":"27","author":[{"given":"Shilian","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongrui","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zengfu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,6,22]]},"reference":[{"key":"444_CR1","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778, (2016)","DOI":"10.1109\/CVPR.2016.90"},{"issue":"11","key":"444_CR2","doi-asserted-by":"publisher","first-page":"3212","DOI":"10.1109\/TNNLS.2018.2876865","volume":"30","author":"Z-Q Zhao","year":"2019","unstructured":"Zhao, Z.-Q., Zheng, P., Shou-tao, X., Xindong, W.: Object detection with deep learning: a review. IEEE Trans. Neural Netw. Learn. Syst. 30(11), 3212\u20133232 (2019)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"444_CR3","first-page":"17721","volume":"33","author":"X Wang","year":"2020","unstructured":"Wang, X., Zhang, R., Kong, T., Li, L., Shen, C.: Solov2: dynamic and fast instance segmentation. Adv. Neural Inf. Process. Syst. 33, 17721\u201317732 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"444_CR4","doi-asserted-by":"publisher","first-page":"343","DOI":"10.1613\/jair.1.12007","volume":"69","author":"F Stahlberg","year":"2020","unstructured":"Stahlberg, F.: Neural machine translation: a review. J. Artif. Intell. Res. 69, 343\u2013418 (2020)","journal-title":"J. Artif. Intell. Res."},{"issue":"11","key":"444_CR5","doi-asserted-by":"publisher","first-page":"2298","DOI":"10.1109\/TPAMI.2016.2646371","volume":"39","author":"B Shi","year":"2016","unstructured":"Shi, B., Bai, X., Yao, C.: An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE Trans. Pattern Anal. Mach. Intell. 39(11), 2298\u20132304 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"444_CR6","doi-asserted-by":"crossref","unstructured":"Shi, B., Wang, X., Lyu, P., Yao, C., Bai, X.: Robust scene text recognition with automatic rectification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4168\u20134176, (2016)","DOI":"10.1109\/CVPR.2016.452"},{"issue":"9","key":"444_CR7","doi-asserted-by":"publisher","first-page":"2035","DOI":"10.1109\/TPAMI.2018.2848939","volume":"41","author":"B Shi","year":"2018","unstructured":"Shi, B., Yang, M., Wang, X., Lyu, P., Yao, C., Bai, X.: Aster: an attentional scene text recognizer with flexible rectification. IEEE Trans. Pattern Anal. Mach. Intell. 41(9), 2035\u20132048 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"444_CR8","unstructured":"Wang, K., Babenko, B., Belongie, S.: End-to-end scene text recognition. In International Conference on Computer Vision, (2011)"},{"key":"444_CR9","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning, pp. 369\u2013376, (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"444_CR10","unstructured":"Bahdanau, D., Cho, K., Bengio, Y.: Neural machine translation by jointly learning to align and translate. In: Yoshua, B. and Yann, L. (eds.), 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7\u20139, 2015, Conference Track Proceedings, (2015)"},{"key":"444_CR11","doi-asserted-by":"crossref","unstructured":"Wang, W., Zhang, J., Du, J., Wang, Z.-R., Zhu, Y.: Denseran for offline handwritten Chinese character recognition. In: International Conference on Frontiers in Handwriting Recognition, (2018)","DOI":"10.1109\/ICFHR-2018.2018.00027"},{"key":"444_CR12","doi-asserted-by":"crossref","unstructured":"Chen, J., Li, B., Xue, X.: Zero-shot Chinese character recognition with stroke-level decomposition. In: International Joint Conference on Artificial Intelligence, (2021)","DOI":"10.24963\/ijcai.2021\/85"},{"issue":"1","key":"444_CR13","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1007\/BF00422382","volume":"31","author":"W K\u00f6hler","year":"1967","unstructured":"K\u00f6hler, W.: Gestalt psychology. Psychol. Forsch. 31(1), 18\u201330 (1967)","journal-title":"Psychol. Forsch."},{"issue":"4","key":"444_CR14","doi-asserted-by":"publisher","first-page":"466","DOI":"10.1016\/j.jecp.2010.06.006","volume":"107","author":"PD Liu","year":"2010","unstructured":"Liu, P.D., Chung, K.K.H., McBride-Chang, C., Tong, X.: Holistic versus analytic processing: evidence for a different approach to processing of Chinese at the word and character levels in Chinese children. J. Exp. Child Psychol. 107(4), 466\u2013478 (2010)","journal-title":"J. Exp. Child Psychol."},{"key":"444_CR15","unstructured":"Chen, H.-C., Song, H., Lau, W.\u00a0Y., Wong, K.F.E., Tang, S.L.: Developmental characteristics of eye movements in reading Chinese. Reading development in Chinese children, pp. 157\u2013169, (2003)"},{"key":"444_CR16","doi-asserted-by":"crossref","unstructured":"Su, B., Lu, S.: Accurate scene text recognition based on recurrent neural network. In: Asian Conference on Computer Vision, (2014)","DOI":"10.1007\/978-3-319-16865-4_3"},{"key":"444_CR17","doi-asserted-by":"crossref","unstructured":"He, P., Huang, W., Qiao, Y., Loy, C.\u00a0C., Tang, X.: Reading scene text in deep convolutional sequences. In: National Conference on Artificial Intelligence, (2016)","DOI":"10.1609\/aaai.v30i1.10465"},{"key":"444_CR18","unstructured":"Diaz, D.H., Qin, S., Ingl,e R.R., Fujii, Y., Bissacco, A.: Rethinking text line recognition models. Computer Vision and Pattern Recognition (2021)"},{"key":"444_CR19","doi-asserted-by":"crossref","unstructured":"Wu, G., Zhang, Z., Xiong, Y.: Carvenet: a channel-wise attention-based network for irregular scene text recognition. International Journal on Document Analysis and Recognition (IJDAR), pp. 1\u201310, (2022)","DOI":"10.1007\/s10032-022-00398-4"},{"issue":"1","key":"444_CR20","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1007\/s10032-021-00388-y","volume":"25","author":"SD Cui","year":"2022","unstructured":"Cui, S.D., YiLa, S., Ji, Y.T., et al.: An end-to-end network for irregular printed Mongolian recognition. Int. J. Doc. Anal. Recognit. (IJDAR) 25(1), 41\u201350 (2022)","journal-title":"Int. J. Doc. Anal. Recognit. (IJDAR)"},{"issue":"3","key":"444_CR21","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1007\/s10032-012-0186-8","volume":"16","author":"N Tagougui","year":"2013","unstructured":"Tagougui, N., Kherallah, M., Alimi, A.M.: Online Arabic handwriting recognition: a survey. Int. J. Doc. Anal. Recognit. (IJDAR) 16(3), 209\u2013226 (2013)","journal-title":"Int. J. Doc. Anal. Recognit. (IJDAR)"},{"issue":"2","key":"444_CR22","doi-asserted-by":"publisher","first-page":"143","DOI":"10.1007\/s10032-019-00320-5","volume":"22","author":"X Liu","year":"2019","unstructured":"Liu, X., Meng, G., Pan, C.: Scene text detection and recognition with advances in deep learning: a survey. Int. J. Doc. Anal. Recognit. (IJDAR) 22(2), 143\u2013162 (2019)","journal-title":"Int. J. Doc. Anal. Recognit. (IJDAR)"},{"key":"444_CR23","doi-asserted-by":"publisher","first-page":"109","DOI":"10.1016\/j.patcog.2019.01.020","volume":"90","author":"C Luo","year":"2019","unstructured":"Luo, C., Jin, L., Sun, Z.: Moran: a multi-object rectified attention network for scene text recognition. Pattern Recognit. 90, 109\u2013118 (2019)","journal-title":"Pattern Recognit."},{"key":"444_CR24","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: Attention is all you need. In: Neural Information Processing Systems, (2017)"},{"key":"444_CR25","doi-asserted-by":"crossref","unstructured":"Lee, J., Park, S., Baek, J., Oh, S.J., Kim, S., Lee, H.: On recognizing texts of arbitrary shapes with 2d self-attention. In: Computer Vision and Pattern Recognition, (2020)","DOI":"10.1109\/CVPRW50498.2020.00281"},{"key":"444_CR26","doi-asserted-by":"publisher","first-page":"107980","DOI":"10.1016\/j.patcog.2021.107980","volume":"117","author":"L Ning","year":"2021","unstructured":"Ning, L., Wenwen, Yu., Qi, X., Chen, Y., Gong, P., Xiao, R., Bai, X.: Master: multi-aspect non-local network for scene text recognition. Pattern Recognit. 117, 107980 (2021)","journal-title":"Pattern Recognit."},{"key":"444_CR27","doi-asserted-by":"crossref","unstructured":"Feng, Z., Du, C., Wang, Y., Xiao, B.: Oster: an orientation sensitive scene text recognizer with centerline rectification. In: Asian Conference on Pattern Recognition, (2019)","DOI":"10.1007\/978-3-030-41404-7_34"},{"key":"444_CR28","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1007\/s10032-019-00348-7","volume":"23","author":"G Tong","year":"2020","unstructured":"Tong, G., Li, Y., Gao, H., Chen, H., Wang, H., Yang, X.: Ma-crnn: a multi-scale attention CRNN for Chinese text line recognition in natural scenes. Int. J. Doc. Anal. Recognit. 23, 103\u2013114 (2020)","journal-title":"Int. J. Doc. Anal. Recognit."},{"key":"444_CR29","unstructured":"Chen, J., Yu, H., Ma, J., Guan, M., Xu, X., Wang, X., Qu, S., Li, B., Xue, X.: Benchmarking Chinese text recognition: datasets, baselines, and an empirical study. arXiv preprint arXiv:2112.15093, (2021)"},{"key":"444_CR30","doi-asserted-by":"crossref","unstructured":"Yang, M., Guan, Y., Liao, M., He, X., Bian, K., Bai, S., Yao, C., Bai, X.: Symmetry-constrained rectification network for scene text recognition. In: International Conference on Computer Vision, (2019)","DOI":"10.1109\/ICCV.2019.00924"},{"key":"444_CR31","doi-asserted-by":"crossref","unstructured":"Zhan, F., Lu, S.: Esir: end-to-end scene text recognition via iterative image rectification. In: Computer Vision and Pattern Recognition, (2019)","DOI":"10.1109\/CVPR.2019.00216"},{"key":"444_CR32","doi-asserted-by":"crossref","unstructured":"Yu, D., Li, X., Zhang, C., Tao, L., Han, J., Liu, J., Ding, E.: Towards accurate scene text recognition with semantic reasoning networks. In: Computer Vision and Pattern Recognition, (2020)","DOI":"10.1109\/CVPR42600.2020.01213"},{"key":"444_CR33","doi-asserted-by":"crossref","unstructured":"Fang, S., Xie, H., Wang, Y., Mao, Z., Zhang, Y.: Read like humans: autonomous, bidirectional and iterative language modeling for scene text recognition. Computer Vision and Pattern Recognition (2021)","DOI":"10.1109\/CVPR46437.2021.00702"},{"key":"444_CR34","doi-asserted-by":"crossref","unstructured":"Wang, Y., Xie, H., Fang, S., Wang, J., Zhu, S., Zhang, Y.: From two to one: a new scene text recognizer with visual language modeling network. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14194\u201314203, (2021)","DOI":"10.1109\/ICCV48922.2021.01393"},{"key":"444_CR35","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929, (2020)"},{"issue":"3","key":"444_CR36","first-page":"1","volume":"8","author":"W Wang","year":"2022","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.-P., Song, K., Liang, D., Tong, L., Luo, P., Shao, L.: Pvtv 2: improved baselines with pyramid vision transformer. Comput. Vis. Media 8(3), 1\u201310 (2022)","journal-title":"Comput. Vis. Media"},{"key":"444_CR37","unstructured":"Alec, R., Jong\u00a0Wook, K., Chris, H., Aditya, R., Gabriel, G., Sandhini, A., Girish, S., Amanda, A., Pamela, M., Jack, C. et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR, (2021)"},{"key":"444_CR38","doi-asserted-by":"crossref","unstructured":"Yuan, T.-L., Zhu, Z., Xu, K., Li, C.-J., Mu, T.-J., Hu, S.-M.: A large chinese text dataset in the wild. J. Comput. Sci. Technol., (2019)","DOI":"10.1007\/s11390-019-1923-y"},{"key":"444_CR39","doi-asserted-by":"crossref","unstructured":"Chng, C.K., Liu, Y., Sun, Y., Ng, C.C., Luo, C., Ni, Z., Fang, C., Zhang, S., Han, J., Ding, E. et\u00a0al.: Icdar2019 robust reading challenge on arbitrary-shaped text-rrc-art. In: ICDAR, (2019)","DOI":"10.1109\/ICDAR.2019.00252"},{"key":"444_CR40","doi-asserted-by":"crossref","unstructured":"Sun, Y., Ni, Z., Chng, C.-K., Liu, Y., Luo, C., Ng, C.C., Han, J., Ding, E., Liu, J., Karatzas, D. et\u00a0al.: Icdar 2019 competition on large-scale street view text with partial labeling-rrc-lsvt. In: ICDAR, (2019)","DOI":"10.1109\/ICDAR.2019.00250"},{"key":"444_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, R., Zhou, Y., Jiang, Q., Song, Q., Li, N., Zhou, K., Wang, L., Wang, D., Liao, M., Yang, M. et\u00a0al.: Icdar 2019 robust reading challenge on reading Chinese text on signboard. In: ICDAR, (2019)","DOI":"10.1109\/ICDAR.2019.00253"},{"key":"444_CR42","doi-asserted-by":"crossref","unstructured":"Shi, B., Yao, C., Liao, M., Yang, M., Xu, P., Cui, L., Belongie, S., Lu, S., Bai, X.: Icdar2017 competition on reading Chinese text in the wild (rctw-17). In: ICDAR, (2017)","DOI":"10.1109\/ICDAR.2017.233"},{"key":"444_CR43","doi-asserted-by":"crossref","unstructured":"He, M., Liu, Y., Yang, Z., Zhang, S., Luo, C., Gao, F., Zheng, Q., Wang, Y., Zhang, X., and Jin, L.: Icpr2018 contest on robust reading for multi-type web images. In: ICPR, 2018","DOI":"10.1109\/ICPR.2018.8546143"},{"key":"444_CR44","doi-asserted-by":"crossref","unstructured":"Yim, M., Kim, Y., Cho, H.-C., Park, S.: Synthtiger: synthetic text image generator towards better text recognition models. In: International Conference on Document Analysis and Recognition, pp. 109\u2013124. Springer, (2021)","DOI":"10.1007\/978-3-030-86337-1_8"},{"key":"444_CR45","doi-asserted-by":"crossref","unstructured":"Zhang, H., Liang, L., Jin, L.: Scut-hccdoc: a new benchmark dataset of handwritten Chinese text in unconstrained camera-captured documents. Pattern Recognition, pp. 107559, (2020)","DOI":"10.1016\/j.patcog.2020.107559"},{"key":"444_CR46","doi-asserted-by":"crossref","unstructured":"Li, H., Wang, P., Shen, C., Zhang, G.: Show, attend and read: a simple and strong baseline for irregular text recognition. In: AAAI, (2019)","DOI":"10.1609\/aaai.v33i01.33018610"},{"key":"444_CR47","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin,Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., and Guo, B.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022, 2021","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"444_CR48","first-page":"9355","volume":"34","author":"X Chu","year":"2021","unstructured":"Chu, X., Tian, Z., Wang, Y., Zhang, B., Ren, H., Wei, X., Xia, H., Shen, C.: Twins: revisiting the design of spatial attention in vision transformers. Adv. Neural Inf. Process. Syst. 34, 9355\u20139366 (2021)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"444_CR49","doi-asserted-by":"crossref","unstructured":"Kuang,Z., Sun, H., Li, Z., Yue, X., Lin, T.H., Chen, J., Wei, H., Zhu, Y., Gao, T., Zhang, W., et\u00a0al.: Mmocr: a comprehensive toolbox for text detection, recognition and understanding. arXiv preprint arXiv:2108.06543, 2021","DOI":"10.1145\/3474085.3478328"},{"key":"444_CR50","first-page":"2579","volume":"9","author":"L van der Maaten","year":"2008","unstructured":"van der Maaten, L., Hinton, G.E.: Visualizing data using T-SNE. J. Mach. Learn. Res. 9, 2579\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."},{"key":"444_CR51","doi-asserted-by":"crossref","unstructured":"Yu, D., Li, X., Zhang, C., Liu, T., Han, J., Liu, J., Ding, E.: Towards accurate scene text recognition with semantic reasoning networks. In: CVPR, (2020)","DOI":"10.1109\/CVPR42600.2020.01213"},{"key":"444_CR52","unstructured":"Yu, Z.Q., Zhou, D.Y., Zhou, Y., and Wang, W.: Semantics enhanced encoder-decoder framework for scene text recognition. In: CVPR, Seed (2020)"},{"key":"444_CR53","doi-asserted-by":"crossref","unstructured":"Chen, J., Li, B., and Xue, X.: Text-focused scene image super-resolution. In: CVPR, Scene text telescope (2021)","DOI":"10.1109\/CVPR46437.2021.01185"},{"key":"444_CR54","doi-asserted-by":"crossref","unstructured":"Yue, X., Kuang, Z., Lin, C., Sun, H., Zhang, W.: Robustscanner: dynamically enhancing positional clues for robust text recognition. In: European Conference on Computer Vision, pp. 135\u2013151. Springer, (2020)","DOI":"10.1007\/978-3-030-58529-7_9"},{"key":"444_CR55","unstructured":"Lyu, P., Zhang, C., Liu, S., Qiao, M., Xu, Y., Wu, L., Yao, K., Han, J., Ding, E., Wang, J.: Maskocr: text recognition with masked encoder-decoder pretraining. arXiv preprint arXiv:2206.00311, (2022)"}],"container-title":["International Journal on Document Analysis and Recognition (IJDAR)"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-023-00444-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10032-023-00444-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-023-00444-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T19:37:10Z","timestamp":1729625830000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10032-023-00444-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,22]]},"references-count":55,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,3]]}},"alternative-id":["444"],"URL":"https:\/\/doi.org\/10.1007\/s10032-023-00444-9","relation":{},"ISSN":["1433-2833","1433-2825"],"issn-type":[{"value":"1433-2833","type":"print"},{"value":"1433-2825","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,6,22]]},"assertion":[{"value":"13 July 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 October 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 June 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 June 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}