{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,26]],"date-time":"2026-08-26T01:07:28Z","timestamp":1787706448279,"version":"build-2784847793"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030863333","type":"print"},{"value":"9783030863340","type":"electronic"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-86334-0_12","type":"book-chapter","created":{"date-parts":[[2021,9,3]],"date-time":"2021-09-03T20:16:02Z","timestamp":1630700162000},"page":"172-187","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["HCADecoder: A Hybrid CTC-Attention Decoder for Chinese Text Recognition"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8012-6820","authenticated-orcid":false,"given":"Siqi","family":"Cai","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7398-5785","authenticated-orcid":false,"given":"Wenyuan","family":"Xue","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3860-4809","authenticated-orcid":false,"given":"Qingyong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8176-9230","authenticated-orcid":false,"given":"Peng","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,9,2]]},"reference":[{"issue":"11","key":"12_CR1","doi-asserted-by":"publisher","first-page":"2298","DOI":"10.1109\/TPAMI.2016.2646371","volume":"39","author":"B Shi","year":"2016","unstructured":"Shi, B., Bai, X., Yao, C.: An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE Trans. Pattern Anal. Mach. Intell. 39(11), 2298\u20132304 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"12_CR2","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Bai, F., Xu, Y., Zheng, G., Pu, S., Zhou, S.: Focusing attention: towards accurate text recognition in natural images. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp. 5076\u20135084 (2017)","DOI":"10.1109\/ICCV.2017.543"},{"key":"12_CR3","unstructured":"Veit, A., Matera, T., Neumann, L., Matas, J., Belongie, S.: Coco-text: dataset and benchmark for text detection and recognition in natural images. arXiv preprint arXiv:1601.07140 (2016)"},{"key":"12_CR4","doi-asserted-by":"crossref","unstructured":"Karatzas, D., et al.: ICDAR 2015 competition on robust reading. In: Proceedings of the International Conference on Document Analysis and Recognition (ICDAR), pp. 1156\u20131160 (2015)","DOI":"10.1109\/ICDAR.2015.7333942"},{"issue":"4","key":"12_CR5","doi-asserted-by":"publisher","first-page":"793","DOI":"10.1109\/TNN.2002.1021881","volume":"13","author":"MR Naphade","year":"2002","unstructured":"Naphade, M.R., Huang, T.S.: Extracting semantics from audio-visual content: the final frontier in multimedia retrieval. IEEE Trans. Neural Netw. 13(4), 793\u2013810 (2002)","journal-title":"IEEE Trans. Neural Netw."},{"key":"12_CR6","doi-asserted-by":"crossref","unstructured":"Sadeghi, H., Valaee, S., Shirani, S.: Ocrapose: an indoor positioning system using smartphone\/tablet cameras and OCR-aided stereo feature matching. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1473\u20131477 (2015)","DOI":"10.1109\/ICASSP.2015.7178215"},{"key":"12_CR7","unstructured":"Wan, Z., Xie, F., Liu, Y., Bai, X., Yao, C.: 2D-CTC for scene text recognition. arXiv preprint arXiv:1907.09705 (2019)"},{"key":"12_CR8","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Xu, Y., Bai, F., Niu, Y., Pu, S., Zhou, S.: AON: towards arbitrarily-oriented text recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5571\u20135579 (2018)","DOI":"10.1109\/CVPR.2018.00584"},{"key":"12_CR9","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning (ICML), pp. 369\u2013376 (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"12_CR10","unstructured":"Bahdanau, D., Cho, K., Bengio, Y.: Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473 (2014)"},{"key":"12_CR11","doi-asserted-by":"crossref","unstructured":"Liu, C.L., Yin, F., Wang, D.H., Wang, Q.F.: CASIA online and offline Chinese handwriting databases. In: Proceedings of the International Conference on Document Analysis and Recognition (ICDAR), pp. 37\u201341 (2011)","DOI":"10.1109\/ICDAR.2011.17"},{"key":"12_CR12","doi-asserted-by":"crossref","unstructured":"Nayef, N., et al.: ICDAR2017 robust reading challenge on multi-lingual scene text detection and script identification-RRC-MLT. In: Proceedings of the International Conference on Document Analysis and Recognition (ICDAR), vol. 1, pp. 1454\u20131459 (2017)","DOI":"10.1109\/ICDAR.2017.237"},{"key":"12_CR13","doi-asserted-by":"crossref","unstructured":"Gupta, A., Vedaldi, A., Zisserman, A.: Synthetic data for text localisation in natural images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2315\u20132324 (2016)","DOI":"10.1109\/CVPR.2016.254"},{"key":"12_CR14","unstructured":"Ding, X., Wang, Y.: Character Recognition: Principles, Methods and Practice (2017)"},{"key":"12_CR15","doi-asserted-by":"publisher","first-page":"107102","DOI":"10.1016\/j.patcog.2019.107102","volume":"100","author":"ZR Wang","year":"2020","unstructured":"Wang, Z.R., Du, J., Wang, J.M.: Writer-aware CNN for parsimonious HMM-based offline handwritten Chinese text recognition. Pattern Recognit. 100, 107102 (2020)","journal-title":"Pattern Recognit."},{"key":"12_CR16","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1007\/s10032-019-00348-7","volume":"23","author":"G Tong","year":"2019","unstructured":"Tong, G., Li, Y., Gao, H., Chen, H., Wang, H., Yang, X.: MA-CRNN: a multi-scale attention CRNN for Chinese text line recognition in natural scenes. Int. J. Doc. Anal. Recognit. (IJDAR) 23, 103\u2013114 (2019)","journal-title":"Int. J. Doc. Anal. Recognit. (IJDAR)"},{"key":"12_CR17","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Xue, W., Li, Q.: A multi-scale CRNN model for Chinese papery medical document recognition. In: Proceedings of the IEEE Fourth International Conference on Multimedia Big Data (BigMM), pp. 1\u20135 (2018)","DOI":"10.1109\/BigMM.2018.8499468"},{"key":"12_CR18","unstructured":"Gao, Y., Chen, Y., Wang, J., Lu, H.: Reading scene text with attention convolutional sequence modeling. arXiv preprint arXiv:1709.04303 (2017)"},{"key":"12_CR19","doi-asserted-by":"crossref","unstructured":"Shigeki, K., Soplin, N., Watanabe, S., Delcroix, D., Ogawa, A., Nakatani, T.: Improving transformer-based end-to-end speech recognition with connectionist temporal classification and language model integration. In: Proceedings of INTERSPEECH, vol. 9, pp. 1408\u20131412 (2019)","DOI":"10.21437\/Interspeech.2019-1938"},{"key":"12_CR20","doi-asserted-by":"crossref","unstructured":"Chorowski, J., Jaitly, N.: Towards better decoding and language model integration in sequence to sequence models. arXiv preprint arXiv:1612.02695 (2016)","DOI":"10.21437\/Interspeech.2017-343"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Lee, C.Y., Osindero, S.: Recursive recurrent nets with attention modeling for OCR in the wild. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2231\u20132239 (2016)","DOI":"10.1109\/CVPR.2016.245"},{"key":"12_CR22","doi-asserted-by":"crossref","unstructured":"Liu, Z., Li, Y., Ren, F., Goh, W.L., Yu, H.: Squeezedtext: a real-time scene text recognition by binary convolutional encoder-decoder network. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)","DOI":"10.1609\/aaai.v32i1.12252"},{"key":"12_CR23","doi-asserted-by":"crossref","unstructured":"Bai, F., Cheng, Z., Niu, Y., Pu, S., Zhou, S.: Edit probability for scene text recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1508\u20131516 (2018)","DOI":"10.1109\/CVPR.2018.00163"},{"key":"12_CR24","doi-asserted-by":"crossref","unstructured":"Kim, S., Hori, T., Watanabe, S.: Joint CTC-attention based end-to-end speech recognition using multi-task learning. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4835\u20134839 (2017)","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"12_CR25","doi-asserted-by":"crossref","unstructured":"Hu, W., Cai, X., Hou, J., Yi, S., Lin, Z.: GTC: guided training of CTC towards efficient and accurate scene text recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11005\u201311012 (2020)","DOI":"10.1609\/aaai.v34i07.6735"},{"key":"12_CR26","doi-asserted-by":"crossref","unstructured":"Fan, D., Gao, G., Wu, H.: Sub-word based Mongolian offline handwriting recognition. In: Proceedings of the International Conference on Document Analysis and Recognition (ICDAR), pp. 246\u2013253 (2019)","DOI":"10.1109\/ICDAR.2019.00048"},{"key":"12_CR27","doi-asserted-by":"crossref","unstructured":"Saluja, R., Punjabi, M., Carman, M., Ramakrishnan, G., Chaudhuri, P.: Sub-word embeddings for OCR corrections in highly fusional Indic languages. In: Proceedings of the International Conference on Document Analysis and Recognition (ICDAR), pp. 160\u2013165 (2019)","DOI":"10.1109\/ICDAR.2019.00034"},{"key":"12_CR28","unstructured":"Jaderberg, M., Simonyan, K., Vedaldi, A., Zisserman, A.: Synthetic data and artificial neural networks for natural scene text recognition. arXiv preprint arXiv:1406.2227 (2014)"},{"key":"12_CR29","doi-asserted-by":"crossref","unstructured":"Wang, S., Chen, L., Xu, L., Fan, W., Sun, J., Naoi, S.: Deep knowledge training and heterogeneous CNN for handwritten Chinese text recognition. In: Proceedings of the 15th IEEE International Conference on Frontiers in Handwriting Recognition (ICFHR), pp. 84\u201389 (2016)","DOI":"10.1109\/ICFHR.2016.0028"},{"issue":"9","key":"12_CR30","doi-asserted-by":"publisher","first-page":"2035","DOI":"10.1109\/TPAMI.2018.2848939","volume":"41","author":"B Shi","year":"2018","unstructured":"Shi, B., Yang, M., Wang, X., Lyu, P., Yao, C., Bai, X.: ASTER: an attentional scene text recognizer with flexible rectification. IEEE Trans. Pattern Anal. Mach. Intell. 41(9), 2035\u20132048 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"12_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1007\/978-3-030-57058-3_4","volume-title":"Document Analysis Systems","author":"C Xie","year":"2020","unstructured":"Xie, C., Lai, S., Liao, Q., Jin, L.: High performance offline handwritten Chinese text recognition with a new data preprocessing and augmentation pipeline. In: Bai, X., Karatzas, D., Lopresti, D. (eds.) DAS 2020. LNCS, vol. 12116, pp. 45\u201359. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-57058-3_4"},{"key":"12_CR32","doi-asserted-by":"publisher","first-page":"251","DOI":"10.1016\/j.patcog.2016.12.026","volume":"65","author":"YC Wu","year":"2017","unstructured":"Wu, Y.C., Yin, F., Liu, C.L.: Improving handwritten Chinese text recognition using neural network language models and convolutional neural network shape models. Pattern Recognit. 65, 251\u2013264 (2017)","journal-title":"Pattern Recognit."},{"key":"12_CR33","doi-asserted-by":"crossref","unstructured":"Wang, Z.X., Wang, Q.F., Yin, F., Liu, C.L.: Weakly supervised learning for over-segmentation based handwritten Chinese text recognition. In: Proceedings of the 17th IEEE International Conference on Frontiers in Handwriting Recognition (ICFHR), pp. 157\u2013162 (2020)","DOI":"10.1109\/ICFHR2020.2020.00038"}],"container-title":["Lecture Notes in Computer Science","Document Analysis and Recognition \u2013 ICDAR 2021"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-86334-0_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,2]],"date-time":"2025-09-02T22:08:57Z","timestamp":1756850937000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-86334-0_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030863333","9783030863340"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-86334-0_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"2 September 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICDAR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Document Analysis and Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lausanne","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Switzerland","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 September 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 September 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icdar2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iapr.org\/icdar2021","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"340","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"182","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"54% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.9","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4.9","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Additionally, 13 competition reports are included.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}