{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,6]],"date-time":"2026-02-06T23:38:50Z","timestamp":1770421130282,"version":"3.49.0"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2022,10,6]],"date-time":"2022-10-06T00:00:00Z","timestamp":1665014400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,10,6]],"date-time":"2022-10-06T00:00:00Z","timestamp":1665014400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2022,12]]},"DOI":"10.1007\/s13735-022-00253-6","type":"journal-article","created":{"date-parts":[[2022,10,6]],"date-time":"2022-10-06T17:36:07Z","timestamp":1665077767000},"page":"669-680","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Visual and semantic ensemble for scene text recognition with gated dual mutual attention"],"prefix":"10.1007","volume":"11","author":[{"given":"Zhiguang","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liangwei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jian","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,10,6]]},"reference":[{"key":"253_CR1","doi-asserted-by":"crossref","unstructured":"Li H, Wang P, Shen C (2017) Towards end-to-end text spotting with convolutional recurrent neural networks. In: Proceedings of the IEEE ICCV, pp 5238\u20135246","DOI":"10.1109\/ICCV.2017.560"},{"key":"253_CR2","first-page":"12120","volume":"34","author":"Z Wan","year":"2020","unstructured":"Wan Z, He M, Chen H, Bai X, Yao C (2020) Textscanner: reading characters in order for robust scene text recognition. Proc AAAI Conf Artif Intell 34:12120\u201312127","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"253_CR3","doi-asserted-by":"crossref","unstructured":"Yue X, Kuang Z, Lin C, Sun H, Zhang W (2020) Robustscanner: dynamically enhancing positional clues for robust text recognition. In: European conference on computer vision, Springer, pp 135\u2013151","DOI":"10.1007\/978-3-030-58529-7_9"},{"key":"253_CR4","doi-asserted-by":"crossref","unstructured":"Wang T, Zhu Y, Jin L, Luo C, Chen X, Wu Y, Wang Q, Cai M (2020) Decoupled attention network for text recognition. In: AAAI, pp 12216\u201312224","DOI":"10.1609\/aaai.v34i07.6903"},{"key":"253_CR5","doi-asserted-by":"crossref","unstructured":"Shi B, Yang M, Wang X, Lyu P, Yao C, Bai X (2018) Aster: an attentional scene text recognizer with flexible rectification. IEEE transactions on pattern analysis and machine intelligence","DOI":"10.1109\/TPAMI.2018.2848939"},{"key":"253_CR6","doi-asserted-by":"publisher","first-page":"8714","DOI":"10.1609\/aaai.v33i01.33018714","volume":"33","author":"M Liao","year":"2019","unstructured":"Liao M, Zhang J, Wan Y, Xie F, Liang J, Lyu P, Yao C, Bai X (2019) Scene text recognition from two-dimensional perspective. Proc AAAI Conf Artif Intell 33:8714\u20138721. https:\/\/doi.org\/10.1609\/aaai.v33i01.33018714","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"253_CR7","doi-asserted-by":"publisher","unstructured":"Yang H, Wang C, Che , Luo S, Meinel C (2015) An improved system for real-time scene text recognition. In: Proceedings of the 5th ACM on international conference on multimedia retrieval. ICMR \u201915, pp 657\u2013660. Association for Computing Machinery, New York, NY, USA https:\/\/doi.org\/10.1145\/2671188.2749352","DOI":"10.1145\/2671188.2749352"},{"key":"253_CR8","doi-asserted-by":"crossref","unstructured":"Yang X, He D, Zhou Z, Kifer D, Giles CL(2017) Learning to read irregular text with attention mechanisms. In: Proceedings of the 26th international joint conference on artificial intelligence. IJCAI\u201917, pp 3280\u20133286. AAAI Press, ??? http:\/\/dl.acm.org\/citation.cfm?id=3172077.3172347","DOI":"10.24963\/ijcai.2017\/458"},{"key":"253_CR9","doi-asserted-by":"crossref","unstructured":"Li H, Wang P, Shen C, Zhang G (2019) Show, attend and read: a simple and strong baseline for irregular text recognition. In: AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v33i01.33018610"},{"key":"253_CR10","doi-asserted-by":"publisher","first-page":"109","DOI":"10.1016\/j.patcog.2019.01.020","volume":"90","author":"C Luo","year":"2019","unstructured":"Luo C, Jin L, Sun Z (2019) Moran: a multi-object rectified attention network for scene text recognition. Pattern Recognit 90:109\u2013118","journal-title":"Pattern Recognit"},{"key":"253_CR11","doi-asserted-by":"crossref","unstructured":"Shi B, Wang X, Lyu P, Yao C, Bai X (2016) Robust scene text recognition with automatic rectification. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4168\u20134176","DOI":"10.1109\/CVPR.2016.452"},{"key":"253_CR12","doi-asserted-by":"crossref","unstructured":"Cheng Z, Bai F, Xu Y, Zheng G, Pu S, Zhou S (2017) Focusing attention: towards accurate text recognition in natural images. In: Proceedings of the IEEE international conference on computer vision, pp 5076\u20135084","DOI":"10.1109\/ICCV.2017.543"},{"key":"253_CR13","doi-asserted-by":"crossref","unstructured":"Liu Z, Li Y, Ren F, Goh WL, Yu H (2018) Squeezedtext: a real-time scene text recognition by binary convolutional encoder-decoder network. In: Thirty-Second AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v32i1.12252"},{"key":"253_CR14","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1016\/j.neucom.2020.07.010","volume":"414","author":"L Yang","year":"2020","unstructured":"Yang L, Wang P, Li H, Li Z, Zhang Y (2020) A holistic representation guided attention network for scene text recognition. Neurocomputing 414:67\u201375","journal-title":"Neurocomputing"},{"key":"253_CR15","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser, \u0141, Polosukhin I (2017) Attention is all you need. In: Advances in Neural Information Processing Systems, pp 5998\u20136008"},{"key":"253_CR16","doi-asserted-by":"crossref","unstructured":"Liu Z, Wang L, Qiao J (2021) Reading scene text by fusing visual attention with semantic representations. In: Proceedings of the 2021 international conference on multimedia retrieval. ICMR \u201921, pp 210\u2013218. Association for Computing Machinery, New York, NY, USA","DOI":"10.1145\/3460426.3463612"},{"key":"253_CR17","doi-asserted-by":"crossref","unstructured":"Wang K, Babenko B, Belongie S (2011) End-to-end scene text recognition. In: 2011 international conference on computer vision, IEEE, pp 1457\u20131464","DOI":"10.1109\/ICCV.2011.6126402"},{"key":"253_CR18","doi-asserted-by":"crossref","unstructured":"Yao C, Bai X, Shi B, Liu W (2014) Strokelets: a learned multi-scale representation for scene text recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4042\u20134049","DOI":"10.1109\/CVPR.2014.515"},{"key":"253_CR19","doi-asserted-by":"crossref","unstructured":"Lee C-Y, Osindero S (2016) Recursive recurrent nets with attention modeling for ocr in the wild. In: Proceedings of the IEEE conference on CVPR, pp 2231\u20132239","DOI":"10.1109\/CVPR.2016.245"},{"key":"253_CR20","doi-asserted-by":"crossref","unstructured":"Bai F, Cheng Z, Niu Y, Pu S, Zhou S (2018) Edit probability for scene text recognition. In: proceedings of the IEEE conference on computer vision and pattern recognition, pp 1508\u20131516","DOI":"10.1109\/CVPR.2018.00163"},{"key":"253_CR21","doi-asserted-by":"crossref","unstructured":"He P, Huang W, Qiao Y, Loy CC, Tang X (2016) Reading scene text in deep convolutional sequences. In: Thirtieth AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v30i1.10465"},{"issue":"11","key":"253_CR22","doi-asserted-by":"publisher","first-page":"2298","DOI":"10.1109\/TPAMI.2016.2646371","volume":"39","author":"B Shi","year":"2016","unstructured":"Shi B, Bai X, Yao C (2016) An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE Trans PAMI 39(11):2298\u20132304","journal-title":"IEEE Trans PAMI"},{"key":"253_CR23","doi-asserted-by":"crossref","unstructured":"Fang S, Xie H, Zha Z-J, Sun N, Tan J, Zhang Y (2018) Attention and language ensemble for scene text recognition with convolutional sequence modeling. In: 2018 ACM multimedia conference on multimedia conference, ACM, pp 248\u2013256","DOI":"10.1145\/3240508.3240571"},{"key":"253_CR24","unstructured":"Jaderberg M, Simonyan K, Zisserman A, et al (2015) Spatial transformer networks. In: Advances in Neural Information Processing Systems, pp 2017\u20132025"},{"key":"253_CR25","doi-asserted-by":"crossref","unstructured":"Liu W, Chen C, Wong K-YK (2018) Char-net: a character-aware neural network for distorted scene text recognition. In: thirty-second AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v32i1.12246"},{"key":"253_CR26","doi-asserted-by":"crossref","unstructured":"Cheng Z, Xu Y, Bai F, Niu Y, Pu S, Zhou S (2018) Aon: towards arbitrarily-oriented text recognition. In: proceedings of the IEEE conference on computer vision and pattern recognition, pp 5571\u20135579","DOI":"10.1109\/CVPR.2018.00584"},{"key":"253_CR27","unstructured":"Dauphin YN, Fan A, Auli M, Grangier D(2017) Language modeling with gated convolutional networks. In: proceedings of the 34th international conference on machine learning-volume 70, pp 933\u2013941 . JMLR.org"},{"key":"253_CR28","unstructured":"Dehghani M, Gouws S, Vinyals O, Uszkoreit J, Kaiser \u0141 (2018) Universal transformers. arXiv preprint arXiv:1807.03819"},{"key":"253_CR29","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: proceedings of the 2019 conference of the NAACL, pp 4171\u20134186"},{"key":"253_CR30","unstructured":"Gehring J, Auli M, Grangier D, Yarats D, Dauphin YN (2017) Convolutional sequence to sequence learning. In: proceedings of the 34th international conference on machine learning-volume 70, pp 1243\u20131252 JMLR.org"},{"key":"253_CR31","doi-asserted-by":"crossref","unstructured":"Ben-Younes H, Cadene R, Thome N, Cord M (2019) Block: Bilinear superdiagonal fusion for visual question answering and visual relationship detection. arXiv preprint arXiv:1902.00038","DOI":"10.1609\/aaai.v33i01.33018102"},{"key":"253_CR32","doi-asserted-by":"publisher","first-page":"570","DOI":"10.1007\/978-3-030-86331-9_37","volume-title":"Document analysis and recognition - ICDAR 2021","author":"W Zhao","year":"2021","unstructured":"Zhao W, Gao L, Yan Z, Peng S, Du L, Zhang Z (2021) Handwritten mathematical expression recognition with bidirectionally trained transformer. In: Llad\u00f3s J, Lopresti D, Uchida S (eds) Document analysis and recognition - ICDAR 2021. Springer, Cham, pp 570\u2013584"},{"key":"253_CR33","doi-asserted-by":"publisher","DOI":"10.3390\/info13020069","author":"L Liao","year":"2022","unstructured":"Liao L, Afedzie Kwofie F, Chen Z, Han G, Wang Y, Lin Y, Hu D (2022) A bidirectional context embedding transformer for automatic speech recognition. Information. https:\/\/doi.org\/10.3390\/info13020069","journal-title":"Information"},{"key":"253_CR34","unstructured":"Jaderberg M, Simonyan K, Vedaldi A, Zisserman A (2014) Synthetic data and artificial neural networks for natural scene text recognition. arXiv preprint arXiv:1406.2227"},{"key":"253_CR35","doi-asserted-by":"crossref","unstructured":"Gupta A, Vedaldi A, Zisserman A (2016) Synthetic data for text localisation in natural images. In: proceedings of the IEEE conference on computer vision and pattern recognition, pp 2315\u20132324","DOI":"10.1109\/CVPR.2016.254"},{"issue":"18","key":"253_CR36","doi-asserted-by":"publisher","first-page":"8027","DOI":"10.1016\/j.eswa.2014.07.008","volume":"41","author":"A Risnumawan","year":"2014","unstructured":"Risnumawan A, Shivakumara P, Chan CS, Tan CL (2014) A robust arbitrary text detection system for natural scene images. Expert Syst Appl 41(18):8027\u20138048","journal-title":"Expert Syst Appl"},{"key":"253_CR37","doi-asserted-by":"crossref","unstructured":"Jaume G, Kemal Ekenel H, Thiran J-P (2019) FUNSD: a dataset for form understanding in noisy scanned documents. arXiv e-prints, 1905\u201313538 arXiv:1905.13538 [cs.IR]","DOI":"10.1109\/ICDARW.2019.10029"},{"key":"253_CR38","unstructured":"Veit A, Matera T, Neumann L, Matas J, Belongie S (2016) Coco-text: dataset and benchmark for text detection and recognition in natural images. arXiv preprint arXiv:1601.07140"},{"key":"253_CR39","doi-asserted-by":"crossref","unstructured":"Elliott D, Frank S, Sima\u2019an K, Specia L (2016) Multi30K: multilingual English-German image descriptions. In: proceedings of the 5th workshop on vision and language, pp 70\u201374. Association for Computational Linguistics, Berlin, Germany","DOI":"10.18653\/v1\/W16-3210"},{"key":"253_CR40","doi-asserted-by":"crossref","unstructured":"Chelba C, Mikolov T, Schuster M, Ge Q, Brants T, Koehn P, Robinson T (2013) One Billion Word Benchmark for Measuring Progress in Statistical Language Modeling. arXiv e-prints, 1312\u20133005 arXiv:1312.3005 [cs.CL]","DOI":"10.21437\/Interspeech.2014-564"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00253-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-022-00253-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00253-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,17]],"date-time":"2022-12-17T14:24:03Z","timestamp":1671287043000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-022-00253-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,6]]},"references-count":40,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2022,12]]}},"alternative-id":["253"],"URL":"https:\/\/doi.org\/10.1007\/s13735-022-00253-6","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,10,6]]},"assertion":[{"value":"31 March 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 July 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 August 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 October 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"This article does not contain any study with human participants performed by any author.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Human and animal rights"}},{"value":"Informed consent was obtained from individual participants included in the study.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed consent"}}]}}