{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,2]],"date-time":"2025-08-02T05:09:05Z","timestamp":1754111345998},"reference-count":36,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2022,2,16]],"date-time":"2022-02-16T00:00:00Z","timestamp":1644969600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,2,16]],"date-time":"2022-02-16T00:00:00Z","timestamp":1644969600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2022,3]]},"DOI":"10.1007\/s11042-022-12024-w","type":"journal-article","created":{"date-parts":[[2022,2,16]],"date-time":"2022-02-16T17:02:46Z","timestamp":1645030966000},"page":"10545-10559","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Text recognition in natural scenes based on deep learning"],"prefix":"10.1007","volume":"81","author":[{"given":"Yi","family":"Jiang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongyu","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liang","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuai","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,2,16]]},"reference":[{"issue":"99","key":"12024_CR1","first-page":"1","volume":"PP","author":"M Alazab","year":"2020","unstructured":"Alazab M, Khan S, Krishnan SSR et al (2020) A multidirectional LSTM model for predicting the stability of a smart grid. IEEE Access PP(99):1\u201311","journal-title":"IEEE Access"},{"key":"12024_CR2","first-page":"89","volume-title":"Neural machine translation by jointly learning to align and translate","author":"D Bahdanau","year":"2015","unstructured":"Bahdanau D, Cho K, Bengio Y (2015) Neural machine translation by jointly learning to align and translate. International Conference on Learning Representations, San Diego, pp 89\u201393"},{"key":"12024_CR3","doi-asserted-by":"crossref","unstructured":"Bahdanau D, Chorowski J, Serdyuk D, et al. End-to-end attention-based large vocabulary speech recognition. Shanghai: The 41st IEEE International Conference on Acoustics, Speech and Signal Processing, 2016:4945\u20134949.","DOI":"10.1109\/ICASSP.2016.7472618"},{"issue":"11","key":"12024_CR4","first-page":"2298","volume":"39","author":"X Bai","year":"2016","unstructured":"Bai X, Shi B, Yao C (2016) An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE Trans Pattern Anal Mach Intell 39(11):2298\u20132304","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"12024_CR5","doi-asserted-by":"publisher","first-page":"132","DOI":"10.1016\/j.ssci.2020.104812","volume":"130","author":"ZJ Chen","year":"2020","unstructured":"Chen ZJ, Chen DP, Zhang YS, Cheng XZ et al (2020) Deep learning for autonomous ship-oriented small ship detection. Saf Sci 130:132\u2013141","journal-title":"Saf Sci"},{"key":"12024_CR6","doi-asserted-by":"crossref","unstructured":"Chen JN, Gao S, Sun HZ et al (2020) An end-to-end speech recognition algorithm based on attention mechanism. Syst Eng Soc China:6\u201314","DOI":"10.23919\/CCC50068.2020.9189026"},{"key":"12024_CR7","first-page":"516","volume":"144","author":"ZJ Chen","year":"2020","unstructured":"Chen ZJ, Cai H, Zhang YS, Wu CZ et al (2020) A novel sparse representation model for pedestrian abnormal trajectory understanding. Expert Syst Appl 144:516\u2013525","journal-title":"Expert Syst Appl"},{"key":"12024_CR8","doi-asserted-by":"crossref","unstructured":"Chen JN, Gao S, Sun HZ et al (2020) An end-to-end speech recognition algorithm based on attention mechanism. Syst Eng Soc China:640\u2013646","DOI":"10.23919\/CCC50068.2020.9189026"},{"key":"12024_CR9","doi-asserted-by":"crossref","unstructured":"Danish V, Alazab M, Sobia W, Hamad N, et al. IMCFN: Image-based malware classification using fine-tuned convolutional neural network architecture. Computer Networks, 2020:171\u2013177.","DOI":"10.1016\/j.comnet.2020.107138"},{"key":"12024_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.engappai.2020.103976","volume":"96","author":"M Fern\u00e1ndez-D\u00edaz","year":"2020","unstructured":"Fern\u00e1ndez-D\u00edaz M, Gallardo-Antol\u00edn A (2020) An attention long short-term memory based system for automatic classification of speech intelligibility. Eng Appl Artif Intell 96:1\u20138","journal-title":"Eng Appl Artif Intell"},{"key":"12024_CR11","doi-asserted-by":"publisher","first-page":"35055","DOI":"10.1007\/s11042-020-08883-w","volume":"79","author":"J Ganesh","year":"2020","unstructured":"Ganesh J, Hubert C (2020) Data augmentation for handwritten digit recognition using generative adversarial networks. Multimed Tools Appl 79:35055\u201335068","journal-title":"Multimed Tools Appl"},{"key":"12024_CR12","doi-asserted-by":"crossref","unstructured":"Girshick R, Donahue J, Darrell T, et al. Rich feature hierarchies for accurate object detection and semantic segmentation. 2014 IEEE conference on computer vision and pattern recognition, Columbus, OH, 2014, pp. 580\u2013587.","DOI":"10.1109\/CVPR.2014.81"},{"key":"12024_CR13","first-page":"742","volume-title":"Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks","author":"A Graves","year":"2016","unstructured":"Graves A, Gomez F (2016) Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. International Conference on Machine Learning, Hong Kong, pp 742\u2013748"},{"key":"12024_CR14","doi-asserted-by":"publisher","first-page":"114","DOI":"10.1016\/j.future.2020.11.022","volume":"117","author":"S Hakak","year":"2021","unstructured":"Hakak S, Alazab M, Khan S, \u2026 Khan WZ (2021) An ensemble machine learning approach through effective feature extraction to classify fake news. Futur Gener Comput Syst 117:114\u2013123","journal-title":"Futur Gener Comput Syst"},{"issue":"8","key":"12024_CR15","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"12024_CR16","first-page":"1672","volume-title":"Advances in joint CTC-attention based end-to-end speech recognition with a deep CNN encoder and RNN-LM","author":"T Hori","year":"2017","unstructured":"Hori T, Watanabe S, Zhang Y et al (2017) Advances in joint CTC-attention based end-to-end speech recognition with a deep CNN encoder and RNN-LM. IEEE International Conference, USA, pp 1672\u20131679"},{"issue":"1","key":"12024_CR17","doi-asserted-by":"publisher","first-page":"66","DOI":"10.2991\/ijcis.d.200120.002","volume":"13","author":"XH Huang","year":"2020","unstructured":"Huang XH, Qiao LS, Yu WT et al (2020) End-to-end sequence labeling via convolutional recurrent neural network with a connectionist temporal classification layer. Int J Comput Intell Syst 13(1):66\u201373","journal-title":"Int J Comput Intell Syst"},{"key":"12024_CR18","first-page":"682","volume-title":"Batch normalization: accelerating deep network training by reducing internal covariate shift","author":"S Ioffe","year":"2015","unstructured":"Ioffe S, Szegedy C (2015) Batch normalization: accelerating deep network training by reducing internal covariate shift. International Conference on Machine Learning, Lille Grand Palais, pp 682\u2013689"},{"key":"12024_CR19","doi-asserted-by":"crossref","unstructured":"Jabbari M, Khushaba RN, Nazarpour K (2020) EMG-based hand gesture classification with long short-term memory deep recurrent neural networks. Ann Conf Canadian Med Biol Eng Soc:3302\u20133305","DOI":"10.1109\/EMBC44109.2020.9175279"},{"key":"12024_CR20","doi-asserted-by":"crossref","unstructured":"Kim S, Hori T, Watanabe S. Joint CTC-attention based end-to-end speech recognition using multi-task learning. New Orleans: The 42nd IEEE International Conference on Acoustics, Speech and Signal Processing, 2017:798\u2013805.","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"12024_CR21","doi-asserted-by":"crossref","unstructured":"Liu W, Anguelov D, Erhan D, et al. SSD: Single shot multibox detector. European Conference on Computer Vision, 2016:21\u201337.","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"12024_CR22","doi-asserted-by":"crossref","unstructured":"Luong MT, Pham H, Manning CD (2015) Effective approaches to attention-based neural machine translation. Lisbon: Empirical Methods Nat Language Process:316\u2013325","DOI":"10.18653\/v1\/D15-1166"},{"key":"12024_CR23","doi-asserted-by":"crossref","unstructured":"Qu S, Xi Y, Ding S (2017) Visual attention based on long-short term memory model for image caption generation. Melbourne: Control Decis Conf:234\u2013239","DOI":"10.1109\/CCDC.2017.7979342"},{"key":"12024_CR24","doi-asserted-by":"crossref","unstructured":"Redmon J, Divvala S, Girshick R et al (2016) You only look once: unified, real-time object detection. Proc IEEE Conf Comput Vis Pattern Recognit:779\u2013788","DOI":"10.1109\/CVPR.2016.91"},{"issue":"6","key":"12024_CR25","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2017","unstructured":"Ren S, He K, Girshick R, \u2026 Sun J (2017 Jun) Faster R-CNN: towards real-time object detection with region proposal networks. IEEE Trans Pattern Anal Mach Intell 39(6):1137\u20131149","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"12024_CR26","first-page":"807","volume":"2018","author":"V Sitalakshmi","year":"2018","unstructured":"Sitalakshmi V, Mamoun A, Qing Y (2018) Use of data visualisation for zero-day malware detection. Security Commun Networks 2018:807\u2013816","journal-title":"Security Commun Networks"},{"key":"12024_CR27","first-page":"272","volume-title":"Rethinking the inception architecture for computer vision","author":"C Szegedy","year":"2016","unstructured":"Szegedy C, Vanhoucke V, Ioffe S et al (2016) Rethinking the inception architecture for computer vision. Computer Vision and Pattern Recognition, Las Vegas, pp 272\u2013281"},{"key":"12024_CR28","first-page":"626","volume-title":"Inception-v4, inception-resnet and the impact of residual connections on learning","author":"C Szegedy","year":"2017","unstructured":"Szegedy C, Ioffe S, Vanhoucke V et al (2017) Inception-v4, inception-resnet and the impact of residual connections on learning. The Thirty-First AAAI Conference on Artificial Intelligence, San Francisco, pp 626\u2013634"},{"key":"12024_CR29","doi-asserted-by":"crossref","unstructured":"Tian Z, Huang W, He T, et al. Detecting text in natural image with connectionist text proposal network. Springer, Cham, 2016. LNCS, vol. 9912, pp. 56\u201372.","DOI":"10.1007\/978-3-319-46484-8_4"},{"issue":"1","key":"12024_CR30","doi-asserted-by":"publisher","first-page":"1015","DOI":"10.1038\/s41467-020-14757-4","volume":"11","author":"ST Tsai","year":"2020","unstructured":"Tsai ST, Kuo EJ, Tiwary P (2020) Learning molecular dynamics with simple language model built upon long short-term memory neural network. Nat Commun 11(1):1015\u20131021","journal-title":"Nat Commun"},{"issue":"6","key":"12024_CR31","first-page":"473","volume":"29","author":"LL Wang","year":"2020","unstructured":"Wang LL, Wang BQ, Zhao PP et al (2020) Malware detection algorithm based on the attention mechanism and ResNet. Chin J Electron 29(6):473\u2013480","journal-title":"Chin J Electron"},{"issue":"1","key":"12024_CR32","first-page":"51","volume":"31","author":"HP Xiong","year":"2018","unstructured":"Xiong HP, Chen XX, Chen CW (2018) Text location in image based on convolution neural network. Electronic Sci Technol 31(1):51\u201359","journal-title":"Electronic Sci Technol"},{"key":"12024_CR33","doi-asserted-by":"crossref","unstructured":"Xu K, Li D, Cassimatis N et al (2018) LCANet: end-to-end lipreading with cascaded attention-CTC. Xi\u2019an: China Automatic Face Gesture Recogn:351\u2013360","DOI":"10.1109\/FG.2018.00088"},{"key":"12024_CR34","unstructured":"Xu MX, Du XY, Wang DH (2019) Super-resolution restoration of single vehicle image based on ESPCN-VISR model. Adv Sci Industry Res Center: Sci Eng Res Center:517\u2013528"},{"issue":"5","key":"12024_CR35","first-page":"20","volume":"28","author":"HT Xue","year":"2015","unstructured":"Xue HT, Yang JD, Tan KD (2015) Application of an improved BP neural network in handwriting recognition. Electronic Sci Technol 28(5):20\u201327","journal-title":"Electronic Sci Technol"},{"issue":"1","key":"12024_CR36","first-page":"124","volume":"29","author":"Z Yin","year":"2016","unstructured":"Yin Z, Tang CH, Zhang XX (2016) Image recognition based on improved sparse auto-encoder. Electronic Sci Technol 29(1):124\u2013127","journal-title":"Electronic Sci Technol"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-12024-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-022-12024-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-12024-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,3,29]],"date-time":"2022-03-29T11:23:11Z","timestamp":1648552991000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-022-12024-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,2,16]]},"references-count":36,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2022,3]]}},"alternative-id":["12024"],"URL":"https:\/\/doi.org\/10.1007\/s11042-022-12024-w","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,2,16]]},"assertion":[{"value":"11 February 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 April 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 January 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 February 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}