{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,5]],"date-time":"2025-02-05T14:40:45Z","timestamp":1738766445444,"version":"3.37.0"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,2,5]],"date-time":"2025-02-05T00:00:00Z","timestamp":1738713600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,2,5]],"date-time":"2025-02-05T00:00:00Z","timestamp":1738713600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62171470"],"award-info":[{"award-number":["62171470"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007847","name":"Natural Science Foundation of Jilin Province","doi-asserted-by":"publisher","award":["232300421240"],"award-info":[{"award-number":["232300421240"]}],"id":[{"id":"10.13039\/100007847","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010037","name":"Chongqing Municipal Youth Science and Technology Talent Training Project","doi-asserted-by":"publisher","award":["234200510019"],"award-info":[{"award-number":["234200510019"]}],"id":[{"id":"10.13039\/501100010037","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J AUDIO SPEECH MUSIC PROC."],"DOI":"10.1186\/s13636-025-00394-6","type":"journal-article","created":{"date-parts":[[2025,2,5]],"date-time":"2025-02-05T14:12:31Z","timestamp":1738764751000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A speech recognition method with enhanced transformer decoder"],"prefix":"10.1186","volume":"2025","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-1172-1152","authenticated-orcid":false,"given":"Hengbo","family":"Hu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tong","family":"Niu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhenhua","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,2,5]]},"reference":[{"issue":"8","key":"394_CR1","doi-asserted-by":"publisher","first-page":"10617","DOI":"10.1007\/s13369-023-07670-7","volume":"48","author":"S Nasr","year":"2023","unstructured":"S. Nasr, R. Duwairi, M. Quwaider, End-to-end speech recognition for arabic dialects. Arab. J. Sci. Eng. 48(8), 10617\u201310633 (2023)","journal-title":"Arab. J. Sci. Eng."},{"key":"394_CR2","unstructured":"B.H. Juang, L.R. Rabiner, Automatic speech recognition\u2013a brief history of the technology development, vol. 1, no. 67 (Georgia Institute of Technology, Atlanta Rutgers University and the University of California, Santa Barbara, 2005), p. 1"},{"key":"394_CR3","unstructured":"A.r. Mohamed, G.\u00a0Dahl, G.\u00a0Hinton, et\u00a0al., in Nips workshop on deep learning for speech recognition and related applications, Deep belief networks for phone recognition, vol.\u00a01 (Vancouver, Canada, 2009), p.\u00a039"},{"key":"394_CR4","doi-asserted-by":"crossref","unstructured":"G.\u00a0Hinton, L.\u00a0Deng, D.\u00a0Yu, G.E. Dahl, A.r. Mohamed, N.\u00a0Jaitly, A.\u00a0Senior, V.\u00a0Vanhoucke, P.\u00a0Nguyen, T.N. Sainath, et\u00a0al., Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups. IEEE Signal Proc. Mag. 29(6), 82\u201397 (2012)","DOI":"10.1109\/MSP.2012.2205597"},{"key":"394_CR5","doi-asserted-by":"crossref","unstructured":"A. Graves, S. Fern\u00e1ndez, F. Gomez, J. Schmidhuber, in Proceedings of the 23rd international conference on Machine learning, Connectionist temporal classification: Labelling unsegmented sequence data with recurrent neural networks (ACM,2006), pp. 369\u2013376","DOI":"10.1145\/1143844.1143891"},{"key":"394_CR6","unstructured":"D.\u00a0Amodei, S.\u00a0Ananthanarayanan, R.\u00a0Anubhai, J.\u00a0Bai, E.\u00a0Battenberg, C.\u00a0Case, J.\u00a0Casper, B.\u00a0Catanzaro, Q.\u00a0Cheng, G.\u00a0Chen, et\u00a0al., in International conference on machine learning, Deep speech 2: End-to-end speech recognition in english and mandarin (PMLR, 2016), pp. 173\u2013182"},{"key":"394_CR7","doi-asserted-by":"crossref","unstructured":"A.\u00a0Graves, Sequence transduction with recurrent neural networks (2012). arXiv preprint arXiv:1211.3711","DOI":"10.1007\/978-3-642-24797-2"},{"key":"394_CR8","doi-asserted-by":"crossref","unstructured":"A.\u00a0Graves, A.r. Mohamed, G.\u00a0Hinton, in 2013 IEEE international conference on acoustics, speech and signal processing, Speech recognition with deep recurrent neural networks (IEEE, 2013), pp. 6645\u20136649","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"394_CR9","doi-asserted-by":"crossref","unstructured":"W.\u00a0Chan, N.\u00a0Jaitly, Q.\u00a0Le, O.\u00a0Vinyals, in 2016 IEEE international conference on acoustics, speech and signal processing (ICASSP), Listen, attend and spell: A neural network for large vocabulary conversational speech recognition (IEEE, 2016), pp. 4960\u20134964","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"394_CR10","doi-asserted-by":"crossref","unstructured":"L.\u00a0Dong, S.\u00a0Xu, B.\u00a0Xu, in 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), Speech-transformer: A no-recurrence sequence-to-sequence model for speech recognition (IEEE, 2018), pp. 5884\u20135888","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"394_CR11","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A.N. Gomez, \u0141. Kaiser, I. Polosukhin, Attention is all you need. Adv. Neural Inf. Process. Syst. 30 (2017), pp. 5998\u20136008"},{"key":"394_CR12","doi-asserted-by":"crossref","unstructured":"Y. Zhao, C. Ni, C.C. Leung, S.R. Joty, E.S. Chng, B. Ma, in Interspeech, Cross attention with monotonic alignment for speech transformer (ISCA, 2020), pp. 5031\u20135035","DOI":"10.21437\/Interspeech.2020-1198"},{"key":"394_CR13","doi-asserted-by":"crossref","unstructured":"A.\u00a0Kannan, Y.\u00a0Wu, P.\u00a0Nguyen, T.N. Sainath, Z.\u00a0Chen, R.\u00a0Prabhavalkar, in 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), An analysis of incorporating an external language model into a sequence-to-sequence model (IEEE, 2018), pp. 1\u20135828","DOI":"10.1109\/ICASSP.2018.8462682"},{"key":"394_CR14","unstructured":"C.\u00a0Gulcehre, O.\u00a0Firat, K.\u00a0Xu, K.\u00a0Cho, L.\u00a0Barrault, H.C. Lin, F.\u00a0Bougares, H.\u00a0Schwenk, Y.\u00a0Bengio, On using monolingual corpora in neural machine translation (2015). arXiv preprint arXiv:1503.03535"},{"key":"394_CR15","doi-asserted-by":"crossref","unstructured":"A.\u00a0Sriram, H.\u00a0Jun, S.\u00a0Satheesh, A.\u00a0Coates, Cold fusion: Training seq2seq models together with language models (2017). arXiv preprint arXiv:1708.06426","DOI":"10.21437\/Interspeech.2018-1392"},{"key":"394_CR16","doi-asserted-by":"crossref","unstructured":"S.\u00a0Toshniwal, A.\u00a0Kannan, C.C. Chiu, Y.\u00a0Wu, T.N. Sainath, K.\u00a0Livescu, in 2018 IEEE spoken language technology workshop (SLT), A comparison of techniques for language model integration in encoder-decoder speech recognition (IEEE, 2018), pp. 369\u2013375","DOI":"10.1109\/SLT.2018.8639038"},{"key":"394_CR17","doi-asserted-by":"crossref","unstructured":"A.\u00a0Gulati, J.\u00a0Qin, C.C. Chiu, N.\u00a0Parmar, Y.\u00a0Zhang, J.\u00a0Yu, W.\u00a0Han, S.\u00a0Wang, Z.\u00a0Zhang, Y.\u00a0Wu, et\u00a0al., Conformer: Convolution-augmented transformer for speech recognition (2020). arXiv preprint arXiv:2005.08100","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"394_CR18","unstructured":"Y.\u00a0Peng, S.\u00a0Dalmia, I.\u00a0Lane, S.\u00a0Watanabe, in International Conference on Machine Learning, Branchformer: Parallel mlp-attention architectures to capture local and global context for speech recognition and understanding (PMLR, 2022), pp. 17627\u201317643"},{"key":"394_CR19","first-page":"9361","volume":"35","author":"S Kim","year":"2022","unstructured":"S. Kim, A. Gholami, A. Shaw, N. Lee, K. Mangalam, J. Malik, M.W. Mahoney, K. Keutzer, Squeezeformer: An efficient transformer for automatic speech recognition. Adv. Neural Inf. Process. Syst. 35, 9361\u20139373 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"394_CR20","doi-asserted-by":"crossref","unstructured":"D.S. Park, Y.\u00a0Zhang, C.C. Chiu, Y.\u00a0Chen, B.\u00a0Li, W.\u00a0Chan, Q.V. Le, Y.\u00a0Wu, in ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Specaugment on large scale datasets (IEEE, 2020), pp. 6879\u20136883","DOI":"10.1109\/ICASSP40776.2020.9053205"},{"key":"394_CR21","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, in Proceedings of the IEEE conference on computer vision and pattern recognition, Deep residual learning for image recognition (IEEE, 2016), pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"394_CR22","doi-asserted-by":"crossref","unstructured":"D.\u00a0Bahdanau, J.\u00a0Chorowski, D.\u00a0Serdyuk, P.\u00a0Brakel, Y.\u00a0Bengio, in 2016 IEEE international conference on acoustics, speech and signal processing (ICASSP), End-to-end attention-based large vocabulary speech recognition (IEEE, 2016), pp. 4945\u20134949","DOI":"10.1109\/ICASSP.2016.7472618"},{"key":"394_CR23","doi-asserted-by":"crossref","unstructured":"P.\u00a0Shaw, J.\u00a0Uszkoreit, A.\u00a0Vaswani, Self-attention with relative position representations (2018). arXiv preprint arXiv:1803.02155","DOI":"10.18653\/v1\/N18-2074"},{"key":"394_CR24","unstructured":"S.\u00a0Ji, Y.\u00a0Xie, H.\u00a0Gao, A mathematical view of attention models in deep learning (Texas A &M University, 2019)"},{"key":"394_CR25","doi-asserted-by":"crossref","unstructured":"C.F.R. Chen, Q. Fan, R. Panda, in Proceedings of the IEEE\/CVF international conference on computer vision, Crossvit: Cross-attention multi-scale vision transformer for image classification (IEEE, 2021), pp. 357\u2013366","DOI":"10.1109\/ICCV48922.2021.00041"},{"issue":"2","key":"394_CR26","doi-asserted-by":"publisher","first-page":"279","DOI":"10.3233\/IDA-183879","volume":"23","author":"J Graovac","year":"2019","unstructured":"J. Graovac, M. Mladenovi\u0107, I. Tanasijevi\u0107, NgramSPD: Exploring optimal n-gram model for sentiment polarity detection in different languages. Intell. Data Anal. 23(2), 279\u2013296 (2019)","journal-title":"Intell. Data Anal."},{"key":"394_CR27","unstructured":"J.\u00a0Devlin, Bert: Pre-training of deep bidirectional transformers for language understanding (2018). arXiv preprint arXiv:1810.04805"},{"key":"394_CR28","unstructured":"A. Radford, Improving language understanding by generative pre-training (2018). OpenAI Blog. https:\/\/cdn.openai.com\/research-covers\/language-unsupervised\/language_understanding_paper.pdf"},{"issue":"8","key":"394_CR29","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"A. Radford, J. Wu, R. Child, D. Luan, D. Amodei, I. Sutskever et al., Language models are unsupervised multitask learners. OpenAI Blog 1(8), 9 (2019)","journal-title":"OpenAI Blog"},{"key":"394_CR30","unstructured":"B.\u00a0Mann, N.\u00a0Ryder, M.\u00a0Subbiah, J.\u00a0Kaplan, P.\u00a0Dhariwal, A.\u00a0Neelakantan, P.\u00a0Shyam, G.\u00a0Sastry, A.\u00a0Askell, S.\u00a0Agarwal, et\u00a0al., Language models are few-shot learners. 1 (2020). arXiv preprint arXiv:2005.14165"},{"key":"394_CR31","unstructured":"J.\u00a0Achiam, S.\u00a0Adler, S.\u00a0Agarwal, L.\u00a0Ahmad, I.\u00a0Akkaya, F.L. Aleman, D.\u00a0Almeida, J.\u00a0Altenschmidt, S.\u00a0Altman, S.\u00a0Anadkat, et\u00a0al., GPT-4 technical report (2023). arXiv preprint arXiv:2303.08774"},{"key":"394_CR32","unstructured":"D.\u00a0Bahdanau, K.\u00a0Cho, Y.\u00a0Bengio, Neural machine translation by jointly learning to align and translate (2014). arXiv preprint arXiv:1409.0473"},{"key":"394_CR33","doi-asserted-by":"crossref","unstructured":"H.\u00a0Bu, J.\u00a0Du, X.\u00a0Na, B.\u00a0Wu, H.\u00a0Zheng, in 2017 20th conference of the oriental chapter of the international coordinating committee on speech databases and speech I\/O systems and assessment (O-COCOSDA), Aishell-1: An open-source mandarin speech corpus and a speech recognition baseline (IEEE, 2017), pp. 1\u20135","DOI":"10.1109\/ICSDA.2017.8384449"},{"key":"394_CR34","unstructured":"J.\u00a0Du, X.\u00a0Na, X.\u00a0Liu, H.\u00a0Bu, Aishell-2: Transforming mandarin asr research into industrial scale (2018). arXiv preprint arXiv:1808.10583"},{"key":"394_CR35","unstructured":"Z.\u00a0Fan, S.\u00a0Zhou, B.\u00a0Xu, Unsupervised pre-training for sequence to sequence speech recognition (2019). arXiv preprint arXiv:1910.12418"},{"key":"394_CR36","first-page":"3586","volume":"2015","author":"T Ko","year":"2015","unstructured":"T. Ko, V. Peddinti, D. Povey, S. Khudanpur, in Interspeech. Audio augmentation for speech recognition 2015, 3586 (2015)","journal-title":"Audio augmentation for speech recognition"},{"issue":"1","key":"394_CR37","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"N. Srivastava, G. Hinton, A. Krizhevsky, I. Sutskever, R. Salakhutdinov, Dropout: A simple way to prevent neural networks from overfitting. J. Mach. Learn. Res. 15(1), 1929\u20131958 (2014)","journal-title":"J. Mach. Learn. Res."},{"key":"394_CR38","unstructured":"P.K. Diederik, Adam: A method for stochastic optimization (2014). arXiv preprint arXiv:1412.6980."},{"key":"394_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/1687-4722-2014-14","volume":"2014","author":"J Sta\u0161","year":"2014","unstructured":"J. Sta\u0161, J. Juh\u00e1r, D. Hl\u00e1dek, Classification of heterogeneous text data for robust domain-specific language modeling. EURASIP J. Audio Speech Music Process. 2014, 1\u201312 (2014)","journal-title":"EURASIP J. Audio Speech Music Process."},{"key":"394_CR40","doi-asserted-by":"crossref","unstructured":"T. Hori, S. Watanabe, J.R. Hershey, in Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), Joint CTC\/attention decoding for end-to-end speech recognition (ACL,2017), pp. 518\u2013529","DOI":"10.18653\/v1\/P17-1048"}],"container-title":["EURASIP Journal on Audio, Speech, and Music Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00394-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s13636-025-00394-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00394-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,5]],"date-time":"2025-02-05T14:12:54Z","timestamp":1738764774000},"score":1,"resource":{"primary":{"URL":"https:\/\/asmp-eurasipjournals.springeropen.com\/articles\/10.1186\/s13636-025-00394-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,5]]},"references-count":40,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["394"],"URL":"https:\/\/doi.org\/10.1186\/s13636-025-00394-6","relation":{},"ISSN":["1687-4722"],"issn-type":[{"value":"1687-4722","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,2,5]]},"assertion":[{"value":"19 August 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 January 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 February 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"6"}}