{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,7]],"date-time":"2025-10-07T08:31:44Z","timestamp":1759825904564,"version":"3.30.2"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,10,23]],"date-time":"2024-10-23T00:00:00Z","timestamp":1729641600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,23]],"date-time":"2024-10-23T00:00:00Z","timestamp":1729641600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s10772-024-10160-2","type":"journal-article","created":{"date-parts":[[2024,10,23]],"date-time":"2024-10-23T15:02:19Z","timestamp":1729695739000},"page":"1111-1120","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Sub-layer feature fusion applied to transformer model for automatic speech recognition"],"prefix":"10.1007","volume":"27","author":[{"given":"Darong","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangguang","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangyong","family":"Wei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fahad","family":"Anwaar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaxin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenxiao","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiafeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,23]]},"reference":[{"key":"10160_CR1","unstructured":"Ba, J. L., Kiros, J. R., & Hinton, G. E. (2016) Layer normalization. arXiv:1607"},{"key":"10160_CR2","doi-asserted-by":"crossref","unstructured":"Bai, Y., Yi, J., Tao, J., Tian, Z., Wen, Z., & Zhang, S. (2020) Listen attentively, and spell once: Whole sentence generation via a non-autoregressive architecture for low-latency speech recognition. arXiv:2005.04862","DOI":"10.21437\/Interspeech.2020-1600"},{"key":"10160_CR3","unstructured":"Beijing DataTang Technology Co., L.: Aidatatang 200zh, a free Chinese Mandarin speech corpus"},{"key":"10160_CR4","doi-asserted-by":"crossref","unstructured":"Bu, H., Du, J., Na, X., Wu, B., & Zheng, H. (2017). Aishell-1: An open-source Mandarin speech corpus and a speech recognition baseline. In 2017 20th Conference of the Oriental chapter of international committee for coordination and standardization of speech databases and assessment techniques, (O-COCOSDA 2017) (pp. 1\u20135)","DOI":"10.1109\/ICSDA.2017.8384449"},{"key":"10160_CR5","doi-asserted-by":"crossref","unstructured":"Chan, W., Jaitly, N., Le, Q., & Vinyals, O. (2016). Listen, attend and spell: A neural network for large vocabulary conversational speech recognition. In IEEE international conference on acoustics, speech and signal processing-proceedings (ICASSP) (Vol. May, pp. 4960\u20134964).","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"10160_CR6","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1109\/LSP.2020.3044547","volume":"28","author":"N Chen","year":"2021","unstructured":"Chen, N., Watanabe, S., Villalba, J., \u017belasko, P., & Dehak, N. (2021). Non-autoregressive transformer for speech recognition. IEEE Signal Processing Letters, 28, 121\u2013125.","journal-title":"IEEE Signal Processing Letters"},{"key":"10160_CR7","doi-asserted-by":"publisher","first-page":"572","DOI":"10.1109\/TASLP.2018.2888814","volume":"27","author":"S Deena","year":"2019","unstructured":"Deena, S., Hasan, M., Doulaty, M., Saz, O., & Hain, T. (2019). Recurrent neural network language model adaptation for multi-genre broadcast speech recognition and alignment. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 27, 572\u2013582.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10160_CR8","doi-asserted-by":"crossref","unstructured":"Domhan, T. (2018). How much attention do you need? A granular analysis of neural machine translation architectures. In ACL Anthology. https:\/\/aclanthology.org\/P18-1167","DOI":"10.18653\/v1\/P18-1167"},{"key":"10160_CR9","doi-asserted-by":"crossref","unstructured":"Dong, L., Wang, F., & Xu, B. (2019) Self-attention aligner: A latency-control end-to-end model for ASR using self-attention network and chunk-hopping. In 2019 IEEE international conference on acoustics, speech and signal processing (ICASSP 2019) (pp. 5656\u20135660). IEEE.","DOI":"10.1109\/ICASSP.2019.8682954"},{"key":"10160_CR10","doi-asserted-by":"crossref","unstructured":"Dong, L., Xu, S., & Xu, B. (2018) Speech-transformer: A no-recurrence sequence-to-sequence model for speech recognition. In  IEEE international conference on acoustics, speech and signal processing\u2014Proceedings (ICASSP)(Vol. April, pp. 5884\u20135888).","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"10160_CR11","doi-asserted-by":"crossref","unstructured":"Dong, L., Zhou, S., Chen, W., & Xu, B. (2018). Extending recurrent neural aligner for streaming end-to-end speech recognition in Mandarin. arXiv:1806.06342","DOI":"10.21437\/Interspeech.2018-1086"},{"key":"10160_CR12","doi-asserted-by":"crossref","unstructured":"Dou, Z.-Y., Tu, Z., Wang, X., Shi, S., & Zhang, T. (2018) Exploiting deep representations for neural machine translation. In Proceedings of the 2018 conference on empirical methods in natural language processing, (EMNLP 2018) (pp. 4253\u20134262).","DOI":"10.18653\/v1\/D18-1457"},{"key":"10160_CR13","doi-asserted-by":"crossref","unstructured":"Dou, Z.-Y., Tu, Z., Wang, X., Wang, L., Shi, S., & Zhang, T. (2019) Dynamic layer aggregation for neural machine translation with routing-by-agreement. In Proceedings of the AAAI conference on artificial intelligence (Vol. 33, pp. 86\u201393).","DOI":"10.1609\/aaai.v33i01.330186"},{"key":"10160_CR14","doi-asserted-by":"crossref","unstructured":"Fujita, Y., Watanabe, S., Omachi, M., & Chan, X. (2020). Insertion-based modeling for end-to-end automatic speech recognition. arXiv:2005.13211","DOI":"10.21437\/Interspeech.2020-1619"},{"key":"10160_CR15","unstructured":"Gehring, J., Auli, M., Grangier, D., Yarats, D., & Dauphin, Y. N. (2017) Convolutional sequence to sequence learning. In 34th international conference on machine learning, (ICML 2017) (vol. 3, pp. 2029\u20132042)."},{"key":"10160_CR16","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, A.-R., & Hinton, G. (2013) Speech recognition with deep recurrent neural networks. In IEEE international conference on acoustics, speech and signal processing\u2014Proceedings (ICASSP) (pp. 6645\u20136649).","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"10160_CR17","doi-asserted-by":"publisher","first-page":"2222","DOI":"10.1109\/TNNLS.2016.2582924","volume":"28","author":"K Greff","year":"2017","unstructured":"Greff, K., Srivastava, R. K., Koutn\u00edk, J., Steunebrink, B. R., & Schmidhuber, J. (2017). LSTM: A search space odyssey. IEEE Transactions on Neural Networks and Learning Systems, 28, 2222\u20132232.","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10160_CR18","doi-asserted-by":"crossref","unstructured":"Gulati, A., Qin, J., Chiu, C.-C., Parmar, N., Zhang, Y., Yu, J., Han, W., Wang, S., Zhang, Z., Wu, Y., & Pang, R. (2020). Conformer: Convolution-augmented transformer for speech recognition. In Proceedings of the annual conference of the international speech communication association, INTERSPEECH (Vol. 2020-October, pp. 5036\u20135040).","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"10160_CR19","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (Vol. 2016-December, pp. 770\u2013778).","DOI":"10.1109\/CVPR.2016.90"},{"key":"10160_CR20","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., Van Der Maaten, L., & Weinberger, K. Q. (2017). Densely connected convolutional networks. In 2017 IEEE conference on computer vision and pattern recognition (CVPR) (pp. 2261\u20132269).","DOI":"10.1109\/CVPR.2017.243"},{"key":"10160_CR21","doi-asserted-by":"crossref","unstructured":"Jie, Z., Ying, C., Wang, X., Peng, L., & Wei, X. (2016). Deep recurrent models with fast-forward connections for neural machine translation. Transactions of the Association for Computational Linguistics, 4.","DOI":"10.1162\/tacl_a_00105"},{"key":"10160_CR22","doi-asserted-by":"publisher","first-page":"257","DOI":"10.1109\/89.568732","volume":"5","author":"BH Juang","year":"1997","unstructured":"Juang, B. H., Hou, W., & Lee, C. H. (1997). Minimum classification error rate methods for speech recognition. IEEE Transactions on Speech & Audio Processing, 5, 257\u2013265.","journal-title":"IEEE Transactions on Speech & Audio Processing"},{"key":"10160_CR23","doi-asserted-by":"crossref","unstructured":"Kaneko, M., Mita, M., Kiyono, S., Suzuki, J., & Inui, K. (2020). Encoder-decoder models can benefit from pre-trained masked language models in grammatical error correction. In Proceedings of the 58th annual meeting of the association for computational linguistics (Vol. July, pp. 4248\u20134254).","DOI":"10.18653\/v1\/2020.acl-main.391"},{"key":"10160_CR24","doi-asserted-by":"crossref","unstructured":"Karita, S., Chen, N., Hayashi, T., Hori, T., Inaguma, H., Jiang, Z., Someki, M., Soplin, N. E. Y., Yamamoto, R., & Wang, X. (2019). A comparative study on transformer vs RNN in speech applications. In 2019 IEEE automatic speech recognition and understanding workshop (ASRU) (pp. 449\u2013456). IEEE.","DOI":"10.1109\/ASRU46091.2019.9003750"},{"key":"10160_CR25","doi-asserted-by":"crossref","unstructured":"Kim, S., Hori, T., & Watanabe, S. (2017). Joint CTC-attention based end-to-end speech recognition using multi-task learning. In  IEEE International conference on acoustics, speech and signal processing\u2014proceedings (ICASSP) (pp. 4835\u20134839).","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"10160_CR26","unstructured":"Kingma, D. P., & Ba, J. L. (2015). Adam: A method for stochastic optimization. In 3rd International conference on learning representations, (ICLR 2015)\u2014Conference track proceedings"},{"key":"10160_CR27","doi-asserted-by":"publisher","first-page":"646","DOI":"10.1109\/TASLP.2019.2959721","volume":"28","author":"R Li","year":"2020","unstructured":"Li, R., Wang, X., Mallidi, S. H., Watanabe, S., Hori, T., & Hermansky, H. (2020). Multi-stream end-to-end speech recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 28, 646\u2013655.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10160_CR28","unstructured":"Liu, X., Wang, L., Wong, D. F., Ding, L., Chao, L. S., & Tu, Z. (2020) Understanding and improving encoder layer fusion in sequence-to-sequence learning. arXiv:2012.14768"},{"issue":"1","key":"10160_CR29","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s10772-024-10092-x","volume":"27","author":"O Mahmoudi","year":"2024","unstructured":"Mahmoudi, O., Filali-Bouami, M., & Benchat, M. (2024). Speech recognition based on the transformer\u2019s multi-head attention in Arabic. International Journal of Speech Technology, 27(1), 211\u2013223.","journal-title":"International Journal of Speech Technology"},{"key":"10160_CR30","unstructured":"Meng, F., Lu, Z., Tu, Z., Li, H., & Liu, Q. (2016). A deep memory-based architecture for sequence-to-sequence learning. In The international conference on learning representations (ICLR)."},{"key":"10160_CR31","doi-asserted-by":"publisher","first-page":"1452","DOI":"10.1109\/TASLP.2020.2987752","volume":"28","author":"H Miao","year":"2020","unstructured":"Miao, H., Cheng, G., Zhang, P., & Yan, Y. (2020). Online hybrid CTC\/attention end-to-end automatic speech recognition architecture. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 28, 1452\u20131465.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10160_CR32","doi-asserted-by":"crossref","unstructured":"Miao, Y., Gowayyed, M., Na, X., Ko, T., Metze, F., & Waibel, A. (2016). An empirical exploration of CTC acoustic models. In 2016 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 2623\u20132627). IEEE.","DOI":"10.1109\/ICASSP.2016.7472152"},{"key":"10160_CR33","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1109\/TASLP.2018.2871755","volume":"27","author":"T Moriya","year":"2019","unstructured":"Moriya, T., Tanaka, T., Shinozaki, T., Watanabe, S., & Duh, K. (2019). Evolution-strategy-based automation of system development for high-performance speech recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 27, 77\u201388.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10160_CR34","unstructured":"Paszke, A., Gross, S., Chintala, S., Chanan, G., Yang, E., Devito, Z., Lin, Z., Desmaison, A., Antiga, L., & Lerer, A. (2017) Automatic differentiation in pytorch"},{"key":"10160_CR35","doi-asserted-by":"crossref","unstructured":"Peters, M. E., Neumann, M., Iyyer, M., Gardner, M., Clark, C., Lee, K., & Zettlemoyer, L. (2018). Deep contextualized word representations. In 2018 conference of the North American chapter of the association for computational linguistics: Human language technologies\u2014Proceedings of the conference (NAACL HLT 2018) (Vol. 1, pp. 2227\u20132237).","DOI":"10.18653\/v1\/N18-1202"},{"key":"10160_CR36","doi-asserted-by":"crossref","unstructured":"Povey, D., & Woodland, P. C. (2002). Minimum phone error and I-smoothing for improved discriminative training. In IEEE international conference on acoustics, speech and signal processing\u2014Proceedings (ICASSP) (Vol. 1, pp. 105\u2013108).","DOI":"10.1109\/ICASSP.2002.5743665"},{"key":"10160_CR37","doi-asserted-by":"crossref","unstructured":"Povey, D., Peddinti, V., Galvez, D., Ghahremani, P., Manohar, V., Na, X., Wang, Y., & Khudanpur, S. (2016). Purely sequence-trained neural networks for ASR based on lattice-free MMI. In Interspeech (pp. 2751\u20132755).","DOI":"10.21437\/Interspeech.2016-595"},{"key":"10160_CR38","unstructured":"Rigoll, G., & Neukirchen, C. (1997). A new approach to hybrid HMM\/ANN speech recognition using mutual information neural networks. In Advances in neural information processing systems (pp. 772\u2013778)."},{"key":"10160_CR39","doi-asserted-by":"crossref","unstructured":"Shabber, S. M., & Bansal, M. (2024). Temporal feature-based approaches for enhancing phoneme boundary detection and masking in speech. International Journal of Speech Technology, 1\u201312.","DOI":"10.1007\/s10772-024-10117-5"},{"key":"10160_CR40","doi-asserted-by":"crossref","unstructured":"Shen, Q., Guo, M., Huang, Y., & Ma, J. (2024). Attentional multi-feature fusion for spoofing-aware speaker verification. International Journal of Speech Technology, 1\u201311.","DOI":"10.1007\/s10772-024-10112-w"},{"key":"10160_CR41","doi-asserted-by":"publisher","first-page":"960","DOI":"10.1109\/TASLP.2019.2907015","volume":"27","author":"K Shimada","year":"2019","unstructured":"Shimada, K., Bando, Y., Mimura, M., Itoyama, K., Yoshii, K., & Kawahara, T. (2019). Unsupervised speech enhancement based on multichannel NMF-informed beamforming for noise-robust automatic speech recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 27, 960\u2013971.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10160_CR42","doi-asserted-by":"crossref","unstructured":"Tian, Z., Yi, J., Bai, Y., Tao, J., Zhang, S., & Wen, Z. (2020). Synchronous transformers for end-to-end speech recognition. In 2020 IEEE international conference on acoustics, speech and signal processing (ICASSP 2020) (pp. 7884\u20137888).","DOI":"10.1109\/ICASSP40776.2020.9054260"},{"key":"10160_CR43","doi-asserted-by":"crossref","unstructured":"Tian, Z., Yi, J., Tao, J., Bai, Y., & Wen, Z. (2019). Self-attention transducers for end-to-end speech recognition. In Proceedings of the annual conference of the international speech communication association, INTERSPEECH (Vol. 2019-September, pp. 4395\u20134399).","DOI":"10.21437\/Interspeech.2019-2203"},{"key":"10160_CR44","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, L., & Polosukhin, I. (2017). Attention is all you need. In Advances in neural information processing systems (pp. 5999\u20136009)."},{"key":"10160_CR45","doi-asserted-by":"crossref","unstructured":"Wang, Q., Li, B., Xiao, T., Zhu, J., Li, C., Wong, D. F., & Chao, L. S. (2019). Learning deep transformer models for machine translation. In 57th Annual meeting of the association for computational linguistics, proceedings of the conference (ACL 2019) (pp. 1810\u20131822).","DOI":"10.18653\/v1\/P19-1176"},{"key":"10160_CR46","unstructured":"Wang, Q., Li, F., Xiao, T., Li, Y., Li, Y., & Zhu, J. (2018) Multi-layer representation fusion for neural machine translation. In Proceedings of the 27th international conference on computational linguistics (pp. 3015\u20133026)."},{"key":"10160_CR47","doi-asserted-by":"crossref","unstructured":"Wang, W., & Tu, Z. (2020) Rethinking the value of transformer components. arXiv:2011.03803","DOI":"10.18653\/v1\/2020.coling-main.529"},{"key":"10160_CR48","doi-asserted-by":"crossref","unstructured":"Wang, X., Wang, L., Tu, Z., & Shi, S. (2019) Exploiting sentential context for neural machine translation. In 57th Annual meeting of the association for computational linguistics, proceedings of the conference (ACL 2019) (pp. 6197\u20136203).","DOI":"10.18653\/v1\/P19-1624"},{"key":"10160_CR49","doi-asserted-by":"crossref","unstructured":"Xiong, H., He, Z., Hu, X., & Wu, H. (2018) Multi-channel encoder for neural machine translation. In 32nd AAAI conference on artificial intelligence (AAAI 2018) (pp. 4962\u20134969).","DOI":"10.1609\/aaai.v32i1.11929"},{"issue":"1","key":"10160_CR50","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1007\/s10772-024-10091-y","volume":"27","author":"C Yang","year":"2024","unstructured":"Yang, C., Yu, X., & Huang, S. (2024). Conditional denoising diffusion implicit model for speech enhancement. International Journal of Speech Technology, 27(1), 201\u2013209.","journal-title":"International Journal of Speech Technology"},{"key":"10160_CR51","doi-asserted-by":"crossref","unstructured":"Yu, F., Wang, D., Shelhamer, E., & Darrell, T. (2018) Deep layer aggregation. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 2403\u20132412).","DOI":"10.1109\/CVPR.2018.00255"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10160-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10160-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10160-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T09:59:27Z","timestamp":1734343167000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10160-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,23]]},"references-count":51,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["10160"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10160-2","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2024,10,23]]},"assertion":[{"value":"6 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 October 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All the authors do not have any possible Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}