{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:24:57Z","timestamp":1757618697947,"version":"3.44.0"},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T00:00:00Z","timestamp":1751673600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T00:00:00Z","timestamp":1751673600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.62277016"],"award-info":[{"award-number":["No.62277016"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s11760-025-04471-3","type":"journal-article","created":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T06:43:53Z","timestamp":1751697833000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["LRAT: Low-Rank Adaptive Transformer for end-to-end speech recognition"],"prefix":"10.1007","volume":"19","author":[{"given":"Xinmin","family":"Cheng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuyang","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruiqin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,7,5]]},"reference":[{"issue":"2","key":"4471_CR1","doi-asserted-by":"publisher","first-page":"257","DOI":"10.1109\/5.18626","volume":"77","author":"LR Rabiner","year":"1989","unstructured":"Rabiner, L.R.: A tutorial on hidden Markov models and selected applications in speech recognition. Proc. IEEE 77(2), 257\u2013286 (1989)","journal-title":"Proc. IEEE"},{"issue":"2","key":"4471_CR2","doi-asserted-by":"publisher","first-page":"179","DOI":"10.1207\/s15516709cog1402_1","volume":"14","author":"JL Elman","year":"1990","unstructured":"Elman, J.L.: Finding structure in time. Cogn. Sci. 14(2), 179\u2013211 (1990)","journal-title":"Cogn. Sci."},{"key":"4471_CR3","doi-asserted-by":"crossref","unstructured":"Chan, W., Jaitly, N., Le, Q., Vinyals, O.: Listen, attend and spell: A neural network for large vocabulary conversational speech recognition. In Proceedings of the 2016 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 4960\u20134964, (2016)","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"4471_CR4","doi-asserted-by":"crossref","unstructured":"Kim, S., Hori, T., Watanabe, S.: Joint CTC-attention based end-to-end speech recognition using multi-task learning. In Proceedings of the 2017 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 4835\u20134839, (2017)","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"4471_CR5","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, ?., Polosukhin, I.: Attention is all you need. Advances in neural information processing systems, 30, (2017)"},{"key":"4471_CR6","doi-asserted-by":"crossref","unstructured":"Dong, L., Xu, S., Xu, B.: Speech-transformer: a no-recurrence sequence-to-sequence model for speech recognition. In Proceedings of the 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 5884\u20135888, (2018)","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"4471_CR7","unstructured":"Li, J., Wang, X., Li, Y., others: The speechtransformer for large-scale mandarin chinese speech recognition. In Proceedings of the ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7095\u20137099, (2019)"},{"key":"4471_CR8","unstructured":"Wang, S., Li, B.Z., Khabsa, M., Fang, H., Ma, H.: Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768, (2020)"},{"key":"4471_CR9","doi-asserted-by":"crossref","unstructured":"Wang, R., Bai, Q., Ao, J., Zhou, L., Xiong, Z., Wei, Z., Zhang, Y., Ko, T., Li, H.: Lighthubert: Lightweight and configurable speech representation learning with once-for-all hidden-unit bert. arXiv preprint arXiv:2203.15610, (2022)","DOI":"10.21437\/Interspeech.2022-10269"},{"key":"4471_CR10","doi-asserted-by":"crossref","unstructured":"Peng, Y., Kim, K., Wu, F., Sridhar, P., Watanabe, S.: Structured pruning of self-supervised pre-trained models for speech recognition and understanding. ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), (pages 1\u20135) (2023)","DOI":"10.1109\/ICASSP49357.2023.10095335"},{"key":"4471_CR11","doi-asserted-by":"crossref","unstructured":"Lv, X., Zhang, P., Li, S., Gan, G., Sun, Y.: Lightformer: Light-weight transformer using svd-based weight transfer and parameter sharing. Findings of the Association for Computational Linguistics: ACL 2023, (pages 10323\u201310335) (2023)","DOI":"10.18653\/v1\/2023.findings-acl.656"},{"key":"4471_CR12","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2023.111207","volume":"152","author":"C Liu","year":"2024","unstructured":"Liu, C., Chen, X., Lin, J., Hu, P., Wang, J., Geng, X.: Evolving masked low-rank transformer for long text understanding. Appl. Soft Comput. 152, 111207 (2024)","journal-title":"Appl. Soft Comput."},{"key":"4471_CR13","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Kingsbury, B., Sindhwani, V., Arisoy, E., Ramabhadran, B.: Low-rank matrix factorization for deep neural network training with high-dimensional output targets. In Proceedings of the 2013 IEEE international conference on acoustics, speech and signal processing, pp. 6655\u20136659, (2013)","DOI":"10.1109\/ICASSP.2013.6638949"},{"key":"4471_CR14","doi-asserted-by":"crossref","unstructured":"Winata, G.I., Cahyawijaya, S., Lin, Z., Liu, Z., Fung, P.: Lightweight and efficient end-to-end speech recognition using low-rank transformer. In Proceedings of the ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6144\u20136148, (2020)","DOI":"10.1109\/ICASSP40776.2020.9053878"},{"key":"4471_CR15","unstructured":"Kuchaiev, O., Ginsburg, B.: Factorization tricks for LSTM networks. arXiv preprint arXiv:1703.10722, (2017)"},{"key":"4471_CR16","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In Proceedings of the European conference on computer vision, pp. 213\u2013229, (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"4471_CR17","doi-asserted-by":"crossref","unstructured":"Peng, Z., Huang, W., Gu, S., Xie, L., Wang, Y., Jiao, J., Ye, Q.: Conformer: Local features coupling global representations for visual recognition. In Proceedings of the IEEE\/CVF international conference on computer vision, pp. 367\u2013376, (2021)","DOI":"10.1109\/ICCV48922.2021.00042"},{"key":"4471_CR18","first-page":"30392","volume":"34","author":"T Xiao","year":"2021","unstructured":"Xiao, T., Singh, M., Mintun, E., Darrell, T., Doll\u00e1r, P., Girshick, R.: Early convolutions help transformers see better. Adv. Neural. Inf. Process. Syst. 34, 30392\u201330400 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4471_CR19","unstructured":"Hassani, A., Walton, S., Shah, N., Abuduweili, A., Li, J., Shi, H.: Escaping the big data paradigm with compact transformers. arXiv preprint arXiv:2104.05704, (2021)"},{"key":"4471_CR20","unstructured":"Xu, W., Wan, Y.: ELA: Efficient local attention for deep convolutional neural networks. arXiv preprint arXiv:2403.01123, (2024)"},{"key":"4471_CR21","doi-asserted-by":"crossref","unstructured":"Sukhbaatar, S., Grave, E., Bojanowski, P., Joulin, A.: Adaptive attention span in transformers. arXiv preprint arXiv:1905.07799, (2019)","DOI":"10.18653\/v1\/P19-1032"},{"key":"4471_CR22","unstructured":"Kumar, S., Parker, J., Naderian, P.: Adaptive transformers in RL. arXiv preprint arXiv:2004.03761, (2020)"},{"key":"4471_CR23","unstructured":"Ioannides, G., Chadha, A., Elkins, A.: Gaussian adaptive attention is all you need: Robust contextual representations across multiple modalities. CoRR, (2024)"},{"key":"4471_CR24","unstructured":"Lou, C., Jia, Z., Zheng, Z., Tu, K.: Sparser is faster and less is more: Efficient sparse attention for long-range transformers. arXiv preprint arXiv:2406.16747, (2024)"},{"key":"4471_CR25","unstructured":"Bahdanau, D., Cho, K., Bengio, Y.: Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473, (2014)"},{"key":"4471_CR26","doi-asserted-by":"crossref","unstructured":"Bu, H., Du, J., Na, X., Wu, B., Zheng, H.: Aishell-1: An open-source mandarin speech corpus and a speech recognition baseline. In Proceedings of the 2017 20th conference of the oriental chapter of the international coordinating committee on speech databases and speech I\/O systems and assessment (O-COCOSDA), pp. 1\u20135, (2017)","DOI":"10.1109\/ICSDA.2017.8384449"},{"key":"4471_CR27","unstructured":"Wang, D., Zhang, X.: Thchs-30: A free chinese speech corpus. arXiv preprint arXiv:1512.01882, (2015)"},{"key":"4471_CR28","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: Librispeech: an asr corpus based on public domain audio books. In Proceedings of the 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 5206\u20135210, (2015)","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"4471_CR29","doi-asserted-by":"crossref","unstructured":"Hori, T., Watanabe, S., Zhang, Y., Chan, W.: Advances in joint CTC-attention based end-to-end speech recognition with a deep CNN encoder and RNN-LM. arXiv preprint arXiv:1706.02737, (2017)","DOI":"10.21437\/Interspeech.2017-1296"},{"key":"4471_CR30","doi-asserted-by":"crossref","unstructured":"Li, M., Cao, Y., Zhou, W., Liu, M.: Framewise Supervised Training Towards End-to-End Speech Recognition Models: First Results. In Proceedings of the Interspeech, pp. 1641\u20131645, (2019)","DOI":"10.21437\/Interspeech.2019-1117"},{"key":"4471_CR31","unstructured":"Liang, S., Yan, W.: Multilingual speech recognition based on the end-to-end framework. Multimed. Tools Appl., (2022)"},{"key":"4471_CR32","unstructured":"Ravanelli, M., Parcollet, T., Plantinga, P., Rouhe, A., Cornell, S., Lugosch, L., Subakan, C., Dawalatabad, N., Heba, A., Zhong, J.: SpeechBrain: A general-purpose speech toolkit. arXiv preprint arXiv:2106.04624, (2021)"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04471-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-025-04471-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04471-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,7]],"date-time":"2025-09-07T01:05:43Z","timestamp":1757207143000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-025-04471-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,5]]},"references-count":32,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["4471"],"URL":"https:\/\/doi.org\/10.1007\/s11760-025-04471-3","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"type":"print","value":"1863-1703"},{"type":"electronic","value":"1863-1711"}],"subject":[],"published":{"date-parts":[[2025,7,5]]},"assertion":[{"value":"19 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 June 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 June 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 July 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no Competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}}],"article-number":"849"}}