{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T15:27:23Z","timestamp":1772119643559,"version":"3.50.1"},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,3,31]],"date-time":"2025-03-31T00:00:00Z","timestamp":1743379200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,3,31]],"date-time":"2025-03-31T00:00:00Z","timestamp":1743379200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"name":"NATIONAL NATURAL SCIENCE FUND PROJECT","award":["U19A2063"],"award-info":[{"award-number":["U19A2063"]}]},{"name":"JILIN PROVINCIAL SCIENCE AND TECHNOLOGY DEVELOPMENT PLAN PROJECT","award":["20220201149GX"],"award-info":[{"award-number":["20220201149GX"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Discov Computing"],"DOI":"10.1007\/s10791-025-09516-2","type":"journal-article","created":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T22:44:13Z","timestamp":1743547453000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["New unit dot product similarity method and parallelized greedy soup algorithm in the end-to-end automatic speech recognition"],"prefix":"10.1007","volume":"28","author":[{"given":"Wei","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zifan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yiming","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunyi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,31]]},"reference":[{"key":"9516_CR1","unstructured":"Jiang Y, Yu J, Yang W, et al. Nextformer: a ConvNeXt augmented conformer for end-to-end speech recognition. arXiv preprint arXiv:2206.14747, 2022."},{"key":"9516_CR2","doi-asserted-by":"crossref","unstructured":"Zeyer A, Bahar P, Irie K, et al. A comparison of transformer and lstm encoder decoder models for asr. 2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU). IEEE, 2019: 8\u201315.","DOI":"10.1109\/ASRU46091.2019.9004025"},{"key":"9516_CR3","unstructured":"Wortsman M, Ilharco G, Gadre SY, et al. Model soups: averaging weights of multiple fine-tuned models improves accuracy without increasing inference time. International Conference on Machine Learning. PMLR, 2022: 23965\u201323998."},{"key":"9516_CR4","unstructured":"Lu Y, Li Z, He D, et al. Understanding and improving transformer from a multi-particle dynamic system point of view. arXiv preprint arXiv:1906.02762, 2019."},{"key":"9516_CR5","doi-asserted-by":"crossref","unstructured":"Chen X, Wu Y, Wang Z, et al. Developing real-time streaming transformer transducer for speech recognition on large-scale dataset. ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2021: 5904\u20135908.","DOI":"10.1109\/ICASSP39728.2021.9413535"},{"key":"9516_CR6","unstructured":"Ren X, Zhu H, Wei L, et al. Improving mandarin speech recognition with block-augmented transformer. arXiv preprint arXiv:2207.11697, 2022."},{"key":"9516_CR7","doi-asserted-by":"crossref","unstructured":"Dai Z, Yang Z, Yang Y, et al. Transformer-xl: attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860, 2019.","DOI":"10.18653\/v1\/P19-1285"},{"key":"9516_CR8","doi-asserted-by":"crossref","unstructured":"Li J. Recent advances in end-to-end automatic speech recognition. APSIPA Trans Signal Inf Process. arXiv preprint arXiv: 2111.01690, 2022.","DOI":"10.1561\/116.00000050"},{"key":"9516_CR9","doi-asserted-by":"crossref","unstructured":"Tjandra A, Liu C, Zhang F, et al. Deja-vu: double feature presentation and iterated loss in deep transformer networks. ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2020: 6899\u20136903.","DOI":"10.1109\/ICASSP40776.2020.9052964"},{"key":"9516_CR10","doi-asserted-by":"crossref","unstructured":"Chen Z, Jain M, Wang Y, et al. End-to-end contextual speech recognition using class language models and a token passing decoder. 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2019: 6186\u20136190.","DOI":"10.1109\/ICASSP.2019.8683573"},{"key":"9516_CR11","doi-asserted-by":"crossref","unstructured":"Tian J, Yu J, Weng C, et al. Consistent training and decoding for end-to-end speech recognition using lattice-free mmi. 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2022: 7782\u20137786.","DOI":"10.1109\/ICASSP43922.2022.9746579"},{"key":"9516_CR12","unstructured":"Weng S Y, Chen B. Effective decoder masking for transformer based end-to-end speech recognition. arXiv preprint arXiv:2010.14764, 2020."},{"key":"9516_CR13","doi-asserted-by":"crossref","unstructured":"Shon S, Ali A, Glass J. Convolutional neural network and language embeddings for end-to-end dialect recognition. Proc. Odyssey 2018 The Speaker and Language Recognition Workshop. 2018: 98\u2013104.","DOI":"10.21437\/Odyssey.2018-14"},{"key":"9516_CR14","doi-asserted-by":"crossref","unstructured":"Winata G I, Cahyawijaya S, Lin Z, et al. Lightweight and efficient end-to-end speech recognition using low-rank transformer. ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2020: 6144\u20136148.","DOI":"10.1109\/ICASSP40776.2020.9053878"},{"key":"9516_CR15","doi-asserted-by":"crossref","unstructured":"Xue B, Yu J, Xu J, et al. Bayesian transformer language models for speech recognition. ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2021: 7378\u20137382.","DOI":"10.1109\/ICASSP39728.2021.9414046"},{"key":"9516_CR16","doi-asserted-by":"crossref","unstructured":"Han W, Zhang Z, Zhang Y, et al. ContextNet: improving convolutional neural networks for automatic speech recognition with global context. arXiv preprint arXiv:2005.03191, 2020.","DOI":"10.21437\/Interspeech.2020-2059"},{"key":"9516_CR17","doi-asserted-by":"crossref","unstructured":"Zhang S, Lei M, Yan Z. Investigation of transformer based spelling correction model for CTC-based end-to-end mandarin speech recognition. INTERSPEECH. 2019: 2180\u20132184.","DOI":"10.21437\/Interspeech.2019-1290"},{"key":"9516_CR18","doi-asserted-by":"crossref","unstructured":"Yoshimura T, Hayashi T, Takeda K, et al. End-to-end automatic speech recognition integrated with ctc-based voice activity detection. ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2020: 6999\u20137003.","DOI":"10.1109\/ICASSP40776.2020.9054358"},{"key":"9516_CR19","doi-asserted-by":"crossref","unstructured":"Seki H, Hori T, Watanabe S, et al. Vectorized Beam Search for CTC-Attention-Based Speech Recognition. INTERSPEECH. 2019: 3825\u20133829.","DOI":"10.21437\/Interspeech.2019-2860"},{"key":"9516_CR20","doi-asserted-by":"crossref","unstructured":"Song X, Wu Z, Huang Y, et al. Non-autoregressive transformer ASR with CTC-enhanced decoder input. ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2021: 5894\u20135898.","DOI":"10.1109\/ICASSP39728.2021.9414694"},{"key":"9516_CR21","doi-asserted-by":"crossref","unstructured":"Higuchi Y, Inaguma H, Watanabe S, et al. Improved mask-CTC for non-autoregressive end-to-end ASR. \/\/ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2021: 8363\u20138367.","DOI":"10.1109\/ICASSP39728.2021.9414198"},{"key":"9516_CR22","doi-asserted-by":"crossref","unstructured":"Pham NQ, Nguyen TS, Niehues J, et al. Very deep self-attention networks for end-to-end speech recognition. arXiv preprint arXiv:1904.13377, 2019.","DOI":"10.21437\/Interspeech.2019-2702"},{"key":"9516_CR23","doi-asserted-by":"crossref","unstructured":"Gulati A, Qin J, Chiu CC, et al. Conformer: convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100, 2020.","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"9516_CR24","doi-asserted-by":"crossref","unstructured":"Yao Z, Wu D, Wang X, et al. Wenet: production oriented streaming and non-streaming end-to-end speech recognition toolkit. arXiv preprint arXiv:2102.01547, 2021.","DOI":"10.21437\/Interspeech.2021-1983"},{"key":"9516_CR25","unstructured":"Vaswani A, Shazeer N, Parmar N, et al. Attention is all you need. Advances in neural information processing systems, arXiv preprint arXiv:1706.03762, 2017."},{"key":"9516_CR26","doi-asserted-by":"crossref","unstructured":"Bu H, Du J, Na X, et al. Aishell-1: An open-source mandarin speech corpus and a speech recognition baseline. 2017 20th conference of the oriental chapter of the international coordinating committee on speech databases and speech I\/O systems and assessment (O-COCOSDA). IEEE, 2017: 1\u20135.","DOI":"10.1109\/ICSDA.2017.8384449"},{"key":"9516_CR27","doi-asserted-by":"crossref","unstructured":"Zhang B, Lv H, Guo P, et al. Wenetspeech: A 10000+ hours multi-domain mandarin corpus for speech recognition. ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2022: 6182\u20136186.","DOI":"10.1109\/ICASSP43922.2022.9746682"},{"key":"9516_CR28","doi-asserted-by":"crossref","first-page":"1897","DOI":"10.1109\/TASLP.2021.3082299","volume":"29","author":"Y Bai","year":"2021","unstructured":"Bai Y, Yi J, Tao J, Tian Z, Wen Z, Zhang S. Fast end-toend speech recognition via non-autoregressive models and crossmodal knowledge transferring from bert. IEEE\/ACM Trans Audio Speech Lang Process. 2021;29:1897\u2013911.","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"9516_CR29","doi-asserted-by":"crossref","unstructured":"Zhang C-F, Liu Y, Zhang T-H, Chen S-L, Chen F, Yin X-C. Non-autoregressive transformer with unified bidirectional decoder for automatic speech recognition. arXiv preprint arXiv:2109.06684, 2021.","DOI":"10.1109\/ICASSP43922.2022.9746903"},{"key":"9516_CR30","doi-asserted-by":"crossref","unstructured":"Tian Z, Yi J, Tao J, Bai Y, Zhang S, Wen Z, Liu X. Tsnat: two-step non-autoregressvie transformer models for speech recognition. arXiv preprint arXiv:2104.01522, 2021.","DOI":"10.21437\/Interspeech.2020-2086"},{"key":"9516_CR31","unstructured":"Yao Z, Guo L, Yang X, Kang W, Kuang F, Jin Z, Lin L, Povey D. Zipformer: a faster and better encoder for automatic speech recognition. arXiv preprint arXiv:2310.11230, 2023."},{"key":"9516_CR32","unstructured":"Zhao M. Research on privatized vertical search engine based on improved TF-IDF and topic clustering. Sichuan Normal University, 2024."}],"container-title":["Discover Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10791-025-09516-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10791-025-09516-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10791-025-09516-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T22:44:36Z","timestamp":1743547476000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10791-025-09516-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,31]]},"references-count":32,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["9516"],"URL":"https:\/\/doi.org\/10.1007\/s10791-025-09516-2","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-4431357\/v1","asserted-by":"object"}]},"ISSN":["2948-2992"],"issn-type":[{"value":"2948-2992","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,31]]},"assertion":[{"value":"16 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 March 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 March 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All authors will abide by academic ethics and agree to publish a statement of compliance with academic ethics.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"The authors declare no competing interests.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"25"}}