{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:09:28Z","timestamp":1757617768660,"version":"3.44.0"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T00:00:00Z","timestamp":1743120000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T00:00:00Z","timestamp":1743120000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62466030","U21B2027"],"award-info":[{"award-number":["62466030","U21B2027"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"High-tech Industry Development Project of Yunnan Province","award":["201606"],"award-info":[{"award-number":["201606"]}]},{"name":"Major Science and Technology Special Project of Yunnan Province","award":["202302AD080003"],"award-info":[{"award-number":["202302AD080003"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s13042-025-02594-0","type":"journal-article","created":{"date-parts":[[2025,3,31]],"date-time":"2025-03-31T01:56:52Z","timestamp":1743386212000},"page":"5707-5725","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Robust speech recognition method based on dense time\u2013frequency convolution and bispectral refinement enhancement"],"prefix":"10.1007","volume":"16","author":[{"given":"Wenjun","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ling","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengtao","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxin","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shangbin","family":"Mo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Linqing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,28]]},"reference":[{"issue":"8","key":"2594_CR1","doi-asserted-by":"publisher","first-page":"1018","DOI":"10.3390\/sym11081018","volume":"11","author":"D Wang","year":"2019","unstructured":"Wang D, Wang X, Lv S (2019) An overview of end-to-end automatic speech recognition. Symmetry 11(8):1018","journal-title":"Symmetry"},{"issue":"1","key":"2594_CR2","doi-asserted-by":"publisher","first-page":"452","DOI":"10.1561\/116.00000050","volume":"11","author":"J Li","year":"2022","unstructured":"Li J et al (2022) Recent advances in end-to-end automatic speech recognition. APSIPA Trans Signal Inf Process 11(1):452","journal-title":"APSIPA Trans Signal Inf Process"},{"issue":"2","key":"2594_CR3","doi-asserted-by":"publisher","first-page":"475","DOI":"10.1007\/s10772-023-10033-0","volume":"26","author":"M Dua","year":"2023","unstructured":"Dua M, Akanksha Dua S (2023) Noise robust automatic speech recognition: review and analysis. Int J Speech Technol 26(2):475\u2013519","journal-title":"Int J Speech Technol"},{"key":"2594_CR4","doi-asserted-by":"publisher","first-page":"131858","DOI":"10.1109\/ACCESS.2021.3112535","volume":"9","author":"S Alharbi","year":"2021","unstructured":"Alharbi S, Alrazgan M, Alrashed A, Alnomasi T, Almojel R, Alharbi R, Alharbi S, Alturki S, Alshehri F, Almojil M (2021) Automatic speech recognition: systematic literature review. IEEE Access 9:131858\u2013131876. https:\/\/doi.org\/10.1109\/ACCESS.2021.3112535","journal-title":"IEEE Access"},{"key":"2594_CR5","doi-asserted-by":"crossref","unstructured":"Prasad A, Jyothi P, Velmurugan R (2021) An investigation of end-to-end models for robust speech recognition. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6893\u20136897","DOI":"10.1109\/ICASSP39728.2021.9414027"},{"key":"2594_CR6","doi-asserted-by":"crossref","unstructured":"Manohar V, Chen S-J, Wang Z, Fujita Y, Watanabe S, Khudanpur S (2019) Acoustic modeling for overlapping speech recognition: JHU CHiME-5 challenge system. In: ICASSP 2019-2019 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6665\u20136669","DOI":"10.1109\/ICASSP.2019.8682556"},{"key":"2594_CR7","doi-asserted-by":"crossref","unstructured":"Park DS, Chan W, Zhang Y, Chiu C-C, Zoph B, Cubuk ED, Le QV (2019) Specaugment: a simple data augmentation method for automatic speech recognition. arXiv preprint arXiv:1904.08779","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"2594_CR8","doi-asserted-by":"crossref","unstructured":"Seltzer ML, Yu D, Wang Y (2013) An investigation of deep neural networks for noise robust speech recognition. In: 2013 IEEE international conference on acoustics, speech and signal processing. IEEE, pp 7398\u20137402","DOI":"10.1109\/ICASSP.2013.6639100"},{"key":"2594_CR9","doi-asserted-by":"crossref","unstructured":"Malek J, Zdansky J, Cerva P (2017) Robust automatic recognition of speech with background music. In: 2017 IEEE International conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5210\u20135214","DOI":"10.1109\/ICASSP.2017.7953150"},{"key":"2594_CR10","doi-asserted-by":"crossref","unstructured":"Iwamoto K, Ochiai T, Delcroix M, Ikeshita R, Sato H, Araki S, Katagiri S (2022) How bad are artifacts?: analyzing the impact of speech enhancement errors on ASR. arXiv preprint arXiv:2201.06685, pp 5418\u20135422","DOI":"10.21437\/Interspeech.2022-318"},{"key":"2594_CR11","unstructured":"Radford A, Kim JW, Xu T, Brockman G, McLeavey C, Sutskever I (2023) Robust speech recognition via large-scale weak supervision. In: International conference on machine learning. PMLR, pp 28492\u201328518"},{"issue":"4","key":"2594_CR12","doi-asserted-by":"publisher","first-page":"796","DOI":"10.1109\/TASLP.2016.2528171","volume":"24","author":"Z-Q Wang","year":"2016","unstructured":"Wang Z-Q, Wang D (2016) A joint training framework for robust automatic speech recognition. IEEE\/ACM Trans Audio Speech Lang Process 24(4):796\u2013806","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"2594_CR13","unstructured":"Ma D, Hou N, Xu H, Chng ES (2021) Multitask-based joint learning approach to robust asr for radio communication speech. In: 2021 Asia-Pacific signal and information processing association annual summit and conference (APSIPA ASC). IEEE, pp 497\u2013502"},{"key":"2594_CR14","doi-asserted-by":"crossref","unstructured":"Sawata R, Kashiwagi Y, Takahashi S (2022) Improving character error rate is not equal to having clean speech: speech enhancement for asr systems with black-box acoustic models. In: ICASSP 2022-2022 IEEE international conference on acoustics, speech and signal processing (ICASSP. IEEE), pp 991\u2013995","DOI":"10.1109\/ICASSP43922.2022.9746398"},{"key":"2594_CR15","doi-asserted-by":"crossref","unstructured":"Wang Q, Moreno IL, Saglam M, Wilson K, Chiao A, Liu R, He Y, Li W, Pelecanos J, Nika M et al (2020) VoiceFilter-Lite: Streaming targeted voice separation for on-device speech recognition. arXiv preprint arXiv:2009.04323","DOI":"10.21437\/Interspeech.2020-1193"},{"key":"2594_CR16","unstructured":"Shi H, Wang L, Li S, Fan C, Dang J, Kawahara T (2021) Spectrograms fusion-based end-to-end robust automatic speech recognition. In: 2021 Asia-Pacific signal and information processing association annual summit and conference (APSIPA ASC). IEEE, pp 438\u2013442"},{"key":"2594_CR17","doi-asserted-by":"crossref","unstructured":"Braun S, Gamper H (2022) Effect of noise suppression losses on speech distortion and ASR performance. In: ICASSP 2022-2022 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 996\u20131000","DOI":"10.1109\/ICASSP43922.2022.9746489"},{"issue":"01","key":"2594_CR18","doi-asserted-by":"publisher","first-page":"2550001","DOI":"10.1142\/S0219467825500019","volume":"25","author":"C Jannu","year":"2025","unstructured":"Jannu C, Vanambathina SD (2025) An overview of speech enhancement based on deep learning techniques. Int J Image Graph 25(01):2550001","journal-title":"Int J Image Graph"},{"key":"2594_CR19","doi-asserted-by":"crossref","unstructured":"Jannu C, Vanambathina SD (2023) An attention based densely connected u-net with convolutional gru for speech enhancement. In: 2023 Third international conference on artificial intelligence and signal processing (AISP). IEEE, pp 1\u20135","DOI":"10.1109\/AISP57993.2023.10134933"},{"issue":"12","key":"2594_CR20","doi-asserted-by":"publisher","first-page":"7467","DOI":"10.1007\/s00034-023-02455-7","volume":"42","author":"C Jannu","year":"2023","unstructured":"Jannu C, Vanambathina SD (2023) Multi-stage progressive learning-based speech enhancement using time\u2013frequency attentive squeezed temporal convolutional networks. Circuits Syst Signal Process 42(12):7467\u20137493","journal-title":"Circuits Syst Signal Process"},{"issue":"1","key":"2594_CR21","first-page":"1195","volume":"45","author":"C Jannu","year":"2023","unstructured":"Jannu C, Vanambathina SD (2023) Dct based densely connected convolutional gru for real-time speech enhancement. J Intell Fuzzy Syst 45(1):1195\u20131208","journal-title":"J Intell Fuzzy Syst"},{"key":"2594_CR22","first-page":"1","volume":"4","author":"V Parisae","year":"2024","unstructured":"Parisae V, Bhavanam SN (2024) Multi scale encoder-decoder network with time frequency attention and s-tcn for single channel speech enhancement. J Intell Fuzzy Syst 4:1\u201316","journal-title":"J Intell Fuzzy Syst"},{"key":"2594_CR23","doi-asserted-by":"publisher","DOI":"10.1142\/S0219467825500676","author":"V Parisae","year":"2024","unstructured":"Parisae V, Nagakishore Bhavanam S (2024) Stacked u-net with time\u2013frequency attention and deep connection net for single channel speech enhancement. Int J Image Graph. https:\/\/doi.org\/10.1142\/S0219467825500676","journal-title":"Int J Image Graph"},{"issue":"1","key":"2594_CR24","first-page":"1","volume":"14","author":"C Jannu","year":"2023","unstructured":"Jannu C, Vanambathina SD (2023) Convolutional transformer based local and global feature learning for speech enhancement. Int J Adv Comput Sci Appl 14(1):1\u201313","journal-title":"Int J Adv Comput Sci Appl"},{"key":"2594_CR25","doi-asserted-by":"crossref","unstructured":"Vanambathina SD, Nandyala S, Jannu C, Sirisha Devi J, Yechuri S, Parisae V (2024) Speech enhancement using u-net-based progressive learning with squeeze-tcn. In: International conference on advances in distributed computing and machine learning. Springer, pp 419\u2013432","DOI":"10.1007\/978-981-97-3523-5_31"},{"key":"2594_CR26","doi-asserted-by":"crossref","unstructured":"Lea C, Vidal R, Reiter A, Hager GD (2016) Temporal convolutional networks: a unified approach to action segmentation. In: Computer vision\u2013ECCV 2016 workshops: Amsterdam, The Netherlands, October 8\u201310 and 15-16, 2016, proceedings, Part III, vol 14. Springer, pp 47\u201354","DOI":"10.1007\/978-3-319-49409-8_7"},{"key":"2594_CR27","doi-asserted-by":"crossref","unstructured":"Sonning S, Sch\u00fcldt C, Erdogan H, Wisdom S (2020) Performance study of a convolutional time-domain audio separation network for real-time speech denoising. In: ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 831\u2013835","DOI":"10.1109\/ICASSP40776.2020.9053846"},{"key":"2594_CR28","doi-asserted-by":"crossref","unstructured":"Kishore V, Tiwari N, Paramasivam P (2020) Improved speech enhancement using TCN with multiple encoder-decoder layers. In: Interspeech, pp 4531\u20134535","DOI":"10.21437\/Interspeech.2020-3122"},{"key":"2594_CR29","doi-asserted-by":"publisher","unstructured":"Fulchiero R, Spanias AS (1993) Speech enhancement using the bispectrum. In: 1993 IEEE international conference on acoustics, speech, and signal processing, vol 4. IEEE, pp 488\u2013491. https:\/\/doi.org\/10.1109\/ICASSP.1993.319701","DOI":"10.1109\/ICASSP.1993.319701"},{"key":"2594_CR30","doi-asserted-by":"crossref","unstructured":"Zhang L, Wang M (2020) Multi-scale TCN: exploring better temporal DNN model for causal speech enhancement. In: Interspeech, pp 2672\u20132676","DOI":"10.21437\/Interspeech.2020-1104"},{"key":"2594_CR31","unstructured":"Jia X, Li D (2022) TFCN: temporal-frequential convolutional network for single-channel speech enhancement. arXiv preprint arXiv:2201.00480"},{"key":"2594_CR32","doi-asserted-by":"publisher","first-page":"198","DOI":"10.1109\/TASLP.2020.3039600","volume":"29","author":"C Fan","year":"2021","unstructured":"Fan C, Yi J, Tao J, Tian Z, Liu B, Wen Z (2021) Gated recurrent fusion with joint training framework for robust end-to-end speech recognition. IEEE\/ACM Trans Audio Speech Lang Process 29:198\u2013209","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"2594_CR33","doi-asserted-by":"crossref","unstructured":"Hu Y, Hou N, Chen C, Chng ES (2022) Interactive feature fusion for end-to-end noise-robust speech recognition. In: ICASSP 2022-2022 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6292\u20136296","DOI":"10.1109\/ICASSP43922.2022.9746314"},{"key":"2594_CR34","doi-asserted-by":"crossref","unstructured":"Hu Y, Hou N, Chen C, Chng ES (2023) Dual-path style learning for end-to-end noise-robust speech recognition. arXiv preprint arXiv:2203.14838","DOI":"10.21437\/Interspeech.2023-101"},{"key":"2594_CR35","doi-asserted-by":"crossref","unstructured":"Zhuang X, Zhang L, Zhang Z, Qian Y, Wang M (2022) Coarse-grained attention fusion with joint training framework for complex speech enhancement and end-to-end speech recognition. In: INTERSPEECH, pp 3794\u20133798","DOI":"10.21437\/Interspeech.2022-698"},{"key":"2594_CR36","doi-asserted-by":"crossref","unstructured":"Wang W, Mo S, Dong L, Yu Z, Guo J, Huang Y (2024) Dgsrn: Noise-robust speech recognition method with dual-path gated spectral refinement network. In: Proc. Interspeech 2024, pp 5018\u20135022","DOI":"10.21437\/Interspeech.2024-1796"},{"key":"2594_CR37","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2021.108382","volume":"185","author":"S Shahnawazuddin","year":"2022","unstructured":"Shahnawazuddin S, Kumar A, Kumar V, Kumar S, Ahmad W (2022) Robust children\u2019s speech recognition in zero resource condition. Appl Acoust 185:108382","journal-title":"Appl Acoust"},{"key":"2594_CR38","doi-asserted-by":"publisher","unstructured":"Alhussein G, Alkhodari M, Khandoker AH, Hadjileontiadis LJ (2023) Deep bispectral analysis of conversational speech towards emotional climate recognition. In: 2023 IEEE international conference on artificial intelligence in engineering and technology (IICAIET). https:\/\/doi.org\/10.1109\/IICAIET59451.2023.10291940. IEEE, pp 170\u2013175","DOI":"10.1109\/IICAIET59451.2023.10291940"},{"key":"2594_CR39","doi-asserted-by":"publisher","first-page":"1829","DOI":"10.1109\/TASLP.2021.3079813","volume":"29","author":"A Li","year":"2021","unstructured":"Li A, Liu W, Zheng C, Fan C, Li X (2021) Two heads are better than one: a two-stage complex spectral mapping approach for monaural speech enhancement. IEEE\/ACM Trans Audio Speech Lang Process 29:1829\u20131843","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"2594_CR40","doi-asserted-by":"crossref","unstructured":"Zhang B, Wu D, Peng Z, Song X, Yao Z, Lv H, Xie L, Yang C, Pan F, Niu J (2022) Wenet 2.0: More productive end-to-end speech recognition toolkit. arXiv preprint arXiv:2203.15455","DOI":"10.21437\/Interspeech.2022-483"},{"key":"2594_CR41","doi-asserted-by":"crossref","unstructured":"Lu H, Li N, Song T, Wang L, Dang J, Wang X, Zhang S (2023) Speech and noise dual-stream spectrogram refine network with speech distortion loss for robust speech recognition. In: ICASSP 2023-2023 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 1\u20135","DOI":"10.1109\/ICASSP49357.2023.10095872"},{"key":"2594_CR42","doi-asserted-by":"crossref","unstructured":"Yamamoto R, Song E, Kim J-M (2020) Parallel WaveGAN: a fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram. In: ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6199\u20136203","DOI":"10.1109\/ICASSP40776.2020.9053795"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-025-02594-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-025-02594-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-025-02594-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T10:59:50Z","timestamp":1757156390000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-025-02594-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,28]]},"references-count":42,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["2594"],"URL":"https:\/\/doi.org\/10.1007\/s13042-025-02594-0","relation":{},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"type":"print","value":"1868-8071"},{"type":"electronic","value":"1868-808X"}],"subject":[],"published":{"date-parts":[[2025,3,28]]},"assertion":[{"value":"13 November 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 February 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 March 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interests"}}]}}