{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T13:43:02Z","timestamp":1784554982763,"version":"3.55.0"},"reference-count":70,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,2,25]],"date-time":"2026-02-25T00:00:00Z","timestamp":1771977600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,25]],"date-time":"2026-02-25T00:00:00Z","timestamp":1771977600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-026-21428-x","type":"journal-article","created":{"date-parts":[[2026,2,25]],"date-time":"2026-02-25T22:43:07Z","timestamp":1772059387000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Speech enhancement using neural free attention with multi-stage squeeze temporal convolutional networks"],"prefix":"10.1007","volume":"85","author":[{"given":"Chaitanya","family":"Jannu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Manaswini","family":"Burra","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sunny Dayal","family":"Vanambathina","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Veeraswamy","family":"Parisae","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,2,25]]},"reference":[{"key":"21428_CR1","unstructured":"Bai S, Kolter JZ, Koltun V (2018) An empirical evaluation of generic convolutional and recurrent networks for sequence modeling. arXiv:1803.01271"},{"key":"21428_CR2","doi-asserted-by":"crossref","unstructured":"Brandstein M, Ward D (2001) Microphone arrays: signal processing techniques and applications. Springer Science & Business Media","DOI":"10.1007\/978-3-662-04619-7"},{"key":"21428_CR3","doi-asserted-by":"crossref","unstructured":"Chen M, Zhang Q, Song Q et al (2023) Neural-free attention for monaural speech enhancement towards voice user interface for consumer electronics. IEEE Trans Consum Electron","DOI":"10.1109\/TCE.2023.3254507"},{"key":"21428_CR4","unstructured":"Fan C, Tao J, Liu B et al (2020) Deep attention fusion feature for speech separation with end-to-end post-filter method. arXiv:2003.07544"},{"key":"21428_CR5","unstructured":"Fu SW, Liao CF, Tsao Y et al (2019) Metricgan: generative adversarial networks based black-box metric scores optimization for speech enhancement. In: International conference on machine learning. PMLR, pp 2031\u20132041"},{"issue":"3","key":"21428_CR6","doi-asserted-by":"publisher","first-page":"399","DOI":"10.1109\/TCE.2018.2867801","volume":"64","author":"M Fukui","year":"2018","unstructured":"Fukui M, Watanabe T, Kanazawa M (2018) Sound source separation for plural passenger speech recognition in smart mobility system. IEEE Trans Consum Electron 64(3):399\u2013405","journal-title":"IEEE Trans Consum Electron"},{"key":"21428_CR7","doi-asserted-by":"crossref","unstructured":"Giri R, Isik U, Krishnaswamy A (2019) Attention wave-u-net for speech enhancement. In: 2019 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA). IEEE, pp 249\u2013253","DOI":"10.1109\/WASPAA.2019.8937186"},{"issue":"6","key":"21428_CR8","doi-asserted-by":"publisher","first-page":"982","DOI":"10.1109\/TASLP.2015.2416653","volume":"23","author":"K Han","year":"2015","unstructured":"Han K, Wang Y, Wang D et al (2015) Learning spectral mapping for speech dereverberation and denoising. IEEE\/ACM Trans Audio Speech Lang Process 23(6):982\u2013992","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR9","doi-asserted-by":"crossref","unstructured":"Hao X, Su X, Wen S et al (2020) Masking and inpainting: a two-stage speech enhancement approach for low snr and non-stationary noise. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 6959\u20136963","DOI":"10.1109\/ICASSP40776.2020.9053188"},{"key":"21428_CR10","doi-asserted-by":"publisher","first-page":"2149","DOI":"10.1109\/LSP.2020.3040693","volume":"27","author":"TA Hsieh","year":"2020","unstructured":"Hsieh TA, Wang HM, Lu X et al (2020) Wavecrn: an efficient convolutional recurrent neural network for end-to-end speech enhancement. IEEE Signal Process Lett 27:2149\u20132153","journal-title":"IEEE Signal Process Lett"},{"key":"21428_CR11","doi-asserted-by":"crossref","unstructured":"Hu J, Shen L, Sun G (2018) Squeeze-and-excitation networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7132\u20137141","DOI":"10.1109\/CVPR.2018.00745"},{"key":"21428_CR12","doi-asserted-by":"crossref","unstructured":"Jannu C, Vanambathina SD (2023) An attention based densely connected u-net with convolutional gru for speech enhancement. In: 2023 3rd International conference on Artificial Intelligence and Signal Processing (AISP). IEEE, pp 1\u20135","DOI":"10.1109\/AISP57993.2023.10134933"},{"key":"21428_CR13","doi-asserted-by":"crossref","unstructured":"Jannu C, Vanambathina SD (2023) Dct based densely connected convolutional gru for real-time speech enhancement. J Intell Fuzzy Syst (Preprint):1\u201314","DOI":"10.14569\/IJACSA.2023.0140181"},{"key":"21428_CR14","doi-asserted-by":"crossref","unstructured":"Jannu C, Vanambathina SD (2023) An overview of speech enhancement based on deep learning techniques. Int J Image Graph, 2550001","DOI":"10.1142\/S0219467825500019"},{"key":"21428_CR15","doi-asserted-by":"crossref","unstructured":"Jannu C, Vanambathina SD (2023) Self-attention-based convolutional gru for enhancement of adversarial speech examples. Int J Image Graph, 2450053","DOI":"10.1142\/S0219467824500530"},{"issue":"2","key":"21428_CR16","doi-asserted-by":"publisher","first-page":"125","DOI":"10.1109\/TCE.2020.2986003","volume":"66","author":"T Kawase","year":"2020","unstructured":"Kawase T, Okamoto M, Fukutomi T et al (2020) Speech enhancement parameter adjustment to maximize accuracy of automatic speech recognition. IEEE Trans Consum Electron 66(2):125\u2013133","journal-title":"IEEE Trans Consum Electron"},{"issue":"5","key":"21428_CR17","doi-asserted-by":"publisher","first-page":"770","DOI":"10.1109\/LSP.2019.2905660","volume":"26","author":"J Kim","year":"2019","unstructured":"Kim J, Hahn M (2019) Speech enhancement using a two-stage network for an efficient boosting strategy. IEEE Signal Process Lett 26(5):770\u2013774","journal-title":"IEEE Signal Process Lett"},{"key":"21428_CR18","doi-asserted-by":"crossref","unstructured":"Kim J, El-Khamy M, Lee J (2020) T-gsa: transformer with gaussian-weighted self-attention for speech enhancement. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 6649\u20136653","DOI":"10.1109\/ICASSP40776.2020.9053591"},{"key":"21428_CR19","unstructured":"Kingma DP, Ba J (2014) Adam: a method for stochastic optimization. arXiv:1412.6980"},{"key":"21428_CR20","doi-asserted-by":"crossref","unstructured":"Kishore V, Tiwari N, Paramasivam P (2020) Improved speech enhancement using tcn with multiple encoder-decoder layers. In: Interspeech, pp 4531\u20134535","DOI":"10.21437\/Interspeech.2020-3122"},{"key":"21428_CR21","doi-asserted-by":"crossref","unstructured":"Koizumi Y, Yatabe K, Delcroix M et al (2020) Speech enhancement using self-adaptation and multi-head self-attention. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 181\u2013185","DOI":"10.1109\/ICASSP40776.2020.9053214"},{"key":"21428_CR22","unstructured":"Koyama Y, Vuong T, Uhlich S et al (2020) Exploring the best loss function for dnn-based low-latency speech enhancement with temporal convolutional networks. arXiv:2005.11611"},{"issue":"12","key":"21428_CR23","doi-asserted-by":"publisher","first-page":"2251","DOI":"10.1109\/TASLP.2016.2602549","volume":"24","author":"M Krawczyk-Becker","year":"2016","unstructured":"Krawczyk-Becker M, Gerkmann T (2016) On mmse-based estimation of amplitude and complex speech spectral coefficients under phase-uncertainty. IEEE\/ACM Trans Audio Speech Lang Process 24(12):2251\u20132262","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR24","doi-asserted-by":"crossref","unstructured":"Le Roux J, Wisdom S, Erdogan H et al (2019) Sdr-half-baked or well done? In: ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 626\u2013630","DOI":"10.1109\/ICASSP.2019.8683855"},{"key":"21428_CR25","doi-asserted-by":"crossref","unstructured":"Li A, Zheng C, Fan C et al (2020) A recursive network with dynamic attention for monaural speech enhancement. Proc Interspeech 2020, pp 2422\u20132426","DOI":"10.21437\/Interspeech.2020-1513"},{"key":"21428_CR26","unstructured":"Li A, Zheng C, Peng R et al (2020) Two heads are better than one: a two-stage approach for monaural noise reduction in the complex domain. arXiv:2011.01561"},{"key":"21428_CR27","doi-asserted-by":"publisher","first-page":"1829","DOI":"10.1109\/TASLP.2021.3079813","volume":"29","author":"A Li","year":"2021","unstructured":"Li A, Liu W, Zheng C et al (2021) Two heads are better than one: a two-stage complex spectral mapping approach for monaural speech enhancement. IEEE\/ACM Trans Audio Speech Lang Process 29:1829\u20131843","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR28","doi-asserted-by":"publisher","first-page":"108499","DOI":"10.1016\/j.apacoust.2021.108499","volume":"187","author":"A Li","year":"2022","unstructured":"Li A, Zheng C, Zhang L et al (2022) Glance and gaze: a collaborative learning framework for single-channel speech enhancement. Appl Acoust 187:108499","journal-title":"Appl Acoust"},{"key":"21428_CR29","doi-asserted-by":"crossref","unstructured":"Lin J, Niu S, Wei Z et al (2019) Speech enhancement using forked generative adversarial networks with spectral subtraction. Proceedings of Interspeech 2019","DOI":"10.21437\/Interspeech.2019-2954"},{"key":"21428_CR30","doi-asserted-by":"crossref","unstructured":"Lin J, Niu S, Wijngaarden AJ et al (2020) Improved speech enhancement using a time-domain gan with mask learning. Proceedings of Interspeech 2020","DOI":"10.21437\/Interspeech.2020-1946"},{"key":"21428_CR31","doi-asserted-by":"publisher","first-page":"3440","DOI":"10.1109\/TASLP.2021.3125143","volume":"29","author":"J Lin","year":"2021","unstructured":"Lin J, van Wijngaarden AJL, Wang KC et al (2021) Speech enhancement using multi-stage self-attentive temporal convolutional networks. IEEE\/ACM Trans Audio Speech Lang Process 29:3440\u20133450","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR32","doi-asserted-by":"publisher","DOI":"10.1201\/9781420015836","volume-title":"Speech enhancement: theory and practice","author":"PC Loizou","year":"2007","unstructured":"Loizou PC (2007) Speech enhancement: theory and practice. CRC Press"},{"issue":"8","key":"21428_CR33","doi-asserted-by":"publisher","first-page":"1256","DOI":"10.1109\/TASLP.2019.2915167","volume":"27","author":"Y Luo","year":"2019","unstructured":"Luo Y, Mesgarani N (2019) Conv-tasnet: surpassing ideal time-frequency magnitude masking for speech separation. IEEE\/ACM Trans Audio Speech Lang Process 27(8):1256\u20131266","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR34","unstructured":"Macartney C, Weyde T (2018) Improved speech enhancement with the wave-u-net. arXiv:1811.11307"},{"key":"21428_CR35","doi-asserted-by":"publisher","first-page":"405","DOI":"10.1007\/978-981-97-3523-5_30","volume":"1015","author":"UN Model","year":"2024","unstructured":"Model UN (2024) Forest aerial image segmentation through satellite images using refine. Adv Distributed Computing Mach Learn Proc ICADCML 2024 Volume 2 1015:405","journal-title":"Adv Distributed Computing Mach Learn Proc ICADCML 2024 Volume 2"},{"key":"21428_CR36","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1016\/j.specom.2020.10.004","volume":"125","author":"A Nicolson","year":"2020","unstructured":"Nicolson A, Paliwal KK (2020) Masked multi-head self-attention for causal speech enhancement. Speech Commun 125:80\u201396","journal-title":"Speech Commun"},{"key":"21428_CR37","doi-asserted-by":"crossref","unstructured":"Panayotov V, Chen G, Povey D et al (2015) Librispeech: an asr corpus based on public domain audio books. In: 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5206\u20135210","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"21428_CR38","doi-asserted-by":"crossref","unstructured":"Pandey A, Wang D (2019) Tcnn: temporal convolutional neural network for real-time speech enhancement in the time domain. In: ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 6875\u20136879","DOI":"10.1109\/ICASSP.2019.8683634"},{"key":"21428_CR39","doi-asserted-by":"publisher","first-page":"1270","DOI":"10.1109\/TASLP.2021.3064421","volume":"29","author":"A Pandey","year":"2021","unstructured":"Pandey A, Wang D (2021) Dense cnn with self-attention for time-domain speech enhancement. IEEE\/ACM Trans Audio Speech Lang Process 29:1270\u20131279","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR40","doi-asserted-by":"crossref","unstructured":"Parisae V, Bhavanam SN (2024) Multi scale encoder-decoder network with time frequency attention and s-tcn for single channel speech enhancement. J Intell Fuzzy Syst (Preprint)","DOI":"10.3233\/JIFS-233312"},{"key":"21428_CR41","doi-asserted-by":"crossref","unstructured":"Pascual S, Bonafonte A, Serra J (2017) Segan: speech enhancement generative adversarial network. arXiv:1703.09452","DOI":"10.21437\/Interspeech.2017-1428"},{"key":"21428_CR42","doi-asserted-by":"publisher","first-page":"1700","DOI":"10.1109\/LSP.2020.3025020","volume":"27","author":"H Phan","year":"2020","unstructured":"Phan H, McLoughlin IV, Pham L et al (2020) Improving gans for speech enhancement. IEEE Signal Process Lett 27:1700\u20131704","journal-title":"IEEE Signal Process Lett"},{"key":"21428_CR43","doi-asserted-by":"crossref","unstructured":"Reddy CK, Dubey H, Gopal V et al (2021) Icassp 2021 deep noise suppression challenge. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 6623\u20136627","DOI":"10.1109\/ICASSP39728.2021.9415105"},{"key":"21428_CR44","doi-asserted-by":"crossref","unstructured":"Rix AW, Beerends JG, Hollier MP et al (2001) Perceptual evaluation of speech quality (pesq)-a new method for speech quality assessment of telephone networks and codecs. In: 2001 IEEE international conference on acoustics, speech, and signal processing. Proceedings (Cat. No. 01CH37221). IEEE, pp 749\u2013752","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"21428_CR45","doi-asserted-by":"crossref","unstructured":"Ruan D, Wang D, Zheng Y et al (2021) Gaussian context transformer. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 15129\u201315138","DOI":"10.1109\/CVPR46437.2021.01488"},{"key":"21428_CR46","doi-asserted-by":"crossref","unstructured":"Soni MH, Shah N, Patil HA (2018) Time-frequency masking-based speech enhancement using generative adversarial network. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5039\u20135043","DOI":"10.1109\/ICASSP.2018.8462068"},{"issue":"7","key":"21428_CR47","doi-asserted-by":"publisher","first-page":"2125","DOI":"10.1109\/TASL.2011.2114881","volume":"19","author":"CH Taal","year":"2011","unstructured":"Taal CH, Hendriks RC, Heusdens R et al (2011) An algorithm for intelligibility prediction of time\u2013frequency weighted noisy speech. IEEE Trans Audio Speech Lang Process 19(7):2125\u20132136","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"21428_CR48","doi-asserted-by":"crossref","unstructured":"Tan K, Wang D (2018) A convolutional recurrent neural network for real-time speech enhancement. In: Interspeech, pp 3229\u20133233","DOI":"10.21437\/Interspeech.2018-1405"},{"key":"21428_CR49","doi-asserted-by":"crossref","unstructured":"Tang C, Luo C, Zhao Z et al (2021) Joint time-frequency and time domain learning for speech enhancement. In: Proceedings of the twenty-ninth international conference on international joint conferences on artificial intelligence, pp 3816\u20133822","DOI":"10.24963\/ijcai.2020\/528"},{"key":"21428_CR50","doi-asserted-by":"crossref","unstructured":"Thiemann J, Ito N, Vincent E (2013) The diverse environments multi-channel acoustic noise database: a database of multichannel environmental noise recordings. J Acoustical Soc America 133(5_Supplement):3591\u20133591","DOI":"10.1121\/1.4806631"},{"key":"21428_CR51","unstructured":"Valentini-Botinhao C et al (2017) Noisy speech database for training speech enhancement algorithms and tts models. University of Edinburgh School of Informatics Centre for Speech Technology Research (CSTR)"},{"key":"21428_CR52","unstructured":"Van Den Oord A, Dieleman S, Zen H et al (2016) Wavenet: a generative model for raw audio. arXiv:1609.03499 12:1"},{"key":"21428_CR53","doi-asserted-by":"crossref","unstructured":"Vanambathina SD, Nandyala S, Jannu C et al (2024) Speech enhancement using u-net-based progressive learning with squeeze-tcn. In: International conference on advances in distributed computing and machine learning. Springer, pp 419\u2013432","DOI":"10.1007\/978-981-97-3523-5_31"},{"issue":"3","key":"21428_CR54","doi-asserted-by":"publisher","first-page":"247","DOI":"10.1016\/0167-6393(93)90095-3","volume":"12","author":"A Varga","year":"1993","unstructured":"Varga A, Steeneken HJ (1993) Assessment for automatic speech recognition: Ii. noisex-92: a database and an experiment to study the effect of additive noise on speech recognition systems. Speech Commun 12(3):247\u2013251","journal-title":"Speech Commun"},{"key":"21428_CR55","unstructured":"Vaswani A, Shazeer N, Parmar N et al (2017) Attention is all you need. Adv Neural Inf Process Syst 30"},{"key":"21428_CR56","doi-asserted-by":"crossref","unstructured":"Veaux C, Yamagishi J, King S (2013) The voice bank corpus: design, collection and data analysis of a large regional accent speech database. In: 2013 international conference oriental COCOSDA held jointly with 2013 conference on Asian spoken language research and evaluation (O-COCOSDA\/CASLRE). IEEE, pp 1\u20134","DOI":"10.1109\/ICSDA.2013.6709856"},{"issue":"10","key":"21428_CR57","doi-asserted-by":"publisher","first-page":"1702","DOI":"10.1109\/TASLP.2018.2842159","volume":"26","author":"D Wang","year":"2018","unstructured":"Wang D, Chen J (2018) Supervised speech separation based on deep learning: an overview. IEEE\/ACM Trans Audio Speech Lang Process 26(10):1702\u20131726","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR58","doi-asserted-by":"crossref","unstructured":"Wang Q, Muckenhirn H, Wilson K et al (2018) Voicefilter: targeted voice separation by speaker-conditioned spectrogram masking. arXiv:1810.04826","DOI":"10.21437\/Interspeech.2019-1101"},{"issue":"7","key":"21428_CR59","doi-asserted-by":"publisher","first-page":"1381","DOI":"10.1109\/TASL.2013.2250961","volume":"21","author":"Y Wang","year":"2013","unstructured":"Wang Y, Wang D (2013) Towards scaling up classification-based speech separation. IEEE Trans Audio Speech Lang Process 21(7):1381\u20131390","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"21428_CR60","doi-asserted-by":"crossref","unstructured":"Weninger F, Eyben F, Schuller B (2014) Single-channel speech separation with memory-enhanced recurrent neural networks. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 3709\u20133713","DOI":"10.1109\/ICASSP.2014.6854294"},{"key":"21428_CR61","doi-asserted-by":"crossref","unstructured":"Weninger F, Erdogan H, Watanabe S et al (2015) Speech enhancement with lstm recurrent neural networks and its application to noise-robust asr. In: Latent Variable Analysis and Signal Separation: 12th International Conference, LVA\/ICA 2015, Liberec, Czech Republic, August 25\u201328, 2015, Proceedings 12. Springer, pp 91\u201399","DOI":"10.1007\/978-3-319-22482-4_11"},{"key":"21428_CR62","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1109\/LSP.2021.3128374","volume":"29","author":"X Xiang","year":"2021","unstructured":"Xiang X, Zhang X, Chen H (2021) A nested u-net with self-attention and dense connectivity for monaural speech enhancement. IEEE Signal Process Lett 29:105\u2013109","journal-title":"IEEE Signal Process Lett"},{"issue":"1","key":"21428_CR63","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","volume":"23","author":"Y Xu","year":"2014","unstructured":"Xu Y, Du J, Dai LR et al (2014) A regression approach to speech enhancement based on deep neural networks. IEEE\/ACM Trans Audio Speech Lang Process 23(1):7\u201319","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR64","unstructured":"Zhang Q, Nicolson A, Wang M et al (2019) Monaural speech enhancement using a multi-branch temporal convolutional network. arXiv:1912.12023"},{"key":"21428_CR65","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1016\/j.dsp.2019.01.019","volume":"88","author":"Q Zhang","year":"2019","unstructured":"Zhang Q, Wang M, Lu Y et al (2019) A novel fast nonstationary noise tracking approach based on mmse spectral power estimator. Digital Signal Process 88:41\u201352","journal-title":"Digital Signal Process"},{"key":"21428_CR66","doi-asserted-by":"publisher","first-page":"1404","DOI":"10.1109\/TASLP.2020.2987441","volume":"28","author":"Q Zhang","year":"2020","unstructured":"Zhang Q, Nicolson A, Wang M et al (2020) Deepmmse: a deep learning approach to mmse-based noise power spectral density estimation. IEEE\/ACM Trans Audio Speech Lang Process 28:1404\u20131415","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR67","doi-asserted-by":"crossref","unstructured":"Zhang Q, Song Q, Nicolson A et al (2021) Temporal convolutional network with frequency dimension adaptive attention for speech enhancement. Proc Interspeech 2021, pp 166\u2013170","DOI":"10.21437\/Interspeech.2021-46"},{"key":"21428_CR68","doi-asserted-by":"crossref","unstructured":"Zhao Y, Wang D (2020) Noisy-reverberant speech enhancement using denseunet with time-frequency attention. In: Interspeech, pp 3261\u20133265","DOI":"10.21437\/Interspeech.2020-2952"},{"key":"21428_CR69","doi-asserted-by":"publisher","first-page":"1598","DOI":"10.1109\/TASLP.2020.2995273","volume":"28","author":"Y Zhao","year":"2020","unstructured":"Zhao Y, Wang D, Xu B et al (2020) Monaural speech dereverberation using temporal convolutional networks with self attention. IEEE\/ACM Trans Audio Speech Lang Process 28:1598\u20131607","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"21428_CR70","doi-asserted-by":"crossref","unstructured":"Zheng C, Peng X, Zhang Y et al (2021) Interactive speech and noise modeling for speech enhancement. In: Proceedings of the AAAI conference on artificial intelligence, pp 14549\u201314557","DOI":"10.1609\/aaai.v35i16.17710"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21428-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-026-21428-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21428-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,25]],"date-time":"2026-02-25T22:43:21Z","timestamp":1772059401000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-026-21428-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,25]]},"references-count":70,"journal-issue":{"issue":"3","published-online":{"date-parts":[[2026,3]]}},"alternative-id":["21428"],"URL":"https:\/\/doi.org\/10.1007\/s11042-026-21428-x","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,25]]},"assertion":[{"value":"1 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 January 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 February 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 February 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"Not applicable","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to Participate"}},{"value":"Not applicable","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to Publish"}},{"value":"No conflict of interest.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"195"}}