{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T10:28:19Z","timestamp":1771064899138,"version":"3.50.1"},"reference-count":110,"publisher":"Springer Science and Business Media LLC","issue":"16","license":[{"start":{"date-parts":[[2024,6,28]],"date-time":"2024-06-28T00:00:00Z","timestamp":1719532800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,6,28]],"date-time":"2024-06-28T00:00:00Z","timestamp":1719532800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-024-19744-1","type":"journal-article","created":{"date-parts":[[2024,6,28]],"date-time":"2024-06-28T05:02:05Z","timestamp":1719550925000},"page":"16455-16479","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Cross channel interaction based ECA-Net using gated recurrent convolutional network for speech enhancement"],"prefix":"10.1007","volume":"84","author":[{"given":"Manaswini","family":"Burra","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sunny Dayal","family":"Vanambathina","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Venkata Adi Lakshmi","family":"A","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Loukya","family":"Ch","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siva Kotiah","family":"N","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,6,28]]},"reference":[{"key":"19744_CR1","doi-asserted-by":"crossref","unstructured":"Aroudi A, Braun S (2020) Dbnet: doa-driven beamforming network for end-to-end farfield sound source separation. arXiv:2010.11566","DOI":"10.1109\/ICASSP39728.2021.9414187"},{"key":"19744_CR2","doi-asserted-by":"crossref","unstructured":"Bastanfard A, Amirkhani D, Hasani M (2019) Increasing the accuracy of automatic speaker age estimation by using multiple ubms. In: 2019 5th conference on knowledge based engineering and innovation (KBEI), IEEE, pp 592\u2013598","DOI":"10.1109\/KBEI.2019.8735005"},{"key":"19744_CR3","doi-asserted-by":"crossref","unstructured":"Berouti M, Schwartz R, Makhoul J (1979) Enhancement of speech corrupted by acoustic noise. In: ICASSP\u201979. IEEE International Conference on Acoustics, Speech, and Signal Processing, IEEE, pp 208\u2013211","DOI":"10.1109\/ICASSP.1979.1170788"},{"issue":"2","key":"19744_CR4","doi-asserted-by":"publisher","first-page":"113","DOI":"10.1109\/TASSP.1979.1163209","volume":"27","author":"S Boll","year":"1979","unstructured":"Boll S (1979) Suppression of acoustic noise in speech using spectral subtraction. IEEE Trans Acoustics Speech Signal Process 27(2):113\u2013120","journal-title":"IEEE Trans Acoustics Speech Signal Process"},{"key":"19744_CR5","first-page":"996","volume-title":"ICASSP 2022\u20132022 IEEE International Conference on Acoustics","author":"S Braun","year":"2022","unstructured":"Braun S, Gamper H (2022) Effect of noise suppression losses on speech distortion and asr performance. ICASSP 2022\u20132022 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 996\u20131000"},{"key":"19744_CR6","doi-asserted-by":"crossref","unstructured":"Burra M, Yerva PKR, Eemani B, et\u00a0al (2023) Densely connected dilated convolutions with time-frequency attention for speech enhancement. In: 2023 2nd International Conference on Applied Artificial Intelligence and Computing (ICAAIC), IEEE, pp 602\u2013607","DOI":"10.1109\/ICAAIC56838.2023.10140871"},{"issue":"6","key":"19744_CR7","doi-asserted-by":"publisher","first-page":"4705","DOI":"10.1121\/1.4986931","volume":"141","author":"J Chen","year":"2017","unstructured":"Chen J, Wang D (2017) Long short-term memory for speaker generalization in supervised speech separation. J Acoustical Soc America 141(6):4705\u20134714","journal-title":"J Acoustical Soc America"},{"key":"19744_CR8","unstructured":"Clevert DA, Unterthiner T, Hochreiter S (2015) Fast and accurate deep network learning by exponential linear units (elus). arXiv:1511.07289"},{"key":"19744_CR9","unstructured":"Commonvoice (2017): https:\/\/commonvoice.mozilla.org\/en"},{"key":"19744_CR10","unstructured":"Dauphin YN, Fan A, Auli M, et\u00a0al (2017) Language modeling with gated convolutional networks. In: International conference on machine learning, PMLR, pp 933\u2013941"},{"key":"19744_CR11","doi-asserted-by":"crossref","unstructured":"Defossez A, Synnaeve G, Adi Y (2020) Real time speech enhancement in the waveform domain. arXiv:2006.12847","DOI":"10.21437\/Interspeech.2020-2409"},{"issue":"3","key":"19744_CR12","doi-asserted-by":"publisher","first-page":"783","DOI":"10.1007\/s11760-022-02288-y","volume":"17","author":"X Duan","year":"2023","unstructured":"Duan X, Sun Y, Wang J (2023) Eca-unet for coronary artery segmentation and three-dimensional reconstruction. Signal Image Video Process 17(3):783\u2013789","journal-title":"Signal Image Video Process"},{"issue":"3","key":"19744_CR13","doi-asserted-by":"publisher","first-page":"572","DOI":"10.1016\/j.patcog.2010.09.020","volume":"44","author":"M El Ayadi","year":"2011","unstructured":"El Ayadi M, Kamel MS, Karray F (2011) Survey on speech emotion recognition: Features, classification schemes, and databases. Pattern Recogn 44(3):572\u2013587","journal-title":"Pattern Recogn"},{"key":"19744_CR14","first-page":"708","volume-title":"2015 IEEE International Conference on Acoustics","author":"H Erdogan","year":"2015","unstructured":"Erdogan H, Hershey JR, Watanabe S et al (2015) Phase-sensitive and recognition-boosted speech separation using deep recurrent neural networks. 2015 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 708\u2013712"},{"key":"19744_CR15","doi-asserted-by":"crossref","unstructured":"Eskimez SE, Wang X, Tang M, et\u00a0al (2021) Human listening and live captioning: multi-task training for speech enhancement. arXiv:2106.02896","DOI":"10.21437\/Interspeech.2021-220"},{"key":"19744_CR16","doi-asserted-by":"publisher","first-page":"60","DOI":"10.1016\/j.neunet.2017.02.013","volume":"92","author":"HM Fayek","year":"2017","unstructured":"Fayek HM, Lech M, Cavedon L (2017) Evaluating deep learning architectures for speech emotion recognition. Neural Netw 92:60\u201368","journal-title":"Neural Netw"},{"key":"19744_CR17","doi-asserted-by":"crossref","unstructured":"Fu SW, Tsao Y, Lu X, et\u00a0al (2016) Snr-aware convolutional neural network modeling for speech enhancement. In: Interspeech, pp 3768\u20133772","DOI":"10.21437\/Interspeech.2016-211"},{"key":"19744_CR18","doi-asserted-by":"crossref","unstructured":"Fu SW, Hu Ty, Tsao Y, et\u00a0al (2017) Complex spectrogram enhancement by convolutional neural network with multi-metrics learning. In: 2017 IEEE 27th international workshop on machine learning for signal processing (MLSP), IEEE, pp 1\u20136","DOI":"10.1109\/MLSP.2017.8168119"},{"key":"19744_CR19","unstructured":"Fu SW, Liao CF, Tsao Y, et\u00a0al (2019) Metricgan: Generative adversarial networks based black-box metric scores optimization for speech enhancement. In: International Conference on Machine Learning, PMLR, pp 2031\u20132041"},{"key":"19744_CR20","first-page":"7417","volume-title":"ICASSP 2022\u20132022 IEEE International Conference on Acoustics","author":"Y Fu","year":"2022","unstructured":"Fu Y, Liu Y, Li J et al (2022) Uformer: A unet based dilated complex & real dual-path conformer network for simultaneous speech enhancement and dereverberation. ICASSP 2022\u20132022 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 7417\u20137421"},{"key":"19744_CR21","doi-asserted-by":"crossref","unstructured":"Fuchs A, Priewald R, Pernkopf F (2019) Recurrent dilated densenets for a time-series segmentation task. In: 2019 18th IEEE International Conference On Machine Learning And Applications (ICMLA), IEEE, pp 75\u201380","DOI":"10.1109\/ICMLA.2019.00021"},{"key":"19744_CR22","doi-asserted-by":"crossref","unstructured":"Giri R, Isik U, Krishnaswamy A (2019) Attention wave-u-net for speech enhancement. In: 2019 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA), IEEE, pp 249\u2013253","DOI":"10.1109\/WASPAA.2019.8937186"},{"key":"19744_CR23","doi-asserted-by":"crossref","unstructured":"Grais EM, Plumbley MD (2017) Single channel audio source separation using convolutional denoising autoencoders. In: 2017 IEEE global conference on signal and information processing (GlobalSIP), IEEE, pp 1265\u20131269","DOI":"10.1109\/GlobalSIP.2017.8309164"},{"key":"19744_CR24","doi-asserted-by":"crossref","unstructured":"Gulati A, Qin J, Chiu CC, et\u00a0al (2020) Conformer: Convolution-augmented transformer for speech recognition. arXiv:2005.08100","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"19744_CR25","first-page":"4628","volume-title":"2014 IEEE International Conference on Acoustics","author":"K Han","year":"2014","unstructured":"Han K, Wang Y, Wang D (2014) Learning spectral mapping for speech dereverberation. 2014 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 4628\u20134632"},{"key":"19744_CR26","first-page":"6959","volume-title":"ICASSP 2020\u20132020 IEEE International Conference on Acoustics","author":"X Hao","year":"2020","unstructured":"Hao X, Su X, Wen S et al (2020) Masking and inpainting: a two-stage speech enhancement approach for low snr and non-stationary noise. ICASSP 2020\u20132020 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 6959\u20136963"},{"key":"19744_CR27","doi-asserted-by":"crossref","unstructured":"Harsh H, Indraganti A, Vanambathina SD, et\u00a0al (2022) Convolutional gru networks based singing voice separation. In: 2022 2nd International Conference on Artificial Intelligence and Signal Processing (AISP), IEEE, pp 1\u20135","DOI":"10.1109\/AISP53593.2022.9760616"},{"issue":"8","key":"19744_CR28","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"issue":"3","key":"19744_CR29","doi-asserted-by":"publisher","first-page":"156","DOI":"10.1049\/iet-spr:20070008","volume":"1","author":"H Hu","year":"2007","unstructured":"Hu H, Yu C (2007) Adaptive noise spectral estimation for spectral subtraction speech enhancement. IET Signal Process 1(3):156\u2013163","journal-title":"IET Signal Process"},{"key":"19744_CR30","doi-asserted-by":"crossref","unstructured":"Hu J, Shen L, Sun G (2018) Squeeze-and-excitation networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7132\u20137141","DOI":"10.1109\/CVPR.2018.00745"},{"issue":"1","key":"19744_CR31","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TASL.2007.911054","volume":"16","author":"Y Hu","year":"2007","unstructured":"Hu Y, Loizou PC (2007) Evaluation of objective quality measures for speech enhancement. IEEE Trans Audio Speech Language Process 16(1):229\u2013238","journal-title":"IEEE Trans Audio Speech Language Process"},{"key":"19744_CR32","doi-asserted-by":"crossref","unstructured":"Hu Y, Liu Y, Lv S, et\u00a0al (2020) Dccrn: Deep complex convolution recurrent network for phase-aware speech enhancement. arXiv:2008.00264","DOI":"10.21437\/Interspeech.2020-2537"},{"key":"19744_CR33","first-page":"1562","volume-title":"2014 IEEE International Conference on Acoustics","author":"PS Huang","year":"2014","unstructured":"Huang PS, Kim M, Hasegawa-Johnson M et al (2014) Deep learning for monaural speech separation. 2014 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 1562\u20131566"},{"key":"19744_CR34","unstructured":"Ioffe S, Szegedy C (2015) Batch normalization: Accelerating deep network training by reducing internal covariate shift. In: International conference on machine learning, pmlr, pp 448\u2013456"},{"key":"19744_CR35","unstructured":"ITU-T P (2003) 835: subjective test methodology for evaluating speech communication systems that include noise suppression algorithms. ITU-T recommendation"},{"key":"19744_CR36","doi-asserted-by":"crossref","unstructured":"Jannu C, Vanambathina SD (2023a) An attention based densely connected u-net with convolutional gru for speech enhancement. In: 2023 3rd International conference on Artificial Intelligence and Signal Processing (AISP), IEEE, pp 1\u20135","DOI":"10.1109\/AISP57993.2023.10134933"},{"key":"19744_CR37","doi-asserted-by":"crossref","unstructured":"Jannu C, Vanambathina SD (2023b) Convolutional transformer based local and global feature learning for speech enhancement. Int J Advan Comput Sci Appl 14(1)","DOI":"10.14569\/IJACSA.2023.0140181"},{"issue":"12","key":"19744_CR38","doi-asserted-by":"publisher","first-page":"7467","DOI":"10.1007\/s00034-023-02455-7","volume":"42","author":"C Jannu","year":"2023","unstructured":"Jannu C, Vanambathina SD (2023) Multi-stage progressive learning-based speech enhancement using time-frequency attentive squeezed temporal convolutional networks. Circuits Syst Signal Process 42(12):7467\u20137493","journal-title":"Circuits Syst Signal Process"},{"key":"19744_CR39","doi-asserted-by":"crossref","unstructured":"Jannu C, Vanambathina SD (2023d) An overview of speech enhancement based on deep learning techniques. Int J Image Graphics:2550001","DOI":"10.1142\/S0219467825500019"},{"issue":"1","key":"19744_CR40","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1007\/s10772-023-10020-5","volume":"26","author":"C Jannu","year":"2023","unstructured":"Jannu C, Vanambathina SD (2023) Weibull and nakagami speech priors based regularized nmf with adaptive wiener filter for speech enhancement. Int J Speech Technol 26(1):197\u2013209","journal-title":"Int J Speech Technol"},{"key":"19744_CR41","unstructured":"Jansson A, Humphrey E, Montecchio N, et\u00a0al (2017) Singing voice separation with deep u-net convolutional networks. ISMIR Conference"},{"key":"19744_CR42","doi-asserted-by":"crossref","unstructured":"Kamath S, Loizou P, et\u00a0al (2002) A multi-band spectral subtraction method for enhancing speech corrupted by colored noise. In: ICASSP, Citeseer, pp 44164\u201344164","DOI":"10.1109\/ICASSP.2002.5745591"},{"key":"19744_CR43","doi-asserted-by":"crossref","unstructured":"Kim Y, Lee H, Provost EM (2013) Deep learning for robust feature generation in audiovisual emotion recognition. In: 2013 IEEE international conference on acoustics, speech and signal processing, IEEE, pp 3687\u20133691","DOI":"10.1109\/ICASSP.2013.6638346"},{"key":"19744_CR44","doi-asserted-by":"crossref","unstructured":"Kishore V, Tiwari N, Paramasivam P (2020) Improved speech enhancement using tcn with multiple encoder-decoder layers. In: Interspeech, pp 4531\u20134535","DOI":"10.21437\/Interspeech.2020-3122"},{"key":"19744_CR45","first-page":"181","volume-title":"ICASSP 2020\u20132020 IEEE International Conference on Acoustics","author":"Y Koizumi","year":"2020","unstructured":"Koizumi Y, Yatabe K, Delcroix M et al (2020) Speech enhancement using self-adaptation and multi-head self-attention. ICASSP 2020\u20132020 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 181\u2013185"},{"issue":"12","key":"19744_CR46","doi-asserted-by":"publisher","first-page":"1931","DOI":"10.1109\/TASLP.2014.2354236","volume":"22","author":"M Krawczyk","year":"2014","unstructured":"Krawczyk M, Gerkmann T (2014) Stft phase reconstruction in voiced speech for an improved single-channel speech enhancement. IEEE\/ACM Trans Audio Speech Language Process 22(12):1931\u20131940","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"issue":"5","key":"19744_CR47","doi-asserted-by":"publisher","first-page":"598","DOI":"10.1109\/LSP.2014.2365040","volume":"22","author":"J Kulmer","year":"2014","unstructured":"Kulmer J, Mowlaee P (2014) Phase estimation in single channel speech enhancement using phase decomposition. IEEE Signal Process Lett 22(5):598\u2013602","journal-title":"IEEE Signal Process Lett"},{"key":"19744_CR48","unstructured":"Kumar A, Daume\u00a0III H (2012) Learning task grouping and overlap in multi-task learning. arXiv:1206.6417"},{"key":"19744_CR49","doi-asserted-by":"publisher","first-page":"312","DOI":"10.1016\/j.neucom.2016.12.012","volume":"230","author":"K Kumar","year":"2017","unstructured":"Kumar K, Cruces S et al (2017) An iterative posterior nmf method for speech enhancement in the presence of additive gaussian noise. Neurocomputing 230:312\u2013315","journal-title":"Neurocomputing"},{"key":"19744_CR50","doi-asserted-by":"crossref","unstructured":"Lalitha V, Prema P, Mathew L (2010) A kepstrum based approach for enhancement of dysarthric speech. In: 2010 3rd International Congress on Image and Signal Processing, IEEE, pp 3474\u20133478","DOI":"10.1109\/CISP.2010.5646752"},{"key":"19744_CR51","doi-asserted-by":"publisher","first-page":"2411","DOI":"10.1109\/TASLP.2022.3190738","volume":"30","author":"X Le","year":"2022","unstructured":"Le X, Lei T, Chen K et al (2022) Inference skipping for more efficient real-time speech enhancement with parallel rnns. IEEE\/ACM Trans Audio Speech Language Process 30:2411\u20132421","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"issue":"12","key":"19744_CR52","doi-asserted-by":"publisher","first-page":"1586","DOI":"10.1109\/PROC.1979.11540","volume":"67","author":"JS Lim","year":"1979","unstructured":"Lim JS, Oppenheim AV (1979) Enhancement and bandwidth compression of noisy speech. Proc IEEE 67(12):1586\u20131604","journal-title":"Proc IEEE"},{"key":"19744_CR53","doi-asserted-by":"publisher","first-page":"3440","DOI":"10.1109\/TASLP.2021.3125143","volume":"29","author":"J Lin","year":"2021","unstructured":"Lin J, van Wijngaarden AJdL, Wang KC et al (2021) Speech enhancement using multi-stage self-attentive temporal convolutional networks. IEEE\/ACM Trans Audio Speech Language Process 29:3440\u20133450","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"key":"19744_CR54","doi-asserted-by":"crossref","unstructured":"Liu JY, Yang YH (2019) Dilated convolution with dilated gru for music source separation. arXiv:1906.01203","DOI":"10.24963\/ijcai.2019\/655"},{"key":"19744_CR55","doi-asserted-by":"crossref","unstructured":"Lu X, Tsao Y, Matsuda S, et\u00a0al (2013) Speech enhancement based on deep denoising autoencoder. In: Interspeech, pp 436\u2013440","DOI":"10.21437\/Interspeech.2013-130"},{"key":"19744_CR56","unstructured":"Macartney C, Weyde T (2018) Improved speech enhancement with the wave-u-net. arXiv:1811.11307"},{"key":"19744_CR57","first-page":"1","volume-title":"2020 25th international computer conference","author":"R Mahdavi","year":"2020","unstructured":"Mahdavi R, Bastanfard A, Amirkhani D (2020) Persian accents identification using modeling of speech articulatory features. 2020 25th international computer conference. Computer Society of Iran (CSICC), IEEE, pp 1\u20139"},{"key":"19744_CR58","doi-asserted-by":"crossref","unstructured":"Mehrish A, Majumder N, Bharadwaj R, et\u00a0al (2023) A review of deep learning techniques for speech processing. Inform Fusion:101869","DOI":"10.1016\/j.inffus.2023.101869"},{"key":"19744_CR59","unstructured":"Michelsanti D (2021) Audio-visual speech enhancement based on deep learning. Aalborg Universitet"},{"key":"19744_CR60","doi-asserted-by":"publisher","first-page":"1368","DOI":"10.1109\/TASLP.2021.3066303","volume":"29","author":"D Michelsanti","year":"2021","unstructured":"Michelsanti D, Tan ZH, Zhang SX et al (2021) An overview of deep-learning-based audio-visual speech enhancement and separation. IEEE\/ACM Trans Audio Speech Language Process 29:1368\u20131396","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"key":"19744_CR61","doi-asserted-by":"crossref","unstructured":"Naithani G, Barker T, Parascandolo G, et\u00a0al (2017) Low latency sound source separation using convolutional recurrent neural networks. In: 2017 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA), IEEE, pp 71\u201375","DOI":"10.1109\/WASPAA.2017.8169997"},{"key":"19744_CR62","first-page":"1","volume-title":"ICASSP 2023\u20132023 IEEE International Conference on Acoustics","author":"J Neri","year":"2023","unstructured":"Neri J, Braun S (2023) Towards real-time single-channel speech separation in noisy and reverberant environments. ICASSP 2023\u20132023 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 1\u20135"},{"key":"19744_CR63","unstructured":"Noizeus (2007) https:\/\/ecs.utdallas.edu\/loizou\/speech\/noizeus"},{"key":"19744_CR64","unstructured":"Van\u00a0den Oord A, Kalchbrenner N, Espeholt L, et\u00a0al (2016) Conditional image generation with pixelcnn decoders. Advan Neural Inform Process Syst 29"},{"key":"19744_CR65","unstructured":"Oord Avd, Dieleman S, Zen H, et\u00a0al (2016) Wavenet: A generative model for raw audio. arXiv:1609.03499"},{"key":"19744_CR66","first-page":"671","volume-title":"ICASSP 2021\u20132021 IEEE International Conference on Acoustics","author":"K Oostermeijer","year":"2021","unstructured":"Oostermeijer K, Du J, Wang Q et al (2021) Speech enhancement autoencoder with hierarchical latent structure. ICASSP 2021\u20132021 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 671\u2013675"},{"issue":"4","key":"19744_CR67","doi-asserted-by":"publisher","first-page":"465","DOI":"10.1016\/j.specom.2010.12.003","volume":"53","author":"K Paliwal","year":"2011","unstructured":"Paliwal K, W\u00f3jcicki K, Shannon B (2011) The importance of phase in speech enhancement. Speech Commun 53(4):465\u2013494","journal-title":"Speech Commun"},{"key":"19744_CR68","doi-asserted-by":"crossref","unstructured":"Parisae V, Bhavanam SN (2024) Adaptive attention mechanism for single channel speech enhancement. Multimed Tool Appl:1\u201326","DOI":"10.1007\/s11042-024-19076-0"},{"key":"19744_CR69","doi-asserted-by":"crossref","unstructured":"Pascual S, Bonafonte A, Serra J (2017) Segan: Speech enhancement generative adversarial network. arXiv:1703.09452","DOI":"10.21437\/Interspeech.2017-1428"},{"key":"19744_CR70","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s10462-012-9356-9","volume":"43","author":"SS Rautaray","year":"2015","unstructured":"Rautaray SS, Agrawal A (2015) Vision based hand gesture recognition for human computer interaction: a survey. Artif Intell Rev 43:1\u201354","journal-title":"Artif Intell Rev"},{"key":"19744_CR71","doi-asserted-by":"crossref","unstructured":"Rim\u00a0Park S, Lee J (2016) A fully convolutional neural network for speech enhancement. pp arXiv\u20131609","DOI":"10.21437\/Interspeech.2017-1465"},{"key":"19744_CR72","doi-asserted-by":"crossref","unstructured":"Saeidi R, Mowlaee P, Martin R (2012) Phase estimation for signal reconstruction in single-channel source separation. Interspeech","DOI":"10.21437\/Interspeech.2012-436"},{"key":"19744_CR73","doi-asserted-by":"crossref","unstructured":"Savargiv M, Bastanfard A (2016) Real-time speech emotion recognition by minimum number of features. In: 2016 Artificial Intelligence and Robotics (IRANOPEN), IEEE, pp 72\u201376","DOI":"10.1109\/RIOS.2016.7529493"},{"key":"19744_CR74","doi-asserted-by":"crossref","unstructured":"Scalart P, et\u00a0al (1996) Speech enhancement based on a priori signal to noise estimation. In: 1996 IEEE International Conference on Acoustics, Speech, and Signal Processing Conference Proceedings, IEEE, pp 629\u2013632","DOI":"10.1109\/ICASSP.1996.543199"},{"key":"19744_CR75","first-page":"5225","volume-title":"2017 IEEE International Conference on Acoustics","author":"S Shahnawazuddin","year":"2017","unstructured":"Shahnawazuddin S, Deepak K, Pradhan G et al (2017) Enhancing noise and pitch robustness of children\u2019s asr. 2017 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 5225\u20135229"},{"key":"19744_CR76","doi-asserted-by":"crossref","unstructured":"Shriberg LD, Paul R, McSweeny JL, et\u00a0al (2001) Speech and prosody characteristics of adolescents and adults with high-functioning autism and asperger syndrome. Journal of Speech, Language, and Hearing Research","DOI":"10.1044\/1092-4388(2001\/087)"},{"issue":"4","key":"19744_CR77","doi-asserted-by":"publisher","first-page":"328","DOI":"10.1109\/89.701361","volume":"6","author":"BL Sim","year":"1998","unstructured":"Sim BL, Tong YC, Chang JS et al (1998) A parametric formulation of the generalized spectral subtraction method. IEEE Trans Speech Audio Process 6(4):328\u2013337","journal-title":"IEEE Trans Speech Audio Process"},{"key":"19744_CR78","doi-asserted-by":"crossref","unstructured":"Soni MH, Shah N, Patil HA (2018) Time-frequency masking-based speech enhancement using generative adversarial network. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), IEEE, pp 5039\u20135043","DOI":"10.1109\/ICASSP.2018.8462068"},{"key":"19744_CR79","doi-asserted-by":"crossref","unstructured":"Srivastava S, Bisht A, Narayan N (2017) Safety and security in smart cities using artificial intelligence\u2013a review. 2017 7th International Conference on Cloud Computing. Data Science & Engineering-Confluence, IEEE, pp 130\u2013133","DOI":"10.1109\/CONFLUENCE.2017.7943136"},{"key":"19744_CR80","unstructured":"Stoller D, Ewert S, Dixon S (2018) Wave-u-net: a multi-scale neural network for end-to-end audio source separation. arXiv:1806.03185"},{"key":"19744_CR81","first-page":"1","volume-title":"2016 International Conference on Microelectronics","author":"V Sunnydayal","year":"2016","unstructured":"Sunnydayal V, Kumar TK (2016) Speech enhancement using $$\\beta $$-divergence based nmf with update bases. 2016 International Conference on Microelectronics. Computing and Communications (MicroCom), IEEE, pp 1\u20136"},{"key":"19744_CR82","doi-asserted-by":"publisher","first-page":"663","DOI":"10.1016\/j.compeleceng.2017.02.021","volume":"62","author":"V Sunnydayal","year":"2017","unstructured":"Sunnydayal V et al (2017) Speech enhancement using posterior regularized nmf with bases update. Comput Electrical Eng 62:663\u2013675","journal-title":"Comput Electrical Eng"},{"issue":"7","key":"19744_CR83","doi-asserted-by":"publisher","first-page":"2125","DOI":"10.1109\/TASL.2011.2114881","volume":"19","author":"CH Taal","year":"2011","unstructured":"Taal CH, Hendriks RC, Heusdens R et al (2011) An algorithm for intelligibility prediction of time-frequency weighted noisy speech. IEEE Trans Audio Speech Language Process 19(7):2125\u20132136","journal-title":"IEEE Trans Audio Speech Language Process"},{"key":"19744_CR84","doi-asserted-by":"crossref","unstructured":"Takahashi N, Mitsufuji Y (2017) Multi-scale multi-band densenets for audio source separation. In: 2017 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA), IEEE, pp 21\u201325","DOI":"10.1109\/WASPAA.2017.8169987"},{"key":"19744_CR85","doi-asserted-by":"crossref","unstructured":"Takahashi N, Goswami N, Mitsufuji Y (2018) Mmdenselstm: An efficient combination of convolutional and recurrent neural networks for audio source separation. In: 2018 16th International workshop on acoustic signal enhancement (IWAENC), IEEE, pp 106\u2013110","DOI":"10.1109\/IWAENC.2018.8521383"},{"key":"19744_CR86","doi-asserted-by":"crossref","unstructured":"Tan K, Wang D (2018) A convolutional recurrent neural network for real-time speech enhancement. In: Interspeech, pp 3229\u20133233","DOI":"10.21437\/Interspeech.2018-1405"},{"key":"19744_CR87","doi-asserted-by":"publisher","first-page":"380","DOI":"10.1109\/TASLP.2019.2955276","volume":"28","author":"K Tan","year":"2019","unstructured":"Tan K, Wang D (2019) Learning complex spectral mapping with gated convolutional recurrent networks for monaural speech enhancement. IEEE\/ACM Trans Audio Speech Language Process 28:380\u2013390","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"issue":"1","key":"19744_CR88","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1109\/TASLP.2018.2876171","volume":"27","author":"K Tan","year":"2018","unstructured":"Tan K, Chen J, Wang D (2018) Gated residual networks with dilated convolutions for monaural speech enhancement. IEEE\/ACM Trans Audio Speech Language Process 27(1):189\u2013198","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"key":"19744_CR89","unstructured":"Tompson JJ, Jain A, LeCun Y, et\u00a0al (2014) Joint training of a convolutional network and a graphical model for human pose estimation. Advan Neural Inform Process Syst 27"},{"key":"19744_CR90","doi-asserted-by":"crossref","unstructured":"Toshev A, Szegedy C (2014) Deeppose: human pose estimation via deep neural networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1653\u20131660","DOI":"10.1109\/CVPR.2014.214"},{"key":"19744_CR91","doi-asserted-by":"crossref","unstructured":"Valentini-Botinhao C, Wang X, Takaki S, et\u00a0al (2016) Investigating rnn-based speech enhancement methods for noise-robust text-to-speech. In: SSW, pp 146\u2013152","DOI":"10.21437\/SSW.2016-24"},{"key":"19744_CR92","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1016\/j.specom.2015.11.004","volume":"77","author":"S Vanambathina","year":"2016","unstructured":"Vanambathina S, Kumar TK (2016) Speech enhancement by bayesian estimation of clean speech modeled as super gaussian given a priori knowledge of phase. Speech Commun 77:8\u201327","journal-title":"Speech Commun"},{"key":"19744_CR93","unstructured":"Vaswani A, Shazeer N, Parmar N, et\u00a0al (2017) Attention is all you need. Advan Neural Inform Process Syst 30"},{"key":"19744_CR94","doi-asserted-by":"crossref","unstructured":"Wang D, Brown GJ (2006) Computational auditory scene analysis: Principles, algorithms, and applications. Wiley-IEEE press","DOI":"10.1109\/9780470043387"},{"issue":"4","key":"19744_CR95","doi-asserted-by":"publisher","first-page":"679","DOI":"10.1109\/TASSP.1982.1163920","volume":"30","author":"D Wang","year":"1982","unstructured":"Wang D, Lim J (1982) The unimportance of phase in speech enhancement. IEEE Trans Acoustics Speech Signal Process 30(4):679\u2013681","journal-title":"IEEE Trans Acoustics Speech Signal Process"},{"key":"19744_CR96","doi-asserted-by":"crossref","unstructured":"Wang Q, Wu B, Zhu P, et\u00a0al (2020) Eca-net: efficient channel attention for deep convolutional neural networks. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11534\u201311542","DOI":"10.1109\/CVPR42600.2020.01155"},{"key":"19744_CR97","first-page":"4390","volume-title":"2015 IEEE International Conference on Acoustics","author":"Y Wang","year":"2015","unstructured":"Wang Y, Wang D (2015) A deep neural network for time-domain signal reconstruction. 2015 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 4390\u20134394"},{"issue":"12","key":"19744_CR98","doi-asserted-by":"publisher","first-page":"1849","DOI":"10.1109\/TASLP.2014.2352935","volume":"22","author":"Y Wang","year":"2014","unstructured":"Wang Y, Narayanan A, Wang D (2014) On training targets for supervised speech separation. IEEE\/ACM Trans Audio Speech Language Process 22(12):1849\u20131858","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"key":"19744_CR99","doi-asserted-by":"crossref","unstructured":"Weninger F, Eyben F, Schuller B (2014a) Single-channel speech separation with memory-enhanced recurrent neural networks. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP), IEEE, pp 3709\u20133713","DOI":"10.1109\/ICASSP.2014.6854294"},{"key":"19744_CR100","doi-asserted-by":"crossref","unstructured":"Weninger F, Hershey JR, Le\u00a0Roux J, et\u00a0al (2014b) Discriminatively trained recurrent neural networks for single-channel speech separation. In: 2014 IEEE global conference on signal and information processing (GlobalSIP), IEEE, pp 577\u2013581","DOI":"10.1109\/GlobalSIP.2014.7032183"},{"issue":"3","key":"19744_CR101","doi-asserted-by":"publisher","first-page":"483","DOI":"10.1109\/TASLP.2015.2512042","volume":"24","author":"DS Williamson","year":"2015","unstructured":"Williamson DS, Wang Y, Wang D (2015) Complex ratio masking for monaural speech separation. IEEE\/ACM Trans Audio Speech Language Process 24(3):483\u2013492","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"key":"19744_CR102","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1109\/LSP.2021.3128374","volume":"29","author":"X Xiang","year":"2021","unstructured":"Xiang X, Zhang X, Chen H (2021) A nested u-net with self-attention and dense connectivity for monaural speech enhancement. IEEE Signal Process Lett 29:105\u2013109","journal-title":"IEEE Signal Process Lett"},{"issue":"1","key":"19744_CR103","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1109\/LSP.2013.2291240","volume":"21","author":"Y Xu","year":"2013","unstructured":"Xu Y, Du J, Dai LR et al (2013) An experimental study on speech enhancement based on deep neural networks. IEEE Signal Process Lett 21(1):65\u201368","journal-title":"IEEE Signal Process Lett"},{"issue":"2","key":"19744_CR104","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1109\/T-AFFC.2012.38","volume":"4","author":"Y Yang","year":"2012","unstructured":"Yang Y, Fairbairn C, Cohn JF (2012) Detecting depression severity from vocal prosody. IEEE Trans Affective Comput 4(2):142\u2013150","journal-title":"IEEE Trans Affective Comput"},{"issue":"06","key":"19744_CR105","doi-asserted-by":"publisher","first-page":"2350054","DOI":"10.1142\/S0219467823500547","volume":"23","author":"S Yechuri","year":"2023","unstructured":"Yechuri S, Vanabathina SD (2023) Genetic algorithm-based adaptive wiener gain for speech enhancement using an iterative posterior nmf. Int J Image Graph 23(06):2350054","journal-title":"Int J Image Graph"},{"key":"19744_CR106","unstructured":"Zhang Q, Nicolson A, Wang M, et\u00a0al (2019) Monaural speech enhancement using a multi-branch temporal convolutional network. arXiv:1912.12023"},{"issue":"1","key":"19744_CR107","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1093\/nsr\/nwx105","volume":"5","author":"Y Zhang","year":"2018","unstructured":"Zhang Y, Yang Q (2018) An overview of multi-task learning. National Sci Rev 5(1):30\u201343","journal-title":"National Sci Rev"},{"key":"19744_CR108","first-page":"2401","volume-title":"2018 IEEE International Conference on Acoustics","author":"H Zhao","year":"2018","unstructured":"Zhao H, Zarar S, Tashev I et al (2018) Convolutional-recurrent neural networks for speech enhancement. 2018 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 2401\u20132405"},{"key":"19744_CR109","first-page":"6648","volume-title":"ICASSP 2021\u20132021 IEEE International Conference on Acoustics","author":"S Zhao","year":"2021","unstructured":"Zhao S, Nguyen TH, Ma B (2021) Monaural speech enhancement with complex convolutional block attention module and joint time frequency losses. ICASSP 2021\u20132021 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 6648\u20136652"},{"key":"19744_CR110","first-page":"9281","volume-title":"ICASSP 2022\u20132022 IEEE International Conference on Acoustics","author":"S Zhao","year":"2022","unstructured":"Zhao S, Ma B, Watcharasupat KN et al (2022) Frcrn: Boosting feature representation using frequency recurrence for monaural speech enhancement. ICASSP 2022\u20132022 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 9281\u20139285"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-19744-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-024-19744-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-19744-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T04:04:19Z","timestamp":1748059459000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-024-19744-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,28]]},"references-count":110,"journal-issue":{"issue":"16","published-online":{"date-parts":[[2025,5]]}},"alternative-id":["19744"],"URL":"https:\/\/doi.org\/10.1007\/s11042-024-19744-1","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,6,28]]},"assertion":[{"value":"14 October 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 May 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 June 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 June 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"No conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}