{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,2]],"date-time":"2025-07-02T04:24:08Z","timestamp":1751430248087,"version":"3.41.0"},"reference-count":36,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2025,3,12]],"date-time":"2025-03-12T00:00:00Z","timestamp":1741737600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,12]],"date-time":"2025-03-12T00:00:00Z","timestamp":1741737600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.61701286"],"award-info":[{"award-number":["No.61701286"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["ZR2022MF330"],"award-info":[{"award-number":["ZR2022MF330"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Circuits Syst Signal Process"],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1007\/s00034-025-03063-3","type":"journal-article","created":{"date-parts":[[2025,3,12]],"date-time":"2025-03-12T18:06:18Z","timestamp":1741802778000},"page":"5260-5278","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A Time-Domain Speech Enhancement Model with Controllable Output Based on Conditional Network"],"prefix":"10.1007","volume":"44","author":[{"given":"Qingyang","family":"Qu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiahui","family":"Song","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuepeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenhao","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,12]]},"reference":[{"key":"3063_CR1","unstructured":"F. Albu, N. Dumitriu, L. D. Stanciu, Speech enhancement by spectral subtraction, in Proceedings of International Symposium on Electronics and Telecommunications (1996), pp. 78\u201383"},{"issue":"1","key":"3063_CR2","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1186\/s40537-023-00727-2","volume":"10","author":"L Alzubaidi","year":"2023","unstructured":"L. Alzubaidi, J. Bai, A. Al-Sabaawi, A survey on deep learning tools dealing with data scarcity: definitions, challenges, solutions, tips, and applications. J. Big Data 10(1), 46 (2023)","journal-title":"J. Big Data"},{"key":"3063_CR3","doi-asserted-by":"publisher","first-page":"106","DOI":"10.1109\/TASLP.2020.3036783","volume":"29","author":"L Chai","year":"2021","unstructured":"L. Chai, J. Du, Q.-F. Liu, C.-H. Lee, A cross-entropy-guided measure (CEGM) for assessing speech recognition performance and optimizing DNN-based speech enhancement. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 106\u2013117 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"3063_CR4","doi-asserted-by":"crossref","unstructured":"A. Defossez, G. Synnaeve, Y. Adi, Real time speech enhancement in the waveform domain, in Proceedings of the Annual Conference of the International Speech Communication Association, INTERSPEECH (2020), pp. 3291\u20133295","DOI":"10.21437\/Interspeech.2020-2409"},{"issue":"2","key":"3063_CR5","doi-asserted-by":"publisher","first-page":"443","DOI":"10.1109\/TASSP.1985.1164550","volume":"33","author":"Y Ephraim","year":"1985","unstructured":"Y. Ephraim, D. Malah, Speech enhancement using a minimum mean-square error log-spectral amplitude estimator. IEEE Trans. Acoust. Speech Signal Process. 33(2), 443\u2013445 (1985)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"issue":"6","key":"3063_CR6","doi-asserted-by":"publisher","first-page":"1109","DOI":"10.1109\/TASSP.1984.1164453","volume":"32","author":"Y Ephraim","year":"1984","unstructured":"Y. Ephraim, D. Malah, Speech enhancement using a minimum-mean square error short-time spectral amplitude estimator. IEEE Trans. Acoust. Speech Signal Process. 32(6), 1109\u20131121 (1984)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"key":"3063_CR7","doi-asserted-by":"crossref","unstructured":"K. Han, Y. Wang, D.L. Wang, Learning spectral mapping for speech dereverberation, in 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2014), pp. 4628\u20134632","DOI":"10.1109\/ICASSP.2014.6854479"},{"key":"3063_CR8","doi-asserted-by":"publisher","first-page":"2149","DOI":"10.1109\/LSP.2020.3040693","volume":"27","author":"T Hsieh","year":"2020","unstructured":"T. Hsieh, H. Wang, X. Lu, WaveCRN: An efficient convolutional recurrent neural network for end-to-end speech enhancement. IEEE Signal Process. Lett. 27, 2149\u20132153 (2020)","journal-title":"IEEE Signal Process. Lett."},{"key":"3063_CR9","doi-asserted-by":"publisher","first-page":"1939","DOI":"10.1109\/TASLP.2022.3180676","volume":"30","author":"M Kim","year":"2022","unstructured":"M. Kim, J.W. Shin, Improved speech enhancement considering speech PSD uncertainty. IEEE\/ACM Trans. Audio Speech Lang. Process. 30, 1939\u20131951 (2022)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"3063_CR10","doi-asserted-by":"crossref","unstructured":"Z. Kong, W. Ping, A. Dantrey, B. Catanzaro, Speech denoising in the waveform domain with self-attention, in ICASSP 2022\u20142022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2022), pp. 7867\u20137871","DOI":"10.1109\/ICASSP43922.2022.9746169"},{"key":"3063_CR11","unstructured":"P.-J. Ku, C.-H.H. Yang, S.M. Siniscalchi, A multi-dimensional deep structured state space approach to speech enhancement using small-footprint models. arXiv preprint arXiv:2306.00331 (2023)"},{"key":"3063_CR12","doi-asserted-by":"crossref","unstructured":"B. Kumar (2021) Comparative performance evaluation of greedy algorithms for speech enhancement system. Fluct. Noise Lett. 20(02): 2150017","DOI":"10.1142\/S0219477521500176"},{"key":"3063_CR13","doi-asserted-by":"crossref","unstructured":"X. Lu, Y. Tsao, S. Matsuda, Speech enhancement based on deep denoising autoencoder, in Proceedings of the Annual Conference of the International Speech Communication Association, INTERSPEECH (2013), pp. 436\u2013440","DOI":"10.21437\/Interspeech.2013-130"},{"key":"3063_CR14","doi-asserted-by":"crossref","unstructured":"Y. Lu, Z. Wang, S. Watanabe, Conditional diffusion probabilistic model for speech enhancement, in ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2022), pp. 7402\u20137406","DOI":"10.1109\/ICASSP43922.2022.9746901"},{"key":"3063_CR15","unstructured":"C. Macartney, T. Weyde, Improved speech enhancement with the Wave-U-Net. arXiv preprint arXiv:1811.11307 (2018)"},{"key":"3063_CR16","unstructured":"A. v. d. Oord, S. Dieleman, H. Zen, Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499 (2016)"},{"key":"3063_CR17","unstructured":"A. Oord, Y. Li, I. Babuschkin, Parallel wavenet: fast high-fidelity speech synthesis, in International Conference on Machine Learning. PMLR (2018), pp. 3918\u20133926"},{"key":"3063_CR18","doi-asserted-by":"crossref","unstructured":"A. Pandey, D. Wang, TCNN: temporal convolutional neural network for real-time speech enhancement in the time domain, in ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2019), pp. 6875\u20136879","DOI":"10.1109\/ICASSP.2019.8683634"},{"key":"3063_CR19","doi-asserted-by":"crossref","unstructured":"A. Pandey, D. Wang, Densely connected neural network with dilated convolutions for real-time speech enhancement in the time domain, in ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). pp. 6629\u20136633(2020)","DOI":"10.1109\/ICASSP40776.2020.9054536"},{"key":"3063_CR20","doi-asserted-by":"publisher","first-page":"1270","DOI":"10.1109\/TASLP.2021.3064421","volume":"29","author":"A Pandey","year":"2021","unstructured":"A. Pandey, D. Wang, Dense CNN With self-attention for time-domain speech enhancement. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 1270\u20131279 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"3063_CR21","doi-asserted-by":"crossref","unstructured":"S. Pascual, A. Bonafonte, J. Serr\u00e0, SEGAN: Speech enhancement generative adversarial network, in Interspeech 2017 (2017), pp. 3642\u20133646","DOI":"10.21437\/Interspeech.2017-1428"},{"key":"3063_CR22","doi-asserted-by":"crossref","unstructured":"C.K. Reddy, V. Gopal, R. Cutler, DNSMOS P. 835: a non-intrusive perceptual objective speech quality metric to evaluate noise suppressors, in ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2022), p. 886\u2013890","DOI":"10.1109\/ICASSP43922.2022.9746108"},{"key":"3063_CR23","doi-asserted-by":"crossref","unstructured":"A.W. Rix, J.G. Beerends, M.P. Hollier, Perceptual evaluation of speech quality (PESQ)\u2014a new method for speech quality assessment of telephone networks and codecs. in 2001 IEEE International Conference on Acoustics, Speech and Signal Processing. Proceedings (Cat.No.01CH37221) (2001) , p. 749\u2013752","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"3063_CR24","unstructured":"D. Stoller, S. Ewert, S. Dixon, Wave-u-net: a multi-scale neural network for end-to-end audio source separation. arXiv preprint arXiv:1806.03185 (2018)"},{"key":"3063_CR25","doi-asserted-by":"crossref","unstructured":"A. Sugiyama, O. Shimada, T. Nomura, User preference between residual noise and speech distortion in speech enhancement, in 2022 International Workshop on Acoustic Signal Enhancement (IWAENC) (2022) , p. 1\u20135","DOI":"10.1109\/IWAENC53105.2022.9914802"},{"key":"3063_CR26","doi-asserted-by":"crossref","unstructured":"C.H. Taal, R.C. Hendriks, R. Heusdens, A short-time objective intelligibility measure for time-frequency weighted noisy speech, in 2010 IEEE International Conference on Acoustics, Speech and Signal Processing (2010), p. 214\u20134217","DOI":"10.1109\/ICASSP.2010.5495701"},{"key":"3063_CR27","unstructured":"C. Valentini-Botinhao, Noisy speech database for training speech enhancement algorithms and TTS models, 2016 [sound]. University of Edinburgh. School of Informatics. Centre for Speech Technology Research (CSTR) (2017)"},{"issue":"12","key":"3063_CR28","doi-asserted-by":"publisher","first-page":"1849","DOI":"10.1109\/TASLP.2014.2352935","volume":"22","author":"Y Wang","year":"2014","unstructured":"Y. Wang, A. Narayanan, D. Wang, On training targets for supervised speech separation. IEEE\/ACM Trans. Audio, Speech Lang. Process. 22(12), 1849\u20131858 (2014)","journal-title":"IEEE\/ACM Trans. Audio, Speech Lang. Process."},{"issue":"7","key":"3063_CR29","doi-asserted-by":"publisher","first-page":"1381","DOI":"10.1109\/TASL.2013.2250961","volume":"21","author":"Y Wang","year":"2013","unstructured":"Y. Wang, D. Wang, Towards scaling up classification-based speech separation. IEEE Trans. Audio Speech Lang. Process. 21(7), 1381\u20131390 (2013)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"3063_CR30","doi-asserted-by":"crossref","unstructured":"Y. Xia, S. Braun, C.K. Reddy, Weighted speech distortion losses for neural-network-based real-time speech enhancement, in ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) pp. 871\u2013875(2020)","DOI":"10.1109\/ICASSP40776.2020.9054254"},{"key":"3063_CR31","doi-asserted-by":"publisher","first-page":"164","DOI":"10.1109\/TASLP.2023.3321975","volume":"32","author":"Y Xiang","year":"2024","unstructured":"Y. Xiang, J.L. H\u00f8jvang, M.H. Rasmussen, M.G. Christensen, A two-Stage deep representation learning-based speech enhancement method using variational autoencoder and adversarial training. IEEE\/ACM Trans. Audio Speech Lang. Process. 32, 164\u2013177 (2024)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"3063_CR32","doi-asserted-by":"publisher","first-page":"1455","DOI":"10.1109\/LSP.2021.3093859","volume":"28","author":"X Xiang","year":"2021","unstructured":"X. Xiang, X. Zhang, H. Chen, A convolutional network with multi-scale and attention mechanisms for end-to-end single-channel speech enhancement. IEEE Signal Process. Lett. 28, 1455\u20131459 (2021)","journal-title":"IEEE Signal Process. Lett."},{"key":"3063_CR33","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1109\/LSP.2021.3128374","volume":"29","author":"X Xiang","year":"2022","unstructured":"X. Xiang, X. Zhang, H. Chen, A nested u-net with self-attention and dense connectivity for monaural speech enhancement. IEEE Signal Process. Lett. 29, 105\u2013109 (2022)","journal-title":"IEEE Signal Process. Lett."},{"issue":"1","key":"3063_CR34","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1109\/LSP.2013.2291240","volume":"21","author":"Y Xu","year":"2014","unstructured":"Y. Xu, J. Du, L.-R. Dai, C.-H. Lee, An experimental study on speech enhancement based on deep neural networks. IEEE Signal Process. Lett. 21(1), 65\u201368 (2014)","journal-title":"IEEE Signal Process. Lett."},{"issue":"1","key":"3063_CR35","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","volume":"23","author":"Y Xu","year":"2015","unstructured":"Y. Xu, J. Du, L.-R. Dai, C.-H. Lee, A regression approach to speech enhancement based on deep neural networks. IEEE\/ACM Trans. Audio Speech Lang. Process. 23(1), 7\u201319 (2015)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"3063_CR36","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2023.109718","volume":"215","author":"W Yuan","year":"2023","unstructured":"W. Yuan, Q. Qu, J. Song, Dynamic controllable speech enhancement models based on quantile loss functions. Appl. Acoust. 215, 109718 (2023)","journal-title":"Appl. Acoust."}],"container-title":["Circuits, Systems, and Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-03063-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00034-025-03063-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-03063-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T21:02:23Z","timestamp":1751403743000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00034-025-03063-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,12]]},"references-count":36,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2025,7]]}},"alternative-id":["3063"],"URL":"https:\/\/doi.org\/10.1007\/s00034-025-03063-3","relation":{},"ISSN":["0278-081X","1531-5878"],"issn-type":[{"type":"print","value":"0278-081X"},{"type":"electronic","value":"1531-5878"}],"subject":[],"published":{"date-parts":[[2025,3,12]]},"assertion":[{"value":"24 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 February 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 February 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 March 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}