{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,13]],"date-time":"2025-05-13T06:46:39Z","timestamp":1747118799998},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T00:00:00Z","timestamp":1709251200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T00:00:00Z","timestamp":1709251200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1007\/s10772-024-10091-y","type":"journal-article","created":{"date-parts":[[2024,3,26]],"date-time":"2024-03-26T16:02:19Z","timestamp":1711468939000},"page":"201-209","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Conditional Denoising Diffusion Implicit Model for Speech Enhancement"],"prefix":"10.1007","volume":"27","author":[{"given":"Chengyong","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiukang","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sheng","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,3,26]]},"reference":[{"key":"10091_CR1","doi-asserted-by":"crossref","unstructured":"Defossez, A., Synnaeve, G., & Adi, Y. (2020). Real time speech enhancement in the waveform domain. arXiv:2006.12847v3.","DOI":"10.21437\/Interspeech.2020-2409"},{"key":"10091_CR2","doi-asserted-by":"crossref","unstructured":"Fang, H., Carbajal, G., Wermter, S., & Gerkmann, T. (2021). Variational autoencoder for speech enhancement with a noise-aware encoder. In ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 676\u2013680).","DOI":"10.1109\/ICASSP39728.2021.9414060"},{"key":"10091_CR3","unstructured":"Fu, S.-W., Liao, C.-F., Tsao, Y., & Lin, S.-D. (2019). MetricGAN: Generative adversarial networks based black-box metric scores optimization for speech enhancement (No. arXiv:1905.04874)."},{"key":"10091_CR4","doi-asserted-by":"crossref","unstructured":"Germain, F.G., Chen, Q., & Koltun, V. (2018). Speech denoising with deep feature losses (No. arXiv:1806.10522).","DOI":"10.21437\/Interspeech.2019-1924"},{"key":"10091_CR5","doi-asserted-by":"crossref","unstructured":"Hang, T., Gu, S., Li, C., Bao, J., Chen, D., Hu, H., Geng, X., & Guo, B. (2023). Efficient diffusion training via Min-SNR weighting strategy (No. arXiv:2303.09556).","DOI":"10.1109\/ICCV51070.2023.00684"},{"key":"10091_CR6","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models (No. arXiv:2006.11239)."},{"issue":"1","key":"10091_CR7","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TASL.2007.911054","volume":"16","author":"Y Hu","year":"2008","unstructured":"Hu, Y., & Loizou, P. C. (2008). Evaluation of objective quality measures for speech enhancement. IEEE Transactions on Audio, Speech, and Language Processing, 16(1), 229\u2013238.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"issue":"11","key":"10091_CR8","doi-asserted-by":"publisher","first-page":"1601","DOI":"10.1109\/LSP.2017.2750979","volume":"24","author":"CKA Reddy","year":"2017","unstructured":"Reddy, C. K. A., Shankar, N., Shreedhar, B. C., Charan, R., & Panahi, I. (2017). An individualized super-gaussian single microphone speech enhancement for hearing aid users with smartphone as an assistive device. IEEE Signal Processing Letters, 24(11), 1601\u20131605. https:\/\/doi.org\/10.1109\/LSP.2017.2750979","journal-title":"IEEE Signal Processing Letters"},{"key":"10091_CR9","unstructured":"Kong, Z., Ping, W., Huang, J., Zhao, K., & Catanzaro, B. (2021). DiffWave: A versatile diffusion model for audio synthesis (No. arXiv:2009.09761)."},{"key":"10091_CR10","doi-asserted-by":"crossref","unstructured":"Leglaive, S., Girin, L., & Horaud, R. (2018). A variance modeling framework based on variational autoencoders for speech enhancement. In 2018 IEEE 28th international workshop on machine learning for signal processing (MLSP) (pp. 1\u20136).","DOI":"10.1109\/MLSP.2018.8516711"},{"key":"10091_CR11","doi-asserted-by":"publisher","DOI":"10.1201\/b14529","volume-title":"Speech enhancement: Theory and practice","author":"PC Loizou","year":"2013","unstructured":"Loizou, P. C. (2013). Speech enhancement: Theory and practice (ed, Vol. 2). CRC Press.","edition":"ed"},{"key":"10091_CR12","first-page":"5775","volume":"35","author":"C Lu","year":"2022","unstructured":"Lu, C., Zhou, Y., Bao, F., Chen, J., Li, C., & Zhu, J. (2022). DPM-solver: A fast ode solver for diffusion probabilistic model sampling in around 10 steps. Advances in Neural Information Processing Systems, 35, 5775\u20135787.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10091_CR13","unstructured":"Lu, Y.-J., Tsao, Y., & Watanabe, S. (2021). A study on speech enhancement based on diffusion probabilistic model (No. arXiv:2107.11876)."},{"key":"10091_CR14","doi-asserted-by":"crossref","unstructured":"Lu, Y.-J., Wang, Z.-Q., Watanabe, S., Richard, A., Yu, C., & Tsao, Y. (2022). Conditional diffusion probabilistic model for speech enhancement. In ICASSP 2022\u20132022 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 7402\u20137406).","DOI":"10.1109\/ICASSP43922.2022.9746901"},{"key":"10091_CR15","doi-asserted-by":"crossref","unstructured":"Michelsanti, D., & Tan, Z.-H. (2017). Conditional generative adversarial networks for speech enhancement and noise-robust speaker verification. Interspeech 2017 (pp. 2008\u20132012).","DOI":"10.21437\/Interspeech.2017-1620"},{"key":"10091_CR16","unstructured":"Nichol, A.Q., & Dhariwal, P. (2021). Improved denoising diffusion probabilistic models. In M. Meila & T. Zhang (Eds.), Proceedings of the 38th international conference on machine learning (ICML) (Vol. 139, pp. 8162\u20138171). PLMR"},{"key":"10091_CR17","doi-asserted-by":"crossref","unstructured":"Nossier, S.A., Wall, J., Moniri, M., Glackin, C., & Cannings, N. (2020). Mapping and masking targets comparison using different deep learning based speech enhancement architectures. In 2020 international joint conference on neural networks (IJCNN) (pp. 1\u20138).","DOI":"10.1109\/IJCNN48605.2020.9206623"},{"key":"10091_CR18","doi-asserted-by":"publisher","first-page":"1104","DOI":"10.1109\/TASLP.2020.2979603","volume":"28","author":"AA Nugraha","year":"2020","unstructured":"Nugraha, A. A., Sekiguchi, K., & Yoshii, K. (2020). A flow-based deep latent variable model for speech spectrogram modeling and enhancement. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 28, 1104\u20131117.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10091_CR19","doi-asserted-by":"crossref","unstructured":"Pascual, S., Bonafonte, A., & Serr\u00e0, J. (2017). SEGAN: Speech enhancement generative adversarial network (No. arXiv:1703.09452).","DOI":"10.21437\/Interspeech.2017-1428"},{"key":"10091_CR20","doi-asserted-by":"crossref","unstructured":"Phan, H., McLoughlin, I.V., Pham, L., Ch\u00e9n, O.Y., Koch, P., De Vos, M., & Mertins, A. (2020). Improving GANs for speech enhancement. In IEEE signal processing letters, (Vol. 27 , pp. 1700\u20131704), arxiv:2001.05532 [cs, eess, stat].","DOI":"10.1109\/LSP.2020.3025020"},{"key":"10091_CR21","doi-asserted-by":"crossref","unstructured":"Rix, A., Beerends, J., Hollier, M., & Hekstra, A. (2001). Perceptual evaluation of speech quality (PESQ): A new method for speech quality assessment of telephone networks and codecs. In 2001 IEEE international conference on acoustics, speech, and signal processing. Proceedings (Cat. No.01CH37221) (Vol. 2, pp. 749-752).","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"10091_CR22","unstructured":"Sohl-Dickstein, J., Weiss, E.A., Maheswaranathan, N., & Ganguli, S. (2015). Deep unsupervised learning using nonequilibrium thermodynamics (No. arXiv:1503.03585)."},{"key":"10091_CR23","unstructured":"Song, J., Meng, C., & Ermon, S. (2022). Denoising diffusion implicit models (No. arXiv:2010.02502)."},{"key":"10091_CR24","doi-asserted-by":"crossref","unstructured":"Soni, M.H., Shah, N., & Patil, H.A. (2018). Time-frequency masking-based speech enhancement using generative adversarial network. In 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 5039\u20135043).","DOI":"10.1109\/ICASSP.2018.8462068"},{"key":"10091_CR25","doi-asserted-by":"crossref","unstructured":"Strauss, M., & Edler, B. (2021). A flow-based neural network for time domain speech enhancement. In ICASSP 2021\u20132021 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 5754\u20135758).","DOI":"10.1109\/ICASSP39728.2021.9413999"},{"issue":"11","key":"10091_CR26","doi-asserted-by":"publisher","first-page":"13627","DOI":"10.1609\/aaai.v37i11.26597","volume":"37","author":"W Tai","year":"2023","unstructured":"Tai, W., Zhou, F., Trajcevski, G., & Zhong, T. (2023). Revisiting denoising diffusion probabilistic models for speech enhancement: Condition collapse, efficiency and refinement. Proceedings of the AAAI Conference on Artificial Intelligence, 37(11), 13627\u201313635.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"issue":"5 Supplement","key":"10091_CR27","doi-asserted-by":"publisher","first-page":"3591","DOI":"10.1121\/1.4806631","volume":"133","author":"J Thiemann","year":"2013","unstructured":"Thiemann, J., Ito, N., & Vincent, E. (2013). The diverse environments multi-channel acoustic noise database: A database of multichannel environmental noise recordings. The Journal of the Acoustical Society of America, 133(5 Supplement), 3591\u20133591.","journal-title":"The Journal of the Acoustical Society of America"},{"key":"10091_CR28","unstructured":"Ulhaq, A., Akhtar, N., & Pogrebna, G. (2022). Efficient diffusion models for vision: A survey (No. arXiv:2210.09292)."},{"key":"10091_CR29","doi-asserted-by":"crossref","unstructured":"Valentini-Botinhao, C., Wang, X., Takaki, S., & Yamagishi, J. (2016). Investigating RNNbased speech enhancement methods for noise-robust text-to-speech. In 9th ISCA workshop on speech synthesis workshop (SSW 9) (pp. 146\u2013152).","DOI":"10.21437\/SSW.2016-24"},{"key":"10091_CR30","doi-asserted-by":"crossref","unstructured":"Veaux, C., Yamagishi, J., & King, S. (2013). The voice bank corpus: Design, collection and data analysis of a large regional accent speech database. In 2013 international conference oriental COCOSDA held jointly with 2013 conference on Asian spoken language research and evaluation (OCOCOSDA\/ CASLRE) (pp. 1\u20134).","DOI":"10.1109\/ICSDA.2013.6709856"},{"key":"10091_CR31","doi-asserted-by":"crossref","unstructured":"Wang, D., & Chen, J. (2018). Supervised speech separation based on deep learning: An overview (No. arXiv:1708.07524).","DOI":"10.1109\/TASLP.2018.2842159"},{"key":"10091_CR32","doi-asserted-by":"crossref","unstructured":"Welker, S., Richter, J., & Gerkmann, T. (2022). Speech enhancement with score-based generative models in the complex STFT domain (No. arXiv:2203.17004).","DOI":"10.21437\/Interspeech.2022-10653"},{"key":"10091_CR33","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4471-5779-3","volume-title":"Automatic speech recognition: A deep learning approach","author":"D Yu","year":"2015","unstructured":"Yu, D., & Deng, L. (2015). Automatic speech recognition: A deep learning approach. Springer."}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10091-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10091-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10091-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T15:15:14Z","timestamp":1715613314000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10091-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3]]},"references-count":33,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,3]]}},"alternative-id":["10091"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10091-y","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,3]]},"assertion":[{"value":"9 January 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 February 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 March 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}