{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T15:48:27Z","timestamp":1778860107250,"version":"3.51.4"},"reference-count":49,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62061136001"],"award-info":[{"award-number":["62061136001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61836014"],"award-info":[{"award-number":["61836014"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/taslp.2023.3321191","type":"journal-article","created":{"date-parts":[[2023,10,4]],"date-time":"2023-10-04T17:44:38Z","timestamp":1696441478000},"page":"3362-3373","source":"Crossref","is-referenced-by-count":8,"title":["A Fast High-Fidelity Source-Filter Vocoder With Lightweight Neural Modules"],"prefix":"10.1109","volume":"31","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-2526-1909","authenticated-orcid":false,"given":"Runxuan","family":"Yang","sequence":"first","affiliation":[{"name":"Department of Computer Science and Technology, State Key Laboratory of Intelligent Technology and Systems, Institute for AI, THBI, BNRist, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1955-2635","authenticated-orcid":false,"given":"Yuyang","family":"Peng","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4907-7354","authenticated-orcid":false,"given":"Xiaolin","family":"Hu","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Technology, State Key Laboratory of Intelligent Technology and Systems, Institute for AI, THBI, BNRist, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"ref12","article-title":"ClariNet: Parallel wave generation in end-to-end text-to-speech","author":"ping","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref15","first-page":"17022","article-title":"HiFi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis","author":"kong","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref14","article-title":"MelGAN: Generative adversarial networks for conditional waveform synthesis","author":"kumar","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref11","first-page":"3918","article-title":"Parallel WaveNet: Fast high-fidelity speech synthesis","author":"oord","year":"0","journal-title":"Proc 35th Int Conf Mach Learn"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746713"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1016"},{"key":"ref19","article-title":"DiffWave: A versatile diffusion model for audio synthesis","author":"kong","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref18","article-title":"WaveGrad: Estimating gradients for waveform generation","author":"chen","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref46","year":"2014","journal-title":"Method for the Subjective Assessment of Intermediate Quality Level of Audio Systems"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475437"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.2982285"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461329"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.195"},{"key":"ref41","article-title":"DDSP: Differentiable digital signal processing","author":"engel","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref44","first-page":"807","article-title":"Rectified linear units improve restricted boltzmann machines","author":"nair","year":"0","journal-title":"Proc 27th Int Conf Mach Learn"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1984.1164317"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2013.2271648"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462431"},{"key":"ref7","article-title":"SampleRNN: An unconditional end-to-end neural audio generation model","author":"mehri","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref9","first-page":"2410","article-title":"Efficient neural audio synthesis","author":"kalchbrenner","year":"0","journal-title":"Proc 35th Int Conf Mach Learn"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2008.4518514"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-68"},{"key":"ref5","article-title":"Fast and reliable F0 estimation method based on the period extraction of vocal fold vibration of singing voice and speech","author":"morise","year":"0","journal-title":"Proc Audio Eng Soc Conf 35th Int Conf Audio Games"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-67"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1121\/1.1911395"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1121\/1.1918949"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1974.1162554"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1121\/1.1912679"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-517"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1018"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1121\/1.1916020"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095298"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1250\/ast.27.349"},{"key":"ref1","first-page":"125","article-title":"WaveNet: A generative model for raw audio","author":"oord","year":"0","journal-title":"Proc 9th ISCA Workshop Speech Synth Workshop"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2016.09.001"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2014.09.003"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682804"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3051765"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683271"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2008"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-301"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3061245"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095749"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682298"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3188"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2956145"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9970249\/10271535.pdf?arnumber=10271535","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,13]],"date-time":"2023-11-13T19:36:43Z","timestamp":1699904203000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10271535\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":49,"URL":"https:\/\/doi.org\/10.1109\/taslp.2023.3321191","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]}}}