{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:38:54Z","timestamp":1776886734110,"version":"3.51.2"},"reference-count":70,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,1,9]],"date-time":"2023-01-09T00:00:00Z","timestamp":1673222400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,1,9]],"date-time":"2023-01-09T00:00:00Z","timestamp":1673222400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,1,9]]},"DOI":"10.1109\/slt54892.2023.10022496","type":"proceedings-article","created":{"date-parts":[[2023,1,27]],"date-time":"2023-01-27T18:54:03Z","timestamp":1674845643000},"page":"884-891","source":"Crossref","is-referenced-by-count":17,"title":["Wavefit: an Iterative and Non-Autoregressive Neural Vocoder Based on Fixed-Point Iteration"],"prefix":"10.1109","author":[{"given":"Yuma","family":"Koizumi","sequence":"first","affiliation":[{"name":"Google Research,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kohei","family":"Yatabe","sequence":"additional","affiliation":[{"name":"Tokyo University of Agriculture and Technology,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Heiga","family":"Zen","sequence":"additional","affiliation":[{"name":"Google Research,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michiel","family":"Bacchiani","sequence":"additional","affiliation":[{"name":"Google Research,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"SampleRNN: An unconditional end-to-end neural audio","volume-title":"Proc. ICLR","author":"Mehri"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2017-314"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"ref4","article-title":"WaveFlow: A com-pact flow-based model for raw audio","volume-title":"Proc. ICML","author":"Ping"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref6","article-title":"Non-attentive Tacotron: Robust and controllable neu-ral TTS synthesis including unsupervised duration modeling","author":"Shen","year":"2020","journal-title":"arXiv"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/icassp39728.2021.9414718"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1757"},{"key":"ref9","article-title":"FastSpeech: Fast, robust and controllable text to speech","volume-title":"Proc. NeurIPS","author":"Ren"},{"key":"ref10","article-title":"FastSpeech 2: Fast and high-quality end-to-end text to speech","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Ren"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3038524"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3193761"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1951"},{"key":"ref14","article-title":"Translatotron 2: High-quality direct speech-to-speech trans-lation with voice preservation","volume-title":"Proc. ICML","author":"Jia"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.235"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2019.8937165"},{"key":"ref17","article-title":"Speaker independence of neural vocoders and their ef-fect on parametric resynthesis speech enhancement","volume-title":"Proc. ICASSP","author":"Maiti"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2143"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA52581.2021.9632770"},{"key":"ref20","article-title":"VoiceFixer: Toward general speech restoration with neural vocoder","author":"Liu","year":"2021","journal-title":"arXiv"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-298"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462529"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639598"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1255"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3129994"},{"key":"ref26","article-title":"WaveNet: A generative model for raw au-dio","author":"van den Oord","year":"2016","journal-title":"arXiv"},{"key":"ref27","article-title":"Efficient neural audio syn-thesis","volume-title":"Proc. ICML","author":"Kalchbrenner"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682804"},{"key":"ref29","article-title":"Parallel WaveNet: Fast high-fidelity speech synthesis","volume-title":"Proc. ICML","author":"van den Oord"},{"key":"ref30","article-title":"Variational inference with normalizing flows","volume-title":"Proc. ICML","author":"Rezende"},{"key":"ref31","article-title":"Generative ad-versarial nets","volume-title":"Proc. NeurIPS","author":"Goodfellow"},{"key":"ref32","article-title":"Adversarial audio synthesis","volume-title":"Proc. ICLR","author":"Donahue"},{"key":"ref33","article-title":"HiFi-GAN: Generative adversar-ial networks for efficient and high fidelity speech synthesis","volume-title":"Proc. NeurIPS","author":"Kong"},{"key":"ref34","article-title":"MelGAN: Generative adversarial networks for conditional waveform synthesis","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","author":"Kumar"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383551"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-41"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1016"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746713"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i11.26479"},{"key":"ref41","article-title":"Big VGAN: A universal neural vocoder with large-scale training","author":"Lee","year":"2022","journal-title":"arXiv"},{"key":"ref42","article-title":"WaveGrad: Estimating gradients for waveform gen-eration","volume-title":"Proc. ICLR","author":"Chen"},{"key":"ref43","article-title":"Diff-Wave: A versatile diffusion model for audio synthesis","volume-title":"Proc. ICLR","author":"Kong"},{"key":"ref44","article-title":"BDDM: Bilateral denoising diffusion models for fast and high-quality speech synthesis","volume-title":"Proc. ICLR","author":"Lam"},{"key":"ref45","article-title":"PriorGrad: Improving con-ditional denoising diffusion models with data-dependent adaptive prior","volume-title":"Proc. ICLR","author":"Lee"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-301"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9415087"},{"key":"ref48","article-title":"Its Raw! audio generation with state-space models","author":"Goel","year":"2022","journal-title":"arXiv"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746690"},{"key":"ref50","article-title":"Tackling the generative learning trilemma with denoising diffusion GANs","volume-title":"Proc. ICLR","author":"Xiao"},{"key":"ref51","article-title":"DiffGAN-TTS: High-fidelity and efficient text-to-speech with denoising diffusion GANs","author":"Liu","year":"2022","journal-title":"arXiv"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2021.3069677"},{"key":"ref53","article-title":"Denoising diffusion probabilis-tic models","volume-title":"Proc. NeurIPS","author":"Ho"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-2409"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1137\/17M1122451"},{"key":"ref56","article-title":"Plug-and-play methods provably converge with properly trained de-noisers","volume-title":"Proc. ICML","author":"Ryu"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1137\/20M1387961"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682744"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2020.3034486"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1137\/20M1337168"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-48311-5"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4419-9569-8_17"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1561\/9781601987174"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1563"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2010.938752"},{"key":"ref66","volume-title":"Parallel WaveGAN implementation with Py-torch","author":"Hayashi"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"ref68","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. ICLR","author":"Kingma"},{"key":"ref69","article-title":"A spectral energy distance for parallel speech synthesis","volume-title":"Proc. NeurIPS","author":"Gritsenko"},{"key":"ref70","article-title":"Deep unsupervised learning using nonequilibrium ther-modynamic","volume-title":"Proc. ICML","author":"Sohl-Dickstein"}],"event":{"name":"2022 IEEE Spoken Language Technology Workshop (SLT)","location":"Doha, Qatar","start":{"date-parts":[[2023,1,9]]},"end":{"date-parts":[[2023,1,12]]}},"container-title":["2022 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10022052\/10022330\/10022496.pdf?arnumber=10022496","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,13]],"date-time":"2024-02-13T06:44:47Z","timestamp":1707806687000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10022496\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,1,9]]},"references-count":70,"URL":"https:\/\/doi.org\/10.1109\/slt54892.2023.10022496","relation":{},"subject":[],"published":{"date-parts":[[2023,1,9]]}}}