{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,14]],"date-time":"2025-05-14T05:26:43Z","timestamp":1747200403732,"version":"3.37.3"},"reference-count":61,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100001691","name":"Japan Society for the Promotion of Science (JSPS) KAKENHI","doi-asserted-by":"publisher","award":["JP18K11163","JP19H04136"],"award-info":[{"award-number":["JP18K11163","JP19H04136"]}],"id":[{"id":"10.13039\/501100001691","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005936","name":"CASIO SCIENCE PROMOTION FOUNDATION","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100005936","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2021]]},"DOI":"10.1109\/access.2021.3118033","type":"journal-article","created":{"date-parts":[[2021,10,6]],"date-time":"2021-10-06T03:31:45Z","timestamp":1633491105000},"page":"137599-137612","source":"Crossref","is-referenced-by-count":6,"title":["PeriodNet: A Non-Autoregressive Raw Waveform Generative Model With a Structure Separating Periodic and Aperiodic Components"],"prefix":"10.1109","volume":"9","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1245-8791","authenticated-orcid":false,"given":"Yukiya","family":"Hono","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7294-7699","authenticated-orcid":false,"given":"Shinji","family":"Takaki","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2081-0396","authenticated-orcid":false,"given":"Kei","family":"Hashimoto","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keiichiro","family":"Oura","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yoshihiko","family":"Nankaku","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6143-0133","authenticated-orcid":false,"given":"Keiichi","family":"Tokuda","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","first-page":"2336","article-title":"Accurate speech decomposition into periodic and aperiodic components based on discrete harmonic transform","author":"zubrycki","year":"2007","journal-title":"Proc 15th Eur Signal Process Conf"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.2307\/3680788"},{"key":"ref33","article-title":"DiffWave: A versatile diffusion model for audio synthesis","author":"kong","year":"2021","journal-title":"Proc ICLR"},{"key":"ref32","article-title":"WaveGrad: Estimating gradients for waveform generation","author":"chen","year":"2021","journal-title":"Proc ICLR"},{"key":"ref31","first-page":"7586","article-title":"Non-autoregressive neural text-to-speech","author":"peng","year":"2020","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref30","article-title":"High fidelity speech synthesis with adversarial networks","author":"bi?kowski","year":"2020","journal-title":"Proc ICLR"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3051765"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3061245"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2019-3"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2956145"},{"key":"ref60","first-page":"1992","article-title":"Using cyclic noise as the source signal for neural source-filter-based speech waveform model","author":"wang","year":"2020","journal-title":"Proc INTERSPEECH"},{"key":"ref61","article-title":"HiFiSinger: Towards high-fidelity neural singing voice synthesis","author":"chen","year":"2020","journal-title":"arXiv 2009 01776"},{"key":"ref28","article-title":"HiFi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis","volume":"33","author":"kong","year":"2020","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1238"},{"key":"ref29","first-page":"492","article-title":"Multi-band MelGAN: Faster waveform generation for high-quality text-to-speech","author":"yang","year":"2021","journal-title":"Proc IEEE Spoken Lang Technol Workshop (SLT)"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639215"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"ref22","first-page":"7706","article-title":"WaveFlow: A compact flow-based model for raw audio","author":"ping","year":"2020","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref21","article-title":"FloWaveNet: A generative flow for raw audio","author":"kim","year":"2018","journal-title":"arXiv 1811 02155"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053202"},{"key":"ref23","article-title":"WaveNODE: A continuous normalizing flow for speech synthesis","author":"kim","year":"2020","journal-title":"arXiv 2006 04598"},{"key":"ref26","first-page":"14910","article-title":"MelGAN: Generative adversarial networks for conditional waveform synthesis","author":"kumar","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.304"},{"key":"ref51","first-page":"1","article-title":"The NITech text-to-speech system for the Blizzard challenge 2016","author":"sawada","year":"2016","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2016.01.007"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1121\/1.1912389"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1998.674420"},{"journal-title":"PeriodNet Demo","year":"2021","author":"hono","key":"ref56"},{"journal-title":"REAPER Robust Epoch And Pitch EstimatoR","year":"2021","key":"ref55"},{"key":"ref54","article-title":"On the variance of the adaptive learning rate and beyond","author":"liu","year":"2020","journal-title":"Proc ICLR"},{"key":"ref53","first-page":"901","article-title":"Weight normalization: A simple reparameterization to accelerate training of deep neural networks","author":"salimans","year":"2016","journal-title":"Proc Adv Neural Inf Process Syst"},{"journal-title":"Pulse Code Modulation (PCM) of Voice Frequencies","year":"1988","key":"ref52"},{"key":"ref10","first-page":"125","article-title":"WaveNet: A generative model for raw audio","author":"van den oord","year":"2016","journal-title":"Proc ISCA SSW"},{"key":"ref11","article-title":"SampleRNN: An unconditional end-to-end neural audio generation model","author":"mehri","year":"2017","journal-title":"Proc ICLR"},{"journal-title":"SMS-Tools Sound Analysis\/Synthesis Tools for Music Applications","year":"2021","key":"ref40"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-314"},{"key":"ref13","first-page":"2962","article-title":"Deep voice 2: Multi-speaker neural text-to-speech","author":"gibiansky","year":"2017","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-986"},{"key":"ref16","article-title":"Efficient neural audio synthesis","author":"kalchbrenner","year":"2018","journal-title":"arXiv 1802 08435"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462431"},{"key":"ref18","first-page":"3918","article-title":"Parallel WaveNet: Fast high-fidelity speech synthesis","author":"van den oord","year":"2018","journal-title":"Proc ICML"},{"key":"ref19","article-title":"ClariNet: Parallel wave generation in end-to-end text-to-speech","author":"ping","year":"2019","journal-title":"Proc ICML"},{"key":"ref4","first-page":"211","article-title":"Recent development of the HMM-based singing voice synthesis system&#x2014;Sinsy","author":"oura","year":"2010","journal-title":"Proc ISCA SSW"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1420"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3104165"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00085-5"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683154"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2008"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2906484"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1757"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682804"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683271"},{"key":"ref42","first-page":"10215","article-title":"Glow: Generative flow with invertible $1\\times1$\n convolutions","author":"kingma","year":"2018","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref41","first-page":"6049","article-title":"Periodnet: A non-autoregressive waveform generation model with a structure separating periodic and aperiodic components","author":"hono","year":"2021","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process (ICASSP)"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1635"},{"key":"ref43","first-page":"2672","article-title":"Generative adversarial nets","author":"goodfellow","year":"2014","journal-title":"Proc Adv Neural Inf Process Syst"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/9312710\/09559963.pdf?arnumber=9559963","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,12,17]],"date-time":"2021-12-17T19:55:52Z","timestamp":1639770952000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9559963\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"references-count":61,"URL":"https:\/\/doi.org\/10.1109\/access.2021.3118033","relation":{},"ISSN":["2169-3536"],"issn-type":[{"type":"electronic","value":"2169-3536"}],"subject":[],"published":{"date-parts":[[2021]]}}}