{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T18:00:54Z","timestamp":1781287254742,"version":"3.54.1"},"reference-count":62,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"JST CREST, Japan,","award":["JPMJCR18A6"],"award-info":[{"award-number":["JPMJCR18A6"]}]},{"name":"MEXT KAKENHI, Japan,","award":["16H06302"],"award-info":[{"award-number":["16H06302"]}]},{"name":"MEXT KAKENHI, Japan,","award":["16K16096"],"award-info":[{"award-number":["16K16096"]}]},{"name":"MEXT KAKENHI, Japan,","award":["17H04687"],"award-info":[{"award-number":["17H04687"]}]},{"name":"MEXT KAKENHI, Japan,","award":["18H04120"],"award-info":[{"award-number":["18H04120"]}]},{"name":"MEXT KAKENHI, Japan,","award":["18H04112"],"award-info":[{"award-number":["18H04112"]}]},{"name":"MEXT KAKENHI, Japan,","award":["18KT0051"],"award-info":[{"award-number":["18KT0051"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2020]]},"DOI":"10.1109\/taslp.2019.2950099","type":"journal-article","created":{"date-parts":[[2019,10,28]],"date-time":"2019-10-28T19:25:32Z","timestamp":1572290732000},"page":"157-170","source":"Crossref","is-referenced-by-count":30,"title":["A Vector Quantized Variational Autoencoder (VQ-VAE) Autoregressive Neural $F_0$ Model for Statistical Parametric Speech Synthesis"],"prefix":"10.1109","volume":"28","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8246-0606","authenticated-orcid":false,"given":"Xin","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shinji","family":"Takaki","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2752-3955","authenticated-orcid":false,"given":"Junichi","family":"Yamagishi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Simon","family":"King","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Keiichi","family":"Tokuda","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","article-title":"Variational lossy autoencoder","author":"chen","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K16-1002"},{"key":"ref33","first-page":"589","article-title":"Better generative models for sequential data problems: Bidirectional recurrent mixture density networks","author":"schuster","year":"0","journal-title":"Proc Neural Inf Process Syst"},{"key":"ref32","article-title":"Mixture density networks","author":"bishop","year":"2004"},{"key":"ref31","doi-asserted-by":"crossref","DOI":"10.1093\/oso\/9780198538493.001.0001","author":"bishop","year":"1995","journal-title":"Neural Networks for Pattern Recognition"},{"key":"ref30","first-page":"325","article-title":"On the prediction of global F0 shape for Japanese text-to-speech","author":"sagisaka","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref37","article-title":"Auto-encoding variational Bayes","author":"kingma","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref36","article-title":"Deep encoder-decoder models for unsupervised learning of controllable speech synthesis","author":"henter","year":"2018","journal-title":"arXiv 1807 11470"},{"key":"ref35","first-page":"1863","article-title":"A clockwork RNN","author":"koutnik","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref34","first-page":"107","article-title":"Generating F0 contours for speech synthesis using the Tilt intonation theory","author":"dusterhoff","year":"1997","journal-title":"ESCA Workshop on Intonation Theory Models and Applications"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.21437\/SpeechProsody.2018-161"},{"key":"ref62","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"0","journal-title":"Proc Neural Inf Process Syst"},{"key":"ref61","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","author":"sutskever","year":"0","journal-title":"Proc Neural Inf Process Syst"},{"key":"ref28","first-page":"141","article-title":"F0 generation with a data base of natural F0 patterns and with a neural network","author":"traber","year":"0","journal-title":"Proc ESCA Workshop Speech Synthesis"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1989.266404"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/89.668817"},{"key":"ref2","author":"shigeto","year":"2015","journal-title":"Handbook of Japanese Phonetics and Phonology"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511616983.001"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2097248"},{"key":"ref22","first-page":"7962","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"zen","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP.2014.6936598"},{"key":"ref24","article-title":"Suprasegmental representations for the modeling of fundamental frequency in statistical parametric speech synthesis","author":"ribeiro","year":"2018"},{"key":"ref23","first-page":"1964","article-title":"TTS synthesis with bidirectional LSTM based recurrent neural networks","author":"fan","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2017.11.002"},{"key":"ref25","first-page":"2273","article-title":"Modeling DCT parameterized F0 trajectory at intonation phrase level with DNN or decision tree","author":"yin","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref50","first-page":"1043","article-title":"Mel-generalized cepstral analysis a unified approach","author":"tokuda","year":"0","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref51","first-page":"547","article-title":"Introducing CURRENNT: The Munich open-source CUDA recurrent neural network toolkit","volume":"16","author":"weninger","year":"2015","journal-title":"J Mach Learn Res"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461569"},{"key":"ref58","article-title":"Fundamental frequency modelling: An articulatory perspective with target approximation and deep learning","author":"liu","year":"2017"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2005.03.014"},{"key":"ref56","first-page":"1627","article-title":"Using decision trees within the Tilt intonation model to predict F0 contours","author":"dusterhoff","year":"0","journal-title":"Proc EUROSPEECH"},{"key":"ref55","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"maaten","year":"2008","journal-title":"J Mach Learn Res"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461452"},{"key":"ref53","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","author":"glorot","year":"0","journal-title":"Proc 13th Int Conf on Artificial Intell"},{"key":"ref52","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00085-5"},{"key":"ref11","first-page":"1118","article-title":"Speaker-dependent WaveNet vocoder","author":"tamamori","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref40","first-page":"867","article-title":"ToBI: A standard for labeling English prosody","author":"silverman","year":"0","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref12","first-page":"2268","article-title":"Prosody contour prediction with long short-term memory, bi-directional, deep recurrent neural networks","author":"fernandez","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2828650"},{"key":"ref14","first-page":"6306","article-title":"Neural discrete representation learning","author":"van den oord","year":"0","journal-title":"Proc Neural Inf Process Syst"},{"key":"ref15","first-page":"455","article-title":"Multi-space probability distribution HMM","volume":"85","author":"tokuda","year":"2002","journal-title":"IEICE Trans Inf Syst"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2076805"},{"key":"ref17","first-page":"2091","article-title":"Context-dependent additive log F0 model for HMM-based speech synthesis","author":"zen","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref18","first-page":"2170","article-title":"A hierarchical F0 modeling method for HMM-based speech synthesis","author":"lei","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref19","first-page":"2274","article-title":"Multilevel parametric-base F0 model for speech synthesis","author":"latorre","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref3","article-title":"WaveNet: A generative model for raw audio","author":"van den oord","year":"2016","journal-title":"arXiv 1609 03499"},{"key":"ref6","article-title":"The CSTR entry to the Blizzard Challenge 2016","author":"merritt","year":"0","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref5","article-title":"Deep voice 3: Scaling text-to-speech with convolutional sequence learning","author":"ping","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2013.2251852"},{"key":"ref7","article-title":"The USTC system for Blizzard Challenge 2017","author":"hu","year":"0","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref46","article-title":"The NII speech synthesis entry for Blizzard Challenge 2016","author":"juvela","year":"0","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref45","first-page":"179","article-title":"XIMERA: A new TTS from ATR based on corpus-based technologies","author":"kawai","year":"0","journal-title":"Proc 5th ISCA Speech Synthesis Workshop"},{"key":"ref48","article-title":"An example of context-dependent label format for HMM-based speech synthesis in Japanese","year":"2015"},{"key":"ref47","article-title":"The Japanese TTS system &#x2018;Open JTalk&#x2019;","year":"2015"},{"key":"ref42","article-title":"Estimating or propagating gradients through stochastic neurons for conditional computation","author":"bengio","year":"2013","journal-title":"arXiv 1308 3432"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1121\/1.428453"},{"key":"ref44","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"hinton","year":"2014","journal-title":"J Mach Learn Res"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8938144\/08884734.pdf?arnumber=8884734","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,25]],"date-time":"2024-07-25T21:46:43Z","timestamp":1721944003000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8884734\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"references-count":62,"URL":"https:\/\/doi.org\/10.1109\/taslp.2019.2950099","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]}}}