{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T15:31:03Z","timestamp":1771515063610,"version":"3.50.1"},"reference-count":85,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2018,8,1]],"date-time":"2018-08-01T00:00:00Z","timestamp":1533081600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/OAPA.html"}],"funder":[{"name":"MEXT KAKENHI","award":["15H01686"],"award-info":[{"award-number":["15H01686"]}]},{"name":"MEXT KAKENHI","award":["16H06302"],"award-info":[{"award-number":["16H06302"]}]},{"name":"MEXT KAKENHI","award":["17H04687"],"award-info":[{"award-number":["17H04687"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2018,8]]},"DOI":"10.1109\/taslp.2018.2828650","type":"journal-article","created":{"date-parts":[[2018,4,19]],"date-time":"2018-04-19T18:13:49Z","timestamp":1524161629000},"page":"1406-1419","source":"Crossref","is-referenced-by-count":21,"title":["Autoregressive Neural F0 Model for Statistical Parametric Speech Synthesis"],"prefix":"10.1109","volume":"26","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8246-0606","authenticated-orcid":false,"given":"Xin","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"Takaki","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2752-3955","authenticated-orcid":false,"given":"Junichi","family":"Yamagishi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref73","article-title":"The Japanese TTS System &#x2018;Open JTalk&#x2019;","year":"2015"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016-20"},{"key":"ref71","first-page":"179","article-title":"XIMERA: A new TTS from ATR based on corpus-based technologies","author":"kawai","year":"0","journal-title":"Proc SSW5"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1159"},{"key":"ref76","article-title":"The NII speech synthesis entry for blizzard challenge 2016","author":"juvela","year":"0","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref77","author":"o\u2019shaughnessy","year":"2000","journal-title":"Speech Communications Human and Machine"},{"key":"ref74","article-title":"MeCab: Yet another part-of-speech and morphological analyzer","author":"kudo","year":"0"},{"key":"ref39","first-page":"2170","article-title":"A hierarchical F0 modeling method for HMM-based speech synthesis","author":"lei","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref75","article-title":"An example of context-dependent label format for HMM-based speech synthesis in Japanese","year":"2015"},{"key":"ref38","first-page":"2091","article-title":"Context-dependent additive log F0 model for HMM-based speech synthesis","author":"zen","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref78","first-page":"1043","article-title":"Mel-generalized cepstral analysis a unified approach","author":"tokuda","year":"0","journal-title":"Proc Int Conf Spoken Lang"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2097248"},{"key":"ref32","first-page":"2029","article-title":"Stylization and trajectory modelling of short and long term speech prosody variations","author":"obin","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref31","first-page":"2274","article-title":"Multilevel parametric-base F0 model for speech synthesis","author":"latorre","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2008.4518524"},{"key":"ref37","first-page":"455","article-title":"Multi-space probability distribution HMM","volume":"85","author":"tokuda","year":"2002","journal-title":"IEICE Trans Inf Syst"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21437\/SpeechProsody.2016-58"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178904"},{"key":"ref34","first-page":"2273","article-title":"Modeling DCT parameterized F0 trajectory at intonation phrase level with DNN or decision tree","author":"yin","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178815"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178816"},{"key":"ref61","first-page":"121","article-title":"The effect of using normalized models in statistical speech synthesis","author":"shannon","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref63","first-page":"246","article-title":"Hierarchical probabilistic neural network language model","volume":"5","author":"morin","year":"0","journal-title":"Proc Artif Intell Stat"},{"key":"ref28","first-page":"2077","article-title":"F0 generation for speech synthesis using a multi-tier approach","author":"sun","year":"0","journal-title":"Proc Int Conf Spoken Lang"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1989.1.2.270"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1121\/1.3037222"},{"key":"ref65","article-title":"Sequence level training with recurrent neural networks","author":"ranzato","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref66","article-title":"Variational lossy autoencoder","author":"chen","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP.2014.6936598"},{"key":"ref67","first-page":"3024","article-title":"Improving multi-step prediction of learned time series models","author":"venkatraman","year":"0","journal-title":"Proc 29th AAAI Conf Artif Intell"},{"key":"ref68","first-page":"1171","article-title":"Scheduled sampling for sequence prediction with recurrent neural networks","author":"bengio","year":"0","journal-title":"Proc Neural Inf Process Syst"},{"key":"ref69","article-title":"How (not) to train your generative model: Scheduled sampling, likelihood, adversary?","author":"husz\u00e1r","year":"2015"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2013.2251852"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511816338"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953087"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1250\/ast.5.233"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-246"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2005.03.014"},{"key":"ref23","first-page":"43","article-title":"A statistical model of speech F0 contours","author":"kameoka","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref26","first-page":"1627","article-title":"Using decision trees within the Tilt intonation model to predict F0 contours","author":"dusterhoff","year":"0","journal-title":"Proc EUROSPEECH"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1121\/1.428453"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472749"},{"key":"ref51","doi-asserted-by":"crossref","DOI":"10.7551\/mitpress\/3348.001.0001","author":"frey","year":"1998","journal-title":"Graphical Models for Machine Learning and Digital Communication"},{"key":"ref59","first-page":"1747","article-title":"Pixel recurrent neural networks","author":"van den oord","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref58","first-page":"1","article-title":"Neural autoregressive distribution estimation","volume":"17","author":"uria","year":"2016","journal-title":"J Mach Learn Res"},{"key":"ref57","first-page":"29","article-title":"The neural autoregressive distribution estimator","author":"larochelle","year":"0","journal-title":"Proc Artif Intell Statist"},{"key":"ref56","first-page":"1242","article-title":"Deep autoregressive networks","author":"gregor","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref55","first-page":"1045","article-title":"Recurrent neural network based language model.","volume":"2","author":"mikolov","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref54","first-page":"2980","article-title":"A recurrent latent variable model for sequential data","author":"chung","year":"0","journal-title":"Proc Neural Inf Process Syst"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2012.2227740"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/PROC.1975.9792"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1121\/1.387033"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/89.890071"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2017.11.002"},{"key":"ref12","first-page":"1385","article-title":"Generating F0 contours from ToBI labels using linear regression","volume":"3","author":"black","year":"0","journal-title":"Proc 4th Int Conf Spoken Lang"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/89.759037"},{"key":"ref14","first-page":"325","article-title":"On the prediction of global F0 shape for Japanese text-to-speech","author":"sagisaka","year":"0","journal-title":"Proc Int Conf Acoust Speech Signal Process"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1989.266404"},{"key":"ref82","first-page":"547","article-title":"Introducing CURRENT: The Munich open-source CUDA recurrent neural network toolkit","volume":"16","author":"weninger","year":"2015","journal-title":"J Mach Learn Res"},{"key":"ref16","first-page":"141","article-title":"F0 generation with a data base of natural F0 patterns and with a neural network","author":"traber","year":"0","journal-title":"Proc ESCA Workshop Speech Synthesis"},{"key":"ref81","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","author":"glorot","year":"0","journal-title":"Proc Artif Intell Stat"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/89.668817"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780199249633.003.0007"},{"key":"ref18","first-page":"179","article-title":"Data driven intonation modelling of 6 languages","author":"buhmann","year":"0","journal-title":"Proc 4th Int Conf Spoken Lang"},{"key":"ref83","first-page":"148","article-title":"CRF-based statistical learning of Japanese accent sandhi for developing Japanese text-to-speech synthesis systems","author":"minematsu","year":"0","journal-title":"Proc SSW6"},{"key":"ref19","first-page":"2268","article-title":"Prosody contour prediction with long short-term memory, bi-directional, deep recurrent neural networks","author":"fernandez","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref80","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"duchi","year":"2011","journal-title":"J Mach Learn Res"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511616983"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2014.2359987"},{"key":"ref5","article-title":"Guidelines for ToBI labelling","volume":"3","author":"beckman","year":"1997","journal-title":"The OSU Research Foundation"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1016\/0004-3702(93)90020-C"},{"key":"ref8","first-page":"1964","article-title":"TTS synthesis with bidirectional LSTM based recurrent neural networks","author":"fan","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref7","first-page":"7962","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"zen","year":"0","journal-title":"Proc Int Conf Acoust Speech Signal Process"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2006.01.002"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2076805"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D15-1176"},{"key":"ref45","article-title":"WaveNet: A generative model for raw audio","author":"van den oord","year":"2016"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2165280"},{"key":"ref47","doi-asserted-by":"crossref","first-page":"357","DOI":"10.1162\/tacl_a_00104","article-title":"Named entity recognition with bidirectional LSTM-CNNs","volume":"4","author":"chiu","year":"2016","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"ref42","article-title":"Mixture Density Networks","author":"bishop","year":"2004"},{"key":"ref41","doi-asserted-by":"crossref","DOI":"10.1093\/oso\/9780198538493.001.0001","author":"bishop","year":"1995","journal-title":"Neural Networks for Pattern Recognition"},{"key":"ref44","author":"moore","year":"2012","journal-title":"An Introduction to the Psychology of Hearing"},{"key":"ref43","first-page":"589","article-title":"Better generative models for sequential data problems: Bidirectional recurrent mixture density networks","author":"schuster","year":"0","journal-title":"Neural Process"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8356719\/08341752.pdf?arnumber=8341752","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,6]],"date-time":"2024-07-06T04:57:25Z","timestamp":1720241845000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8341752\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,8]]},"references-count":85,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2018.2828650","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,8]]}}}