{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T01:10:01Z","timestamp":1751245801464,"version":"3.41.0"},"reference-count":41,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017,12]]},"DOI":"10.1109\/asru.2017.8268954","type":"proceedings-article","created":{"date-parts":[[2018,1,25]],"date-time":"2018-01-25T21:43:53Z","timestamp":1516916633000},"page":"331-337","source":"Crossref","is-referenced-by-count":3,"title":["The blizzard machine learning challenge 2017"],"prefix":"10.1109","author":[{"given":"Kei","family":"Sawada","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keiichi","family":"Tokuda","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Simon","family":"King","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alan W","family":"Black","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","author":"duchi","year":"2011","journal-title":"The Journal of Machine Learning Research"},{"key":"ref38","doi-asserted-by":"crossref","DOI":"10.21437\/Blizzard.2017-11","article-title":"The NITech text-to-speech system for the Blizzard Challenge 2017","author":"sawada","year":"2017","journal-title":"Blizzard Challenge 2017 Workshop"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268999"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268997"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1250\/ast.21.79"},{"key":"ref30","first-page":"455","article-title":"Multi-space probability distribution HMM","volume":"e85 d","author":"tokuda","year":"2002","journal-title":"IEICE Transactions on Information and Systems"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472749"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854321"},{"journal-title":"Adam A method for stochastic optimization","year":"2014","author":"kingma","key":"ref35"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268998"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-134"},{"journal-title":"Recurrent highway networks","year":"2016","author":"zilly","key":"ref40"},{"key":"ref11","article-title":"Char2Wav: End-to-end speech synthesis","author":"sotelo","year":"2017","journal-title":"International Conference on Learning Representations"},{"journal-title":"Deep Voice Real-time neural text-to-speech","year":"2017","author":"arik","key":"ref12"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref14","doi-asserted-by":"crossref","first-page":"77","DOI":"10.21437\/Interspeech.2005-72","article-title":"The Blizzard Challenge&#x2013;2005: Evaluating corpus-based speech synthesis on common datasets","author":"black","year":"2005","journal-title":"InterSpeech 2005"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.3989\/loquens.2014.006"},{"key":"ref16","doi-asserted-by":"crossref","DOI":"10.21437\/Blizzard.2017-1","article-title":"The Blizzard Challenge 2017","author":"king","year":"2017","journal-title":"Blizzard Challenge 2017 Workshop"},{"key":"ref17","doi-asserted-by":"crossref","DOI":"10.21437\/Blizzard.2016-1","article-title":"The Blizzard Challenge 2016","author":"king","year":"2016","journal-title":"Proceedings of The Blizzard Challenge 2016 Workshop"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1992.225953"},{"key":"ref19","first-page":"495","article-title":"A robust algorithm for pitch tracking (RAPT)","author":"talkin","year":"1995","journal-title":"Speech Coding and Synthesis"},{"key":"ref28","first-page":"2347","article-title":"Simultaneous modeling of spectrum, pitch and duration in HMM-based speech synthesis","author":"yoshimura","year":"1999","journal-title":"Eurospeech 1999"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"7962","DOI":"10.1109\/ICASSP.2013.6639215","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"zen","year":"2013","journal-title":"2013 IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref27","first-page":"1185","article-title":"Hidden semi-Markov model based speech synthesis","author":"zen","year":"2004","journal-title":"8th International Conference on Spoken Language Processing"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2013.2251852"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639225"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.861820"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639187"},{"journal-title":"WaveNet A Generative Model for Raw Audio","year":"2016","author":"van den oord","key":"ref8"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638996"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"journal-title":"Samplernn An unconditional end-to-end neural audio generation model","year":"2016","author":"mehri","key":"ref9"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1996.541110"},{"journal-title":"SWIPE A sawtooth waveform inspired pitch estimator for speech and music","year":"2007","author":"camacho","key":"ref20"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"journal-title":"REAPER","year":"0","key":"ref21"},{"journal-title":"Festival","year":"0","key":"ref24"},{"key":"ref41","first-page":"2672","article-title":"Generative adversarial nets","author":"goodfellow","year":"2014","journal-title":"Advances in neural information processing systems"},{"journal-title":"SPTK","year":"0","key":"ref23"},{"journal-title":"HTS","year":"0","key":"ref26"},{"journal-title":"CMU pronouncing dictionary","year":"0","key":"ref25"}],"event":{"name":"2017 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","start":{"date-parts":[[2017,12,16]]},"location":"Okinawa, Japan","end":{"date-parts":[[2017,12,20]]}},"container-title":["2017 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8260578\/8268903\/08268954.pdf?arnumber=8268954","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:47:49Z","timestamp":1751244469000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/8268954\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,12]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/asru.2017.8268954","relation":{},"subject":[],"published":{"date-parts":[[2017,12]]}}}