{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T15:02:55Z","timestamp":1729609375008,"version":"3.28.0"},"reference-count":28,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016,10]]},"DOI":"10.1109\/iscslp.2016.7918441","type":"proceedings-article","created":{"date-parts":[[2017,5,12]],"date-time":"2017-05-12T22:22:40Z","timestamp":1494627760000},"page":"1-5","source":"Crossref","is-referenced-by-count":1,"title":["Learning auxiliary categorical information for speech synthesis based on deep and recurrent neural networks"],"prefix":"10.1109","author":[{"given":"Zhengqi","family":"Wen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kehuang","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhen","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianhua","family":"Tao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chin-Hui","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","first-page":"1964","author":"fan","year":"2014","journal-title":"TTS Synthesis with Bidirectional LSTM based Recurrent Neural Networks"},{"key":"ref11","doi-asserted-by":"crossref","first-page":"2673","DOI":"10.1109\/78.650093","article-title":"Bidirectional recurrent neural networks","volume":"45","author":"mike","year":"1997","journal-title":"IEEE Transactions on Signal Processing"},{"key":"ref12","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"sepp","year":"1997","journal-title":"Neural Computation"},{"key":"ref13","first-page":"2347","author":"yoshimura","year":"1999","journal-title":"Simultaneous Modeling of Spectrum Pitch and Duration in Hmm-based Speech Synthesis"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2012.11.008"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2013.2238591"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1023\/A:1007379606734"},{"key":"ref17","first-page":"4460","author":"wu","year":"2015","journal-title":"Deep neural networks employing multi-task learning and stacked bottleneck features for speech synthesis"},{"key":"ref18","doi-asserted-by":"crossref","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","article-title":"A Fast Learning Algorithm for Deep Belief Nets","volume":"18","author":"hinton","year":"2006","journal-title":"Neural Computation"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1162\/089976602760128018"},{"journal-title":"WEB-based listening test system for speech synthesis and speech conversion evaluation","year":"2008","author":"blin","key":"ref28"},{"key":"ref4","first-page":"160","author":"collobert","year":"2008","journal-title":"A Unified Architecture for Natural Language Processing Deep Neural Networks with Multitask Learning"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1985.1164550"},{"key":"ref3","first-page":"1106","author":"krizhevsky","year":"2012","journal-title":"ImageNet Classification with Deep Convolutional Neural Networks"},{"key":"ref6","first-page":"7825","author":"ling","year":"2013","journal-title":"Modeling spectral envelopes using restricted Boltzmann machines for statistical parametric speech synthesis"},{"key":"ref5","first-page":"7962","author":"kang","year":"2013","journal-title":"Multi-distribution deep belief network for speech synthesis"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref9","first-page":"3844","author":"zen","year":"2014","journal-title":"Deep mixture density networks for acoustic modeling in statistical parametric speech synthesis"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.tics.2007.09.004"},{"key":"ref20","first-page":"1315","author":"tokuda","year":"2000","journal-title":"Speech parameter generation algorithms for hmm-based speech synthesis"},{"journal-title":"Nonlinear programming Athena Scientific","year":"1999","author":"bertsekas","key":"ref22"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-3210-1"},{"key":"ref24","first-page":"1.10.1","volume":"1","author":"soong","year":"1984","journal-title":"Line spectrum pair (UP) and speech data compression"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1038\/323533a0"},{"journal-title":"Speech Recognition Toolkit","year":"2011","author":"povey","key":"ref26"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00085-5"}],"event":{"name":"2016 10th International Symposium on Chinese Spoken Language Processing (ISCSLP)","start":{"date-parts":[[2016,10,17]]},"location":"Tianjin, China","end":{"date-parts":[[2016,10,20]]}},"container-title":["2016 10th International Symposium on Chinese Spoken Language Processing (ISCSLP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7912121\/7918361\/07918441.pdf?arnumber=7918441","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,24]],"date-time":"2019-09-24T11:53:01Z","timestamp":1569325981000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7918441\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,10]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/iscslp.2016.7918441","relation":{},"subject":[],"published":{"date-parts":[[2016,10]]}}}