{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T11:34:49Z","timestamp":1766748889272},"reference-count":43,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2015,11,1]],"date-time":"2015-11-01T00:00:00Z","timestamp":1446336000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/3.0\/legalcode"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2015,11]]},"DOI":"10.1109\/taslp.2015.2461448","type":"journal-article","created":{"date-parts":[[2015,7,28]],"date-time":"2015-07-28T18:43:18Z","timestamp":1438108998000},"page":"2003-2014","source":"Crossref","is-referenced-by-count":31,"title":["A Deep Generative Architecture for Postfiltering in Statistical Parametric Speech Synthesis"],"prefix":"10.1109","volume":"23","author":[{"given":"Ling-Hui","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tuomo","family":"Raitio","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cassia","family":"Valentini-Botinhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhen-Hua","family":"Ling","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junichi","family":"Yamagishi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","first-page":"1043","article-title":"Mel-generalized cepstral analysis?A unified approach to speech spectral estimation","volume":"3","author":"tokuda","year":"1994","journal-title":"Proc ICSLP"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00085-5"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6855135"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.79.8.2554"},{"key":"ref31","author":"salakhutdinov","year":"2009","journal-title":"Learning deep generative models"},{"key":"ref30","doi-asserted-by":"crossref","first-page":"599","DOI":"10.1007\/978-3-642-35289-8_32","author":"hinton","year":"2012","journal-title":"Neural Networks Tricks of the Trade"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.861820"},{"key":"ref36","first-page":"825","article-title":"Products of experts","volume":"1","author":"hinton","year":"1999","journal-title":"Proc 9th Int Conf Artif Neural Netw"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1561\/2200000006"},{"key":"ref34","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2010-487","article-title":"Binary coding of speech spectrograms using a deep auto-encoder","author":"deng","year":"2010","journal-title":"Proc INTERSPEECH"},{"key":"ref10","first-page":"2268","article-title":"Prosody contour prediction with long short-term memory, bi-directional, deep recurrent neural networks","author":"fernandez","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref40","article-title":"Method for the subjective assessment of intermediate quality level of coding systems","year":"2003","journal-title":"ITU Rec ITU-R BS 1534-1 Int Telecomm Union Radiocommunication Assembly"},{"key":"ref11","first-page":"1964","article-title":"TTS synthesis with bidirectional LSTM based recurrent neural networks","author":"fan","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178816"},{"key":"ref13","article-title":"USTC system for blizzard challenge 2006an improved HMM-based speech synthesis method","author":"ling","year":"2006","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1002\/scj.20354"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e90-d.5.816"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853604"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.907344"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1998.674423"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1995.479266"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1126\/science.1127647"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2014.2359987"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/21.87054"},{"key":"ref3","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2005-72","article-title":"The blizzard challenge 2005: Evaluating corpus-based speech synthesis on common databases","author":"black","year":"2005","journal-title":"Proc INTERSPEECH"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2013.2269291"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1162\/089976602760128018"},{"key":"ref5","first-page":"7962","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"zen","year":"2013","journal-title":"Proc ICASSP"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638996"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639225"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2010.5495546"},{"key":"ref1","first-page":"1954","article-title":"DNN-based stochastic postfilter for HMM-based speech synthesis","author":"chen","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref20","doi-asserted-by":"crossref","first-page":"825","DOI":"10.21437\/Interspeech.2010-183","article-title":"Global variance modeling on the log power spectrum of LSPs for HMM-based speech synthesis","author":"ling","year":"2010","journal-title":"Proc INTERSPEECH"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1992.225953"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2013.2251852"},{"key":"ref42","article-title":"The USTC system for blizzard challenge 2014","author":"chen","year":"2014","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref24","first-page":"2313","article-title":"Voice conversion using generative trained deep neural networks with multiple frame spectral envelopes","author":"chen","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref41","first-page":"225","article-title":"IEEE recommended practice for speech quality measurement","volume":"ae 17","year":"1969","journal-title":"IEEE Trans Audio Electroacoust"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2047683"},{"key":"ref26","first-page":"194","volume":"1","author":"smolensky","year":"1986","journal-title":"Parallel Distributed Processing Explorations in the Microstructure of Cognition"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2014.2326991"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2353991"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/7140863\/07169536.pdf?arnumber=7169536","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,19]],"date-time":"2022-05-19T10:10:36Z","timestamp":1652955036000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7169536\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,11]]},"references-count":43,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2015.2461448","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,11]]}}}