{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,7]],"date-time":"2025-08-07T21:24:46Z","timestamp":1754601886648,"version":"3.28.0"},"reference-count":29,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016,3]]},"DOI":"10.1109\/icassp.2016.7472736","type":"proceedings-article","created":{"date-parts":[[2016,6,24]],"date-time":"2016-06-24T01:58:30Z","timestamp":1466733510000},"page":"5535-5539","source":"Crossref","is-referenced-by-count":19,"title":["A deep auto-encoder based low-dimensional feature extraction from FFT spectral envelopes for statistical parametric speech synthesis"],"prefix":"10.1109","author":[{"given":"Shinji","family":"Takaki","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junichi","family":"Yamagishi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","first-page":"1","article-title":"Learning the speech front-end with raw waveform cldnns","author":"sainath","year":"2015","journal-title":"Proceedings of Interspeech"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00085-5"},{"key":"ref12","first-page":"289","article-title":"An attempt to develop a singing synthesizer by collaborative creation","author":"morise","year":"2015","journal-title":"The Stockholm Music Acoustics Conference 2013 (SMAC2013)"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2014.09.003"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDL8015"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2012.6288833"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638284"},{"key":"ref17","doi-asserted-by":"crossref","first-page":"22","DOI":"10.21437\/Interspeech.2012-6","article-title":"Recurrent neural networks for noise reduction in robust ASR","author":"maas","year":"2012","journal-title":"Proceedings of Interspeech"},{"key":"ref18","doi-asserted-by":"crossref","first-page":"3512","DOI":"10.21437\/Interspeech.2013-267","article-title":"Reverberant speech recognition based on denoising autoencoder","author":"ishii","year":"2013","journal-title":"Proceedings of Interspeech"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853900"},{"key":"ref28","doi-asserted-by":"crossref","first-page":"2801","DOI":"10.21437\/Interspeech.2005-617","article-title":"Speech parameter generation algorithm considering global variance for HMM-based speech synthesis","author":"toda","year":"2005","journal-title":"Proceedings of Interspeech 2005"},{"key":"ref4","first-page":"2268","article-title":"Prosody contour prediction with long short-term memory, bidirectional, deep recurrent neural networks","author":"fernandez","year":"2014","journal-title":"Proceedings of Interspeech"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref3","first-page":"1964","article-title":"TTS synthesis with bidirectional LSTM based recurrent neural networks","author":"fan","year":"2014","journal-title":"Proceedings of Interspeech"},{"key":"ref6","first-page":"1954","article-title":"DNN-based stochastic postfilter for HMM-based speech synthesis","author":"chen","year":"2014","journal-title":"Proceedings of Interspeech"},{"key":"ref29","doi-asserted-by":"crossref","first-page":"1974","DOI":"10.21437\/Interspeech.2010-560","article-title":"On generating combilex pronunciations via morphological analysis","author":"richmond","year":"2010","journal-title":"Proceedings of Interspeech"},{"key":"ref5","article-title":"A function-wise pre-training technique for constructing a deep neural network based spectral model in statistical parametric speech synthesis","author":"takaki","year":"2015","journal-title":"Machine Learning in Spoken Language Processing (ML-SLP)"},{"key":"ref8","first-page":"890","article-title":"Acoustic modeling with deep neural networks using raw time signal for lvcsr","author":"tuske","year":"2014","journal-title":"Proceedings of Interspeech"},{"key":"ref7","first-page":"2242","article-title":"Multiple feed- forward deep neural networks for statistical parametric speech synthesis","author":"takaki","year":"2015","journal-title":"Proceedings of Interspeech"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2013.2269291"},{"article-title":"Convolutional neural networks-based continuous speech recognition using raw speech signal2, journal","year":"0","author":"palaz","key":"ref9"},{"key":"ref1","first-page":"7962","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"zen","year":"2013","journal-title":"Proceedings of ICASSP"},{"key":"ref20","doi-asserted-by":"crossref","first-page":"1692","DOI":"10.21437\/Interspeech.2010-487","article-title":"Binary coding of speech spectrograms using a deep auto-encoder","author":"deng","year":"2010","journal-title":"Proceedings of Interspeech"},{"key":"ref22","first-page":"4614","article-title":"An au-toencoder neural-network based low-dimensionality approach to excitation modeling for hmm-based text-to-speech","author":"vishnubhotla","year":"2010","journal-title":"Proceedings of ICASSP"},{"key":"ref21","doi-asserted-by":"crossref","first-page":"436","DOI":"10.21437\/Interspeech.2013-130","article-title":"Speech enhancement based on deep denoising autoencoder","author":"lu","year":"2013","journal-title":"Proceedings of Interspeech"},{"key":"ref24","article-title":"A deep learning approach to data-driven parameterizations for statistical parametric speech synthesis","volume":"abs 1409 8558","author":"muthukumar","year":"2014","journal-title":"CoRR"},{"key":"ref23","first-page":"1969","article-title":"Deep neural network based trainable voice source model for synthesis of speech with varying vocal effort","author":"raitio","year":"2014","journal-title":"Proceedings of Interspeech"},{"key":"ref26","doi-asserted-by":"crossref","DOI":"10.21437\/Blizzard.2011-1","article-title":"The blizzard challenge 2011","author":"king","year":"2011"},{"key":"ref25","doi-asserted-by":"crossref","first-page":"504","DOI":"10.1126\/science.1127647","article-title":"Reducing the dimensionality of data with neural networks","volume":"313","author":"hinton","year":"2006","journal-title":"Science"}],"event":{"name":"2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2016,3,20]]},"location":"Shanghai","end":{"date-parts":[[2016,3,25]]}},"container-title":["2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7465907\/7471614\/07472736.pdf?arnumber=7472736","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,17]],"date-time":"2024-06-17T21:25:55Z","timestamp":1718659555000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7472736\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,3]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/icassp.2016.7472736","relation":{},"subject":[],"published":{"date-parts":[[2016,3]]}}}