{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,18]],"date-time":"2026-01-18T00:03:05Z","timestamp":1768694585083,"version":"3.49.0"},"reference-count":64,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2018,9,1]],"date-time":"2018-09-01T00:00:00Z","timestamp":1535760000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"funder":[{"DOI":"10.13039\/501100002341","name":"Academy of Finland","doi-asserted-by":"publisher","award":["312490"],"award-info":[{"award-number":["312490"]}],"id":[{"id":"10.13039\/501100002341","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002341","name":"Academy of Finland","doi-asserted-by":"publisher","award":["284671"],"award-info":[{"award-number":["284671"]}],"id":[{"id":"10.13039\/501100002341","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2018,9]]},"DOI":"10.1109\/taslp.2018.2835720","type":"journal-article","created":{"date-parts":[[2018,5,18]],"date-time":"2018-05-18T21:39:54Z","timestamp":1526679594000},"page":"1658-1670","source":"Crossref","is-referenced-by-count":38,"title":["A Comparison Between STRAIGHT, Glottal, and Sinusoidal Vocoding in Statistical Parametric Speech Synthesis"],"prefix":"10.1109","volume":"26","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8031-2260","authenticated-orcid":false,"given":"Manu","family":"Airaksinen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2201-103X","authenticated-orcid":false,"given":"Lauri","family":"Juvela","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bajibabu","family":"Bollepalli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2752-3955","authenticated-orcid":false,"given":"Junichi","family":"Yamagishi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8173-9418","authenticated-orcid":false,"given":"Paavo","family":"Alku","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1288"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-712"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/89.799695"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1121\/1.1995189"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1121\/1.4812756"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(93)90019-H"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-342"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(94)00054-E"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/89.928922"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1121\/1.384992"},{"key":"ref60","first-page":"334","article-title":"Comparison of formant enhancement methods for HMM-based speech\n synthesis","author":"raitio","year":"0","journal-title":"Proc 8th Int Speech Commun Assoc Speech Synthesis Workshop"},{"key":"ref62","article-title":"Crowdflower","year":"2017"},{"key":"ref61","first-page":"337","article-title":"Uniform\n speech parameterization for multiform segment synthesis","author":"sorin","year":"0","journal-title":"Proc Int Speech Commun Assoc"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-479"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2013.2294585"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-363"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016-35"},{"key":"ref29","first-page":"1","author":"d\u2019alessandro","year":"2007","journal-title":"Phase-Based Methods for Voice Source Analysis"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2013.08.003"},{"key":"ref20","article-title":"Wavenet: A generative model for raw audio","author":"van den oord","year":"2016","journal-title":"CoRR"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2013.03.003"},{"key":"ref21","article-title":"Tacotron: A fully end-to-end text-to-speech\n synthesis model","author":"wang","year":"2017","journal-title":"CoRR"},{"key":"ref24","first-page":"1969","article-title":"Deep neural network based trainable voice source model for\n synthesis of speech with varying vocal effort","author":"raitio","year":"0","journal-title":"Proc Int Speech Commun Assoc"},{"key":"ref23","first-page":"2290","article-title":"Voice source modeling using deep neural\n networks for statistical parametric speech synthesis","author":"raitio","year":"0","journal-title":"Proc 22nd Eur Signal Process Conf"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2017.2665687"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472653"},{"key":"ref50","first-page":"1420","article-title":"Wideband parametric speech synthesis using\n warped linear prediction","author":"raitio","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1121\/1.2951592"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1002\/scj.20354"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.861820"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2013.2251852"},{"key":"ref56","first-page":"1974","article-title":"On generating\n combilex pronunciations via morphological analysis","author":"richmond","year":"0","journal-title":"Proc Int Speech Commun Assoc"},{"key":"ref55","article-title":"Festival","year":"2014"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016-33"},{"key":"ref53","article-title":"Objective measurement of active speech level)","year":"2011"},{"key":"ref52","first-page":"497","article-title":"A robust algorithm for pitch tracking (RAPT)","author":"talkin","year":"0","journal-title":"Speech Coding and Synthesis"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00085-5"},{"key":"ref11","article-title":"Aperiodicity extraction and control using mixed mode excitation and group delay manipulation for a high quality speech\n analysis, modification and synthesis system STRAIGHT","author":"kawahara","year":"0","journal-title":"Proc 7th Int Workshop Models Anal Vocal Emissions Biomed Appl"},{"key":"ref40","first-page":"1043","article-title":"Mel-generalized cepstral analysis&#x2013;a\n unified approach to speech spectral estimation","author":"tokuda","year":"0","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2169787"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2045239"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2014.2307274"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1186\/s13636-014-0038-1"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2013.2283471"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178768"},{"key":"ref19","first-page":"155","article-title":"An experimental comparison of multiple vocoder\n types","author":"hu","year":"0","journal-title":"Proc 8th Int Speech Commun Assoc Speech Synthesis Workshop"},{"key":"ref4","article-title":"Toward the perfect audio morph? singing voice synthesis and\n processing","author":"cook","year":"0","journal-title":"Proc Digital Audio Effects Workshop"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2009.4960401"},{"key":"ref6","author":"rabiner","year":"1978","journal-title":"Digital Processing of Speech Signals"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/89.397089"},{"key":"ref8","author":"b\u00e4ckstr\u00f6m","year":"2017","journal-title":"Speech Coding with Code-Excited Linear Prediction"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1002\/j.1538-7305.1966.tb01706.x"},{"key":"ref49","article-title":"Methods for subjective determination of transmission quality","year":"1996"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1202"},{"key":"ref46","article-title":"Hurricane natural speech corpus","author":"cooke","year":"2013"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953090"},{"key":"ref48","article-title":"Method for the subjective assessment of intermediate quality levels of coding systems)","year":"2015"},{"key":"ref47","article-title":"The Blizzard challenge 2011","author":"king","year":"0","journal-title":"Proceedings of Blizzard Challenge Workshop 2011"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1996.541114"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(92)90005-R"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472657"},{"key":"ref43","first-page":"1964","article-title":"TTS synthesis with bidirectional LSTM based recurrent\n neural networks","author":"fan","year":"0","journal-title":"Proc Int Speech Commun Assoc"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8361959\/08357921.pdf?arnumber=8357921","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T16:12:19Z","timestamp":1642003939000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8357921\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,9]]},"references-count":64,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2018.2835720","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,9]]}}}