{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,15]],"date-time":"2025-06-15T07:40:06Z","timestamp":1749973206019,"version":"3.41.0"},"reference-count":27,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016,12]]},"DOI":"10.1109\/slt.2016.7846337","type":"proceedings-article","created":{"date-parts":[[2017,2,10]],"date-time":"2017-02-10T15:58:30Z","timestamp":1486742310000},"page":"686-692","source":"Crossref","is-referenced-by-count":2,"title":["Median-based generation of synthetic speech durations using a non-parametric approach"],"prefix":"10.1109","author":[{"given":"Srikanth","family":"Ronanki","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Oliver","family":"Watts","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Simon","family":"King","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gustav Eje","family":"Henter","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/s12046-011-0048-y"},{"key":"ref11","doi-asserted-by":"crossref","first-page":"1393","DOI":"10.21437\/Interspeech.2004-460","article-title":"Hidden semi-markov model based speech synthesis","author":"zen","year":"2004","journal-title":"Proc INTERSPEECH"},{"key":"ref12","first-page":"294","article-title":"The HMM-based speech synthesis system (HTS) version 2.0","volume":"6","author":"zen","year":"2007","journal-title":"Proc of SSW6"},{"key":"ref13","doi-asserted-by":"crossref","first-page":"2698","DOI":"10.21437\/Eurospeech.1989-328","article-title":"Syllable-level duration determination","author":"nick campbell","year":"1989","journal-title":"Proc EUROSPEECH"},{"key":"ref14","first-page":"1127","article-title":"A statistical model of duration control for speech synthesis","author":"huber","year":"1990","journal-title":"Proc EUSIPCO"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-522"},{"key":"ref16","first-page":"1504","article-title":"Measuring the perceptual effects of modelling assumptions in speech synthesis using stimuli constructed from repeated natural speech","author":"henter","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1111\/j.1467-842X.1981.tb00784.x"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1137\/S0040585X97975447"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.861820"},{"key":"ref4","article-title":"A study of speaker adaptation for DNN-based speech synthesis","author":"wu","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472657"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178816"},{"key":"ref6","article-title":"Wavenet: A generative model for raw audio","author":"van den oord","year":"2016","journal-title":"ArXiv"},{"key":"ref5","first-page":"2217","article-title":"Sentence-level control vectors for deep neural network speech synthesis","author":"watts","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.941031"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1121\/1.395275"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178814"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref1","first-page":"7962","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"zen","year":"2013","journal-title":"Proc ICASSP"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472655"},{"key":"ref22","first-page":"2268","article-title":"Prosody contour prediction with long short-term memory, bi-directional, deep recurrent neural networks","author":"fernandez","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref21","article-title":"The NST-Glott HMM entry to the Blizzard Challenge 2015","author":"watts","year":"2015","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/PACRIM.1993.407206"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-96"},{"key":"ref26","first-page":"-853i","article-title":"Sub-phonetic modeling for capturing pronunciation variations for conversational speech synthesis","author":"kishore","year":"2006","journal-title":"Proc ICASSP"},{"key":"ref25","article-title":"The Blizzard Challenge 2016","author":"king","year":"2016","journal-title":"Blizzard Challenge Workshop"}],"event":{"name":"2016 IEEE Spoken Language Technology Workshop (SLT)","start":{"date-parts":[[2016,12,13]]},"location":"San Diego, CA","end":{"date-parts":[[2016,12,16]]}},"container-title":["2016 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7836849\/7846230\/07846337.pdf?arnumber=7846337","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,15]],"date-time":"2025-06-15T07:19:48Z","timestamp":1749971988000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7846337\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,12]]},"references-count":27,"URL":"https:\/\/doi.org\/10.1109\/slt.2016.7846337","relation":{},"subject":[],"published":{"date-parts":[[2016,12]]}}}