{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,24]],"date-time":"2026-02-24T20:15:14Z","timestamp":1771964114537,"version":"3.50.1"},"reference-count":42,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2012,8,1]],"date-time":"2012-08-01T00:00:00Z","timestamp":1343779200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2012,8]]},"DOI":"10.1109\/tasl.2012.2187195","type":"journal-article","created":{"date-parts":[[2012,2,7]],"date-time":"2012-02-07T22:03:33Z","timestamp":1328652213000},"page":"1713-1724","source":"Crossref","is-referenced-by-count":48,"title":["Statistical Parametric Speech Synthesis Based on Speaker and Language Factorization"],"prefix":"10.1109","volume":"20","author":[{"given":"H.","family":"Zen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"N.","family":"Braunschweiler","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"S.","family":"Buchholz","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"M. J. F.","family":"Gales","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"K.","family":"Knill","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"S.","family":"Krstulovic","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"J.","family":"Latorre","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"crossref","first-page":"410","DOI":"10.21437\/Interspeech.2010-172","article-title":"Speaker and language adaptive training for HMM-based polyglot speech synthesis","author":"zen","year":"2010","journal-title":"Proc INTERSPEECH"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2011.03.003"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2004.1325908"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2010.5495562"},{"key":"ref31","author":"yamagishi","year":"2006","journal-title":"Average-voice-based speech synthesis"},{"key":"ref30","doi-asserted-by":"crossref","first-page":"528","DOI":"10.21437\/Interspeech.2009-192","article-title":"State mapping based method for cross-lingual speaker adaptation in HMM-based speech synthesis","author":"wu","year":"2009","journal-title":"Proc INTERSPEECH"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2004.1325986"},{"key":"ref36","doi-asserted-by":"crossref","first-page":"2347","DOI":"10.21437\/Eurospeech.1999-596","article-title":"Simultaneous modeling of spectrum, pitch and duration in HMM-based speech synthesis","author":"yoshimura","year":"1999","journal-title":"Proc EUROSPEECH"},{"key":"ref35","article-title":"The HTS2007' system: Yet another evaluation of the speaker-adaptive HMM-based speech synthesis system in the 2008 Blizzard Challenge","author":"yamagishi","year":"2008","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref34","article-title":"The CSTR\/EMIME HTS system for Blizzard Challenge","author":"yamagishi","year":"2010","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref10","first-page":"4469","article-title":"Acoustic modeling with contextual additive structure for HMM-based speech recognition","author":"nankaku","year":"2008","journal-title":"Proc ICASSP"},{"key":"ref40","doi-asserted-by":"crossref","first-page":"2091","DOI":"10.21437\/Interspeech.2009-599","article-title":"Context-dependent additive <ref_formula> <tex Notation=\"TeX\">$\\log F_{0}$<\/tex><\/ref_formula> model for HMM-based speech synthesis","author":"zen","year":"2009","journal-title":"Proc INTERSPEECH"},{"key":"ref11","author":"odell","year":"1995","journal-title":"The use of context in large vocabulary speech recognition"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICOSP.2010.5656849"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2009.2015708"},{"key":"ref14","author":"saino","year":"2008","journal-title":"A clustering technique for factor analyzed voice models"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2003.07.011"},{"key":"ref16","first-page":"345","article-title":"Globalphone: A multilingual speech and text database developed at Karlsruhe University","author":"schultz","year":"2002","journal-title":"Proc ICSLP"},{"key":"ref17","doi-asserted-by":"crossref","first-page":"1097","DOI":"10.21437\/Interspeech.2011-415","article-title":"Separating speaker and environmental variability using factored transforms","author":"seltzer","year":"2011","journal-title":"Proc INTERSPEECH"},{"key":"ref18","first-page":"1269","article-title":"Eigenvoices for HMM-based speech synthesis","author":"shichiri","year":"2002","journal-title":"Proc ICSLP"},{"key":"ref19","doi-asserted-by":"crossref","first-page":"99","DOI":"10.21437\/Eurospeech.1997-52","article-title":"Acoustic modeling based on the MDL criterion for speech recognition","author":"shinoda","year":"1997","journal-title":"Proc EUROSPEECH"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5947375"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1998.0043"},{"key":"ref27","doi-asserted-by":"crossref","first-page":"835","DOI":"10.21437\/Eurospeech.1999-216","article-title":"From multilingual to polyglot speech synthesis","author":"traber","year":"1999","journal-title":"Proc EUROSPEECH"},{"key":"ref3","doi-asserted-by":"crossref","first-page":"3053","DOI":"10.21437\/Interspeech.2011-764","article-title":"Crowdsourcing preference tests, and how to detect cheating","author":"buchholz","year":"2011","journal-title":"Proc INTERSPEECH"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2001.1034593"},{"key":"ref29","author":"wester","year":"2010","journal-title":"The EMIME Bilingual Database"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/89.848223"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2006.05.003"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/89.876308"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSLP.1996.607807"},{"key":"ref9","doi-asserted-by":"crossref","first-page":"622","DOI":"10.21437\/Interspeech.2010-238","article-title":"An analysis of language mismatch in HMM state mapping-based cross-lingual speaker adaptation","author":"liang","year":"2010","journal-title":"Proc INTERSPEECH"},{"key":"ref1","year":"1999","journal-title":"Handbook of the International Phonetic Association"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2006.1660231"},{"key":"ref22","author":"sproat","year":"1998","journal-title":"Multilingual Text-to-Speech Synthesis The Bell Labs Approach"},{"key":"ref21","first-page":"97","article-title":"Adaptation of precision matrix models on large vocabulary continuous speech recognition","author":"sim","year":"2005","journal-title":"Proc ICASSP"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref24","first-page":"455","article-title":"Multi-space probability distribution HMM","volume":"e85 d","author":"tokuda","year":"2002","journal-title":"IEICE Trans Inf Syst"},{"key":"ref41","first-page":"186","article-title":"HMM-based polyglot speech synthesis by speaker and language adaptive training","author":"zen","year":"2010","journal-title":"Proc ISCA SSW7"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e90-d.5.816"},{"key":"ref26","article-title":"An HMM-based speech synthesis system applied to English","author":"tokuda","year":"2002","journal-title":"Proc IEEE Workshop Speech Synth"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.861820"}],"container-title":["IEEE Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx5\/10376\/6176018\/06148263.pdf?arnumber=6148263","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,20]],"date-time":"2025-03-20T01:36:56Z","timestamp":1742434616000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/6148263\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012,8]]},"references-count":42,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/tasl.2012.2187195","relation":{},"ISSN":["1558-7916","1558-7924"],"issn-type":[{"value":"1558-7916","type":"print"},{"value":"1558-7924","type":"electronic"}],"subject":[],"published":{"date-parts":[[2012,8]]}}}