{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T07:42:23Z","timestamp":1782805343737,"version":"3.54.5"},"reference-count":67,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2009,1,1]],"date-time":"2009-01-01T00:00:00Z","timestamp":1230768000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2009,1]]},"DOI":"10.1109\/tasl.2008.2006647","type":"journal-article","created":{"date-parts":[[2009,1,14]],"date-time":"2009-01-14T22:35:19Z","timestamp":1231972519000},"page":"66-83","source":"Crossref","is-referenced-by-count":201,"title":["Analysis of Speaker Adaptation Algorithms for HMM-Based Speech Synthesis and a Constrained SMAPLR Adaptation Algorithm"],"prefix":"10.1109","volume":"17","author":[{"given":"Junichi","family":"Yamagishi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Takao","family":"Kobayashi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuji","family":"Nakano","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Katsumi","family":"Ogata","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Juri","family":"Isogai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/89.906001"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1998.0043"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICSLP.1996.607807"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1016\/S0885-2308(86)80009-2"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1985.1168477"},{"key":"ref30","first-page":"143","article-title":"variable duration models for speech","author":"ferguson","year":"1980","journal-title":"Proc Symp Applic Hidden Markov Models Text Speech"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/89.466659"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2006.05.003"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/11939993_27"},{"key":"ref34","first-page":"534","article-title":"a context clustering technique for average voice models","volume":"e86 d","author":"yamagishi","year":"2003","journal-title":"IEICE Trans Inf Syst"},{"key":"ref60","first-page":"361","article-title":"multiple-cluster adaptive training schemes","author":"gales","year":"2001","journal-title":"Proc ICASSP'01"},{"key":"ref62","first-page":"1027","article-title":"improved bayesian learning of hidden markov models for speaker adaptation","author":"chien","year":"1997","journal-title":"Proc ICASSP'97"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/978-94-011-1646-6"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1992.225953"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1002\/scj.10332"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1250\/ast.21.79"},{"key":"ref27","first-page":"455","article-title":"multi-space probability distribution hmm","volume":"e85 d","author":"tokuda","year":"2002","journal-title":"IEICE Trans Inf Syst"},{"key":"ref65","first-page":"1089","article-title":"a large-scale japanese speech database","author":"sagisaka","year":"1990","journal-title":"Proc ICSLP'96"},{"key":"ref66","author":"tokuda","year":"0","journal-title":"The HMM-Based Speech Synthesis System (HTS)"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e90-d.5.825"},{"key":"ref67","first-page":"294","article-title":"the hmm-based speech synthesis system (hts) version 2.0","author":"zen","year":"2007","journal-title":"Proc 5th ISCA Speech Synthesis Workshop"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1996.541110"},{"key":"ref1","doi-asserted-by":"crossref","first-page":"581","DOI":"10.21437\/Eurospeech.1995-148","article-title":"optimising selection of units from speech database for concatenative synthesis","author":"black","year":"1995","journal-title":"Proc EuroSpeech'95"},{"key":"ref20","doi-asserted-by":"crossref","first-page":"533","DOI":"10.1093\/ietisy\/e90-d.2.533","article-title":"average-voice-based speech synthesis using hsmm-based speaker adaptation and adaptive training","volume":"e90 d","author":"yamagishi","year":"2007","journal-title":"IEICE Trans Inf Syst"},{"key":"ref22","first-page":"2509","article-title":"voice characteristics conversion for hmm-based speech synthesis system using map-vfs","volume":"j83 d ii","author":"masuko","year":"2000","journal-title":"IEICE Trans"},{"key":"ref21","first-page":"273","article-title":"speaker adaptation for hmm-based speech synthesis system using mllr","author":"tamura","year":"1998","journal-title":"Proc 3rd ESCA\/COCOSDA Workshop Speech Synth"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/89.279278"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1995.0010"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1996.0025"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1996.0008"},{"key":"ref50","first-page":"2812","article-title":"robust f0 estimation of speech signal using harmonicity measure based on instantaneous frequency","volume":"e87 d","author":"arifianto","year":"2004","journal-title":"IEICE Trans Inf Syst"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e90-d.5.816"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.1996.481449"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1998.675351"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/89.759034"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e91-d.6.1764"},{"key":"ref55","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1111\/j.2517-6161.1977.tb01600.x","article-title":"maximum likelihood from incomplete data via the em algorithm","volume":"39","author":"dempster","year":"1977","journal-title":"J R Statist Soc Series B"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1996.0013"},{"key":"ref53","article-title":"a speaker-adaptive hmm-based speech synthesis for the blizzard challenge 2007","author":"yamagishi","year":"2008","journal-title":"IEEE Audio Speech Lang Process"},{"key":"ref52","first-page":"125","article-title":"improved average-voice-based speech synthesis using gender-mixed modeling and a parameter generation algorithm considering gv","author":"yamagishi","year":"2007","journal-title":"Proc 5th ISCA Speech Synthesis Workshop"},{"key":"ref10","first-page":"2374","article-title":"simultaneous modeling of spectrum, pitch and duration in hmm-based speech synthesis","author":"yoshimura","year":"1999","journal-title":"Proc Eurospeech'99"},{"key":"ref11","first-page":"2099","article-title":"simultaneous modeling of spectrum, pitch and duration in hmm-based speech synthesis","volume":"j83 d ii","author":"yoshimura","year":"2000","journal-title":"IEICE Trans"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1006\/csla.2001.0181"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.861820"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e88-d.3.502"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e89-d.3.1092"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e88-d.11.2484"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e90-d.9.1406"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1997.598807"},{"key":"ref18","first-page":"805","article-title":"adaptation of pitch and spectrum for hmm-based speech synthesis using mllr","author":"tamura","year":"2001","journal-title":"Proc ICASSP'01"},{"key":"ref19","first-page":"1956","article-title":"a training method of average voice model for hmm-based speech synthesis","volume":"e86 a","author":"yamagishi","year":"2003","journal-title":"IEICE Trans Fundamentals"},{"key":"ref4","first-page":"411","article-title":"corpus-based techniques in the at&t nextgen synthesis system","author":"syrdal","year":"2000","journal-title":"Proc ICSLP'00"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1999.0123"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1995.479684"},{"key":"ref5","doi-asserted-by":"crossref","first-page":"1649","DOI":"10.21437\/Eurospeech.2003-473","article-title":"unit selection and emotional speech","author":"black","year":"2003","journal-title":"Proc Eurospeech'03"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1996.541114"},{"key":"ref7","first-page":"192","article-title":"an algorithm for speech parameter generation from hmm using dynamic features","volume":"53","author":"tokuda","year":"1997","journal-title":"J Acoust Soc Jpn"},{"key":"ref49","year":"2007","journal-title":"Speech Signal Processing Toolkit (SPTK) Version 3 1"},{"key":"ref9","first-page":"2184","article-title":"hmm-based speech synthesis using dynamic features","volume":"j79 d ii","author":"masuko","year":"1996","journal-title":"IEICE Trans"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2007.367299"},{"key":"ref45","first-page":"1328","article-title":"acoustic model training based on linear transformation and map modification for hsmm-based speech synthesis","author":"ogata","year":"2006","journal-title":"Proc ICSLP'06"},{"key":"ref48","first-page":"1240","article-title":"spectral estimation of speech based on mel-cepstral representation","volume":"j74 a","author":"tokuda","year":"1991","journal-title":"IEICE Trans Fundamentals"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1983.1172250"},{"key":"ref42","first-page":"2597","article-title":"model adaptation and adaptive training using esat algorithm for hmm-based speech synthesis","author":"isogai","year":"2005","journal-title":"Proc EUROSPEECH'05"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/89.506933"},{"key":"ref44","first-page":"2286","article-title":"constrained structural maximum a posteriori linear regression for average-voice-based speech synthesis","author":"nakano","year":"2006","journal-title":"Proc ICSLP'06"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2006.1659961"}],"container-title":["IEEE Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx5\/10376\/4740138\/04740153.pdf?arnumber=4740153","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,6]],"date-time":"2025-02-06T23:18:46Z","timestamp":1738883926000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/4740153\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2009,1]]},"references-count":67,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/tasl.2008.2006647","relation":{},"ISSN":["1558-7916"],"issn-type":[{"value":"1558-7916","type":"print"}],"subject":[],"published":{"date-parts":[[2009,1]]}}}