{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T04:29:20Z","timestamp":1688444960767},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2013,8,15]],"date-time":"2013-08-15T00:00:00Z","timestamp":1376524800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2014,11]]},"DOI":"10.1007\/s11042-013-1633-3","type":"journal-article","created":{"date-parts":[[2013,8,14]],"date-time":"2013-08-14T08:12:11Z","timestamp":1376467931000},"page":"377-396","source":"Crossref","is-referenced-by-count":14,"title":["A statistical parametric approach to video-realistic text-driven talking avatar"],"prefix":"10.1007","volume":"73","author":[{"given":"Lei","family":"Xie","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Naicai","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Fan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,8,15]]},"reference":[{"key":"1633_CR1","doi-asserted-by":"crossref","unstructured":"Berger MA, Hofer G, Shimodaira H (2011) Carnival\u2014combining speech technology and computer animation. IEEE Comput Graph Appl 80\u201389","DOI":"10.1109\/MCG.2011.71"},{"key":"1633_CR2","doi-asserted-by":"crossref","unstructured":"Blanz V, Vetter T (1999) A morphable model for the synthesis of 3d faces. In: Siggraph, pp 187\u2013194","DOI":"10.1145\/311535.311556"},{"key":"1633_CR3","doi-asserted-by":"crossref","unstructured":"Blanz V, Basso C, Poggio T, Vetter T (2003) Reanimating faces in images and video. In: Eurographics, pp 641\u2013650","DOI":"10.1111\/1467-8659.t01-1-00712"},{"key":"1633_CR4","doi-asserted-by":"crossref","unstructured":"Brand M (1999) Voice puppetry. In: Siggraph, pp 21\u201328","DOI":"10.1145\/311535.311537"},{"key":"1633_CR5","unstructured":"Bregler C, Covell M, Slaney M (2007) Video rewrite: driving visual speech with audio. In: Siggraph, pp 353\u2013360"},{"issue":"1","key":"1633_CR6","doi-asserted-by":"crossref","first-page":"9","DOI":"10.1109\/79.911195","volume":"18","author":"T Chen","year":"2001","unstructured":"Chen T (2001) Audiovisual speech processing: lip reading and lip synchronization. IEEE Signal Proc Mag 18(1):9\u201321","journal-title":"IEEE Signal Proc Mag"},{"key":"1633_CR7","unstructured":"Choi K, Hwang JN (1999) Baum\u2013welch hidden markov model inversion for reliable audio-to-visual conversion. In: Proc. IEEE 3rd workshop multimedia signal processing, pp 175\u2013180"},{"issue":"6","key":"1633_CR8","doi-asserted-by":"crossref","first-page":"681","DOI":"10.1109\/34.927467","volume":"23","author":"TG Cootes","year":"2001","unstructured":"Cootes TG, Edwards GJ, Taylor CJ (2001) Active appearance models. IEEE Trans Pattern Anal Mach Intell 23(6):681\u2013685","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"9","key":"1633_CR9","doi-asserted-by":"crossref","first-page":"1406","DOI":"10.1109\/JPROC.2003.817141","volume":"91","author":"E Cosatto","year":"2003","unstructured":"Cosatto E, Ostermann J, Graf HP, Schroeter J (2003) Lifelike talking faces for interactive services. Proc IEEE 91(9):1406\u20131428","journal-title":"Proc IEEE"},{"key":"1633_CR10","volume-title":"Data-driven 3D facial animation","year":"2008","unstructured":"Deng Z, Neumann U (eds) (2008) Data-driven 3D facial animation. Springer, New York"},{"issue":"1","key":"1633_CR11","doi-asserted-by":"crossref","first-page":"45","DOI":"10.1023\/A:1008166717597","volume":"38","author":"T Ezzat","year":"2000","unstructured":"Ezzat T, Poggio T (2000) Visual speech synthesis by morphing visemes. Int J Comput Vis 38(1):45\u201357","journal-title":"Int J Comput Vis"},{"key":"1633_CR12","doi-asserted-by":"crossref","unstructured":"Ezzat T, Geiger G, Poggio T (2002) Trainable videorealistic speech animation. In: Siggraph, pp\u00a0388\u2013397","DOI":"10.1145\/566570.566594"},{"key":"1633_CR13","unstructured":"Fagel S, Bailly GB, Theobald B-J (2009) Animating virtual speakers or singers fromaudio: lip-synching facial animation. In: EURASIP journal on audio, speech, and music processing 2009, pp 1\u20132"},{"key":"1633_CR14","doi-asserted-by":"crossref","unstructured":"Fu S, Gutierrez-Osuna R, Esposito A, Kakumanu KP, Garcia ON (2005) Audio\/visual mapping with cross-modal hidden markov models. IEEE Trans Multimedia 7:243\u2013251","DOI":"10.1109\/TMM.2005.843341"},{"key":"1633_CR15","doi-asserted-by":"crossref","unstructured":"Hofer G, Yamagishi J, Shimodaira H (2008) Speech-driven lip motion generation with a trajectory hmm. In: Proc. of interspeech","DOI":"10.21437\/Interspeech.2008-591"},{"key":"1633_CR16","unstructured":"Hura S, Leathem C, Shaked N (2010) Avatars meet the challenge. Speech Technol 30\u201332"},{"issue":"3","key":"1633_CR17","doi-asserted-by":"crossref","first-page":"570","DOI":"10.1109\/TASL.2010.2052246","volume":"19","author":"J Jia","year":"2011","unstructured":"Jia J, Zhang S, Meng F, Wang Y, Cai L (2011) Emotional audio-visual speech synthesis based on pad. EURASIP J Audio Speech Music Process 19(3):570\u2013582","journal-title":"EURASIP J Audio Speech Music Process"},{"key":"1633_CR18","doi-asserted-by":"crossref","unstructured":"Jia J, Wu Z, Zhang S, Meng H, Cai L (2013) Head and facial gestures synthesis using pad model for an expressive talking avatar. Multimed Tools Appl. doi: 10.1007\/S11042-013-1604-8","DOI":"10.1007\/s11042-013-1604-8"},{"key":"1633_CR19","doi-asserted-by":"crossref","unstructured":"Kessentini Y, Paquet T, Hamadou AB (2010) Off-line handwritten word recognition using multi-stream hidden markov models. Pattern Recogn Lett 31(1):60\u201370","DOI":"10.1016\/j.patrec.2009.08.009"},{"key":"1633_CR20","doi-asserted-by":"crossref","unstructured":"Liu K, Ostermann J (2009) Optimization of an image-based talking head system. In: EURASIP journal on audio, speech, and music processing, vol 2009","DOI":"10.1155\/2009\/174192"},{"key":"1633_CR21","doi-asserted-by":"crossref","unstructured":"Meng F, Wu Z, Jia J, Meng H, Cai L (2013) Synthesizing english emphatic speech for multimodal correstive feedback in computer-aided pronunciation training. Multimed Tools Appl. doi: 10.1007\/s11042-013-1601-y","DOI":"10.1007\/s11042-013-1601-y"},{"key":"1633_CR22","doi-asserted-by":"crossref","first-page":"746","DOI":"10.1038\/264746a0","volume":"264","author":"H McGurk","year":"1976","unstructured":"McGurk H, MacDonald J (1976) Hearing lips and seeing voices. Nature 264:746\u2013748","journal-title":"Nature"},{"issue":"1\u20132","key":"1633_CR23","first-page":"45","volume":"40","author":"T Ohman","year":"1999","unstructured":"Ohman T, Salvi G (1999) Using hmms and anns formapping acoustic to visual speech. TMH-QPSR 40(1\u20132):45\u201350","journal-title":"TMH-QPSR"},{"key":"1633_CR24","doi-asserted-by":"crossref","unstructured":"Ostermann J, Weissenfeld A (2004) Talking faces - technologies and applications. In: Proc. of ICPR, vol 3, pp 826\u2013833","DOI":"10.1109\/ICPR.2004.1334656"},{"key":"1633_CR25","volume-title":"MPEG-4 facial animation the standard, implementation and applications","year":"2002","unstructured":"Pandzic IS, Forchheimer R (eds) (2002) MPEG-4 facial animation the standard, implementation and applications. Wiley, New York"},{"key":"1633_CR26","doi-asserted-by":"crossref","unstructured":"P\u00e8rez P, Gangnet M, Blake A (2003) Poisson image editing. In: ACM Trans. Graphics, vol 22, pp 313\u2013318","DOI":"10.1145\/882262.882269"},{"key":"1633_CR27","doi-asserted-by":"crossref","unstructured":"Pighin F, Hecker J, Lischinski D, Szeliski R, Salesin DH (1998) Synthesizing realistic facial expressions from photographs. In: Siggraph, pp 75\u201384","DOI":"10.1145\/280814.280825"},{"key":"1633_CR28","unstructured":"Potamianos G, Neti C, Luettin J, Matthews I (2004) Issues in visual and audio-visual speech processing. Ch. Audio-visual automatic speech recognition: an overview. MIT Press, pp 121\u2013148"},{"key":"1633_CR29","doi-asserted-by":"crossref","unstructured":"Salvi G, Beskow J, Moubayed SA, Granstrom B (2009) Synface\u2013speech-driven facial animation for virtual speech-reading support. In: EURASIP journal on audio, speech, and music processing, vol 2009","DOI":"10.1155\/2009\/191940"},{"key":"1633_CR30","doi-asserted-by":"crossref","unstructured":"Shinji\u00a0Sako KT, Masuko T, Kobayashi T, Kitamura T (2000) Hmm-based text-to-audio-visual speech synthesis. In: Interspeech","DOI":"10.21437\/ICSLP.2000-469"},{"key":"1633_CR31","unstructured":"Summereld AQ (1987) Some preliminaries to a comprehensive account of audio-visual speech perception. Lawrence Erlbaum Associates, Ch. Hearing by Eye: The Psychology of Lip-Reading, pp 97\u2013113"},{"key":"1633_CR32","doi-asserted-by":"crossref","unstructured":"Tamura M, Kondo S, Masuko T, Kobayashi T (1999) Text to audio-visual speech synthesis based on parameter generation from HMM. In: Eurospeech, pp 959\u2013962","DOI":"10.21437\/Eurospeech.1999-234"},{"key":"1633_CR33","unstructured":"Theobald B-J, Wilkinson N (2007) A real-time speech-driven talking head using active appearance models. In: AVSP"},{"key":"1633_CR34","doi-asserted-by":"crossref","unstructured":"Theobald B-J, Fagel S, Bailly G, Elisei F (2008) Lips2008: visual speech synthesis challenge. In: Proc. of interspeech","DOI":"10.21437\/Interspeech.2008-590"},{"key":"1633_CR35","unstructured":"Theobald B, Matthews I, Wilkinson N, Cohn JF, Boker S (2007) Animating faces using appearance models. In: Proceedings of the workshop on vision, video and graphics"},{"key":"1633_CR36","unstructured":"Tokuda K, Yoshimura T, Masuko T, Kobayashi T, Kitamura T (2000) Speech parameter generation algorigthms for hmm-based speech synthesis. In: ICASSP, pp 1315\u20131318"},{"key":"1633_CR37","doi-asserted-by":"crossref","unstructured":"Wang L, Qian X, Han W, Soong FK (2010) Synthesizing photo-real talking head via trajectory-guided sample selection. In: Interspeech","DOI":"10.21437\/Interspeech.2010-194"},{"key":"1633_CR38","unstructured":"Wang L, Han W, Soong FK, Huo Q (2011) Text driven 3d photo-realistic talking head. In: Interspeech, pp 3307\u20133310"},{"key":"1633_CR39","doi-asserted-by":"crossref","unstructured":"Weise T, Bouaziz S, Li H, Pauly M (2011) Realtime performance-based facial animation. In: Siggraph","DOI":"10.1145\/1964921.1964972"},{"key":"1633_CR40","doi-asserted-by":"crossref","unstructured":"Wu Z, Zhang S, Cai L, Meng H (2006) Real-time synthesis of chinese visual speech and facial expressions using mpeg-4 fap features in a three-dimensional avatar. In: Proc. Interspeech, pp 1802\u20131805","DOI":"10.21437\/Interspeech.2006-498"},{"issue":"23","key":"1633_CR41","doi-asserted-by":"crossref","first-page":"500","DOI":"10.1109\/TMM.2006.888009","volume":"9","author":"L Xie","year":"2007","unstructured":"Xie L, Liu Z-Q (2007) Realistic mouth-synching for speech-driven talking face using articulatory modelling. IEEE Trans Multimed 9(23):500\u2013510","journal-title":"IEEE Trans Multimed"},{"issue":"1\u20132","key":"1633_CR42","doi-asserted-by":"crossref","first-page":"105","DOI":"10.1016\/S0167-6393(98)00054-5","volume":"26","author":"E Yamamoto","year":"1998","unstructured":"Yamamoto E, Nakamura S, Shikano K (1998) Lip movement synthesis from speech based on hidden markov models. Speech Comm 26(1\u20132):105\u2013115","journal-title":"Speech Comm"},{"key":"1633_CR43","doi-asserted-by":"crossref","unstructured":"Yamagishi J, Masuko T, Tokuda K, Kobayashi T (2003) A training method for average voice model based on shared decision tree context clustering and speaker adaptive training. In: ICASSP, pp 716\u2013719","DOI":"10.1109\/ICASSP.2003.1198881"},{"issue":"4","key":"1633_CR44","doi-asserted-by":"crossref","first-page":"570","DOI":"10.1109\/TMM.2008.921737","volume":"10","author":"Z Zeng","year":"2008","unstructured":"Zeng Z, Tu J, Pianfetti BM, Huang TS (2008) Audio-visual affective expression recognition through multistream fused hmm. IEEE Trans Multimed 10(4):570\u2013577","journal-title":"IEEE Trans Multimed"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-013-1633-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-013-1633-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-013-1633-3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,3]],"date-time":"2023-07-03T21:15:43Z","timestamp":1688418943000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-013-1633-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,8,15]]},"references-count":44,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2014,11]]}},"alternative-id":["1633"],"URL":"https:\/\/doi.org\/10.1007\/s11042-013-1633-3","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2013,8,15]]}}}