{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,6,5]],"date-time":"2024-06-05T11:38:21Z","timestamp":1717587501853},"reference-count":28,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2013,7,27]],"date-time":"2013-07-27T00:00:00Z","timestamp":1374883200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2014,11]]},"DOI":"10.1007\/s11042-013-1610-x","type":"journal-article","created":{"date-parts":[[2013,7,26]],"date-time":"2013-07-26T12:03:49Z","timestamp":1374840229000},"page":"397-415","source":"Crossref","is-referenced-by-count":8,"title":["Speech driven photo realistic facial animation based on an articulatory DBN model and AAM features"],"prefix":"10.1007","volume":"73","author":[{"given":"Dongmei","family":"Jiang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yong","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hichem","family":"Sahli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanning","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,7,27]]},"reference":[{"key":"1610_CR1","author":"B Abboud","year":"2004","unstructured":"Abboud B, Davoine F, Dang M (2004) Facial expression recognition and synthesis based on an appearance model. Sig Process: Image Commun. doi: 10.1016\/j.image.2004.05.009","journal-title":"Sig Process: Image Commun"},{"key":"1610_CR2","first-page":"3916","volume":"4","author":"J Bilmes","year":"2002","unstructured":"Bilmes J, Zweig G (2002) The graphical models toolkit: an open source software system for speech and time series processing. Proc IEEE Int Conf Acoust, Speech, Signal Process 4:3916\u20133919","journal-title":"Proc IEEE Int Conf Acoust, Speech, Signal Process"},{"key":"1610_CR3","doi-asserted-by":"crossref","unstructured":"Brand M (1999) Voice puppetry. SIGGRAPH \u201999 proceedings of the 26th annual conference on computer graphics and interactive techniques, pp. 21\u201328","DOI":"10.1145\/311535.311537"},{"key":"1610_CR4","unstructured":"Bregler C, Covell M, Slaney M (2006) Video rewrite: driving visual speech with audio. Computer graphics annual conference series (SIGGRAPH), 353\u2013360, Los Angeles, California"},{"key":"1610_CR5","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1023\/A:1011171430700","volume":"29","author":"K Choi","year":"2001","unstructured":"Choi K, Luo Y, Hwang J (2001) Hidden Markov model inversion for audio-to-visual conversion in an MPEG-4 facial animation system. J VLSI Signal Process 29:51\u201361","journal-title":"J VLSI Signal Process"},{"key":"1610_CR6","first-page":"484","volume":"2","author":"TF Cootes","year":"1998","unstructured":"Cootes TF, Edwards GJ, Taylor CJ (1998) Active appearance models. Proc European Conf Comput Vis 2:484\u2013498","journal-title":"Proc European Conf Comput Vis"},{"issue":"3","key":"1610_CR7","doi-asserted-by":"crossref","first-page":"152","DOI":"10.1109\/6046.865480","volume":"2","author":"E Cossato","year":"2000","unstructured":"Cossato E, Graf HP (2000) Photo-realistic talking heads from image samples. IEEE Trans Multimedia 2(3):152\u2013163","journal-title":"IEEE Trans Multimedia"},{"key":"1610_CR8","doi-asserted-by":"crossref","unstructured":"Ezzat T, Geiger G, Poggio T (2002) Trainable video realistic speech animation. SIGGRAPH \u201902 proceedings of the 29th annual conference on computer graphics and interactive techniques, pp. 388\u2013398","DOI":"10.1145\/566570.566594"},{"key":"1610_CR9","doi-asserted-by":"crossref","unstructured":"Gowdy JN, Subramanya A, et al. (2004) DBN based multi-stream models for audio-visual speech recognition. Proc. International Conference on Acoustics, Speech and Signal Processing, pp. 993\u2013996","DOI":"10.1109\/ICASSP.2004.1326155"},{"issue":"1","key":"1610_CR10","doi-asserted-by":"crossref","first-page":"33","DOI":"10.1109\/TMM.2004.840611","volume":"7","author":"R Gutierrez-Osuna","year":"2005","unstructured":"Gutierrez-Osuna R, Kakumanu PK, Esposito A et al (2005) Speech-driven facial animation with realistic dynamics. IEEE Trans Multimedia 7(1):33\u201342","journal-title":"IEEE Trans Multimedia"},{"key":"1610_CR11","doi-asserted-by":"crossref","first-page":"340","DOI":"10.1007\/978-3-540-74607-2_31","volume":"4678","author":"Y Hou","year":"2007","unstructured":"Hou Y, Sahli H, Ravyse I, Zhang Y, Zhao R (2007) Robust shape based head tracking. Proc Adv Concepts Intell Vis Syst LNCS 4678:340\u2013351","journal-title":"Proc. Adv Concepts Intell Vis Syst LNCS"},{"key":"1610_CR12","unstructured":"http:\/\/personalpages.manchester.ac.uk\/staff\/timothy.f.cootes\/software\/am_tools_doc\/index.html . Accessed on February 23, 2013"},{"key":"1610_CR13","unstructured":"http:\/\/www.reallusion.com\/crazytalk\/ , accessed on February 23, 2013"},{"key":"1610_CR14","doi-asserted-by":"crossref","unstructured":"Jiang D, Ravyse I, Liu P, Sahli H, Verhelst W (2010) Realistic mouth animation based on an articulatory DBN model with constrained asynchrony. Proc. 35th IEEE Int. Conf. Audio, speech and signal processing (ICASSP), March 14\u201319, Texas, USA, pp. 2478\u20132481","DOI":"10.1109\/ICASSP.2010.5494894"},{"issue":"3","key":"1610_CR15","doi-asserted-by":"crossref","first-page":"542","DOI":"10.1109\/TMM.2006.870732","volume":"8","author":"Y Li","year":"2006","unstructured":"Li Y, Shum H-Y (2006) Learning dynamic audio-visual mapping with input-output hidden Markov models. IEEE Trans Multimedia 8(3):542\u2013549","journal-title":"IEEE Trans Multimedia"},{"key":"1610_CR16","doi-asserted-by":"crossref","unstructured":"Livescu K, Centin O, H J Mark, et al (2006). Articulatory feature-based methods for acoustic and audio-visual speech recognition: 2006 JHU summer workshop final report. Center for Language and Speech Processing, Johns Hopkins University","DOI":"10.1109\/ICASSP.2007.366989"},{"key":"1610_CR17","doi-asserted-by":"crossref","unstructured":"Massaro W (2003) A computer-animated tutor for spoken and written language learning. Int. Conf. Multimodal Interfaces, 172\u2013175","DOI":"10.1145\/958432.958466"},{"key":"1610_CR18","doi-asserted-by":"crossref","unstructured":"Mattheyses W, Latacz L, Verhelst V (2010) Active appearance models for photorealistic visual speech synthesis. Proc. INTERSPEECH 2010, pp. 1113\u20131116","DOI":"10.21437\/Interspeech.2010-354"},{"key":"1610_CR19","doi-asserted-by":"crossref","unstructured":"Mattheyses W, Latacz L, Verhelst V, Sahli H (2008) Multimodal unit selection for 2D audiovisual text-to-speech synthesis. Proceedings of the 5th international workshop on machine learning for multimodal interaction, In: Popescu-Belis A, Stiefelhagen R (Eds.), Lecture notes in computer science, Vol. 5237, pp. 125\u2013136","DOI":"10.1007\/978-3-540-85853-9_12"},{"key":"1610_CR20","doi-asserted-by":"crossref","unstructured":"Ning L, Ning F, Kamata S (2010) 3D reconstruction from a single image for a Chinese talking face. TENCON 2010\u20132010 I.E. Region 10 Conference, pp. 1613\u20131616, Nov. 21\u201324, Shanghai, China","DOI":"10.1109\/TENCON.2010.5686036"},{"issue":"1","key":"1610_CR21","doi-asserted-by":"crossref","first-page":"15","DOI":"10.1109\/41.661300","volume":"45","author":"RR Rao","year":"1998","unstructured":"Rao RR, Chen T, Mersereau RM (1998) Audio-to-visual conversion for multimedia communication. IEEE Trans Ind Electron 45(1):15\u201322","journal-title":"IEEE Trans Ind Electron"},{"key":"1610_CR22","author":"G Salvi","year":"2009","unstructured":"Salvi G, Beskow J et al (2009) SynFace-speech-driven facial animation for virtual speech-reading support. EURASIP J Audio, Speech, Music Process. doi: 10.1155\/2009\/191940","journal-title":"EURASIP J Audio, Speech, Music Process"},{"key":"1610_CR23","first-page":"33","volume":"5249","author":"LD Terissi","year":"2008","unstructured":"Terissi LD, Gomez JC (2008) Audio-to-visual conversion via HMM inversion for speech-driven facial animation. Lecture Notes on Artif Intell, LNAI 5249:33\u201342","journal-title":"Lecture Notes on Artif Intell, LNAI"},{"key":"1610_CR24","unstructured":"Wu P, Jiang D, Zhang H, Sahli H (2011) Realistic visual speech synthesis based on AAM features and an articulatory DBN model with constrained asynchrony. Proc. Audio-Visual Speech Processing (AVSP), pp. 59\u201364"},{"issue":"8","key":"1610_CR25","doi-asserted-by":"crossref","first-page":"2325","DOI":"10.1016\/j.patcog.2006.12.001","volume":"40","author":"L Xie","year":"2007","unstructured":"Xie L, Liu Z (2007) Speech animation using coupled hidden Markov models. Pattern Recognit 40(8):2325\u20132340","journal-title":"Pattern Recognit"},{"issue":"3","key":"1610_CR26","doi-asserted-by":"crossref","first-page":"500","DOI":"10.1109\/TMM.2006.888009","volume":"9","author":"L Xie","year":"2007","unstructured":"Xie L, Liu Z (2007) Realistic mouth-synching for speech driven talking face using articulatory modelling. IEEE Trans Multimedia 9(3):500\u2013510","journal-title":"IEEE Trans Multimedia"},{"issue":"1\u20132","key":"1610_CR27","doi-asserted-by":"crossref","first-page":"105","DOI":"10.1016\/S0167-6393(98)00054-5","volume":"26","author":"E Yamamoto","year":"1998","unstructured":"Yamamoto E, Nakamura S, Shikano K (1998) Lip movement synthesis from speech based on hidden Markov models. Speech Commun 26(1\u20132):105\u2013115","journal-title":"Speech Commun"},{"key":"1610_CR28","volume-title":"The HTK book","author":"S Young","year":"2001","unstructured":"Young S (2001) The HTK book. Cambridge University Engineering Department, UK"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-013-1610-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-013-1610-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-013-1610-x","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,3,2]],"date-time":"2022-03-02T05:08:33Z","timestamp":1646197713000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-013-1610-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,7,27]]},"references-count":28,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2014,11]]}},"alternative-id":["1610"],"URL":"https:\/\/doi.org\/10.1007\/s11042-013-1610-x","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2013,7,27]]}}}