{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,10]],"date-time":"2025-02-10T08:40:33Z","timestamp":1739176833564,"version":"3.37.0"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"3-4","license":[{"start":{"date-parts":[[2008,12,1]],"date-time":"2008-12-01T00:00:00Z","timestamp":1228089600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["J Multimodal User Interfaces"],"published-print":{"date-parts":[[2008,12]]},"DOI":"10.1007\/s12193-009-0015-7","type":"journal-article","created":{"date-parts":[[2009,6,23]],"date-time":"2009-06-23T10:10:38Z","timestamp":1245751838000},"page":"157-169","source":"Crossref","is-referenced-by-count":6,"title":["Speech driven realistic mouth animation based on multi-modal unit selection"],"prefix":"10.1007","volume":"2","author":[{"given":"Dongmei","family":"Jiang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ilse","family":"Ravyse","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hichem","family":"Sahli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Werner","family":"Verhelst","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2009,6,24]]},"reference":[{"key":"15_CR1","doi-asserted-by":"crossref","first-page":"746","DOI":"10.1038\/264746a0","volume":"264","author":"H McGurk","year":"1976","unstructured":"McGurk H, MacDonaldd J (1976) Hearing lips and seeing voices. Nature 264:746\u2013748","journal-title":"Nature"},{"key":"15_CR2","volume-title":"Perceiving talking faces","author":"D Massaro","year":"1998","unstructured":"Massaro D (1998) Perceiving talking faces. MIT Press, Cambridge"},{"key":"15_CR3","doi-asserted-by":"crossref","first-page":"127","DOI":"10.1016\/j.specom.2004.07.002","volume":"44","author":"BJ Theobald","year":"2004","unstructured":"Theobald BJ, Bangham JA, Matthews IA, Cawley GC (2004) Near-videorealistic synthetic talking faces: implementation and evaluation. Speech Commun. 44:127\u2013140","journal-title":"Speech Commun."},{"key":"15_CR4","doi-asserted-by":"crossref","unstructured":"Wu Z, Zhang S, Cai L, Meng H (2006) Real-time synthesis of chinese visual speech and facial expressions using MPEG-4 FAP features in a three-dimensional avatar. In: Proc of the international conference on spoken language processing (ICSLP), Pittsburg, USA, Sep 17\u201321","DOI":"10.21437\/Interspeech.2006-498"},{"key":"15_CR5","doi-asserted-by":"crossref","unstructured":"Bregler C, Covell M, Slaney M (1997) Video rewrite: driving visual speech with audio. In: Computer graphics annual conference series (SIGGRAPH), pp\u00a0353\u2013360, Los Angeles, California","DOI":"10.1145\/258734.258880"},{"key":"15_CR6","doi-asserted-by":"crossref","unstructured":"Cosatto E, Graf H (1998) Sample-based synthesis of photorealistic talking heads. In: Proc. of computer animation, pp 103\u2013110, Philadelphia, Pennsylvania","DOI":"10.1109\/CA.1998.681914"},{"key":"15_CR7","doi-asserted-by":"crossref","first-page":"45","DOI":"10.1023\/A:1008166717597","volume":"38","author":"T Ezzat","year":"2000","unstructured":"Ezzat T, Poggio T (2000) Visual speech synthesis by morphing visemes. Int J Comput Vis 38:45\u201357","journal-title":"Int J Comput Vis"},{"key":"15_CR8","doi-asserted-by":"crossref","unstructured":"Ezzat T, Geiger G, Poggio T (2002) Trainable videorealistic speech animation. In: Proc of the international conference on computer graphics and interactive techniques (SIGGRAPH), pp\u00a0388\u2013398, San Antonio, Texas","DOI":"10.1145\/566570.566594"},{"key":"15_CR9","doi-asserted-by":"crossref","unstructured":"Huang F, Cosatto E, Graf H (2002) Triphone based unit selection for concatenative visual speech synthesis. In: Proc of the IEEE international conference on acoustics, speech, and signal processing (ICASSP), vol\u00a0II, pp\u00a02037\u20132040, Orlando, Florida, USA","DOI":"10.1109\/ICASSP.2002.5745033"},{"key":"15_CR10","doi-asserted-by":"crossref","unstructured":"Fagel S (2004) Video-realistic synthetic speech with a parametric visual speech synthesizer. In: Proc of the 8th international conference on spoken language processing (INTERSPEECH), pp\u00a02033\u20132036","DOI":"10.21437\/Interspeech.2004-422"},{"issue":"1","key":"15_CR11","doi-asserted-by":"crossref","first-page":"105","DOI":"10.1016\/S0167-6393(98)00054-5","volume":"26","author":"E Yamamoto","year":"1998","unstructured":"Yamamoto E, Nakamura S, Shikano K (1998) Lip movement synthesis from speech based on hidden Markov models. Speech Commun 26(1):105\u2013115","journal-title":"Speech Commun"},{"key":"15_CR12","doi-asserted-by":"crossref","unstructured":"Nakamura S, Yamamoto E, Shikano K (1998) Speech-to-lip movement synthesis by maximizing audio-visual joint probability based on the EM algorithm. In: Proc of the IEEE second workshop on multimedia signal processing (MMSP), pp 53\u201358","DOI":"10.1109\/MMSP.1998.738912"},{"key":"15_CR13","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1023\/A:1011171430700","volume":"29","author":"K Choi","year":"2001","unstructured":"Choi K, Luo Y, Hwang J (2001) Hidden Markov model inversion for audio-to-visual conversion in an MPEG-4 facial animation system. J\u00a0VLSI Signal Process 29:51\u201361","journal-title":"J\u00a0VLSI Signal Process"},{"key":"15_CR14","doi-asserted-by":"crossref","unstructured":"Aleksic PS, Katsaggelos AK (2003) Speech-to-video synthesis using facial animation parameters. In: Proc of the 2003 international conference on image processing (ICIP03), vol\u00a02, issue III, pp\u00a01\u20134","DOI":"10.1109\/ICIP.2003.1247166"},{"key":"15_CR15","doi-asserted-by":"crossref","unstructured":"Cosker D, Marshall D, Rosin P, Hicks Y (2004) Speech driven facial animation using a hidden Markov coarticulation model. In: Proc of the 17th international conference on pattern recognition 2004 (ICPR2004), vol\u00a01, pp\u00a0128\u2013131","DOI":"10.1109\/ICPR.2004.1334024"},{"issue":"3","key":"15_CR16","doi-asserted-by":"crossref","first-page":"500","DOI":"10.1109\/TMM.2006.888009","volume":"9","author":"L Xie","year":"2007","unstructured":"Xie L, Liu Z-Q (2007) Realistic mouth-synching for speech-driven talking face using articulatory modelling. IEEE Trans Multimedia 9(3):500\u2013510","journal-title":"IEEE Trans Multimedia"},{"key":"15_CR17","doi-asserted-by":"crossref","unstructured":"Jiang D, Xie L, Ravyse I, Zhao R, Sahli H, Cornelis J (2002) Triseme decision trees in the continuous speech recognition system for a talking head. In: Proc of the 1st IEEE international conference on machine learning and cybernetics, pp\u00a02097\u20132100","DOI":"10.1109\/ICMLC.2002.1175408"},{"key":"15_CR18","unstructured":"Verma A, Rajput N, Subramaniam L (2003) Using viseme based acoustic models for speech driven lip synthesis. In: Proc of the IEEE international conference on acoustic speech and signal processing (ICASSP), pp\u00a0720\u2013723"},{"issue":"6","key":"15_CR19","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TVCG.2006.86","volume":"12","author":"Z Deng","year":"2006","unstructured":"Deng Z, Neumann U, Lewiss JP et al. (2006) Expressive facial animation synthesis by learning speech coarticulation and expression spaces. IEEE Trans Vis Comput Graph 12(6):1\u201312","journal-title":"IEEE Trans Vis Comput Graph"},{"key":"15_CR20","doi-asserted-by":"crossref","unstructured":"Cao Y, Faloutsos P, Kohler E, Pighin F (2004) Real-time speech motion synthesis from recorded motions. In: Proc of the ACM SIGGRAPH\/Eurographics symposium on computer animation, pp\u00a0347\u2013355","DOI":"10.1145\/1028523.1028570"},{"key":"15_CR21","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TVCG.2006.8","volume":"12","author":"J Ma","year":"2006","unstructured":"Ma J, Cole R, Pellom B, Ward W, Wise B (2006) Accurate visible speech synthesis based on concatenating variable length motion capture data. IEEE Trans Vis Comput Graph 12:1\u201311","journal-title":"IEEE Trans Vis Comput Graph"},{"key":"15_CR22","doi-asserted-by":"crossref","unstructured":"Ravyse I, Enescu V, Sahli H (2005) Kernel-based head tracker for videophony. In: Proc of the IEEE international conference on image processing 2005 (ICIP2005), Genoa, Italy, vol\u00a03, pp\u00a01068\u20131071","DOI":"10.1109\/ICIP.2005.1530580"},{"key":"15_CR23","doi-asserted-by":"crossref","unstructured":"Hou Y, Sahli H, Ravyse I, Zhang Y, Zhao R (2007) Robust shape-based head tracking. In:\u00a0Proc of the advanced concepts for intelligent vision systems. LNCS, vol\u00a04678, pp\u00a0340\u2013351","DOI":"10.1007\/978-3-540-74607-2_31"},{"key":"15_CR24","doi-asserted-by":"crossref","first-page":"485","DOI":"10.1002\/cav.11","volume":"15","author":"J Ma","year":"2004","unstructured":"Ma J, Cole R, Pellom B, Ward W, Wise B (2004) Accurate automatic visible speech synthesis of arbitrary 3D models based on concatenation of diviseme motion capture data. Comput Animat Virtual Worlds 15:485\u2013500","journal-title":"Comput Animat Virtual Worlds"},{"key":"15_CR25","unstructured":"http:\/\/www.ldc.upenn.edu\/Catalog\/CatalogEntry.jsp?catalogId=LDC93S1 . Accessed 5 December 2008"},{"key":"15_CR26","unstructured":"Young SJ (1993) The HTK hidden Markov model toolkit: design and philosophy. Technical Report, University of Cambridge, Department of Engineering, Cambridge, UK"},{"key":"15_CR27","doi-asserted-by":"crossref","unstructured":"Jiang D, Ravyse I, Sahli H, Zhang Y (2008) Accurate visual speech synthesis based on diviseme unit selection and concatenation. In: Proc of the IEEE 10th workshop on multimedia signal processing (MMSP2008), pp\u00a0906\u2013909","DOI":"10.1109\/MMSP.2008.4665203"},{"key":"15_CR28","first-page":"477","volume-title":"Computer aided geometric design III","author":"R Schaback","year":"1995","unstructured":"Schaback R (1995) Computer aided geometric design III. Vanderbilt University Press, Nashville, pp\u00a0477\u2013496"},{"key":"15_CR29","unstructured":"Ravyse I (2006) Facial analysis and synthesis. Ph.D. Thesis, Dept. Electronics and Informatics, Vrije Universiteit Brussel, Belgium"},{"key":"15_CR30","doi-asserted-by":"crossref","unstructured":"Cosatto E, Potamianos G, Graf HP (2000) Audio visual unit selection for the synthesis of photo-realistic talking heads. In: Proc of the IEEE international conference on multimedia and expo (ICME), vol\u00a02, pp\u00a0619\u2013622","DOI":"10.1109\/ICME.2000.871439"},{"key":"15_CR31","doi-asserted-by":"crossref","unstructured":"Lv G, Jiang D, Zhao R, Hou Y (2007) Multi-stream asynchrony modeling for audio-visual speech recognition. In: Proc of the IEEE international symposium on multimedia (ISM), pp\u00a037\u201344","DOI":"10.1109\/ISM.2007.4412354"}],"container-title":["Journal on Multimodal User Interfaces"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s12193-009-0015-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s12193-009-0015-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s12193-009-0015-7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,10]],"date-time":"2025-02-10T08:24:43Z","timestamp":1739175883000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s12193-009-0015-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2008,12]]},"references-count":31,"journal-issue":{"issue":"3-4","published-print":{"date-parts":[[2008,12]]}},"alternative-id":["15"],"URL":"https:\/\/doi.org\/10.1007\/s12193-009-0015-7","relation":{},"ISSN":["1783-7677","1783-8738"],"issn-type":[{"type":"print","value":"1783-7677"},{"type":"electronic","value":"1783-8738"}],"subject":[],"published":{"date-parts":[[2008,12]]}}}