{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T21:13:58Z","timestamp":1740172438807,"version":"3.37.3"},"reference-count":73,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U1736123","61572450","61303150"],"award-info":[{"award-number":["U1736123","61572450","61303150"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003995","name":"Natural Science Foundation of Anhui Province","doi-asserted-by":"publisher","award":["1708085QF138"],"award-info":[{"award-number":["1708085QF138"]}],"id":[{"id":"10.13039\/501100003995","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Fundamental Research Funds for the Central Universities under","award":["WK2350000002"],"award-info":[{"award-number":["WK2350000002"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1109\/taslp.2019.2935843","type":"journal-article","created":{"date-parts":[[2019,8,22]],"date-time":"2019-08-22T18:56:39Z","timestamp":1566500199000},"page":"2223-2233","source":"Crossref","is-referenced-by-count":1,"title":["Synthesizing 3D Trump: Predicting and Visualizing the Relationship Between Text, Speech, and Articulatory Movements"],"prefix":"10.1109","volume":"27","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3197-8103","authenticated-orcid":false,"given":"Jun","family":"Yu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5688-4130","authenticated-orcid":false,"given":"Qiang","family":"Ling","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Changwei","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chang Wen","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref73","first-page":"217","article-title":"The extended Cohn-Kande dataset (CK+): A complete facial expression dataset for action unit and emotion-specified expression","author":"lucey","year":"2010","journal-title":"Proc Comput Vision Pattern Recognit"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-013-5023-2"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-74048-3"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1044\/jshr.0703.209"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2008.4650814"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/37402.37405"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/2929464.2929475"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/2766943"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10593-2_52"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/2461912.2462019"},{"key":"ref37","first-page":"261","article-title":"Simulating speech with a physics-based facial muscle model","author":"sifakis","year":"2006","journal-title":"Proc Symp Comput Animation"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2017.8019362"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/2816795.2818056"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/3099564.3099581"},{"key":"ref60","first-page":"104","article-title":"A three-dimensional linear articulatory model based on MRI data","author":"badin","year":"1998","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref62","first-page":"395","article-title":"Three-dimensional linear modeling of tongue: Articulatory data and models","author":"badin","year":"2006","journal-title":"Proc Int Seminar Studies Speech Prod"},{"key":"ref61","first-page":"901","article-title":"A 3D tongue model based on MRI data","author":"engwall","year":"2000","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1044\/1092-4388(2001\/009)"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1016\/0093-934X(87)90058-7"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-662-03589-4"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1111\/1467-8659.t01-1-00712"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1063\/1.1712836"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbiomech.2009.01.021"},{"journal-title":"Computer Facial Animation","year":"1996","author":"parke","key":"ref29"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref68","first-page":"786","article-title":"A realistic and reliable 3D pronunciation visualization instruction system for computer-assisted language learning","author":"yu","year":"2016","journal-title":"Proc IEEE Int Conf Bioinf Biomed"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-23123-0_11"},{"key":"ref2","first-page":"1","article-title":"Photo-real lips synthesis with trajectory-guided sample selection","author":"wang","year":"2010","journal-title":"Speech Synth Workshop Int Speech Comm Assoc"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.434"},{"key":"ref20","first-page":"156","article-title":"Speech-driven articulator motion synthesis with deep neural networks","volume":"6","author":"tang","year":"2016","journal-title":"ACTA Automatica Sinica"},{"key":"ref22","first-page":"78","article-title":"A deep neural network for acoustic-articulatory speech inversion","author":"uria","year":"2011","journal-title":"Proc Int Conf Workshop Neural Inf Process Syst Workshop Deep Learn Unsupervised Feature Learn"},{"key":"ref21","doi-asserted-by":"crossref","first-page":"2835","DOI":"10.21437\/Interspeech.2009-724","article-title":"Preliminary inversion mapping results with a new EMA corpus","author":"richmond","year":"2009","journal-title":"Proc INTERSPEECH"},{"key":"ref24","first-page":"2192","article-title":"Articulatory movement prediction using deep bidirectional long short-term memory based recurrent neural networks and word\/phone embeddings","author":"zhu","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178812"},{"key":"ref26","doi-asserted-by":"crossref","first-page":"1505","DOI":"10.21437\/Interspeech.2011-316","article-title":"Announcing the electromagnetic articulography (day 1) subset of the mngu0 articulatory corpus","author":"richmond","year":"2011","journal-title":"Proc INTERSPEECH"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2016.7820703"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016-33"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178814"},{"year":"2018","key":"ref59"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2014.6890231"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1145\/1291233.1291453"},{"key":"ref56","doi-asserted-by":"crossref","first-page":"701","DOI":"10.1016\/j.patcog.2007.05.012","article-title":"A region based stereo matching algorithm using cooperative optimization","author":"wang","year":"2008","journal-title":"Proc Comput Vision Pattern Recognit"},{"key":"ref55","first-page":"115","article-title":"Learning precise timing with LSTM recurrent networks","volume":"3","author":"gers","year":"2002","journal-title":"J Mach Learn Res"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2014.04.028"},{"key":"ref53","first-page":"550","article-title":"Residual networks behave like ensembles of relatively shallow networks","author":"veit","year":"2016","journal-title":"Proc Neural Inf Process Syst"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2014.7025408"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2018.8451618"},{"key":"ref40","first-page":"457","article-title":"Art-directed muscle simulation for high-end facial animation","author":"matthew","year":"2016","journal-title":"Proc Symp Comput Animation"},{"key":"ref12","doi-asserted-by":"crossref","first-page":"867","DOI":"10.21437\/Interspeech.2012-263","article-title":"Deep architectures for articulatory inversion","author":"uria","year":"2012","journal-title":"Proc INTERSPEECH"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073699"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2887027"},{"key":"ref15","first-page":"2458","article-title":"Visual speech synthesis using dynamic visemes, contextual features and DNNs","author":"ausdang","year":"2016","journal-title":"Proc INTERSPEECH"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073658"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073640"},{"key":"ref18","first-page":"1876","article-title":"VisemeNet: Audio-driven animator-centric speech animation","volume":"37","author":"zhou","year":"2017","journal-title":"ACM TOG"},{"article-title":"ObamaNet: Photo-realistic lip-sync from text","year":"2017","author":"kumar","key":"ref19"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/1274940.1274947"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2011.6011835"},{"key":"ref6","first-page":"245","article-title":"Dynamic units of visual speech","author":"taylor","year":"2012","journal-title":"Proc ACM SIGGRAPH\/Eurographics Symp Comput Animation"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/2522628.2522904"},{"key":"ref8","first-page":"261","article-title":"Lip motion synthesis using a context dependent trajectory HMM","author":"hofer","year":"2007","journal-title":"Proc Symp Comput Animation"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2012.02.003"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICDSP.2016.7868609"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2007.09.001"},{"key":"ref46","doi-asserted-by":"crossref","first-page":"237","DOI":"10.21437\/Interspeech.2011-91","article-title":"Improved bottleneck features using pretrained deep neural networks","author":"yu","year":"2011","journal-title":"Proc INTERSPEECH"},{"key":"ref45","doi-asserted-by":"crossref","first-page":"920","DOI":"10.1109\/TCSVT.2016.2643504","article-title":"Real-time 3D facial animation: From appearance to internal articulators","volume":"28","author":"yu","year":"2018","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2903724"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbiomech.2012.08.031"},{"key":"ref41","first-page":"107","article-title":"A 3D parametric tongue model for animated speech","volume":"12","author":"scott","year":"2001","journal-title":"Comput Animation Virtual Worlds"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/VCIP.2017.8305052"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1002\/cnm.1423"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8817316\/08805138.pdf?arnumber=8805138","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,26]],"date-time":"2022-09-26T04:21:57Z","timestamp":1664166117000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8805138\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12]]},"references-count":73,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2019.2935843","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"type":"print","value":"2329-9290"},{"type":"electronic","value":"2329-9304"}],"subject":[],"published":{"date-parts":[[2019,12]]}}}