{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,12]],"date-time":"2026-02-12T17:38:18Z","timestamp":1770917898042,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,10,15]],"date-time":"2019-10-15T00:00:00Z","timestamp":1571097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China","award":["U1736123 61572450 61303150"],"award-info":[{"award-number":["U1736123 61572450 61303150"]}]},{"name":"The Anhui Provincial Natural Science Foundation","award":["1708085QF138"],"award-info":[{"award-number":["1708085QF138"]}]},{"name":"The Fundamental Research Funds for the Central Universities","award":["WK2350000002"],"award-info":[{"award-number":["WK2350000002"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,10,15]]},"DOI":"10.1145\/3343031.3350865","type":"proceedings-article","created":{"date-parts":[[2019,10,21]],"date-time":"2019-10-21T16:32:26Z","timestamp":1571675546000},"page":"945-952","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["3D Singing Head for Music VR"],"prefix":"10.1145","author":[{"given":"Jun","family":"Yu","sequence":"first","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chang Wen","family":"Chen","sequence":"additional","affiliation":[{"name":"The State University of New York at Buffalo, Buffalo, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zengfu","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"R. Anderson B. Stenger V. Wan etal 2013. Expressive Visual Text-To-Speech Using Active Appearance Models. CVPR. 146--152.  R. Anderson B. Stenger V. Wan et al. 2013. Expressive Visual Text-To-Speech Using Active Appearance Models. CVPR. 146--152.","DOI":"10.1109\/CVPR.2013.434"},{"key":"e_1_3_2_1_2_1","unstructured":"P. Badin A. Serrurier etal 2006. Three-dimensional linear modeling of tongue: articulatory data and models. ISSP. 395--402.  P. Badin A. Serrurier et al. 2006. Three-dimensional linear modeling of tongue: articulatory data and models. ISSP. 395--402."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-017-1009-7"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"G. Fanelli T. Weise etal 2011. Real Time head pose estimation from consumer depth cameras. DAGM. 101--110.  G. Fanelli T. Weise et al. 2011. Real Time head pose estimation from consumer depth cameras. DAGM. 101--110.","DOI":"10.1007\/978-3-642-23123-0_11"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"P. Garrido etal 2016. Reconstruction of Personalized 3D Face Rigs from Monocular Video. TOG 35 3 (2016) 28:1--28:15.  P. Garrido et al. 2016. Reconstruction of Personalized 3D Face Rigs from Monocular Video. TOG 35 3 (2016) 28:1--28:15.","DOI":"10.1145\/2890493"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"P. Hsieh etal 2015. Unconstrained realtime facial performance capture. CVPR. 1675--1683.  P. Hsieh et al. 2015. Unconstrained realtime facial performance capture. CVPR. 1675--1683.","DOI":"10.1109\/CVPR.2015.7298776"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"T. Karras etal 2017. Audio-Driven Facial Animation by Joint End-to-End Learning of Pose and Emotion. TOG 36 4 (2017) 94:1--94:12.  T. Karras et al. 2017. Audio-Driven Facial Animation by Joint End-to-End Learning of Pose and Emotion. TOG 36 4 (2017) 94:1--94:12.","DOI":"10.1145\/3072959.3073658"},{"key":"#cr-split#-e_1_3_2_1_8_1.1","doi-asserted-by":"crossref","unstructured":"C. W. Luo J. Yu etal 2019. Real-Time Head Pose Estimation and Face Modeling from a Depth Image. TMM DOI: 10.1109\/TMM.2019.2903724. 10.1109\/TMM.2019.2903724","DOI":"10.1109\/TMM.2019.2903724"},{"key":"#cr-split#-e_1_3_2_1_8_1.2","doi-asserted-by":"crossref","unstructured":"C. W. Luo J. Yu et al. 2019. Real-Time Head Pose Estimation and Face Modeling from a Depth Image. TMM DOI: 10.1109\/TMM.2019.2903724.","DOI":"10.1109\/TMM.2019.2903724"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"S. Laine etal 2017. Production-level facial performance capture using deep convolutional neural networks. SCA. 1--9.  S. Laine et al. 2017. Production-level facial performance capture using deep convolutional neural networks. SCA. 1--9.","DOI":"10.1145\/3099564.3099581"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"R. Li J. Yu Z. F. Wang. 2014. A Mass-Spring Tongue Model with Efficient Collision Detection and Response During Speech. ISCSLP. 354--358.  R. Li J. Yu Z. F. Wang. 2014. A Mass-Spring Tongue Model with Efficient Collision Detection and Response During Speech. ISCSLP. 354--358.","DOI":"10.1109\/ISCSLP.2014.6936586"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"K. Liu and J. Ostermann. 2011. Realistic facial expression synthesis for an image based talking head. ICME. 1--6.  K. Liu and J. Ostermann. 2011. Realistic facial expression synthesis for an image based talking head. ICME. 1--6.","DOI":"10.1109\/ICME.2011.6011835"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1486525.1486527"},{"key":"e_1_3_2_1_13_1","first-page":"229","article-title":"An electromyographic study of the tongue during vowel production","volume":"7","author":"MacNeilage P. F.","year":"1964","journal-title":"JSHR"},{"key":"e_1_3_2_1_14_1","unstructured":"C. Matthew K. S. Bhat R. Fedkiw. 2016. Art-directed muscle simulation for high-end facial animation. SCA. 457--465.  C. Matthew K. S. Bhat R. Fedkiw. 2016. Art-directed muscle simulation for high-end facial animation. SCA. 457--465."},{"key":"e_1_3_2_1_15_1","volume-title":"Information Retrieval for Music and Motion","author":"Muller M."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"A. Owens A. A. Efros. 2018. Audio-Visual Scene Analysis with Self-Supervised Multisensory Features. ECCV. 631--648.  A. Owens A. A. Efros. 2018. Audio-Visual Scene Analysis with Self-Supervised Multisensory Features. ECCV. 631--648.","DOI":"10.1007\/978-3-030-01231-1_39"},{"key":"e_1_3_2_1_17_1","first-page":"107","article-title":"A 3D parametric tongue model for animated speech","volume":"12","author":"Scott A. K.","year":"2001","journal-title":"JVCA"},{"key":"e_1_3_2_1_18_1","unstructured":"E. Sifakis A. Selle A. Robinson-Mosher R. Fedkiw. 2006. Simulating speech with a physics-based facial muscle model. SCA. 261--270.  E. Sifakis A. Selle A. Robinson-Mosher R. Fedkiw. 2006. Simulating speech with a physics-based facial muscle model. SCA. 261--270."},{"key":"e_1_3_2_1_19_1","first-page":"2841","article-title":"Automatic prediction of tongue muscle activations using a finite element model","volume":"45","author":"Stavness I.","year":"2012","journal-title":"JB"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073640"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"S. Suwajanakorn etal 2014. Total Moving Face Reconstruction. ECCV. 796--812.  S. Suwajanakorn et al. 2014. Total Moving Face Reconstruction. ECCV. 796--812.","DOI":"10.1007\/978-3-319-10593-2_52"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1044\/1092-4388(2001\/009)"},{"key":"e_1_3_2_1_23_1","unstructured":"H. Tal etal 2015. Effective face frontalization in unconstrained images. CVPR. 710--713.  H. Tal et al. 2015. Effective face frontalization in unconstrained images. CVPR. 710--713."},{"key":"e_1_3_2_1_24_1","first-page":"865","article-title":"A 3D skeletal muscle model coupled with active contraction of muscle fibers and hyperelastic behavior","volume":"42","author":"Tang C. Y.","year":"2009","journal-title":"JB"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"S. Taylor etal 2017. A deep learning approach for generalized speech animation. TOG 36 4 (2017) 93:1--93:11.  S. Taylor et al. 2017. A deep learning approach for generalized speech animation. TOG 36 4 (2017) 93:1--93:11.","DOI":"10.1145\/3072959.3073699"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"J. Thies etal 2016. Face2Face: Real-time face capture and reenactment of RGB videos. CVPR. 12--20.  J. Thies et al. 2016. Face2Face: Real-time face capture and reenactment of RGB videos. CVPR. 12--20.","DOI":"10.1109\/CVPR.2016.262"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/2816795.2818056"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"B. Uria I. Murray S. Renals K. Richmond. 2012. Deep architectures for articulatory inversion. Interspeech. 867--870.  B. Uria I. Murray S. Renals K. Richmond. 2012. Deep architectures for articulatory inversion. Interspeech. 867--870.","DOI":"10.21437\/Interspeech.2012-263"},{"key":"e_1_3_2_1_29_1","unstructured":"A. Veit etal 2016. Residual networks behave like ensembles of relatively shallow networks. NIPS. 550--558.  A. Veit et al. 2016. Residual networks behave like ensembles of relatively shallow networks. NIPS. 550--558."},{"key":"e_1_3_2_1_30_1","first-page":"74","article-title":"Phoneme-level articulatory animation in pronunciation training","volume":"54","author":"Wang L.","year":"2012","journal-title":"SC"},{"key":"e_1_3_2_1_31_1","unstructured":"Z. F. Wang Z. G. Zheng. 2008. A Region Based Stereo Matching Algorithm Using Cooperative Optimization. CVPR. 701--708.  Z. F. Wang Z. G. Zheng. 2008. A Region Based Stereo Matching Algorithm Using Cooperative Optimization. CVPR. 701--708."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016-33"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Z. Wu etal 2015. Deep neural networks employing multi-task learning and stacked bottleneck features for speech synthesis. ICASSP. 4460--4464.  Z. Wu et al. 2015. Deep neural networks employing multi-task learning and stacked bottleneck features for speech synthesis. ICASSP. 4460--4464.","DOI":"10.1109\/ICASSP.2015.7178814"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"X. Xiong F. De la Torre. 2013. Supervised descent method and its applications to face alignment. CVPR. 532--539.  X. Xiong F. De la Torre. 2013. Supervised descent method and its applications to face alignment. CVPR. 532--539.","DOI":"10.1109\/CVPR.2013.75"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"D. Yu M. L. Seltzer. 2011. Improved bottleneck features using pretrained deep neural networks. ISCA. 457--462.  D. Yu M. L. Seltzer. 2011. Improved bottleneck features using pretrained deep neural networks. ISCA. 457--462.","DOI":"10.21437\/Interspeech.2011-91"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"J. Yu etal 2017. From talking head to singing head: a significant enhancement for more natural human computer interaction. ICME. 511--516.  J. Yu et al. 2017. From talking head to singing head: a significant enhancement for more natural human computer interaction. ICME. 511--516.","DOI":"10.1109\/ICME.2017.8019362"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-013-5023-2"},{"key":"e_1_3_2_1_38_1","first-page":"1621","article-title":"BLTRCNN Based 3D Articulatory Movement Prediction","volume":"21","author":"Yu L. Y.","year":"2018","journal-title":"TMM."},{"key":"e_1_3_2_1_39_1","unstructured":"Y. Zhou etal 2018. VisemeNet: Audio-Driven Animator-Centric Speech Animation. arXiv: 1805.09488.  Y. Zhou et al. 2018. VisemeNet: Audio-Driven Animator-Centric Speech Animation. arXiv: 1805.09488."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"W. Zhou Z. F. Wang. 2007. A Speech Rate Related Lip Movement Model for Speech Animation. Interspeech. 710--713.  W. Zhou Z. F. Wang. 2007. A Speech Rate Related Lip Movement Model for Speech Animation. Interspeech. 710--713.","DOI":"10.21437\/Interspeech.2007-296"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"P. C. Zhu etal 2015. Articulatory Movement Prediction Using Deep Bidirectional Long Short-Term Memory Based Recurrent Neural Networks and Word\/Phone Embeddings. Interspeech. 104--108.  P. C. Zhu et al. 2015. Articulatory Movement Prediction Using Deep Bidirectional Long Short-Term Memory Based Recurrent Neural Networks and Word\/Phone Embeddings. Interspeech. 104--108.","DOI":"10.21437\/Interspeech.2015-493"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"crossref","unstructured":"H. Zhao etal 2018. The Sound of Pixels. ECCV. 570--586.  H. Zhao et al. 2018. The Sound of Pixels. ECCV. 570--586.","DOI":"10.1007\/978-3-030-01246-5_35"}],"event":{"name":"MM '19: The 27th ACM International Conference on Multimedia","location":"Nice France","acronym":"MM '19","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 27th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3350865","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3343031.3350865","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:13:25Z","timestamp":1750202005000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3350865"}},"subtitle":["Learning External and Internal Articulatory Synchronicity from Lyric, Audio and Notes"],"short-title":[],"issued":{"date-parts":[[2019,10,15]]},"references-count":43,"alternative-id":["10.1145\/3343031.3350865","10.1145\/3343031"],"URL":"https:\/\/doi.org\/10.1145\/3343031.3350865","relation":{},"subject":[],"published":{"date-parts":[[2019,10,15]]},"assertion":[{"value":"2019-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}