{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T09:15:18Z","timestamp":1742980518815,"version":"3.40.3"},"publisher-location":"Cham","reference-count":27,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031500718"},{"type":"electronic","value":"9783031500725"}],"license":[{"start":{"date-parts":[[2023,12,29]],"date-time":"2023-12-29T00:00:00Z","timestamp":1703808000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,12,29]],"date-time":"2023-12-29T00:00:00Z","timestamp":1703808000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-50072-5_2","type":"book-chapter","created":{"date-parts":[[2023,12,28]],"date-time":"2023-12-28T08:02:17Z","timestamp":1703750537000},"page":"15-26","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Audio-Driven Lips and\u00a0Expression on\u00a03D Human Face"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4952-9207","authenticated-orcid":false,"given":"Le","family":"Ma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6684-9480","authenticated-orcid":false,"given":"Zhihao","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3221-4981","authenticated-orcid":false,"given":"Weiliang","family":"Meng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4037-9900","authenticated-orcid":false,"given":"Shibiao","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0092-6474","authenticated-orcid":false,"given":"Xiaopeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,12,29]]},"reference":[{"key":"2_CR1","unstructured":"Amodei, D., Ananthanarayanan, S., et al.: Deep speech 2: end-to-end speech recognition in English and mandarin. In: Proceedings of the 33rd International Conference on International Conference on Machine Learning, pp. 173\u2013182 (2016)"},{"key":"2_CR2","doi-asserted-by":"crossref","unstructured":"Blanz, V., Vetter, T.: A morphable model for the synthesis of 3D faces. In: Proceedings of the 26th Annual Conference on Computer Graphics and Interactive Techniques, pp. 187\u2013194 (1999)","DOI":"10.1145\/311535.311556"},{"issue":"3","key":"2_CR3","first-page":"413","volume":"20","author":"C Cao","year":"2014","unstructured":"Cao, C., Weng, Y., et al.: FaceWarehouse: a 3D facial expression database for visual computing. TVCG 20(3), 413\u2013425 (2014)","journal-title":"TVCG"},{"key":"2_CR4","doi-asserted-by":"crossref","unstructured":"Cheng, S., Kotsia, I., et al.: 4DFAB: a large scale 4D database for facial expression analysis and biometric applications. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00537"},{"key":"2_CR5","doi-asserted-by":"crossref","unstructured":"Cosker, D., Krumhuber, E., Hilton, A.: A FACS valid 3D dynamic action unit database with applications to 3D dynamic morphable facial modeling. In: ICCV, pp. 2296\u20132303 (2011)","DOI":"10.1109\/ICCV.2011.6126510"},{"key":"2_CR6","unstructured":"Cosker, D.P., Marshall, A.D., et al.: Video realistic talking heads using hierarchical non-linear speech-appearance models. In: In MIRAGE (2003)"},{"key":"2_CR7","doi-asserted-by":"crossref","unstructured":"Cudeiro, D., Bolkart, T., et al.: Capture, learning, and synthesis of 3D speaking styles. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01034"},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"Ezzat, T., Geiger, G., Poggio, T.: Trainable videorealistic speech animation. In: Proceedings of the 29th Annual Conference on Computer Graphics and Interactive Techniques, pp. 388\u2013398 (2002)","DOI":"10.1145\/566570.566594"},{"issue":"6","key":"2_CR9","doi-asserted-by":"publisher","first-page":"591","DOI":"10.1109\/TMM.2010.2052239","volume":"12","author":"G Fanelli","year":"2010","unstructured":"Fanelli, G., Gall, J., et al.: A 3-D audio-visual corpus of affective communication. IEEE Trans. Multimed. 12(6), 591\u2013598 (2010)","journal-title":"IEEE Trans. Multimed."},{"issue":"4","key":"2_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3450626.3459936","volume":"40","author":"Y Feng","year":"2021","unstructured":"Feng, Y., Feng, H., Black, M.J., Bolkart, T.: Learning an animatable detailed 3D face model from in-the-wild images. ACM Trans. Graph. 40(4), 1\u201313 (2021)","journal-title":"ACM Trans. Graph."},{"key":"2_CR11","doi-asserted-by":"crossref","unstructured":"H. Li, J. Yu, Y.Y., Bregler, C.: Realtime facial animation with on-the-fly correctives. ACM Trans. Graph. 32(4), 1\u201310 (2013). Article No. 42","DOI":"10.1145\/2461912.2462019"},{"key":"2_CR12","doi-asserted-by":"crossref","unstructured":"Hussen Abdelaziz, A., Theobald, B.J., et al.: Modality dropout for improved performance-driven talking faces. In: Proceedings of the 2020 International Conference on Multimodal Interaction, pp. 378\u2013386 (2020)","DOI":"10.1145\/3382507.3418840"},{"key":"2_CR13","doi-asserted-by":"publisher","first-page":"1432","DOI":"10.1007\/s00371-022-02460-y","volume":"39","author":"J Xu","year":"2023","unstructured":"Xu, J., Liu, W., Xing, W., Wei, X.: MSPENet: multi-scale adaptive fusion and position enhancement network for human pose estimation. Vis. Comput. 39, 1432\u20132315 (2023). https:\/\/doi.org\/10.1007\/s00371-022-02460-y","journal-title":"Vis. Comput."},{"key":"2_CR14","doi-asserted-by":"publisher","first-page":"1330","DOI":"10.1109\/TMM.2020.2999181","volume":"23","author":"A Kamel","year":"2021","unstructured":"Kamel, A., Sheng, B., Li, P., Kim, J., Feng, D.D.: Hybrid refinement-correction heatmaps for human pose estimation. IEEE Trans. Multimed. 23, 1330\u20131342 (2021). https:\/\/doi.org\/10.1109\/TMM.2020.2999181","journal-title":"IEEE Trans. Multimed."},{"issue":"4","key":"2_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073658","volume":"36","author":"T Karras","year":"2017","unstructured":"Karras, T., Aila, T., et al.: Audio-driven facial animation by joint end-to-end learning of pose and emotion. ACM Trans. Graph. 36(4), 1\u201312 (2017)","journal-title":"ACM Trans. Graph."},{"key":"2_CR16","doi-asserted-by":"publisher","first-page":"1283","DOI":"10.1145\/1095878.1095881","volume":"24","author":"SP Konstantinos Vougioukas","year":"2005","unstructured":"Konstantinos Vougioukas, S.P., Pantic, M.: Expressive speech-driven facial animation. ACM Trans. Graph. 24, 1283\u20131302 (2005)","journal-title":"ACM Trans. Graph."},{"key":"2_CR17","doi-asserted-by":"crossref","unstructured":"Li, T., Bolkart, T., et al.: Learning a model of facial shape and expression from 4D scans. ACM Trans. Graph. (Proc. SIGGRAPH Asia) 36(6), 194:1\u2013194:17 (2017)","DOI":"10.1145\/3130800.3130813"},{"issue":"5","key":"2_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1371\/journal.pone.0196391","volume":"13","author":"SR Livingstone","year":"2018","unstructured":"Livingstone, S.R., Russo, F.A.: The Ryerson audio-visual database of emotional speech and song (RAVDESS): a dynamic, multimodal set of facial and vocal expressions in North American English. PLoS ONE 13(5), 1\u201335 (2018)","journal-title":"PLoS ONE"},{"key":"2_CR19","doi-asserted-by":"crossref","unstructured":"Loper, M., Mahmood, N., et al.: SMPL: a skinned multi-person linear model. ACM Trans. Graph. (Proc. SIGGRAPH Asia) 34(6), 248:1\u2013248:16 (2015)","DOI":"10.1145\/2816795.2818013"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Paysan, P., Knothe, R., Amberg, B., Romdhani, S., Vetter, T.: A 3D face model for pose and illumination invariant face recognition (2009)","DOI":"10.1109\/AVSS.2009.58"},{"key":"2_CR21","doi-asserted-by":"crossref","unstructured":"Pham, H.X., Wang, Y., Pavlovic, V.: End-to-end learning for 3D facial animation from speech. In: International Conference on Multimodal Interaction (2018)","DOI":"10.1145\/3242969.3243017"},{"key":"2_CR22","doi-asserted-by":"crossref","unstructured":"Pham, H.X., Cheung, S., Pavlovic, V.: Speech-driven 3D facial animation with implicit emotional awareness: a deep learning approach. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition Workshops (CVPRW), pp. 2328\u20132336 (2017)","DOI":"10.1109\/CVPRW.2017.287"},{"key":"2_CR23","doi-asserted-by":"crossref","unstructured":"Richard, A., Zollh\u00f6fer, M., Wen, Y., de la Torre, F., Sheikh, Y.: MeshTalk: 3D face animation from speech using cross-modality disentanglement. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 1173\u20131182 (2021)","DOI":"10.1109\/ICCV48922.2021.00121"},{"key":"2_CR24","doi-asserted-by":"crossref","unstructured":"Richardson, E., Sela, M., Or-El, R., Kimmel, R.: Learning detailed face reconstruction from a single image. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.589"},{"key":"2_CR25","doi-asserted-by":"publisher","unstructured":"Yang, S., et al.: EnNeRFACE: improving the generalization of face reenactment with adaptive ensemble neural radiance fields. Vis. Comput. 1432\u20132315 (2022). https:\/\/doi.org\/10.1007\/s00371-022-02709-6","DOI":"10.1007\/s00371-022-02709-6"},{"key":"2_CR26","doi-asserted-by":"publisher","first-page":"1432","DOI":"10.1007\/s00371-022-02409-1","volume":"39","author":"GP Simone Cammarasana","year":"2023","unstructured":"Simone Cammarasana, G.P.: Spatio-temporal analysis and comparison of 3D videos. Vis. Comput. 39, 1432\u20132315 (2023). https:\/\/doi.org\/10.1007\/s00371-022-02409-1","journal-title":"Vis. Comput."},{"key":"2_CR27","doi-asserted-by":"crossref","unstructured":"Zhang, C., Zhao, Y., et al.: FACIAL: synthesizing dynamic talking face with implicit attribute learning. In: ICCV, pp. 3867\u20133876 (2021)","DOI":"10.1109\/ICCV48922.2021.00384"}],"container-title":["Lecture Notes in Computer Science","Advances in Computer Graphics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-50072-5_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,28]],"date-time":"2023-12-28T08:02:57Z","timestamp":1703750577000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-50072-5_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,29]]},"ISBN":["9783031500718","9783031500725"],"references-count":27,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-50072-5_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2023,12,29]]},"assertion":[{"value":"29 December 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CGI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Computer Graphics International Conference","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Shanghai","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 September 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cgi2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"385","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"149","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"39% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}