{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T04:26:24Z","timestamp":1772771184116,"version":"3.50.1"},"reference-count":71,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,10,15]],"date-time":"2023-10-15T00:00:00Z","timestamp":1697328000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,10,15]],"date-time":"2023-10-15T00:00:00Z","timestamp":1697328000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Xiangwai Economic College teaching word [2022]\uff0cHunan International Economics College School-level Educational Reform Project","award":["64"],"award-info":[{"award-number":["64"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Soft Comput"],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s00500-023-09292-5","type":"journal-article","created":{"date-parts":[[2023,10,15]],"date-time":"2023-10-15T09:01:37Z","timestamp":1697360497000},"page":"363-379","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["3D head-talk: speech synthesis 3D head movement face animation"],"prefix":"10.1007","volume":"28","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3876-5722","authenticated-orcid":false,"given":"Daowu","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruihui","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuyi","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xibei","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Zou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,10,15]]},"reference":[{"key":"9292_CR1","doi-asserted-by":"crossref","unstructured":"Aksan E, Kaufmann M, Cao P et al (2021) A spatio-temporal transformer for 3d human motion prediction. 2021 International conference on 3D vision (3DV). IEEE, pp 565\u2013574","DOI":"10.1109\/3DV53792.2021.00066"},{"issue":"3","key":"9292_CR2","doi-asserted-by":"publisher","first-page":"393","DOI":"10.1016\/S0092-8674(00)81878-4","volume":"88","author":"S Arber","year":"1997","unstructured":"Arber S, Hunter JJ, Ross J Jr et al (1997) MLP-deficient mice exhibit a disruption of cardiac cytoarchitectural organization, dilated cardiomyopathy, and heart failure. Cell 88(3):393\u2013403","journal-title":"Cell"},{"key":"9292_CR3","unstructured":"Baevski A, Hsu W N, Xu Q et al (2022) Data2vec: a general framework for self-supervised learning in speech, vision and language. International conference on machine learning. PMLR, pp 1298\u20131312"},{"issue":"1","key":"9292_CR4","doi-asserted-by":"publisher","first-page":"5494","DOI":"10.1038\/s41598-022-09293-8","volume":"12","author":"H Basak","year":"2022","unstructured":"Basak H, Kundu R, Singh PK et al (2022) A union of deep learning and swarm-based optimization for 3D human action recognition. Sci Rep 12(1):5494","journal-title":"Sci Rep"},{"key":"9292_CR5","doi-asserted-by":"crossref","unstructured":"Bhattacharya U, Rewkowski N, Banerjee A, et al (2021) Text2gestures: a transformer-based network for generating emotive body gestures for virtual agents. 2021 IEEE virtual reality and 3D user interfaces (VR). IEEE, pp 1\u201310","DOI":"10.1109\/VR50410.2021.00037"},{"issue":"3","key":"9292_CR6","doi-asserted-by":"publisher","first-page":"1075","DOI":"10.1109\/TASL.2006.885910","volume":"15","author":"C Busso","year":"2007","unstructured":"Busso C, Deng Z, Grimm M et al (2007) Rigid head motion in expressive speech animation: analysis and synthesis. IEEE Trans Audio Speech Lang Process 15(3):1075\u20131086","journal-title":"IEEE Trans Audio Speech Lang Process"},{"issue":"4","key":"9292_CR7","doi-asserted-by":"publisher","first-page":"1283","DOI":"10.1145\/1095878.1095881","volume":"24","author":"Y Cao","year":"2005","unstructured":"Cao Y, Tien WC, Faloutsos P et al (2005) Expressive speech-driven facial animation. ACM Trans Graph (TOG) 24(4):1283\u20131302","journal-title":"ACM Trans Graph (TOG)"},{"issue":"3","key":"9292_CR8","first-page":"413","volume":"20","author":"C Cao","year":"2013","unstructured":"Cao C, Weng Y, Zhou S et al (2013) Facewarehouse: a 3d facial expression database for visual computing. IEEE Trans vis Comput Graph 20(3):413\u2013425","journal-title":"IEEE Trans vis Comput Graph"},{"issue":"3","key":"9292_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11704-020-0133-7","volume":"16","author":"Y Chai","year":"2022","unstructured":"Chai Y, Weng Y, Wang L et al (2022) Speech-driven facial animation with spectral gathering and temporal attention. Front Comput Sci 16(3):1\u201310","journal-title":"Front Comput Sci"},{"key":"9292_CR10","doi-asserted-by":"crossref","unstructured":"Chang Y, Vieira M, Turk M et al (2005) Automatic 3D facial expression analysis in videos. Analysis and modelling of faces and gestures: second international workshop, AMFG 2005, Beijing, China, October 16, 2005. Proceedings 2. Springer, Berlin, pp 293\u2013307","DOI":"10.1007\/11564386_23"},{"key":"9292_CR11","doi-asserted-by":"crossref","unstructured":"Chen L, Maddox RK, Duan Z et al (2019) Hierarchical cross-modal talking face generation with dynamic pixel-wise loss. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. Long Beach, CA, pp 7832\u20137841","DOI":"10.1109\/CVPR.2019.00802"},{"key":"9292_CR12","unstructured":"Chen M, Radford A, Child R et al (2020) Generative pretraining from pixels. International conference on machine learning. PMLR, pp 1691\u20131703"},{"key":"9292_CR13","doi-asserted-by":"crossref","unstructured":"Cheng S, Kotsia I, Pantic M et al (2018) 4dfab: a large scale 4d database for facial expression analysis and biometric applications. Proceedings of the IEEE conference on computer vision and pattern recognition. Salt Lake City, Utah, pp 5117\u20135126","DOI":"10.1109\/CVPR.2018.00537"},{"key":"9292_CR14","unstructured":"Chowdhery A, Narang S, Devlin J et al (2022) Palm: scaling language modeling with pathways. arXiv preprint arXiv:2204.02311"},{"key":"9292_CR15","unstructured":"Chung JS, Jamaludin A, Zisserman A (2017) You said that? arXiv preprint arXiv:1705.02966"},{"key":"9292_CR16","doi-asserted-by":"crossref","unstructured":"Cosker D, Krumhuber E, Hilton A (2011) A FACS valid 3D dynamic action unit database with applications to 3D dynamic morphable facial modeling. 2011 international conference on computer vision. IEEE, pp 2296\u20132303","DOI":"10.1109\/ICCV.2011.6126510"},{"key":"9292_CR17","doi-asserted-by":"crossref","unstructured":"Cudeiro D, Bolkart T, Laidlaw C et al (2019) Capture, learning, and synthesis of 3D speaking styles. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. Long Beach, CA, pp 10101\u201310111","DOI":"10.1109\/CVPR.2019.01034"},{"key":"9292_CR18","doi-asserted-by":"crossref","unstructured":"Dai Z, Yang Z, Yang Y et al (2019) Transformer-xl: attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860","DOI":"10.18653\/v1\/P19-1285"},{"key":"9292_CR19","unstructured":"Dehghani M, Gouws S, Vinyals O et al (2018) Universal transformers. arXiv preprint arXiv:1807.03819"},{"issue":"6","key":"9292_CR20","doi-asserted-by":"publisher","first-page":"1523","DOI":"10.1109\/TVCG.2006.90","volume":"12","author":"Z Deng","year":"2006","unstructured":"Deng Z, Neumann U, Lewis JP et al (2006) Expressive facial animation synthesis by learning speech coarticulation and expression spaces. IEEE Trans vis Comput Graph 12(6):1523\u20131534","journal-title":"IEEE Trans vis Comput Graph"},{"key":"9292_CR21","unstructured":"Devlin J, Chang M W, Lee K et al (2018) Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"9292_CR22","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A et al (2020) An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929"},{"key":"9292_CR23","doi-asserted-by":"crossref","unstructured":"Fan Y, Lin Z, Saito J et al (2022) FaceFormer: speech-driven 3D facial animation with transformers. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. New Orleans, Louisiana, pp 18770\u201318780","DOI":"10.1109\/CVPR52688.2022.01821"},{"issue":"6","key":"9292_CR24","doi-asserted-by":"publisher","first-page":"591","DOI":"10.1109\/TMM.2010.2052239","volume":"12","author":"G Fanelli","year":"2010","unstructured":"Fanelli G, Gall J, Romsdorfer H et al (2010) A 3-d audio-visual corpus of affective communication. IEEE Trans Multimed 12(6):591\u2013598","journal-title":"IEEE Trans Multimed"},{"key":"9292_CR25","doi-asserted-by":"crossref","unstructured":"Habibie I, Xu W, Mehta D et al (2021) Learning speech-driven 3d conversational gestures from video. Proceedings of the 21st ACM international conference on intelligent virtual agents. Virtual Event Japan, pp 101\u2013108","DOI":"10.1145\/3472306.3478335"},{"key":"9292_CR26","unstructured":"Hendrycks D, Gimpel K (2016) Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415"},{"issue":"21","key":"9292_CR27","doi-asserted-by":"publisher","first-page":"16453","DOI":"10.1007\/s00500-020-04954-0","volume":"24","author":"P Hewage","year":"2020","unstructured":"Hewage P, Behera A, Trovati M et al (2020) Temporal convolutional neural (TCN) network for an effective weather forecasting using time-series data from the local weather station. Soft Comput 24(21):16453\u201316482","journal-title":"Soft Comput"},{"issue":"23","key":"9292_CR28","doi-asserted-by":"publisher","first-page":"4941","DOI":"10.3390\/rs13234941","volume":"13","author":"R Hussain","year":"2021","unstructured":"Hussain R, Karbhari Y, Ijaz MF et al (2021) Revise-net: exploiting reverse attention mechanism for salient object detection. Remote Sens 13(23):4941","journal-title":"Remote Sens"},{"key":"9292_CR29","doi-asserted-by":"crossref","unstructured":"Jonell P, Kucherenko T, Henter GE et al (2020) Let's face it: probabilistic multi-modal interlocutor-aware generation of facial gestures in dyadic settings. Proceedings of the 20th ACM international conference on intelligent virtual agents. New York, pp 1\u20138","DOI":"10.1145\/3383652.3423911"},{"issue":"4","key":"9292_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073658","volume":"36","author":"T Karras","year":"2017","unstructured":"Karras T, Aila T, Laine S et al (2017) Audio-driven facial animation by joint end-to-end learning of pose and emotion. ACM Trans Graph (TOG) 36(4):1\u201312","journal-title":"ACM Trans Graph (TOG)"},{"issue":"10s","key":"9292_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3505244","volume":"54","author":"S Khan","year":"2022","unstructured":"Khan S, Naseer M, Hayat M et al (2022) Transformers in vision: a survey. ACM Comput Surv (CSUR) 54(10s):1\u201341","journal-title":"ACM Comput Surv (CSUR)"},{"key":"9292_CR32","unstructured":"Kingma DP, Ba J (2014) Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"issue":"6","key":"9292_CR33","doi-asserted-by":"publisher","first-page":"194:1","DOI":"10.1145\/3130800.3130813","volume":"36","author":"T Li","year":"2017","unstructured":"Li T, Bolkart T, Black MJ et al (2017) Learning a model of facial shape and expression from 4D scans. ACM Trans Graph 36(6):194:1-194:17","journal-title":"ACM Trans Graph"},{"key":"9292_CR34","unstructured":"Li J, Yin Y, Chu H et al (2020) Learning to generate diverse dance motions with transformer. arXiv preprint arXiv:2008.08171"},{"key":"9292_CR35","unstructured":"Li R, Yang S, Ross DA et al (2021) Learn to dance with aist++: music conditioned 3d dance generation. arXiv preprint arXiv:2101.08779, 2(3)"},{"issue":"6","key":"9292_CR36","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2816795.2818130","volume":"34","author":"Y Liu","year":"2015","unstructured":"Liu Y, Xu F, Chai J et al (2015) Video-audio driven real-time facial animation. ACM Trans Graph (TOG) 34(6):1\u201310","journal-title":"ACM Trans Graph (TOG)"},{"key":"9292_CR37","doi-asserted-by":"publisher","first-page":"4873","DOI":"10.1109\/TVCG.2021.3107669","volume":"28","author":"J Liu","year":"2021","unstructured":"Liu J, Hui B, Li K et al (2021) Geometry-guided dense perspective network for speech-driven facial animation. IEEE Trans vis Comput Graph 28:4873\u20134886","journal-title":"IEEE Trans vis Comput Graph"},{"key":"9292_CR38","doi-asserted-by":"crossref","unstructured":"Meyer GP (2021) An alternative probabilistic interpretation of the huber loss. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition. virtually, pp 5261\u20135269","DOI":"10.1109\/CVPR46437.2021.00522"},{"key":"9292_CR39","doi-asserted-by":"crossref","unstructured":"Mittal G, Wang B (2020) Animating face using disentangled audio representations. Proceedings of the IEEE\/CVF winter conference on applications of computer vision. Snowmass village, Colorado, pp 3290\u20133298.","DOI":"10.1109\/WACV45572.2020.9093527"},{"key":"9292_CR40","doi-asserted-by":"crossref","unstructured":"Panayotov V, Chen G, Povey D et al (2015) Librispeech: an asr corpus based on public domain audio books. 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5206\u20135210","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"9292_CR41","doi-asserted-by":"crossref","unstructured":"Petrovich M, Black MJ, Varol G (2021) Action-conditioned 3D human motion synthesis with transformer VAE. Proceedings of the IEEE\/CVF international conference on computer vision. virtually, pp 10985\u201310995","DOI":"10.1109\/ICCV48922.2021.01080"},{"key":"9292_CR42","doi-asserted-by":"crossref","unstructured":"Pham H X, Cheung S, Pavlovic V (2017) Speech-driven 3D facial animation with implicit emotional awareness: a deep learning approach. Proceedings of the IEEE conference on computer vision and pattern recognition workshops. Hawaii Convention Center, pp 80\u201388","DOI":"10.1109\/CVPRW.2017.287"},{"key":"9292_CR43","unstructured":"Press O, Smith NA, Lewis M (2021) Train short, test long: attention with linear biases enables input length extrapolation. arXiv preprint arXiv:2108.12409"},{"key":"9292_CR44","unstructured":"Radford A, Narasimhan K, Salimans T et al (2018) Improving language understanding by generative pre-training. 1\u201312."},{"key":"9292_CR45","doi-asserted-by":"crossref","unstructured":"Richard A, Zollh\u00f6fer M, Wen Y et al (2021) Meshtalk: 3d face animation from speech using cross-modality disentanglement. Proceedings of the IEEE\/CVF international conference on computer vision. virtually, pp 1173\u20131182.","DOI":"10.1109\/ICCV48922.2021.00121"},{"key":"9292_CR46","doi-asserted-by":"crossref","unstructured":"Sadoughi N, Busso C (2016) Head motion generation with synthetic speech: a data driven approach. INTERSPEECH. San Francisco, USA, pp 52\u201356.","DOI":"10.21437\/Interspeech.2016-419"},{"key":"9292_CR47","doi-asserted-by":"crossref","unstructured":"Sadoughi N, Busso C (2018) Novel realizations of speech-driven head movements with generative adversarial networks. 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6169\u20136173","DOI":"10.1109\/ICASSP.2018.8461967"},{"key":"9292_CR48","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1016\/j.specom.2017.07.004","volume":"95","author":"N Sadoughi","year":"2017","unstructured":"Sadoughi N, Liu Y, Busso C (2017) Meaningful head movements driven by emotional synthetic speech. Speech Commun 95:87\u201399","journal-title":"Speech Commun"},{"key":"9292_CR49","doi-asserted-by":"publisher","first-page":"166518","DOI":"10.1109\/ACCESS.2021.3135658","volume":"9","author":"KK Sahoo","year":"2021","unstructured":"Sahoo KK, Dutta I, Ijaz MF et al (2021) TLEFuzzyNet: fuzzy rank-based ensemble of transfer learning models for emotion recognition from human speeches. IEEE Access 9:166518\u2013166530","journal-title":"IEEE Access"},{"key":"9292_CR50","first-page":"47","volume-title":"Bosphorus database for 3D face analysis\/\/Biometrics and Identity Management: First European Workshop, BIOID 2008, Roskilde, Denmark, May 7-9, 2008. Revised Selected Papers 1","author":"A Savran","year":"2008","unstructured":"Savran A, Aly\u00fcz N, Dibeklio\u011flu H et al (2008) Bosphorus database for 3D face analysis\/\/Biometrics and Identity Management: First European Workshop, BIOID 2008, Roskilde, Denmark, May 7-9, 2008. Revised Selected Papers 1. Springer, Berlin, pp 47\u201356"},{"key":"9292_CR51","unstructured":"Su J, Lu Y, Pan S et al (2021) Roformer: enhanced transformer with rotary position embedding. arXiv preprint arXiv:2104.09864"},{"issue":"4","key":"9292_CR52","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073640","volume":"36","author":"S Suwajanakorn","year":"2017","unstructured":"Suwajanakorn S, Seitz SM, Kemelmacher-Shlizerman I (2017) Synthesizing obama: learning lip sync from audio. ACM Trans Graph (ToG) 36(4):1\u201313","journal-title":"ACM Trans Graph (ToG)"},{"key":"9292_CR53","doi-asserted-by":"publisher","first-page":"1482","DOI":"10.21437\/Interspeech.2016-483","volume-title":"Proceedings of the interspeech conference 2016","author":"S Taylor","year":"2016","unstructured":"Taylor S, Kato A, Milner B, Matthews I (2016) Audio-to-visual speech conversion using deep neural networks. In: Proceedings of the interspeech conference 2016. International Speech Communication Association, USA, pp 1482\u20131486"},{"issue":"4","key":"9292_CR54","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073699","volume":"36","author":"S Taylor","year":"2017","unstructured":"Taylor S, Kim T, Yue Y et al (2017) A deep learning approach for generalized speech animation. ACM Trans Graph (TOG) 36(4):1\u201311","journal-title":"ACM Trans Graph (TOG)"},{"key":"9292_CR55","unstructured":"Touvron H, Cord M, Douze M et al (2021) Training data-efficient image transformers and distillation through attention. International conference on machine learning. PMLR, pp 10347\u201310357"},{"issue":"6","key":"9292_CR56","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3478513.3480570","volume":"40","author":"G Valle-P\u00e9rez","year":"2021","unstructured":"Valle-P\u00e9rez G, Henter GE, Beskow J et al (2021) Transflower: probabilistic autoregressive dance generation with multimodal attention. ACM Trans Graph (TOG) 40(6):1\u201314","journal-title":"ACM Trans Graph (TOG)"},{"key":"9292_CR57","first-page":"5998","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani A, Shazeer N, Parmar N et al (2017) Attention is all you need[J]. Adv Neural Inf Process Syst (NIPS) 30:5998\u20136008","journal-title":"Adv Neural Inf Process Syst (NIPS)"},{"key":"9292_CR58","doi-asserted-by":"crossref","unstructured":"Vlasic D et al (2006) Face transfer with multilinear models. ACM SIGGRAPH 2006 Courses. 24-es","DOI":"10.1145\/1185657.1185864"},{"key":"9292_CR59","unstructured":"Wang B, Komatsuzaki A (2021) GPT-J-6B: A 6 billion parameter autoregressive language model. https:\/\/github.com\/kingoflolz\/mesh-transformer-jax"},{"key":"9292_CR60","unstructured":"Wang Q, Fan Z, Xia S (2021a) 3D-TalkEmo: learning to synthesize 3D emotional talking head. arXiv preprint arXiv:2104.12051"},{"key":"9292_CR61","doi-asserted-by":"crossref","unstructured":"Wang S, Li L, Ding Y et al (2021b) Audio2head: audio-driven one-shot talking-head generation with natural head motion. arXiv preprint arXiv:2107.09293","DOI":"10.24963\/ijcai.2021\/152"},{"key":"9292_CR62","doi-asserted-by":"crossref","unstructured":"Wiles O, Koepke A, Zisserman A (2018) X2face: a network for controlling face generation using images, audio, and pose codes. Proceedings of the European conference on computer vision (ECCV). Munich, Germany, pp 670\u2013686.","DOI":"10.1007\/978-3-030-01261-8_41"},{"issue":"1","key":"9292_CR63","doi-asserted-by":"publisher","first-page":"79","DOI":"10.3354\/cr030079","volume":"30","author":"CJ Willmott","year":"2005","unstructured":"Willmott CJ, Matsuura K (2005) Advantages of the mean absolute error (MAE) over the root mean square error (RMSE) in assessing average model performance. Clim Res 30(1):79\u201382","journal-title":"Clim Res"},{"key":"9292_CR64","unstructured":"Yin L, Wei X, Sun Y et al (2006) A 3D facial expression database for facial behavior research. 7th international conference on automatic face and gesture recognition (FGR06). IEEE, pp 211\u2013216"},{"key":"9292_CR65","doi-asserted-by":"crossref","unstructured":"Zeng D, Liu H, Lin H, et al (2020) Talking face generation with expression-tailored generative adversarial network. Proceedings of the 28th ACM international conference on multimedia. Seattle WA USA, pp 1716\u20131724","DOI":"10.1145\/3394171.3413844"},{"key":"9292_CR66","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1016\/j.neucom.2012.01.019","volume":"89","author":"Y Zhag","year":"2012","unstructured":"Zhag Y, Wei W (2012) A realistic dynamic facial expression transfer method. Neurocomputing 89:21\u201329","journal-title":"Neurocomputing"},{"key":"9292_CR67","doi-asserted-by":"crossref","unstructured":"Zhang X, Yin L, Cohn JF et al (2013) A high-resolution spontaneous 3d dynamic facial expression database. 2013 10th IEEE international conference and workshops on automatic face and gesture recognition (FG). IEEE, pp 1\u20136","DOI":"10.1109\/FG.2013.6553788"},{"key":"9292_CR68","doi-asserted-by":"crossref","unstructured":"Zhang Z, Girard J M, Wu Y, et al (2016) Multimodal spontaneous emotion corpus for human behavior analysis. Proceedings of the IEEE conference on computer vision and pattern recognition. Las Vegas, Nevada, pp 3438\u20133446","DOI":"10.1109\/CVPR.2016.374"},{"issue":"2","key":"9292_CR69","doi-asserted-by":"publisher","first-page":"1438","DOI":"10.1109\/TVCG.2021.3117484","volume":"29","author":"C Zhang","year":"2021","unstructured":"Zhang C, Ni S, Fan Z et al (2021) 3d talking face with personalized pose dynamics. IEEE Trans Vis Comput Graph 29(2):1438\u20131449","journal-title":"IEEE Trans Vis Comput Graph"},{"issue":"4","key":"9292_CR70","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3197517.3201292","volume":"37","author":"Y Zhou","year":"2018","unstructured":"Zhou Y, Xu Z, Landreth C et al (2018) Visemenet: audio-driven animator-centric speech animation. ACM Trans Graph (TOG) 37(4):1\u201310","journal-title":"ACM Trans Graph (TOG)"},{"issue":"01","key":"9292_CR71","first-page":"9299","volume":"33","author":"H Zhou","year":"2019","unstructured":"Zhou H, Liu Y, Liu Z et al (2019) Talking face generation by adversarially disentangled audio-visual representation. Proc AAAI Conf Artif Intell 33(01):9299\u20139306","journal-title":"Proc AAAI Conf Artif Intell"}],"container-title":["Soft Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00500-023-09292-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00500-023-09292-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00500-023-09292-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,4]],"date-time":"2024-01-04T15:09:39Z","timestamp":1704380979000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00500-023-09292-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,15]]},"references-count":71,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["9292"],"URL":"https:\/\/doi.org\/10.1007\/s00500-023-09292-5","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-1865787\/v1","asserted-by":"object"}]},"ISSN":["1432-7643","1433-7479"],"issn-type":[{"value":"1432-7643","type":"print"},{"value":"1433-7479","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10,15]]},"assertion":[{"value":"19 September 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 October 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interests regarding the publication of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Our proposed method can synthesize facial animation for anyone based on the input 3D facial data and speech. This can be widely used in several scenarios, such as virtual reality and human\u2013computer interaction. This makes this forward-looking technology potentially open to misuse. We, therefore, want to improve users\u2019 insight into the potential misuse risks and encourage the public to want to report any suspicious videos to the relevant authorities.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical considerations"}},{"value":"This article does not contain any studies with human participants or animals performed by any of the authors.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed consent"}}]}}