{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T12:03:30Z","timestamp":1784894610813,"version":"3.55.0"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"New Generation Artificial Intelligence-National Science and Technology Major Project","award":["2025ZD0124001"],"award-info":[{"award-number":["2025ZD0124001"]}]},{"name":"Beijing Natural Science Foundation under Grant","award":["4252018"],"award-info":[{"award-number":["4252018"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62572032"],"award-info":[{"award-number":["62572032"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s00371-026-04623-7","type":"journal-article","created":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T19:46:30Z","timestamp":1783971990000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["EmoDiffuser: emotional diffuser for speech-driven 3D facial animation"],"prefix":"10.1007","volume":"42","author":[{"given":"Xin","family":"Zhao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9397-8539","authenticated-orcid":false,"given":"Ju","family":"Dai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feng","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haofei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aimin","family":"Hao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junjun","family":"Pan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,13]]},"reference":[{"issue":"4","key":"4623_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3592097","volume":"42","author":"T Ao","year":"2023","unstructured":"Ao, T., Zhang, Z., Liu, L.: Gesturediffuclip: gesture diffusion model with clip latents. ACM TOG 42(4), 1\u201318 (2023)","journal-title":"ACM TOG"},{"key":"4623_CR2","first-page":"12449","volume":"33","author":"A Baevski","year":"2020","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., Auli, M.: wav2vec 2.0: a framework for self-supervised learning of speech representations. Adv. Neural. Inf. Process. Syst. 33, 12449\u201312460 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4623_CR3","doi-asserted-by":"crossref","unstructured":"Blattmann, A., Rombach, R., Ling, H., Dockhorn, T., Kim, S.W., Fidler, S., Kreis, K.: Align your latents: high-resolution video synthesis with latent diffusion models, pp. 22563\u201322575. CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.02161"},{"key":"4623_CR4","doi-asserted-by":"crossref","unstructured":"Cao, X., Quan, P., Mao, Y., Cao, R., Su, L., Li, K.: Trrs-dm: two-stage resampling and residual shifting for high-fidelity texture inpainting of terracotta warriors utilizing diffusion models. Pattern Recognition, p. 111753 (2025)","DOI":"10.1016\/j.patcog.2025.111753"},{"key":"4623_CR5","unstructured":"Chen, H., Zhang, H., Zhang, S., Liu, X., Zhuang, S., Wan, P., ZHANG, D., Li, S.: Cafe-talk: generating 3d talking face animation with multimodal coarse-and fine-grained control. In: ICLR, vol. 2025, pp. 17,529\u201317,549 (2025)"},{"key":"4623_CR6","unstructured":"Cohen, M.M., Clark, R., Massaro, D.W.: Animated speech: research progress and applications. In: Auditory-Visual Speech Processing, p. 200 (2001)"},{"key":"4623_CR7","doi-asserted-by":"crossref","unstructured":"Cudeiro, D., Bolkart, T., Laidlaw, C., Ranjan, A., Black, M.J.: Capture, learning, and synthesis of 3d speaking styles, pp. 10101\u201310111. CVPR (2019)","DOI":"10.1109\/CVPR.2019.01034"},{"key":"4623_CR8","doi-asserted-by":"crossref","unstructured":"Dan\u011b\u010dek, R., Black, M.J., Bolkart, T.: Emoca: emotion driven monocular face capture and animation, pp. 20311\u201320322. CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01967"},{"key":"4623_CR9","doi-asserted-by":"crossref","unstructured":"Dan\u011b\u010dek, R., Chhatre, K., Tripathi, S., Wen, Y., Black, M., Bolkart, T.: Emotional speech-driven animation with content-emotion disentanglement. In: SIGGRAPH Asia Conference Papers, pp. 1\u201313. (2023)","DOI":"10.1145\/3610548.3618183"},{"issue":"6","key":"4623_CR10","doi-asserted-by":"publisher","first-page":"971","DOI":"10.1016\/j.cag.2006.08.017","volume":"30","author":"JM De Martino","year":"2006","unstructured":"De Martino, J.M., Magalh\u00e3es, L.P., Violaro, F.: Facial animation based on context-dependent visemes. Comput. Gr. 30(6), 971\u2013980 (2006)","journal-title":"Comput. Gr."},{"issue":"4","key":"4623_CR11","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2897824.2925984","volume":"35","author":"P Edwards","year":"2016","unstructured":"Edwards, P., Landreth, C., Fiume, E., Singh, K.: Jali: an animator-centric viseme model for expressive lip synchronization. ACM TOG 35(4), 1\u201311 (2016)","journal-title":"ACM TOG"},{"key":"4623_CR12","doi-asserted-by":"crossref","unstructured":"Ezzat, T., Poggio, T.: Miketalk: a talking facial display based on morphing visemes. In: Proceedings Computer Animation, pp. 96\u2013102. (1998)","DOI":"10.1109\/CA.1998.681913"},{"key":"4623_CR13","doi-asserted-by":"crossref","unstructured":"Fan, X., Li, J., Lin, Z., Xiao, W., Yang, L.: Unitalker: scaling up audio-driven 3d facial animation through a unified model. In: ECCV, pp. 204\u2013221. Springer (2024)","DOI":"10.1007\/978-3-031-72940-9_12"},{"key":"4623_CR14","doi-asserted-by":"crossref","unstructured":"Fan, Y., Lin, Z., Saito, J., Wang, W., Komura, T.: Faceformer: speech-driven 3d facial animation with transformers, pp. 18770\u201318780. CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01821"},{"issue":"6","key":"4623_CR15","doi-asserted-by":"publisher","first-page":"591","DOI":"10.1109\/TMM.2010.2052239","volume":"12","author":"G Fanelli","year":"2010","unstructured":"Fanelli, G., Gall, J., Romsdorfer, H., Weise, T., Van Gool, L.: A 3-d audio-visual corpus of affective communication. IEEE Trans. Multimed. 12(6), 591\u2013598 (2010)","journal-title":"IEEE Trans. Multimed."},{"key":"4623_CR16","doi-asserted-by":"crossref","unstructured":"Fu, H., Wang, Z., Gong, K., Wang, K., Chen, T., Li, H., Zeng, H., Kang, W.: Mimic: speaking style disentanglement for speech-driven 3d facial animation. In: AAAI, vol.\u00a038, pp. 1770\u20131777 (2024)","DOI":"10.1609\/aaai.v38i2.27945"},{"key":"4623_CR17","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4623_CR18","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"WN Hsu","year":"2021","unstructured":"Hsu, W.N., Bolte, B., Tsai, Y.H.H., Lakhotia, K., Salakhutdinov, R., Mohamed, A.: Hubert: self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 3451\u20133460 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"4","key":"4623_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073658","volume":"36","author":"T Karras","year":"2017","unstructured":"Karras, T., Aila, T., Laine, S., Herva, A., Lehtinen, J.: Audio-driven facial animation by joint end-to-end learning of pose and emotion. ACM TOG 36(4), 1\u201312 (2017)","journal-title":"ACM TOG"},{"key":"4623_CR20","unstructured":"Kong, Z., Ping, W., Huang, J., Zhao, K., Catanzaro, B.: Diffwave: a versatile diffusion model for audio synthesis. In: ICLR. (2021)"},{"issue":"6","key":"4623_CR21","first-page":"1","volume":"36","author":"T Li","year":"2017","unstructured":"Li, T., Bolkart, T., Black, M.J., Li, H., Romero, J.: Learning a model of facial shape and expression from 4d scans. ACM TOG 36(6), 1\u2013194 (2017)","journal-title":"ACM TOG"},{"key":"4623_CR22","doi-asserted-by":"crossref","unstructured":"Liang, H., Bao, J., Zhang, R., Ren, S., Xu, Y., Yang, S., Chen, X., Yu, J., Xu, L.: Omg: towards open-vocabulary motion generation via mixture of controllers. In: CVPR, pp. 482\u2013493 (2024)","DOI":"10.1109\/CVPR52733.2024.00053"},{"issue":"5","key":"4623_CR23","doi-asserted-by":"publisher","first-page":"e0196391","DOI":"10.1371\/journal.pone.0196391","volume":"13","author":"S Livingstone","year":"2018","unstructured":"Livingstone, S., Russo, F.: Ryerson audiovisual database of emotional speeches and songs (ravdess): a dynamic, multimodal set of north American English face and voice expressions. PLoS ONE 13(5), e0196391 (2018)","journal-title":"PLoS ONE"},{"key":"4623_CR24","doi-asserted-by":"crossref","unstructured":"Mughal, M.H., Dabral, R., Habibie, I., Donatelli, L., Habermann, M., Theobalt, C.: Convofusion: multi-modal conversational diffusion for co-speech gesture synthesis, pp. 1388\u20131398. CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.00138"},{"key":"4623_CR25","doi-asserted-by":"crossref","unstructured":"Nocentini, F., Ferrari, C., Berretti, S.: Emovoca: speech-driven emotional 3d talking heads. In: WACV, pp. 2859\u20132868. IEEE (2025)","DOI":"10.1109\/WACV61041.2025.00283"},{"key":"4623_CR26","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers, pp. 4195\u20134205. ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"4623_CR27","doi-asserted-by":"crossref","unstructured":"Peng, Z., Wu, H., Song, Z., Xu, H., Zhu, X., He, J., Liu, H., Fan, Z.: Emotalk: speech-driven emotional disentanglement for 3d face animation, pp. 20687\u201320697. ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01891"},{"key":"4623_CR28","unstructured":"Popov, V., Vovk, I., Gogoryan, V., Sadekova, T., Kudinov, M.: Grad-tts: a diffusion probabilistic model for text-to-speech, pp. 8599\u20138608. ICLR (2021)"},{"key":"4623_CR29","doi-asserted-by":"crossref","unstructured":"Richard, A., Lea, C., Ma, S., Gall, J., De\u00a0la Torre, F., Sheikh, Y.: Audio-and gaze-driven facial animation of codec avatars. In: WACV, pp. 41\u201350 (2021)","DOI":"10.1109\/WACV48630.2021.00009"},{"key":"4623_CR30","doi-asserted-by":"crossref","unstructured":"Richard, A., Zollh\u00f6fer, M., Wen, Y., De\u00a0la Torre, F., Sheikh, Y.: Meshtalk: 3d face animation from speech using cross-modality disentanglement. In: ICCV, pp. 1173\u20131182 (2021)","DOI":"10.1109\/ICCV48922.2021.00121"},{"key":"4623_CR31","unstructured":"R\u00f6ssler, A., Cozzolino, D., Verdoliva, L., Riess, C., Thies, J., Nie\u00dfner, M.: Faceforensics: a large-scale video dataset for forgery detection in human faces, (2018) arXiv preprint. arXiv:1803.09179"},{"key":"4623_CR32","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: Dreambooth: fine tuning text-to-image diffusion models for subject-driven generation, pp. 22500\u201322510. CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"4623_CR33","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., Ganguli, S.: Deep unsupervised learning using nonequilibrium thermodynamics. In: ICML, pp. 2256\u20132265. (2015)"},{"key":"4623_CR34","doi-asserted-by":"crossref","unstructured":"Song, W., Wang, X., Zheng, S., Li, S., Hao, A., Hou, X.: Talkingstyle: personalized speech-driven 3d facial animation with style preservation. TVCG. (2024)","DOI":"10.1109\/TVCG.2024.3409568"},{"key":"4623_CR35","doi-asserted-by":"crossref","unstructured":"Song, W., Wang, X., Zheng, S., Li, S., Hao, A., Hou, X., Qin, H.: Expressive 3d facial animation generation based on local-to-global latent diffusion. TVCG. (2024)","DOI":"10.1109\/TVCG.2024.3456213"},{"key":"4623_CR36","doi-asserted-by":"crossref","unstructured":"Stan, S., Haque, K.I., Yumak, Z.: Facediffuser: speech-driven 3d facial animation synthesis using diffusion, pp. 1\u201311. ACM SIGGRAPH MIG (2023)","DOI":"10.1145\/3623264.3624447"},{"issue":"4","key":"4623_CR37","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3658221","volume":"43","author":"Z Sun","year":"2024","unstructured":"Sun, Z., Lv, T., Ye, S., Lin, M., Sheng, J., Wen, Y.H., Yu, M., Liu, Y.J.: Diffposetalk: speech-driven stylistic 3d facial animation and head pose generation via diffusion models. ACM TOG 43(4), 1\u20139 (2024)","journal-title":"ACM TOG"},{"key":"4623_CR38","unstructured":"Tevet, G., Raab, S., andMDM Daniel\u00a0Cohen-Or, B.G., Bermano, A.H.: Human motion diffusion model. In: ICLR (2023)"},{"key":"4623_CR39","doi-asserted-by":"crossref","unstructured":"Thambiraja, B., Aliakbarian, S., Cosker, D., Thies, J.: 3diface: diffusion-based speech-driven 3d facial animation and editing. (2023). arXiv preprint. arXiv:2312.00870","DOI":"10.1109\/ICCV51070.2023.01885"},{"key":"4623_CR40","first-page":"244","volume":"15141","author":"L Tian","year":"2024","unstructured":"Tian, L., Wang, Q., Zhang, B., Bo, L.: EMO: emote portrait alive generating expressive portrait videos with audio2video diffusion model under weak conditions. ECCV 15141, 244\u2013260 (2024)","journal-title":"ECCV"},{"key":"4623_CR41","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. Advances in neural information processing systems 30 (2017)"},{"key":"4623_CR42","doi-asserted-by":"crossref","unstructured":"Wu, J.Z., Ge, Y., Wang, X., Lei, S.W., Gu, Y., Shi, Y., Hsu, W., Shan, Y., Qie, X., Shou, M.Z.: Tune-a-video: one-shot tuning of image diffusion models for text-to-video generation, pp. 7623\u20137633. ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00701"},{"key":"4623_CR43","doi-asserted-by":"crossref","unstructured":"Wu, S., Haque, K.I., Yumak, Z.: Probtalk3d: non-deterministic emotion controllable speech-driven 3d facial animation synthesis using vq-vae. In: ACM SIGGRAPH MIG, pp. 1\u201312. (2024)","DOI":"10.1145\/3677388.3696320"},{"key":"4623_CR44","doi-asserted-by":"crossref","unstructured":"Xing, J., Xia, M., Zhang, Y., Cun, X., Wang, J., Wong, T.T.: Codetalker: speech-driven 3d facial animation with discrete motion prior, pp. 12780\u201312790. CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01229"},{"key":"4623_CR45","doi-asserted-by":"publisher","first-page":"1720","DOI":"10.1109\/TASLP.2023.3268730","volume":"31","author":"D Yang","year":"2023","unstructured":"Yang, D., Yu, J., Wang, H., Wang, W., Weng, C., Zou, Y., Yu, D.: Diffsound: discrete diffusion model for text-to-sound generation. IEEE\/ACM Trans. Audio Speech Lang. Process. 31, 1720\u20131733 (2023)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"4","key":"4623_CR46","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3626235","volume":"56","author":"L Yang","year":"2023","unstructured":"Yang, L., Zhang, Z., Song, Y., Hong, S., Xu, R., Zhao, Y., Zhang, W., Cui, B., Yang, M.H.: Diffusion models: a comprehensive survey of methods and applications. ACM Comput. Surv. 56(4), 1\u201339 (2023)","journal-title":"ACM Comput. Surv."},{"issue":"2","key":"4623_CR47","first-page":"1438","volume":"29","author":"C Zhang","year":"2021","unstructured":"Zhang, C., Ni, S., Fan, Z., Li, H., Zeng, M., Budagavi, M., Guo, X.: 3d talking face with personalized pose dynamics. TVCG 29(2), 1438\u20131449 (2021)","journal-title":"TVCG"},{"issue":"6","key":"4623_CR48","doi-asserted-by":"publisher","first-page":"4115","DOI":"10.1109\/TPAMI.2024.3355414","volume":"46","author":"M Zhang","year":"2024","unstructured":"Zhang, M., Cai, Z., Pan, L., Hong, F., Guo, X., Yang, L., Liu, Z.: Motiondiffuse: text-driven human motion generation with diffusion model. IEEE TPAMI 46(6), 4115\u20134128 (2024)","journal-title":"IEEE TPAMI"},{"key":"4623_CR49","doi-asserted-by":"crossref","unstructured":"Zhao, Q., Long, P., Zhang, Q., Qin, D., Liang, H., Zhang, L., Zhang, Y., Yu, J., Xu, L.: Media2face: co-speech facial animation generation with multi-modality guidance. In: ACM SIGGRAPH Conference Papers, p.\u00a018 (2024)","DOI":"10.1145\/3641519.3657413"},{"issue":"4","key":"4623_CR50","first-page":"1","volume":"37","author":"Y Zhou","year":"2018","unstructured":"Zhou, Y., Xu, Z., Landreth, C., Kalogerakis, E., Maji, S., Singh, K.: Visemenet: audio-driven animator-centric speech animation. ACM TOG 37(4), 1\u201310 (2018)","journal-title":"ACM TOG"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04623-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04623-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04623-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T11:38:41Z","timestamp":1784893121000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04623-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":50,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["4623"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04623-7","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"29 April 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 June 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no conflict of interest.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"409"}}