{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T09:08:41Z","timestamp":1779527321916,"version":"3.53.1"},"reference-count":61,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T00:00:00Z","timestamp":1776038400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T00:00:00Z","timestamp":1776038400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1007\/s11263-026-02800-8","type":"journal-article","created":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T09:59:15Z","timestamp":1776074355000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Identity-Preserving Video Dubbing Using Motion Warping"],"prefix":"10.1007","volume":"134","author":[{"given":"Runzhen","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qinjie","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunfei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lijian","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ye","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chuhua","family":"Xian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0108-8596","authenticated-orcid":false,"given":"Fa-Ting","family":"Hong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,13]]},"reference":[{"key":"2800_CR1","doi-asserted-by":"crossref","unstructured":"Blanz V, Vetter T, Rockwood A (2002) A morphable model for the synthesis of 3d faces. ACM SIGGRAPH pp 187\u2013194","DOI":"10.1145\/311535.311556"},{"key":"2800_CR2","doi-asserted-by":"crossref","unstructured":"Chen L, Maddox RK, Duan Z, et\u00a0al (2019) Hierarchical cross-modal talking face generation with dynamic pixel-wise loss. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7832\u20137841","DOI":"10.1109\/CVPR.2019.00802"},{"key":"2800_CR3","doi-asserted-by":"crossref","unstructured":"Chen Z, Cao J, Chen Z, et\u00a0al (2025) Echomimic: Lifelike audio-driven portrait animations through editable landmark conditions. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 2403\u20132410","DOI":"10.1609\/aaai.v39i3.32241"},{"key":"2800_CR4","doi-asserted-by":"crossref","unstructured":"Cheng K, Cun X, Zhang Y, et\u00a0al (2022) Videoretalking: Audio-based lip synchronization for talking head video editing in the wild. In: SIGGRAPH Asia 2022 Conference Papers, pp 1\u20139","DOI":"10.1145\/3550469.3555399"},{"key":"2800_CR5","doi-asserted-by":"crossref","unstructured":"Du C, Chen Q, He T, et\u00a0al (2023) Dae-talker: High fidelity speech-driven talking face generation with diffusion autoencoder. In: ACMMM","DOI":"10.1145\/3581783.3613753"},{"key":"2800_CR6","doi-asserted-by":"crossref","unstructured":"Gafni G, Thies J, Zollhofer M, et\u00a0al (2021) Dynamic neural radiance fields for monocular 4d facial avatar reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 8649\u20138658","DOI":"10.1109\/CVPR46437.2021.00854"},{"key":"2800_CR7","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, et\u00a0al (2014) Generative adversarial nets. NeurIPS"},{"key":"2800_CR8","unstructured":"Guo J, Zhang D, Liu X, et\u00a0al (2024) Liveportrait: Efficient portrait animation with stitching and retargeting control. arXiv:2407.03168"},{"key":"2800_CR9","doi-asserted-by":"crossref","unstructured":"Guo Y, Chen K, Liang S, et\u00a0al (2021) Ad-nerf: Audio driven neural radiance fields for talking head synthesis. In: ICCV","DOI":"10.1109\/ICCV48922.2021.00573"},{"key":"2800_CR10","unstructured":"Hannun A, Case C, Casper J, et\u00a0al (2014) Deep speech: Scaling up end-to-end speech recognition. arXiv preprint arXiv:1412.5567"},{"key":"2800_CR11","unstructured":"Ho J, Jain A, Abbeel P (2020) Denoising diffusion probabilistic models. NeurIPS"},{"key":"2800_CR12","doi-asserted-by":"crossref","unstructured":"Hong FT, Zhang L, Shen L, et\u00a0al (2022) Depth-aware generative adversarial network for talking head video generation. In: CVPR","DOI":"10.1109\/CVPR52688.2022.00339"},{"key":"2800_CR13","doi-asserted-by":"crossref","unstructured":"Hong FT, Shen L, Xu D (2023) Dagan++: Depth-aware generative adversarial network for talking head video generation. TPAMI","DOI":"10.1109\/CVPR52688.2022.00339"},{"key":"2800_CR14","doi-asserted-by":"crossref","unstructured":"Huang X, Belongie S (2017) Arbitrary style transfer in real-time with adaptive instance normalization. In: ICCV","DOI":"10.1109\/ICCV.2017.167"},{"key":"2800_CR15","doi-asserted-by":"crossref","unstructured":"Ji X, Zhou H, Wang K, et\u00a0al (2022) Eamm: One-shot emotional talking face via audio-based emotion-aware motion model. In: ACM SIGGRAPH 2022 Conference Proceedings, pp 1\u201310","DOI":"10.1145\/3528233.3530745"},{"key":"2800_CR16","doi-asserted-by":"crossref","unstructured":"Johnson J, Alahi A, Fei-Fei L (2016) Perceptual losses for real-time style transfer and super-resolution. In: ECCV","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"2800_CR17","doi-asserted-by":"crossref","unstructured":"Karras T, Laine S, Aila T (2019) A style-based generator architecture for generative adversarial networks. In: CVPR","DOI":"10.1109\/CVPR.2019.00453"},{"key":"2800_CR18","unstructured":"KR P, Mukhopadhyay R, Philip J, et\u00a0al (2019) Towards automatic face-to-face translation. In: ACMMM"},{"key":"2800_CR19","unstructured":"Li C, Zhang C, Xu W, et\u00a0al (2024) Latentsync: Audio conditioned latent diffusion models for lip sync. arXiv preprint arXiv:2412.09262"},{"key":"2800_CR20","doi-asserted-by":"crossref","unstructured":"Liang B, Pan Y, Guo Z, et\u00a0al (2022) Expressive talking head generation with granular audio-visual control. In: CVPR","DOI":"10.1109\/CVPR52688.2022.00338"},{"key":"2800_CR21","doi-asserted-by":"crossref","unstructured":"Liu X, Xu Y, Wu Q, et\u00a0al (2022) Semantic-aware implicit neural audio-driven video portrait generation. In: ECCV","DOI":"10.1007\/978-3-031-19836-6_7"},{"key":"2800_CR22","unstructured":"Lugaresi C, Tang J, Nash H, et\u00a0al (2019) Mediapipe: A framework for building perception pipelines. arXiv preprint arXiv:1906.08172"},{"key":"2800_CR23","doi-asserted-by":"crossref","unstructured":"Ma H, Zhang T, Sun S, et\u00a0al (2024) Cvthead: One-shot controllable head avatar with vertex-feature transformer. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp 6131\u20136141","DOI":"10.1109\/WACV57701.2024.00602"},{"key":"2800_CR24","doi-asserted-by":"crossref","unstructured":"Ma Y, Wang S, Hu Z, et\u00a0al (2023) Styletalk: One-shot talking head generation with controllable speaking styles. In: AAAI","DOI":"10.1609\/aaai.v37i2.25280"},{"key":"2800_CR25","unstructured":"Ma Y, Zhang S, Wang J, et\u00a0al (2023) Dreamtalk: When expressive talking head generation meets diffusion probabilistic models. arXiv preprint arXiv:2312.09767"},{"key":"2800_CR26","doi-asserted-by":"crossref","unstructured":"Mao X, Li Q, Xie H, et\u00a0al (2017) Least squares generative adversarial networks. In: ICCV","DOI":"10.1109\/ICCV.2017.304"},{"issue":"1","key":"2800_CR27","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3503250","volume":"65","author":"B Mildenhall","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P. P., Tancik, M., et al. (2021). Nerf: Representing scenes as neural radiance fields for view synthesis. Communications of the ACM, 65(1), 99\u2013106.","journal-title":"Communications of the ACM"},{"key":"2800_CR28","doi-asserted-by":"crossref","unstructured":"Park T, Liu MY, Wang TC, et\u00a0al (2019) Semantic image synthesis with spatially-adaptive normalization. In: CVPR","DOI":"10.1109\/CVPR.2019.00244"},{"key":"2800_CR29","doi-asserted-by":"crossref","unstructured":"Prajwal K, Mukhopadhyay R, Namboodiri VP, et\u00a0al (2020) A lip sync expert is all you need for speech to lip generation in the wild. In: ACMMM","DOI":"10.1145\/3394171.3413532"},{"key":"2800_CR30","doi-asserted-by":"crossref","unstructured":"Ren Y, Li G, Chen Y, et\u00a0al (2021) Pirenderer: Controllable portrait image generation via semantic neural rendering. In: ICCV","DOI":"10.1109\/ICCV48922.2021.01350"},{"key":"2800_CR31","doi-asserted-by":"crossref","unstructured":"Shen S, Li W, Zhu Z, et\u00a0al (2022) Learning dynamic facial radiance fields for few-shot talking head synthesis. In: ECCV","DOI":"10.1007\/978-3-031-19775-8_39"},{"key":"2800_CR32","doi-asserted-by":"crossref","unstructured":"Shen S, Zhao W, Meng Z, et\u00a0al (2023) Difftalk: Crafting diffusion models for generalized audio-driven portraits animation. In: CVPR","DOI":"10.1109\/CVPR52729.2023.00197"},{"key":"2800_CR33","unstructured":"Siarohin A, Lathuili\u00e8re S, Tulyakov S, et\u00a0al (2019) First order motion model for image animation. NeurIPS"},{"key":"2800_CR34","unstructured":"Simonyan K (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556"},{"key":"2800_CR35","doi-asserted-by":"crossref","unstructured":"Stypu\u0142kowski M, Vougioukas K, He S, et\u00a0al (2024) Diffused heads: Diffusion models beat gans on talking-face generation. In: WACV","DOI":"10.1109\/WACV57701.2024.00502"},{"key":"2800_CR36","doi-asserted-by":"crossref","unstructured":"Tan S, Ji B, Ding Y, et\u00a0al (2024) Say anything with any style. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 5088\u20135096","DOI":"10.1609\/aaai.v38i5.28314"},{"key":"2800_CR37","doi-asserted-by":"crossref","unstructured":"Tan S, Ji B, Pan Y (2024) Style2talker: High-resolution talking head generation with emotion style and art style. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 5079\u20135087","DOI":"10.1609\/aaai.v38i5.28313"},{"key":"2800_CR38","doi-asserted-by":"crossref","unstructured":"Thies J, Elgharib M, Tewari A, et\u00a0al (2020) Neural voice puppetry: Audio-driven facial reenactment. In: ECCV","DOI":"10.1007\/978-3-030-58517-4_42"},{"key":"2800_CR39","unstructured":"Vaswani A (2017) Attention is all you need. NeurIPS"},{"key":"2800_CR40","doi-asserted-by":"crossref","unstructured":"Wang S, Li L, Ding Y, et\u00a0al (2021) Audio2head: Audio-driven one-shot talking-head generation with natural head motion. arXiv preprint arXiv:2107.09293","DOI":"10.24963\/ijcai.2021\/152"},{"key":"2800_CR41","doi-asserted-by":"crossref","unstructured":"Wang S, Li L, Ding Y, et\u00a0al (2022) One-shot talking face generation from single-speaker audio-visual correlation learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 2531\u20132539","DOI":"10.1609\/aaai.v36i3.20154"},{"issue":"4","key":"2800_CR42","first-page":"604","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z. (2004). Image quality assessment: Form error visibility to structural similarity. TIP, 13(4), 604\u2013606.","journal-title":"TIP"},{"key":"2800_CR43","unstructured":"Wei H, Yang Z, Wang Z (2024) Aniportrait: Audio-driven synthesis of photorealistic portrait animation. arXiv preprint arXiv:2403.17694"},{"key":"2800_CR44","doi-asserted-by":"crossref","unstructured":"Xie L, Wang X, Zhang H, et\u00a0al (2022) Vfhq: A high-quality dataset and benchmark for video face super-resolution. In: CVPR","DOI":"10.1109\/CVPRW56347.2022.00081"},{"key":"2800_CR45","doi-asserted-by":"crossref","unstructured":"Xie T, Liao L, Bi C, et\u00a0al (2021) Towards realistic visual dubbing with heterogeneous sources. In: ACMMM","DOI":"10.1145\/3474085.3475318"},{"key":"2800_CR46","unstructured":"Xu M, Li H, Su Q, et\u00a0al (2024) Hallo: Hierarchical audio-driven visual synthesis for portrait image animation. arXiv preprint arXiv:2406.08801"},{"key":"2800_CR47","first-page":"660","volume":"37","author":"S Xu","year":"2024","unstructured":"Xu, S., Chen, G., Guo, Y. X., et al. (2024). Vasa-1: Lifelike audio-driven talking faces generated in real time. Advances in Neural Information Processing Systems, 37, 660\u2013684.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2800_CR48","doi-asserted-by":"crossref","unstructured":"Xu Z, Yu Z, Zhou Z, et\u00a0al (2025) Hunyuanportrait: Implicit condition control for enhanced portrait animation. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp 15909\u201315919","DOI":"10.1109\/CVPR52734.2025.01483"},{"key":"2800_CR49","unstructured":"Ye Z, Zhong T, Ren Y, et\u00a0al (2024) Real3d-portrait: One-shot realistic 3d talking portrait synthesis. arXiv preprint arXiv:2401.08503"},{"key":"2800_CR50","doi-asserted-by":"crossref","unstructured":"Yin F, Zhang Y, Cun X, et\u00a0al (2022) Styleheat: One-shot high-resolution editable talking face generation via pre-trained stylegan. In: ECCV","DOI":"10.1007\/978-3-031-19790-1_6"},{"key":"2800_CR51","first-page":"22451","volume":"35","author":"B Zeng","year":"2022","unstructured":"Zeng, B., Liu, B., Li, H., et al. (2022). Fnevr: Neural volume rendering for face animation. Advances in Neural Information Processing Systems, 35, 22451\u201322462.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2800_CR52","doi-asserted-by":"crossref","unstructured":"Zhang C, Zhao Y, Huang Y, et\u00a0al (2021) Facial: Synthesizing dynamic talking face with implicit attribute learning. In: ICCV","DOI":"10.1109\/ICCV48922.2021.00384"},{"key":"2800_CR53","doi-asserted-by":"crossref","unstructured":"Zhang R, Isola P, Efros AA, et\u00a0al (2018) The unreasonable effectiveness of deep features as a perceptual metric. In: CVPR","DOI":"10.1109\/CVPR.2018.00068"},{"key":"2800_CR54","doi-asserted-by":"crossref","unstructured":"Zhang W, Cun X, Wang X, et\u00a0al (2023) Sadtalker: Learning realistic 3d motion coefficients for stylized audio-driven single image talking face animation. In: CVPR","DOI":"10.1109\/CVPR52729.2023.00836"},{"key":"2800_CR55","doi-asserted-by":"crossref","unstructured":"Zhang Z, Ding Y (2022) Adaptive affine transformation: A simple and effective operation for spatial misaligned image generation. In: ACMMM","DOI":"10.1145\/3503161.3548330"},{"key":"2800_CR56","doi-asserted-by":"crossref","unstructured":"Zhang Z, Li L, Ding Y, et\u00a0al (2021) Flow-guided one-shot talking face generation with a high-resolution audio-visual dataset. In: CVPR","DOI":"10.1109\/CVPR46437.2021.00366"},{"key":"2800_CR57","doi-asserted-by":"crossref","unstructured":"Zhang Z, Hu Z, Deng W, et\u00a0al (2023) Dinet: Deformation inpainting network for realistic face visually dubbing on high resolution video. In: AAAI","DOI":"10.1609\/aaai.v37i3.25464"},{"key":"2800_CR58","doi-asserted-by":"crossref","unstructured":"Zhao S, Hong FT, Huang X, et\u00a0al (2025) Synergizing motion and appearance: Multi-scale compensatory codebooks for talking head video generation. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp 26232\u201326241","DOI":"10.1109\/CVPR52734.2025.02443"},{"key":"2800_CR59","doi-asserted-by":"crossref","unstructured":"Zhong W, Fang C, Cai Y, et\u00a0al (2023) Identity-preserving talking face generation with landmark and appearance priors. In: CVPR","DOI":"10.1109\/CVPR52729.2023.00938"},{"key":"2800_CR60","doi-asserted-by":"crossref","unstructured":"Zhou H, Sun Y, Wu W, et\u00a0al (2021) Pose-controllable talking face generation by implicitly modularized audio-visual representation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4176\u20134186","DOI":"10.1109\/CVPR46437.2021.00416"},{"key":"2800_CR61","doi-asserted-by":"crossref","unstructured":"Zhou Y, Han X, Shechtman E, et\u00a0al (2020) Makelttalk: speaker-aware talking-head animation. TOG","DOI":"10.1145\/3414685.3417774"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02800-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02800-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02800-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T08:42:20Z","timestamp":1779525740000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02800-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,13]]},"references-count":61,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2026,5]]}},"alternative-id":["2800"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02800-8","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,13]]},"assertion":[{"value":"10 February 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"226"}}