{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T07:45:43Z","timestamp":1782805543107,"version":"3.54.5"},"reference-count":51,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.cviu.2026.104856","type":"journal-article","created":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T23:36:31Z","timestamp":1782257791000},"page":"104856","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["LiveNeRF: Efficient face replacement through Neural Radiance Fields integration"],"prefix":"10.1016","volume":"270","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-7188-5972","authenticated-orcid":false,"given":"Tung","family":"Vu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4445-2811","authenticated-orcid":false,"given":"Hai","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9467-4978","authenticated-orcid":false,"given":"Cong","family":"Tran","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104856_b1","series-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","first-page":"1","article-title":"wav2vec 2.0: a framework for self-supervised learning of speech representations","author":"Baevski","year":"2020"},{"key":"10.1016\/j.cviu.2026.104856_b2","doi-asserted-by":"crossref","unstructured":"Blanz, V., Vetter, T., 1999. A Morphable Model for the Synthesis of 3D Faces. In: SIGGRAPH. pp. 187\u2013194.","DOI":"10.1145\/311535.311556"},{"key":"10.1016\/j.cviu.2026.104856_b3","series-title":"Computer Vision\u2013ECCV 2018: 15th European Conference","first-page":"538","article-title":"Lip movements generation at a glance","author":"Chen","year":"2018"},{"key":"10.1016\/j.cviu.2026.104856_b4","doi-asserted-by":"crossref","unstructured":"Chen, L., Maddox, R.K., Duan, Z., Xu, C., 2019. Hierarchical cross-modal talking face generation with dynamic pixel-wise loss. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 7832\u20137841.","DOI":"10.1109\/CVPR.2019.00802"},{"key":"10.1016\/j.cviu.2026.104856_b5","doi-asserted-by":"crossref","unstructured":"Cho, K., et al., 2024. GaussianTalker: Real-time talking head synthesis with 3D Gaussian splatting. In: Proceedings of the 32nd ACM International Conference on Multimedia. pp. 9614\u20139623.","DOI":"10.1145\/3664647.3681627"},{"issue":"9","key":"10.1016\/j.cviu.2026.104856_b6","doi-asserted-by":"crossref","first-page":"8953","DOI":"10.1109\/TCSVT.2024.3386836","article-title":"CorrTalk: Correlation between hierarchical speech and facial activity variances for 3D animation","volume":"34","author":"Chu","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104856_b7","series-title":"Computer Vision\u2013ACCV 2016: 13th Asian Conference on Computer Vision","first-page":"87","article-title":"Lip reading in the wild","author":"Chung","year":"2017"},{"key":"10.1016\/j.cviu.2026.104856_b8","doi-asserted-by":"crossref","unstructured":"Drobyshev, N., Casademunt, A.B., Vougioukas, K., Landgraf, Z., Petridis, S., Pantic, M., 2024. EmoPortraits: Emotion-Enhanced Multimodal One-Shot Head Avatars. In: CVPR. pp. 4899\u20134909.","DOI":"10.1109\/CVPR52733.2024.00812"},{"key":"10.1016\/j.cviu.2026.104856_b9","doi-asserted-by":"crossref","unstructured":"Drobyshev, N., Chelishev, J., Khakhulin, T., Ivakhnenko, A., Lempitsky, V., Zakharov, E., 2022. MegaPortraits: One-Shot Megapixel Neural Head Avatars. In: ACM MM. pp. 2663\u20132672.","DOI":"10.1145\/3503161.3547838"},{"issue":"3","key":"10.1016\/j.cviu.2026.104856_b10","doi-asserted-by":"crossref","first-page":"388","DOI":"10.1145\/566654.566594","article-title":"Trainable videorealistic speech animation","volume":"21","author":"Ezzat","year":"2002","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.cviu.2026.104856_b11","doi-asserted-by":"crossref","unstructured":"Ghosh, P., Gupta, P.S., Uziel, R., Ranjan, A., Black, M.J., Bolkart, T., 2020. GIF: Generative Interpretable Faces. In: 3DV. pp. 868\u2013878.","DOI":"10.1109\/3DV50981.2020.00097"},{"key":"10.1016\/j.cviu.2026.104856_b12","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105341","article-title":"3D human avatar reconstruction with neural fields: A recent survey","volume":"154","author":"Gu","year":"2025","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.cviu.2026.104856_b13","doi-asserted-by":"crossref","unstructured":"Guo, Y., Chen, K., Liang, S., Liu, Y.J., Bao, H., Zhang, J., 2021. AD-NeRF: Audio Driven Neural Radiance Fields for Talking Head Synthesis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV, pp. 5784\u20135794.","DOI":"10.1109\/ICCV48922.2021.00573"},{"key":"10.1016\/j.cviu.2026.104856_b14","series-title":"LivePortrait: Efficient portrait animation with stitching and retargeting control","author":"Guo","year":"2024"},{"key":"10.1016\/j.cviu.2026.104856_b15","doi-asserted-by":"crossref","unstructured":"He, Q., Cao, J., Lu, H., Zhang, P., 2024. Dynamic Region Fusion Neural Radiance Fields for Audio-Driven Talking Head Generation. In: 2024 7th International Conference on Machine Learning and Natural Language Processing. MLNLP, pp. 1\u20137.","DOI":"10.1109\/MLNLP63328.2024.10799951"},{"key":"10.1016\/j.cviu.2026.104856_b16","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S., 2017. GANs trained by a two time-scale update rule converge to a local Nash equilibrium. In: NeurIPS. pp. 6626\u20136637."},{"key":"10.1016\/j.cviu.2026.104856_b17","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio, Speech Lang. Proc."},{"issue":"11","key":"10.1016\/j.cviu.2026.104856_b18","doi-asserted-by":"crossref","first-page":"1767","DOI":"10.1007\/s11263-019-01150-y","article-title":"You said that?: Synthesising talking faces from audio","volume":"127","author":"Jamaludin","year":"2019","journal-title":"Int. J. Comput. Vis."},{"issue":"4","key":"10.1016\/j.cviu.2026.104856_b19","doi-asserted-by":"crossref","first-page":"9315","DOI":"10.1109\/TCE.2025.3596239","article-title":"SMACNet: A unified framework for one-shot talking head synthesis via subtle motion and appearance compensation","volume":"71","author":"Ji","year":"2025","journal-title":"IEEE Trans. Consum. Electron."},{"key":"10.1016\/j.cviu.2026.104856_b20","doi-asserted-by":"crossref","unstructured":"Khakhulin, T., Sklyarova, V., Lempitsky, V., Zakharov, E., 2022. Realistic One-Shot Mesh-Based Head Avatars. In: ECCV. pp. 345\u2013362.","DOI":"10.1007\/978-3-031-20086-1_20"},{"key":"10.1016\/j.cviu.2026.104856_b21","doi-asserted-by":"crossref","unstructured":"Lee, D., Kim, C., Yu, S., Yoo, J., Park, G.M., 2024. RADIO: Reference-Agnostic Dubbing Video Synthesis. In: 2024 IEEE\/CVF Winter Conference on Applications of Computer Vision. WACV, pp. 4156\u20134166.","DOI":"10.1109\/WACV57701.2024.00412"},{"key":"10.1016\/j.cviu.2026.104856_b22","doi-asserted-by":"crossref","unstructured":"Li, J., Zhang, J., Bai, X., Zhou, J., Gu, L., 2023. Efficient Region-Aware Neural Radiance Fields for High-Fidelity Talking Portrait Synthesis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV, pp. 7568\u20137578.","DOI":"10.1109\/ICCV51070.2023.00696"},{"key":"10.1016\/j.cviu.2026.104856_b23","series-title":"Computer Vision\u2013ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXXVII","first-page":"106","article-title":"Semantic-aware implicit neural audio-driven video portrait generation","author":"Liu","year":"2022"},{"key":"10.1016\/j.cviu.2026.104856_b24","doi-asserted-by":"crossref","unstructured":"Meng, R., Zhang, X., Li, Y., Ma, C., 2025. EchoMimicV2: Towards Striking, Simplified, and Semi-Body Human Animation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 5489\u20135498.","DOI":"10.1109\/CVPR52734.2025.00516"},{"key":"10.1016\/j.cviu.2026.104856_b25","series-title":"European Conference on Computer Vision","first-page":"405","article-title":"NeRF: Representing scenes as neural radiance fields for view synthesis","author":"Mildenhall","year":"2020"},{"key":"10.1016\/j.cviu.2026.104856_b26","doi-asserted-by":"crossref","unstructured":"Peng, Z., Hu, W., Shi, Y., Zhu, X., Zhang, X., Zhao, H., He, J., Liu, H., Fan, Z., 2024. SyncTalk: The Devil is in the Synchronization for Talking Head Synthesis. In: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 666\u2013676.","DOI":"10.1109\/CVPR52733.2024.00070"},{"key":"10.1016\/j.cviu.2026.104856_b27","series-title":"Proceedings of the 28th ACM International Conference on Multimedia","first-page":"484","article-title":"A lip sync expert is all you need for speech to lip generation in the wild","author":"Prajwal","year":"2020"},{"key":"10.1016\/j.cviu.2026.104856_b28","series-title":"DiffTalker: Co-driven audio-image diffusion for talking faces via intermediate landmarks","author":"Qi","year":"2023"},{"key":"10.1016\/j.cviu.2026.104856_b29","series-title":"Computer Vision\u2013ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XII","first-page":"666","article-title":"Learning dynamic facial radiance fields for few-shot talking head synthesis","author":"Shen","year":"2022"},{"key":"10.1016\/j.cviu.2026.104856_b30","doi-asserted-by":"crossref","unstructured":"Shen, S., Zhao, W., Meng, Z., Li, W., Zhu, Z., Zhou, J., Lu, J., 2023. DiffTalk: Crafting Diffusion Models for Generalized Audio-Driven Portraits Animation. In: CVPR. pp. 1982\u20131991.","DOI":"10.1109\/CVPR52729.2023.00197"},{"key":"10.1016\/j.cviu.2026.104856_b31","unstructured":"Siarohin, A., Lathuili\u00e8re, S., Tulyakov, S., Ricci, E., Sebe, N., 2019. First Order Motion Model for Image Animation. In: NeurIPS. pp. 7137\u20137147."},{"key":"10.1016\/j.cviu.2026.104856_b32","doi-asserted-by":"crossref","unstructured":"Siarohin, A., Woodford, O., Ren, J., Chai, M., Tulyakov, S., 2021. Motion Representations for Articulated Animation. In: CVPR. pp. 13653\u201313662.","DOI":"10.1109\/CVPR46437.2021.01344"},{"issue":"1","key":"10.1016\/j.cviu.2026.104856_b33","doi-asserted-by":"crossref","first-page":"479","DOI":"10.3390\/app15010479","article-title":"Multi-level feature dynamic fusion neural radiance fields for audio-driven talking head generation","volume":"15","author":"Song","year":"2025","journal-title":"Appl. Sci."},{"key":"10.1016\/j.cviu.2026.104856_b34","doi-asserted-by":"crossref","unstructured":"Su, Y., Wang, S., Wang, H., 2024. DT-NeRF: Decomposed Triplane-Hash Neural Radiance Fields For High-Fidelity Talking Portrait Synthesis. In: ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 3975\u20133979.","DOI":"10.1109\/ICASSP48485.2024.10448446"},{"issue":"12","key":"10.1016\/j.cviu.2026.104856_b35","doi-asserted-by":"crossref","first-page":"8758","DOI":"10.1109\/TPAMI.2024.3409380","article-title":"Memories are one-to-many mapping alleviators in talking face generation","volume":"46","author":"Tang","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104856_b36","series-title":"Real-time neural radiance talking portrait synthesis via audio-spatial decomposition","author":"Tang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104856_b37","series-title":"Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XVI","first-page":"716","article-title":"Neural voice puppetry: Audio-driven facial reenactment","author":"Thies","year":"2020"},{"key":"10.1016\/j.cviu.2026.104856_b38","doi-asserted-by":"crossref","unstructured":"Wang, T.C., Mallya, A., Liu, M.Y., 2021. One-Shot Free-View Neural Talking-Head Synthesis for Video Conferencing. In: CVPR. pp. 10039\u201310049.","DOI":"10.1109\/CVPR46437.2021.00991"},{"key":"10.1016\/j.cviu.2026.104856_b39","series-title":"Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXI","first-page":"700","article-title":"MEAD: A large-scale audio-visual dataset for emotional talking-face generation","author":"Wang","year":"2020"},{"key":"10.1016\/j.cviu.2026.104856_b40","article-title":"High-fidelity and high-efficiency talking portrait synthesis with detail-aware neural radiance fields","author":"Wang","year":"2024","journal-title":"IEEE Trans. Vis. Comput. Graphics"},{"key":"10.1016\/j.cviu.2026.104856_b41","series-title":"Computer Vision \u2013 ECCV 2018: 15th European Conference, Munich, Germany, September 8-14, 2018, Proceedings, Part XIII","first-page":"690","article-title":"X2Face: A network for controlling face generation using images, audio, and pose codes","author":"Wiles","year":"2018"},{"key":"10.1016\/j.cviu.2026.104856_b42","doi-asserted-by":"crossref","unstructured":"Xu, S., Chen, G., Guo, Y.X., Yang, J., Li, C., Zang, Z., Zhang, Y., Tong, X., Guo, B., 2024. VASA-1: Lifelike Audio-Driven Talking Faces Generated in Real Time. In: The Thirty-Eighth Annual Conference on Neural Information Processing Systems. pp. 1\u201325.","DOI":"10.52202\/079017-0021"},{"key":"10.1016\/j.cviu.2026.104856_b43","doi-asserted-by":"crossref","unstructured":"Xu, Z., Zhang, J., Liew, J.H., Yan, H., Liu, J.W., Zhang, C., Feng, J., Shou, M.Z., 2024. MagicAnimate: Temporally Consistent Human Image Animation Using Diffusion Model. In: CVPR. pp. 1481\u20131490.","DOI":"10.1109\/CVPR52733.2024.00147"},{"key":"10.1016\/j.cviu.2026.104856_b44","series-title":"2024 IEEE International Conference on Multimedia and Expo","first-page":"1","article-title":"Talking portrait with discrete motion priors in neural radiation field","author":"Yang","year":"2024"},{"key":"10.1016\/j.cviu.2026.104856_b45","series-title":"DFA-NeRF: Personalized talking head generation via disentangled face attributes neural rendering","author":"Yao","year":"2022"},{"key":"10.1016\/j.cviu.2026.104856_b46","doi-asserted-by":"crossref","unstructured":"Yu, Z., Yin, Z., Zhou, D., Wang, D., Wong, F., Wang, B., 2023. Talking Head Generation with Probabilistic Audio-to-Visual Diffusion Priors. In: International Conference on Computer Vision. ICCV, pp. 19842\u201319851.","DOI":"10.1109\/ICCV51070.2023.00703"},{"key":"10.1016\/j.cviu.2026.104856_b47","doi-asserted-by":"crossref","unstructured":"Zhang, W., Cun, X., Wang, X., Zhang, Y., Shen, X., Guo, Y., Shan, Y., Wang, F., 2023. Sadtalker: Learning realistic 3d motion coefficients for stylized audio-driven single image talking face animation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8652\u20138661.","DOI":"10.1109\/CVPR52729.2023.00836"},{"key":"10.1016\/j.cviu.2026.104856_b48","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A., Shechtman, E., Wang, O., 2018. The unreasonable effectiveness of deep features as a perceptual metric. In: CVPR. pp. 586\u2013595.","DOI":"10.1109\/CVPR.2018.00068"},{"key":"10.1016\/j.cviu.2026.104856_b49","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zheng, R., Li, B., Han, C., Li, T., Wang, M., Guo, T., Chen, J., Liu, Z., Yang, M., 2024. Learning Dynamic Tetrahedra for High-Quality Talking Head Synthesis. In: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 5209\u20135219.","DOI":"10.1109\/CVPR52733.2024.00498"},{"issue":"6","key":"10.1016\/j.cviu.2026.104856_b50","first-page":"1","article-title":"MakeItTalk: Speaker-aware talking-head animation","volume":"39","author":"Zhou","year":"2020","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.cviu.2026.104856_b51","doi-asserted-by":"crossref","unstructured":"Zhou, H., Sun, Y., Wu, W., Loy, C.C., Wang, X., Liu, Z., 2021. Pose-controllable talking face generation by implicitly modularized audio-visual representation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4176\u20134186.","DOI":"10.1109\/CVPR46437.2021.00416"}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226002237?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226002237?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T07:09:15Z","timestamp":1782803355000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226002237"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":51,"alternative-id":["S1077314226002237"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104856","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"LiveNeRF: Efficient face replacement through Neural Radiance Fields integration","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104856","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104856"}}