{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T22:34:44Z","timestamp":1775082884684,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Science Fund of China","award":["62361166670"],"award-info":[{"award-number":["62361166670"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680619","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:27Z","timestamp":1729925967000},"page":"3964-3973","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["ConsistentAvatar: Learning to Diffuse Fully Consistent Talking Head Avatar with Temporal Guidance"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3489-0438","authenticated-orcid":false,"given":"Haijie","family":"Yang","sequence":"first","affiliation":[{"name":"Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5727-9450","authenticated-orcid":false,"given":"Zhenyu","family":"Zhang","sequence":"additional","affiliation":[{"name":"Nanjing University, Suzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2077-1246","authenticated-orcid":false,"given":"Hao","family":"Tang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0968-8556","authenticated-orcid":false,"given":"Jianjun","family":"Qian","sequence":"additional","affiliation":[{"name":"Nanjing University of Science and Techonology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4800-832X","authenticated-orcid":false,"given":"Jian","family":"Yang","sequence":"additional","affiliation":[{"name":"Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"ShahRukh Athar Zexiang Xu Kalyan Sunkavalli Eli Shechtman and Zhixin Shu. 2022. RigNeRF: Fully Controllable Neural 3D Portraits. arxiv: 2206.06481 [cs.CV]","DOI":"10.1109\/CVPR52688.2022.01972"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/311535.311556"},{"key":"e_1_3_2_1_3_1","volume-title":"EMOCA: Emotion Driven Monocular Face Capture and Animation. arxiv: 2204.11312 [cs.CV]","author":"Danecek Radek","year":"2022","unstructured":"Radek Danecek, Michael J. Black, and Timo Bolkart. 2022. EMOCA: Emotion Driven Monocular Face Capture and Animation. arxiv: 2204.11312 [cs.CV]"},{"key":"e_1_3_2_1_4_1","volume-title":"Disentangled and Controllable Face Image Generation via 3D Imitative-Contrastive Learning. arxiv","author":"Deng Yu","year":"2004","unstructured":"Yu Deng, Jiaolong Yang, Dong Chen, Fang Wen, and Xin Tong. 2020. Disentangled and Controllable Face Image Generation via 3D Imitative-Contrastive Learning. arxiv: 2004.11660 [cs.CV]"},{"key":"e_1_3_2_1_5_1","volume-title":"Accurate 3D Face Reconstruction with Weakly-Supervised Learning: From Single Image to Image Set. arxiv","author":"Deng Yu","year":"1903","unstructured":"Yu Deng, Jiaolong Yang, Sicheng Xu, Dong Chen, Yunde Jia, and Xin Tong. 2020. Accurate 3D Face Reconstruction with Weakly-Supervised Learning: From Single Image to Image Set. arxiv: 1903.08527 [cs.CV]"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Zheng Ding Xuaner Zhang Zhihao Xia Lars Jebe Zhuowen Tu and Xiuming Zhang. 2023. DiffusionRig: Learning Personalized Priors for Facial Appearance Editing. arxiv: 2304.06711 [cs.CV]","DOI":"10.1109\/CVPR52729.2023.01225"},{"key":"e_1_3_2_1_7_1","volume-title":"Learning an Animatable Detailed 3D Face Model from In-The-Wild Images. arxiv","author":"Feng Yao","year":"2012","unstructured":"Yao Feng, Haiwen Feng, Michael J. Black, and Timo Bolkart. 2021. Learning an Animatable Detailed 3D Face Model from In-The-Wild Images. arxiv: 2012.04012 [cs.CV]"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00854"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Simon Giebenhain Tobias Kirschstein Markos Georgopoulos Martin R\u00fcnz Lourdes Agapito and Matthias Nie\u00dfner. 2023. Learning Neural Parametric Head Models. arxiv: 2212.02761 [cs.CV]","DOI":"10.1109\/CVPR52729.2023.02012"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01810"},{"key":"e_1_3_2_1_11_1","volume-title":"FLNet: Landmark Driven Fetching and Learning Network for Faithful Talking Facial Animation Synthesis. arxiv","author":"Gu Kuangxiao","year":"1911","unstructured":"Kuangxiao Gu, Yuqian Zhou, and Thomas Huang. 2019. FLNet: Landmark Driven Fetching and Learning Network for Faithful Talking Facial Animation Synthesis. arxiv: 1911.09224 [cs.CV]"},{"key":"e_1_3_2_1_12_1","unstructured":"Yue Han Jiangning Zhang Junwei Zhu Xiangtai Li Yanhao Ge Wei Li Chengjie Wang Yong Liu Xiaoming Liu and Ying Tai. 2023. A Generalist FaceX via Learning Unified Facial Representation. arxiv: 2401.00551 [cs.CV]"},{"key":"e_1_3_2_1_13_1","volume-title":"Denoising Diffusion Probabilistic Models. arxiv","author":"Ho Jonathan","year":"2006","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising Diffusion Probabilistic Models. arxiv: 2006.11239 [cs.LG]"},{"key":"e_1_3_2_1_14_1","volume-title":"Xun Cao, and Feng Xu.","author":"Ji Xinya","year":"2021","unstructured":"Xinya Ji, Hang Zhou, Kaisiyuan Wang, Wayne Wu, Chen Change Loy, Xun Cao, and Feng Xu. 2021. Audio-Driven Emotional Video Portraits. arxiv: 2104.07452 [cs.CV]"},{"key":"e_1_3_2_1_15_1","volume-title":"A Style-Based Generator Architecture for Generative Adversarial Networks. arxiv","author":"Karras Tero","year":"1812","unstructured":"Tero Karras, Samuli Laine, and Timo Aila. 2019. A Style-Based Generator Architecture for Generative Adversarial Networks. arxiv: 1812.04948 [cs.NE]"},{"key":"e_1_3_2_1_16_1","volume-title":"Analyzing and Improving the Image Quality of StyleGAN. arxiv","author":"Karras Tero","year":"1912","unstructured":"Tero Karras, Samuli Laine, Miika Aittala, Janne Hellsten, Jaakko Lehtinen, and Timo Aila. 2020. Analyzing and Improving the Image Quality of StyleGAN. arxiv: 1912.04958 [cs.CV]"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Taras Khakhulin Vanessa Sklyarova Victor Lempitsky and Egor Zakharov. 2022. Realistic One-shot Mesh-based Head Avatars. arxiv: 2206.08343 [cs.CV]","DOI":"10.1007\/978-3-031-20086-1_20"},{"key":"e_1_3_2_1_18_1","unstructured":"Gyeongman Kim Hajin Shim Hyunsu Kim Yunjey Choi Junho Kim and Eunho Yang. 2023. Diffusion Video Autoencoders: Toward Temporally Consistent Face Video Editing via Disentangled Video Encoding. arxiv: 2212.02802 [cs.CV]"},{"key":"e_1_3_2_1_19_1","unstructured":"Minchul Kim Feng Liu Anil Jain and Xiaoming Liu. 2023. DCFace: Synthetic Face Generation with Dual Condition Diffusion Model. arxiv: 2304.07060 [cs.CV]"},{"key":"e_1_3_2_1_20_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2015","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7--9, 2015, Conference Track Proceedings, Yoshua Bengio and Yann LeCun (Eds.)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Tobias Kirschstein Simon Giebenhain and Matthias Nie\u00dfner. 2023. DiffusionAvatars: Deferred Diffusion for High-fidelity 3D Head Avatars. arxiv: 2311.18635 [cs.CV]","DOI":"10.1109\/CVPR52733.2024.00524"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3130800.3130813"},{"key":"e_1_3_2_1_23_1","unstructured":"Simian Luo Yiqin Tan Longbo Huang Jian Li and Hang Zhao. 2023. Latent Consistency Models: Synthesizing High-Resolution Images with Few-Step Inference. arxiv: 2310.04378 [cs.CV]"},{"key":"e_1_3_2_1_24_1","volume-title":"Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020","author":"Nguyen-Phuoc Thu","year":"2020","unstructured":"Thu Nguyen-Phuoc, Christian Richardt, Long Mai, Yong-Liang Yang, and Niloy J. Mitra. 2020. BlockGAN: Learning 3D Object-aware Scene Representations from Unlabelled Images. In Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, December 6--12, 2020, virtual, Hugo Larochelle, Marc'Aurelio Ranzato, Raia Hadsell, Maria-Florina Balcan, and Hsuan-Tien Lin (Eds.)."},{"key":"e_1_3_2_1_25_1","unstructured":"Alex Nichol and Prafulla Dhariwal. 2021. Improved Denoising Diffusion Probabilistic Models. arxiv: 2102.09672 [cs.LG]"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Atsuhiro Noguchi Xiao Sun Stephen Lin and Tatsuya Harada. 2022. Unsupervised Learning of Efficient Geometry-Aware Neural Articulated Representations. arxiv: 2204.08839 [cs.CV]","DOI":"10.1007\/978-3-031-19790-1_36"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Ziqiao Peng Haoyu Wu Zhenbo Song Hao Xu Xiangyu Zhu Jun He Hongyan Liu and Zhaoxin Fan. 2023. EmoTalk: Speech-Driven Emotional Disentanglement for 3D Face Animation. arxiv: 2303.11089 [cs.CV]","DOI":"10.1109\/ICCV51070.2023.01891"},{"key":"e_1_3_2_1_28_1","volume-title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. arxiv: 2307.01952 [cs.CV]","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2023. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. arxiv: 2307.01952 [cs.CV]"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Shenhan Qian Tobias Kirschstein Liam Schoneveld Davide Davoli Simon Giebenhain and Matthias Nie\u00dfner. 2023. GaussianAvatars: Photorealistic Head Avatars with Rigged 3D Gaussians. arxiv: 2312.02069 [cs.CV]","DOI":"10.1109\/CVPR52733.2024.01919"},{"key":"e_1_3_2_1_30_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. arxiv: 2103.00020 [cs.CV]"},{"key":"e_1_3_2_1_31_1","volume-title":"Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems 2021","author":"Ranzato Marc'Aurelio","year":"2021","unstructured":"Marc'Aurelio Ranzato, Alina Beygelzimer, Yann N. Dauphin, Percy Liang, and Jennifer Wortman Vaughan (Eds.). 2021. Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems 2021, NeurIPS 2021, December 6--14, 2021, virtual."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Yurui Ren Ge Li Yuanqi Chen Thomas H. Li and Shan Liu. 2021. PIRenderer: Controllable Portrait Image Generation via Semantic Neural Rendering. arxiv: 2109.08379 [cs.CV]","DOI":"10.1109\/ICCV48922.2021.01350"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2022. High-resolution image synthesis with latent diffusion models. In CVPR. 10684--10695.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_34_1","volume":"201","author":"Sanyal Soubhik","unstructured":"Soubhik Sanyal, Timo Bolkart, Haiwen Feng, and Michael J. Black. 2019. Learning to Regress 3D Face Shape and Expression from an Image without 3D Supervision. arxiv: 1905.06817 [cs.CV]","journal-title":"Michael J. Black."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Shuai Shen Wenliang Zhao Zibin Meng Wanhua Li Zheng Zhu Jie Zhou and Jiwen Lu. 2023. DiffTalk: Crafting Diffusion Models for Generalized Audio-Driven Portraits Animation. arxiv: 2301.03786 [cs.CV]","DOI":"10.1109\/CVPR52729.2023.00197"},{"key":"e_1_3_2_1_36_1","volume-title":"First Order Motion Model for Image Animation. arxiv","author":"Siarohin Aliaksandr","year":"2003","unstructured":"Aliaksandr Siarohin, St\u00e9phane Lathuili\u00e8re, Sergey Tulyakov, Elisa Ricci, and Nicu Sebe. 2020. First Order Motion Model for Image Animation. arxiv: 2003.00196 [cs.CV]"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Sanjana Sinha Sandika Biswas Ravindra Yadav and Brojeshwar Bhowmick. 2022. Emotion-Controllable Generalized Talking Face Generation. arxiv: 2205.01155 [cs.CV]","DOI":"10.24963\/ijcai.2022\/184"},{"key":"e_1_3_2_1_38_1","volume-title":"EMMN: Emotional Motion Memory Network for Audio-driven Emotional Talking Face Generation. In 2023 IEEE\/CVF International Conference on Computer Vision (ICCV). 22089--22099","author":"Tan Shuai","year":"2023","unstructured":"Shuai Tan, Bin Ji, and Ye Pan. 2023. EMMN: Emotional Motion Memory Network for Audio-driven Emotional Talking Face Generation. In 2023 IEEE\/CVF International Conference on Computer Vision (ICCV). 22089--22099."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00618"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings, Part XXI (Lecture Notes in Computer Science","volume":"717","author":"Wang Kaisiyuan","year":"2020","unstructured":"Kaisiyuan Wang, Qianyi Wu, Linsen Song, Zhuoqian Yang, Wayne Wu, Chen Qian, Ran He, Yu Qiao, and Chen Change Loy. 2020. MEAD: A Large-Scale Audio-Visual Dataset for Emotional Talking-Face Generation. In Computer Vision - ECCV 2020 - 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part XXI (Lecture Notes in Computer Science, Vol. 12366), Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.). Springer, 700--717."},{"key":"e_1_3_2_1_41_1","unstructured":"Yibo Xia Lizhen Wang Xiang Deng Xiaoyan Luo and Yebin Liu. 2023. GMTalker: Gaussian Mixture based Emotional talking video Portraits. arxiv: 2312.07669 [cs.CV]"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591567"},{"key":"e_1_3_2_1_43_1","unstructured":"Hu Ye Jun Zhang Sibo Liu Xiao Han and Wei Yang. 2023. IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models. arxiv: 2308.06721 [cs.CV]"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"crossref","unstructured":"Lvmin Zhang Anyi Rao and Maneesh Agrawala. 2023. Adding Conditional Control to Text-to-Image Diffusion Models. arxiv: 2302.05543 [cs.CV]","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Wenxuan Zhang Xiaodong Cun Xuan Wang Yong Zhang Xi Shen Yu Guo Ying Shan and Fei Wang. 2023. SadTalker: Learning Realistic 3D Motion Coefficients for Stylized Audio-Driven Single Image Talking Face Animation. arxiv: 2211.12194 [cs.CV]","DOI":"10.1109\/CVPR52729.2023.00836"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"Ruiqi Zhao Tianyi Wu and Guodong Guo. 2021. Sparse to Dense Motion Transfer for Face Image Animation. arxiv: 2109.00471 [cs.CV]","DOI":"10.1109\/ICCVW54120.2021.00226"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01318"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"Yufeng Zheng Wang Yifan Gordon Wetzstein Michael J. Black and Otmar Hilliges. 2023. PointAvatar: Deformable Point-based Head Avatars from Videos. arxiv: 2212.08377 [cs.CV]","DOI":"10.1109\/CVPR52729.2023.02017"},{"key":"e_1_3_2_1_49_1","volume-title":"Instant Volumetric Head Avatars. In 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 4574--4584","author":"Zielonka Wojciech","year":"2023","unstructured":"Wojciech Zielonka, Timo Bolkart, and Justus Thies. 2023. Instant Volumetric Head Avatars. In 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 4574--4584."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680619","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680619","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:56Z","timestamp":1750295876000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680619"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":49,"alternative-id":["10.1145\/3664647.3680619","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680619","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}