{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T12:19:11Z","timestamp":1783685951485,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,13]],"date-time":"2024-07-13T00:00:00Z","timestamp":1720828800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,13]]},"DOI":"10.1145\/3641519.3657459","type":"proceedings-article","created":{"date-parts":[[2024,7,12]],"date-time":"2024-07-12T06:39:28Z","timestamp":1720766368000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":51,"title":["X-Portrait: Expressive Portrait Animation with Hierarchical Motion Attention"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7320-6518","authenticated-orcid":false,"given":"You","family":"Xie","sequence":"first","affiliation":[{"name":"ByteDance Inc., United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4455-5632","authenticated-orcid":false,"given":"Hongyi","family":"Xu","sequence":"additional","affiliation":[{"name":"ByteDance Inc., United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3664-572X","authenticated-orcid":false,"given":"Guoxian","family":"Song","sequence":"additional","affiliation":[{"name":"ByteDance Inc., United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4378-0193","authenticated-orcid":false,"given":"Chao","family":"Wang","sequence":"additional","affiliation":[{"name":"ByteDance Inc., United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3121-8785","authenticated-orcid":false,"given":"Yichun","family":"Shi","sequence":"additional","affiliation":[{"name":"ByteDance Inc., United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6322-1175","authenticated-orcid":false,"given":"Linjie","family":"Luo","sequence":"additional","affiliation":[{"name":"ByteDance Inc., United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,7,13]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Stable diffusion v1.5 model card. https:\/\/huggingface.co\/runwayml\/stable-diffusion-v1-5","author":"Stability AI.","year":"2022","unstructured":"Stability AI. 2022. Stable diffusion v1.5 model card. https:\/\/huggingface.co\/runwayml\/stable-diffusion-v1-5 (2022)."},{"key":"e_1_3_2_2_2_1","unstructured":"Apple. 2023. ARFaceAnchor.BlendShapeLocation. https:\/\/developer.apple.com\/documentation\/arkit\/arfaceanchor\/blendshapelocation"},{"key":"e_1_3_2_2_3_1","unstructured":"Andreas Blattmann Tim Dockhorn Sumith Kulal Daniel Mendelevitch Maciej Kilian Dominik Lorenz Yam Levi Zion English Vikram Voleti Adam Letts Varun Jampani and Robin Rombach. 2023. Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets. arxiv:2311.15127\u00a0[cs.CV]"},{"key":"e_1_3_2_2_4_1","volume-title":"MasaCtrl: Tuning-Free Mutual Self-Attention Control for Consistent Image Synthesis and Editing. arXiv preprint arXiv:2304.08465","author":"Cao Mingdeng","year":"2023","unstructured":"Mingdeng Cao, Xintao Wang, Zhongang Qi, Ying Shan, Xiaohu Qie, and Yinqiang Zheng. 2023. MasaCtrl: Tuning-Free Mutual Self-Attention Control for Consistent Image Synthesis and Editing. arXiv preprint arXiv:2304.08465 (2023)."},{"key":"e_1_3_2_2_5_1","volume-title":"OpenPose: Realtime Multi-Person 2D Pose Estimation using Part Affinity Fields","author":"Cao Z.","year":"2019","unstructured":"Z. Cao, G. Hidalgo Martinez, T. Simon, S. Wei, and Y.\u00a0A. Sheikh. 2019. OpenPose: Realtime Multi-Person 2D Pose Estimation using Part Affinity Fields. IEEE TPAMI (2019)."},{"key":"e_1_3_2_2_6_1","unstructured":"Di Chang Yichun Shi Quankai Gao Jessica Fu Hongyi Xu Guoxian Song Qing Yan Yizhe Zhu Xiao Yang and Mohammad Soleymani. 2024. MagicPose: Realistic Human Poses and Facial Expressions Retargeting with Identity-aware Diffusion. arxiv:2311.12052\u00a0[cs.CV]"},{"key":"e_1_3_2_2_7_1","volume-title":"Arcface: Additive angular margin loss for deep face recognition. In CVPR. 4690\u20134699.","author":"Deng Jiankang","year":"2019","unstructured":"Jiankang Deng, Jia Guo, Niannan Xue, and Stefanos Zafeiriou. 2019. Arcface: Additive angular margin loss for deep face recognition. In CVPR. 4690\u20134699."},{"key":"e_1_3_2_2_8_1","volume-title":"Disentangled and Controllable Face Image Generation via 3D Imitative-Contrastive Learning","author":"Deng Yu","unstructured":"Yu Deng, Jiaolong Yang, Dong Chen, Fang Wen, and Xin Tong. 2020. Disentangled and Controllable Face Image Generation via 3D Imitative-Contrastive Learning. In IEEE Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_2_9_1","volume-title":"deviantart. https:\/\/www.deviantart.com","year":"2024","unstructured":"DeviantArt. 2024. deviantart. https:\/\/www.deviantart.com (2024)."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547838"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","unstructured":"Yao Feng Haiwen Feng Michael\u00a0J. Black and Timo Bolkart. 2021. Learning an Animatable Detailed 3D Face Model from In-The-Wild Images. ACM Transactions on Graphics (Proc. SIGGRAPH) 40 8. https:\/\/doi.org\/10.1145\/3450626.3459936","DOI":"10.1145\/3450626.3459936"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3272127.3275043"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"crossref","unstructured":"Yuan Gong Yong Zhang Xiaodong Cun Fei Yin Yanbo Fan Xuan Wang Baoyuan Wu and Yujiu Yang. 2023. ToonTalker: Cross-Domain Face Reenactment. arxiv:2308.12866\u00a0[cs.CV]","DOI":"10.1109\/ICCV51070.2023.00707"},{"key":"e_1_3_2_2_14_1","unstructured":"Yuming Gu You Xie Hongyi Xu Guoxian Song Yichun Shi Di Chang Jing Yang and Linjie Luo. 2023. DiffPortrait3D: Controllable Diffusion for Zero-Shot Portrait View Synthesis. arxiv:2312.13016\u00a0[cs.CV]"},{"key":"e_1_3_2_2_15_1","volume-title":"AD-NeRF: Audio Driven Neural Radiance Fields for Talking Head Synthesis. In IEEE\/CVF International Conference on Computer Vision (ICCV).","author":"Guo Yudong","year":"2021","unstructured":"Yudong Guo, Keyu Chen, Sen Liang, Yongjin Liu, Hujun Bao, and Juyong Zhang. 2021. AD-NeRF: Audio Driven Neural Radiance Fields for Talking Head Synthesis. In IEEE\/CVF International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_2_16_1","volume-title":"SparseCtrl: Adding Sparse Controls to Text-to-Video Diffusion Models. arXiv preprint arXiv:2311.16933","author":"Guo Yuwei","year":"2023","unstructured":"Yuwei Guo, Ceyuan Yang, Anyi Rao, Maneesh Agrawala, Dahua Lin, and Bo Dai. 2023a. SparseCtrl: Adding Sparse Controls to Text-to-Video Diffusion Models. arXiv preprint arXiv:2311.16933 (2023)."},{"key":"e_1_3_2_2_17_1","volume-title":"AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning. arXiv preprint arXiv:2307.04725","author":"Guo Yuwei","year":"2023","unstructured":"Yuwei Guo, Ceyuan Yang, Anyi Rao, Yaohui Wang, Yu Qiao, Dahua Lin, and Bo Dai. 2023b. AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning. arXiv preprint arXiv:2307.04725 (2023)."},{"key":"e_1_3_2_2_18_1","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. NeurIPS 33 (2020), 6840\u20136851.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","unstructured":"Fa-Ting Hong and Dan Xu. 2023. Implicit Identity Representation Conditioned Memory Compensation Network for Talking Head video Generation. In ICCV.","DOI":"10.1109\/ICCV51070.2023.02108"},{"key":"e_1_3_2_2_20_1","volume-title":"Depth-Aware Generative Adversarial Network for Talking Head Video Generation. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Hong Fa-Ting","year":"2022","unstructured":"Fa-Ting Hong, Longhao Zhang, Li Shen, and Dan Xu. 2022. Depth-Aware Generative Adversarial Network for Talking Head Video Generation. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/tip.2020.2967829"},{"key":"e_1_3_2_2_22_1","volume-title":"Animate Anyone: Consistent and Controllable Image-to-Video Synthesis for Character Animation. arXiv preprint arXiv:2311.17117","author":"Hu Li","year":"2023","unstructured":"Li Hu, Xin Gao, Peng Zhang, Ke Sun, Bang Zhang, and Liefeng Bo. 2023. Animate Anyone: Consistent and Controllable Image-to-Video Synthesis for Character Animation. arXiv preprint arXiv:2311.17117 (2023)."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"e_1_3_2_2_24_1","volume-title":"Realistic One-shot Mesh-based Head Avatars. In European Conference of Computer vision (ECCV).","author":"Khakhulin Taras","year":"2022","unstructured":"Taras Khakhulin, Vanessa Sklyarova, Victor Lempitsky, and Egor Zakharov. 2022. Realistic One-shot Mesh-based Head Avatars. In European Conference of Computer vision (ECCV)."},{"key":"e_1_3_2_2_25_1","unstructured":"Yukang Lin Haonan Han Chaoqun Gong Zunnan Xu Yachao Zhang and Xiu Li. 2023. Consistent123: One Image to Highly Consistent 3D Asset Using Case-Aware Diffusion Priors. arxiv:2309.17261\u00a0[cs.CV]"},{"key":"e_1_3_2_2_26_1","unstructured":"Minghua Liu Chao Xu Haian Jin Linghao Chen Mukund\u00a0Varma T Zexiang Xu and Hao Su. 2023b. One-2-3-45: Any Single Image to 3D Mesh in 45 Seconds without Per-Shape Optimization. arxiv:2306.16928\u00a0[cs.CV]"},{"key":"e_1_3_2_2_27_1","unstructured":"Ruoshi Liu Rundi Wu Basile\u00a0Van Hoorick Pavel Tokmakov Sergey Zakharov and Carl Vondrick. 2023a. Zero-1-to-3: Zero-shot One Image to 3D Object. arxiv:2303.11328\u00a0[cs.CV]"},{"key":"e_1_3_2_2_28_1","volume-title":"Latent consistency models: Synthesizing high-resolution images with few-step inference. arXiv preprint arXiv:2310.04378","author":"Luo Simian","year":"2023","unstructured":"Simian Luo, Yiqin Tan, Longbo Huang, Jian Li, and Hang Zhao. 2023. Latent consistency models: Synthesizing high-resolution images with few-step inference. arXiv preprint arXiv:2310.04378 (2023)."},{"key":"e_1_3_2_2_29_1","volume-title":"midjourney. https:\/\/www.midjourney.com","year":"2024","unstructured":"Midjourney. 2024. midjourney. https:\/\/www.midjourney.com (2024)."},{"key":"e_1_3_2_2_30_1","volume-title":"pexels. https:\/\/www.pexels.com\/","year":"2024","unstructured":"Pexels. 2024. pexels. https:\/\/www.pexels.com\/ (2024)."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2023.3253184"},{"key":"e_1_3_2_2_32_1","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2021. High-Resolution Image Synthesis with Latent Diffusion Models. arxiv:2112.10752\u00a0[cs.CV]"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2022. High-resolution image synthesis with latent diffusion models. In CVPR. 10684\u201310695.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_2_34_1","first-page":"36479","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume":"35","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily\u00a0L Denton, Kamyar Ghasemipour, Raphael Gontijo\u00a0Lopes, Burcu Karagol\u00a0Ayan, Tim Salimans, 2022. Photorealistic text-to-image diffusion models with deep language understanding. NeurIPS 35 (2022), 36479\u201336494.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00248"},{"key":"e_1_3_2_2_36_1","unstructured":"Aliaksandr Siarohin St\u00e9phane Lathuili\u00e8re Sergey Tulyakov Elisa Ricci and Nicu Sebe. 2019b. First Order Motion Model for Image Animation. In NeurIPS."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"crossref","unstructured":"Aliaksandr Siarohin Oliver Woodford Jian Ren Menglei Chai and Sergey Tulyakov. 2021. Motion Representations for Articulated Animation. In CVPR.","DOI":"10.1109\/CVPR46437.2021.01344"},{"key":"e_1_3_2_2_38_1","volume-title":"Denoising diffusion implicit models. arXiv preprint arXiv:2010.02502","author":"Song Jiaming","year":"2020","unstructured":"Jiaming Song, Chenlin Meng, and Stefano Ermon. 2020a. Denoising diffusion implicit models. arXiv preprint arXiv:2010.02502 (2020)."},{"key":"e_1_3_2_2_39_1","volume-title":"Score-based generative modeling through stochastic differential equations. arXiv preprint arXiv:2011.13456","author":"Song Yang","year":"2020","unstructured":"Yang Song, Jascha Sohl-Dickstein, Diederik\u00a0P Kingma, Abhishek Kumar, Stefano Ermon, and Ben Poole. 2020b. Score-based generative modeling through stochastic differential equations. arXiv preprint arXiv:2011.13456 (2020)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00372"},{"key":"e_1_3_2_2_41_1","unstructured":"Jingxiang Sun Xuan Wang Lizhen Wang Xiaoyu Li Yong Zhang Hongwen Zhang and Yebin Liu. 2023. Next3D: Generative Neural Texture Rasterization for 3D-Aware Head Avatars. In CVPR."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3550469.3555393"},{"key":"e_1_3_2_2_43_1","volume-title":"EDGE: Editable Dance Generation From Music. arxiv:2211.10658\u00a0[cs.SD]","author":"Tseng Jonathan","year":"2022","unstructured":"Jonathan Tseng, Rodrigo Castellon, and C.\u00a0Karen Liu. 2022. EDGE: Editable Dance Generation From Music. arxiv:2211.10658\u00a0[cs.SD]"},{"key":"e_1_3_2_2_44_1","unstructured":"Ting-Chun Wang Arun Mallya and Ming-Yu Liu. 2021. One-Shot Free-View Neural Talking-Head Synthesis for Video Conferencing. In CVPR."},{"key":"e_1_3_2_2_45_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=7r6kDq0mK_","author":"Wang Yaohui","year":"2022","unstructured":"Yaohui Wang, Di Yang, Francois Bremond, and Antitza Dantcheva. 2022. Latent Image Animator: Learning to Animate Images via Latent Space Navigation. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=7r6kDq0mK_"},{"key":"e_1_3_2_2_46_1","volume-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Xu H.","unstructured":"H. Xu, G. Song, Z. Jiang, J. Zhang, Y. Shi, J. Liu, W. Ma, J. Feng, and L. Luo. 2023a. OmniAvatar: Geometry-Guided Controllable 3D Head Synthesis. In 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_2_47_1","unstructured":"Zhongcong Xu Jianfeng Zhang Jun\u00a0Hao Liew Hanshu Yan Jia-Wei Liu Chenxu Zhang Jiashi Feng and Mike\u00a0Zheng Shou. 2023b. MagicAnimate: Temporally Consistent Human Image Animation using Diffusion Model."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00070"},{"key":"e_1_3_2_2_49_1","unstructured":"Chenxu Zhang Chao Wang Jianfeng Zhang Hongyi Xu Guoxian Song You Xie Linjie Luo Yapeng Tian Xiaohu Guo and Jiashi Feng. 2023b. DREAM-Talk: Diffusion-based Realistic Emotional Audio-driven Method for Single Image Talking Face Generation. arxiv:2312.13578\u00a0[cs.CV]"},{"key":"e_1_3_2_2_50_1","unstructured":"Lyumin Zhang. 2023. [major update] reference-only control \u00b7 Mikubill\/SD-webui-controlnet \u00b7 discussion #1236. https:\/\/github.com\/Mikubill\/sd-webui-controlnet\/discussions\/1236"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"crossref","unstructured":"Lvmin Zhang Anyi Rao and Maneesh Agrawala. 2023a. Adding conditional control to text-to-image diffusion models. In ICCV. 3836\u20133847.","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"crossref","unstructured":"Jian Zhao and Hui Zhang. 2022. Thin-Plate Spline Motion Model for Image Animation. cvpr:2203.14367\u00a0[cs.CV]","DOI":"10.1109\/CVPR52688.2022.00364"}],"event":{"name":"SIGGRAPH '24: Special Interest Group on Computer Graphics and Interactive Techniques Conference","location":"Denver CO USA","acronym":"SIGGRAPH '24","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers 24"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641519.3657459","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3641519.3657459","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T19:17:10Z","timestamp":1755890230000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641519.3657459"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,13]]},"references-count":52,"alternative-id":["10.1145\/3641519.3657459","10.1145\/3641519"],"URL":"https:\/\/doi.org\/10.1145\/3641519.3657459","relation":{},"subject":[],"published":{"date-parts":[[2024,7,13]]},"assertion":[{"value":"2024-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}