{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:59:07Z","timestamp":1776931147750,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":73,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763854","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:27:29Z","timestamp":1765211249000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Audio Driven Real-Time Facial Animation for Social Telepresence"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8948-0589","authenticated-orcid":false,"given":"Jiye","family":"Lee","sequence":"first","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7055-5757","authenticated-orcid":false,"given":"Chenghui","family":"Li","sequence":"additional","affiliation":[{"name":"Codec Avatars Lab, Meta, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8715-1076","authenticated-orcid":false,"given":"Linh","family":"Tran","sequence":"additional","affiliation":[{"name":"Codec Avatars Lab, Meta, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0092-3441","authenticated-orcid":false,"given":"Shih-En","family":"Wei","sequence":"additional","affiliation":[{"name":"Codec Avatars Lab, Meta, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6218-5029","authenticated-orcid":false,"given":"Jason","family":"Saragih","sequence":"additional","affiliation":[{"name":"Codec Avatars Lab, Meta, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5173-6438","authenticated-orcid":false,"given":"Alexander","family":"Richard","sequence":"additional","affiliation":[{"name":"Codec Avatars Lab, Meta, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6666-7460","authenticated-orcid":false,"given":"Hanbyul","family":"Joo","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7969-9429","authenticated-orcid":false,"given":"Shaojie","family":"Bai","sequence":"additional","affiliation":[{"name":"Codec Avatars Lab, Meta, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_3_2_1","unstructured":"Shivangi Aneja Artem Sevastopolsky Tobias Kirschstein Justus Thies Angela Dai and Matthias Nie\u00dfner. 2024a. GaussianSpeech: Audio-Driven Gaussian Avatars. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.18675 (2024)."},{"key":"e_1_3_3_3_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02009"},{"key":"e_1_3_3_3_4_1","volume-title":"ECCV","author":"Athar ShahRukh","year":"2024","unstructured":"ShahRukh Athar, Shunsuke Saito, Zhengyu Yang, Stanislav Pidhorsky, and Chen Cao. 2024. Bridging the Gap: Studio-like Avatar Creation from a Monocular Phone Capture. In ECCV."},{"key":"e_1_3_3_3_5_1","unstructured":"Alexei Baevski Yuhao Zhou Abdelrahman Mohamed and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. NeurIPS (2020)."},{"key":"e_1_3_3_3_6_1","unstructured":"Shaojie Bai Te-Li Wang Chenghui Li Akshay Venkatesh Tomas Simon Chen Cao Gabriel Schwartz Jason Saragih Yaser Sheikh and Shih-En Wei. 2024. Universal Facial Encoding of Codec Avatars from VR Headsets. ACM Trans. Graph. (2024)."},{"key":"e_1_3_3_3_7_1","doi-asserted-by":"crossref","unstructured":"Thabo Beeler Fabian Hahn Derek Bradley Bernd Bickel Paul Beardsley Craig Gotsman Robert\u00a0W. Sumner and Markus Gross. 2011. High-quality Passive Facial Performance Capture Using Anchor Frames. ACM Trans. Graph. (2011).","DOI":"10.1145\/1964921.1964970"},{"key":"e_1_3_3_3_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687580"},{"key":"e_1_3_3_3_9_1","doi-asserted-by":"crossref","unstructured":"Chen Cao Tomas Simon Jin\u00a0Kyu Kim Gabe Schwartz Michael Zollhoefer Shun-Suke Saito Stephen Lombardi Shih-En Wei Danielle Belko Shoou-I Yu Yaser Sheikh and Jason Saragih. 2022. Authentic volumetric avatars from a phone scan. ACM Trans. Graph. (2022).","DOI":"10.1145\/3528223.3530143"},{"key":"e_1_3_3_3_10_1","unstructured":"Bo Chen Shoukang Hu Qi Chen Chenpeng Du Ran Yi Yanmin Qian and Xie Chen. 2024. GSTalker: Real-time Audio-Driven Talking Face Generation via Deformable Gaussian Splatting. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.19040 (2024)."},{"key":"e_1_3_3_3_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681627"},{"key":"e_1_3_3_3_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01034"},{"key":"e_1_3_3_3_13_1","unstructured":"Jiahao Cui Hui Li Yao Yao Hao Zhu Hanlin Shang Kaihui Cheng Hang Zhou Siyu Zhu and Jingdong Wang. 2024. Hallo2: Long-Duration and High-Resolution Audio-Driven Portrait Image Animation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.07718 (2024)."},{"key":"e_1_3_3_3_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618183"},{"key":"e_1_3_3_3_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01821"},{"key":"e_1_3_3_3_16_1","doi-asserted-by":"crossref","unstructured":"Graham Fyffe Andrew Jones Oleg Alexander Ryosuke Ichikari and Paul Debevec. 2014. Driving High-Resolution Facial Scans with Video Performance Capture. ACM Trans. Graph. (2014).","DOI":"10.1145\/2638549"},{"key":"e_1_3_3_3_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02012"},{"key":"e_1_3_3_3_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00573"},{"key":"e_1_3_3_3_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3577190.3614157"},{"key":"e_1_3_3_3_20_1","volume-title":"ECCV","author":"He Qianyun","year":"2024","unstructured":"Qianyun He, Xinya Ji, Yicheng Gong, Yuanxun Lu, Zhengyu Diao, Linjia Huang, Yao Yao, Siyu Zhu, Zhan Ma, Songchen Xu, Xiaofei Wu, Zixiao Zhang, Xun Cao, and Hao Zhu. 2024. EmoTalk3D: High-Fidelity Free-View Synthesis of Emotional 3D Talking Head. In ECCV."},{"key":"e_1_3_3_3_21_1","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. NeurIPS (2020)."},{"key":"e_1_3_3_3_22_1","unstructured":"Jonathan Ho and Tim Salimans. 2022. Classifier-free diffusion guidance. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2207.12598 (2022)."},{"key":"e_1_3_3_3_23_1","unstructured":"Wei-Ning Hsu Benjamin Bolte Yao-Hung\u00a0Hubert Tsai Kushal Lakhotia Ruslan Salakhutdinov and Abdelrahman Mohamed. 2021. Hubert: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM transactions on audio speech and language processing (2021)."},{"key":"e_1_3_3_3_24_1","unstructured":"HTC. 2021. HTC VIVE Facial Tracker. https:\/\/www.vive.com\/eu\/accessory\/facial-tracker\/."},{"key":"e_1_3_3_3_25_1","doi-asserted-by":"crossref","unstructured":"Amir Jamaludin Joon\u00a0Son Chung and Andrew Zisserman. 2019. You said that?: Synthesising talking faces from audio. IJCV (2019).","DOI":"10.1007\/s11263-019-01150-y"},{"key":"e_1_3_3_3_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01386"},{"key":"e_1_3_3_3_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00868"},{"key":"e_1_3_3_3_28_1","doi-asserted-by":"crossref","unstructured":"Tero Karras Timo Aila Samuli Laine Antti Herva and Jaakko Lehtinen. 2017. Audio-driven facial animation by joint end-to-end learning of pose and emotion. ACM Trans. Graph. (2017).","DOI":"10.1145\/3072959.3073658"},{"key":"e_1_3_3_3_29_1","doi-asserted-by":"crossref","unstructured":"Bernhard Kerbl Georgios Kopanas Thomas Leimk\u00fchler and George Drettakis. 2023. 3d gaussian splatting for real-time radiance field rendering. ACM Trans. Graph. (2023).","DOI":"10.1145\/3592433"},{"key":"e_1_3_3_3_30_1","doi-asserted-by":"crossref","unstructured":"Tobias Kirschstein Shenhan Qian Simon Giebenhain Tim Walter and Matthias Nie\u00dfner. 2023. NeRSemble: Multi-View Radiance Field Reconstruction of Human Heads. ACM Trans. Graph. (2023).","DOI":"10.1145\/3592455"},{"key":"e_1_3_3_3_31_1","doi-asserted-by":"crossref","unstructured":"Hao Li Laura Trutoiu Kyle Olszewski Lingyu Wei Tristan Trutna Pei-Lun Hsieh Aaron Nicholls and Chongyang Ma. 2015. Facial Performance Sensing Head-mounted Display. ACM Trans. Graph. (2015).","DOI":"10.1145\/2766939"},{"key":"e_1_3_3_3_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687653"},{"key":"e_1_3_3_3_33_1","volume-title":"ECCV","author":"Li Jiahe","year":"2024","unstructured":"Jiahe Li, Jiawei Zhang, Xiao Bai, Jin Zheng, Xin Ning, Jun Zhou, and Lin Gu. 2024b. Talkinggaussian: Structure-persistent 3d talking head synthesis via gaussian splatting. In ECCV."},{"key":"e_1_3_3_3_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00696"},{"key":"e_1_3_3_3_35_1","unstructured":"Tianye Li Timo Bolkart Michael.\u00a0J. Black Hao Li and Javier Romero. 2017. Learning a model of facial shape and expression from 4D scans. ACM Transactions on Graphics (Proc. SIGGRAPH Asia) (2017)."},{"key":"e_1_3_3_3_36_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19836-6_7"},{"key":"e_1_3_3_3_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00013"},{"key":"e_1_3_3_3_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00101"},{"key":"e_1_3_3_3_39_1","volume-title":"ICML","author":"Nichol Alexander\u00a0Quinn","year":"2021","unstructured":"Alexander\u00a0Quinn Nichol and Prafulla Dhariwal. 2021. Improved denoising diffusion probabilistic models. In ICML."},{"key":"e_1_3_3_3_40_1","volume-title":"ECCV","author":"Nocentini Federico","year":"2024","unstructured":"Federico Nocentini, Thomas Besnier, Claudio Ferrari, Sylvain Arguillere, Stefano Berretti, and Mohamed Daoudi. 2024. Scantalk: 3d talking heads from unregistered scans. In ECCV."},{"key":"e_1_3_3_3_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01891"},{"key":"e_1_3_3_3_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413532"},{"key":"e_1_3_3_3_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01919"},{"key":"e_1_3_3_3_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00121"},{"key":"e_1_3_3_3_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_5"},{"key":"e_1_3_3_3_46_1","unstructured":"Shunsuke Saito Stanislav Pidhorskyi Igor Santesteban Forrest Iandola Divam Gupta Anuj Pahuja Nemanja Bartolovic Frank Yu Emanuel Garbin and Tomas Simon. 2024a. SqueezeMe: Efficient Gaussian Avatars for VR. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.15171 (2024)."},{"key":"e_1_3_3_3_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00021"},{"key":"e_1_3_3_3_48_1","doi-asserted-by":"crossref","unstructured":"Steffen Schneider Alexei Baevski Ronan Collobert and Michael Auli. 2019. wav2vec: Unsupervised pre-training for speech recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1904.05862 (2019).","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"e_1_3_3_3_49_1","doi-asserted-by":"crossref","unstructured":"Gabriel Schwartz Shih-En Wei Te-Li Wang Stephen Lombardi Tomas Simon Jason Saragih and Yaser Sheikh. 2020. The Eyes Have It: An Integrated Eye and Face Model for Photorealistic Facial Animation. ACM Trans. Graph. (2020).","DOI":"10.1145\/3386569.3392493"},{"key":"e_1_3_3_3_50_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19775-8_39"},{"key":"e_1_3_3_3_51_1","doi-asserted-by":"crossref","unstructured":"Shuai Shen Wenliang Zhao Zibin Meng Wanhua Li Zheng Zhu Jie Zhou and Jiwen Lu. 2023. Difftalk: Crafting diffusion models for generalized talking head synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.03786 (2023).","DOI":"10.1109\/CVPR52729.2023.00197"},{"key":"e_1_3_3_3_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00502"},{"key":"e_1_3_3_3_53_1","unstructured":"Jianlin Su Murtadha Ahmed Yu Lu Shengfeng Pan Wen Bo and Yunfeng Liu. 2024. Roformer: Enhanced transformer with rotary position embedding. Neurocomputing (2024)."},{"key":"e_1_3_3_3_54_1","doi-asserted-by":"crossref","unstructured":"Zhiyao Sun Tian Lv Sheng Ye Matthieu Lin Jenny Sheng Yu-Hui Wen Minjing Yu and Yong-Jin Liu. 2024. DiffPoseTalk: Speech-Driven Stylistic 3D Facial Animation and Head Pose Generation via Diffusion Models. ACM Trans. Graph. (2024).","DOI":"10.1145\/3658221"},{"key":"e_1_3_3_3_55_1","doi-asserted-by":"crossref","unstructured":"Supasorn Suwajanakorn Steven\u00a0M Seitz and Ira Kemelmacher-Shlizerman. 2017. Synthesizing obama: learning lip sync from audio. ACM Trans. Graph. (2017).","DOI":"10.1145\/3072959.3073640"},{"key":"e_1_3_3_3_56_1","volume-title":"ICLR","author":"Tevet Guy","year":"2023","unstructured":"Guy Tevet, Sigal Raab, Brian Gordon, Yoni Shafir, Daniel Cohen-or, and Amit\u00a0Haim Bermano. 2023. Human Motion Diffusion Model. In ICLR."},{"key":"e_1_3_3_3_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01885"},{"key":"e_1_3_3_3_58_1","doi-asserted-by":"crossref","unstructured":"Justus Thies Michael Zollh\u00f6fer Marc Stamminger Christian Theobalt and Matthias Nie\u00dfner. 2018. FaceVR: Real-Time Gaze-Aware Facial Reenactment in Virtual Reality. ACM Trans. Graph. (2018).","DOI":"10.1145\/3182644"},{"key":"e_1_3_3_3_59_1","doi-asserted-by":"crossref","unstructured":"Phong Tran Egor Zakharov Long-Nhat Ho Liwen Hu Adilbek Karmanov Aviral Agarwal McLean Goldwhite Ariana\u00a0Bermudez Venegas Anh\u00a0Tuan Tran and Hao Li. 2024. VOODOO XP: Expressive One-Shot Head Reenactment for VR Telepresence. ACM Trans. Graph. (2024).","DOI":"10.1145\/3687974"},{"key":"e_1_3_3_3_60_1","unstructured":"Shih-En Wei Jason Saragih Tomas Simon Adam\u00a0W. Harley Stephen Lombardi Michal Perdoch Alexander Hypes Dawei Wang Hernan Badino and Yaser Sheikh. 2019. VR Facial Animation via Multiview Image Translation. ACM Trans. Graph. (2019)."},{"key":"e_1_3_3_3_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3677388.3696320"},{"key":"e_1_3_3_3_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01229"},{"key":"e_1_3_3_3_63_1","unstructured":"Mingwang Xu Hui Li Qingkun Su Hanlin Shang Liwei Zhang Ce Liu Jingdong Wang Yao Yao and Siyu Zhu. 2024b. Hallo: Hierarchical audio-driven visual synthesis for portrait image animation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.08801 (2024)."},{"key":"e_1_3_3_3_64_1","doi-asserted-by":"crossref","unstructured":"Sicheng Xu Guojun Chen Yu-Xiao Guo Jiaolong Yang Chong Li Zhenyu Zang Yizhong Zhang Xin Tong and Baining Guo. 2024a. Vasa-1: Lifelike audio-driven talking faces generated in real time. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.10667 (2024).","DOI":"10.52202\/079017-0021"},{"key":"e_1_3_3_3_65_1","unstructured":"Shunyu Yao RuiZhe Zhong Yichao Yan Guangtao Zhai and Xiaokang Yang. 2022. Dfa-nerf: Personalized talking head generation via disentangled face attributes neural rendering. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2201.00791 (2022)."},{"key":"e_1_3_3_3_66_1","unstructured":"Zhenhui Ye Jinzheng He Ziyue Jiang Rongjie Huang Jiawei Huang Jinglin Liu Yi Ren Xiang Yin Zejun Ma and Zhou Zhao. 2023a. Geneface++: Generalized and stable real-time audio-driven 3d talking face generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.00787 (2023)."},{"key":"e_1_3_3_3_67_1","unstructured":"Zhenhui Ye Ziyue Jiang Yi Ren Jinglin Liu Jinzheng He and Zhou Zhao. 2023b. Geneface: Generalized and high-fidelity audio-driven 3d talking face synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.13430 (2023)."},{"key":"e_1_3_3_3_68_1","volume-title":"ICLR","author":"Ye Zhenhui","year":"2024","unstructured":"Zhenhui Ye, Tianyun Zhong, Yi Ren, Jiaqi Yang, Weichuang Li, Jiangwei Huang, Ziyue Jiang, Jinzheng He, Rongjie Huang, Jinglin Liu, Chen Zhang, Xiang Yin, Zejun Ma, and Zhou Zhao. 2024. Real3D-Portrait: One-shot Realistic 3D Talking Portrait Synthesis. In ICLR."},{"key":"e_1_3_3_3_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00053"},{"key":"e_1_3_3_3_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00632"},{"key":"e_1_3_3_3_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681675"},{"key":"e_1_3_3_3_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00836"},{"key":"e_1_3_3_3_73_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657413"},{"key":"e_1_3_3_3_74_1","doi-asserted-by":"crossref","unstructured":"Yang Zhou Xintong Han Eli Shechtman Jose Echevarria Evangelos Kalogerakis and Dingzeyu Li. 2020. Makelttalk: speaker-aware talking-head animation. ACM TOG (2020).","DOI":"10.1145\/3414685.3417774"}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763854","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:29:49Z","timestamp":1765250989000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763854"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":73,"alternative-id":["10.1145\/3757377.3763854","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763854","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}