{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T22:28:34Z","timestamp":1784413714247,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":75,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,12,10]],"date-time":"2023-12-10T00:00:00Z","timestamp":1702166400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100000781","name":"European Research Council","doi-asserted-by":"publisher","award":["860768"],"award-info":[{"award-number":["860768"]}],"id":[{"id":"10.13039\/501100000781","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,12,10]]},"DOI":"10.1145\/3610548.3618183","type":"proceedings-article","created":{"date-parts":[[2023,12,11]],"date-time":"2023-12-11T12:28:40Z","timestamp":1702297720000},"page":"1-13","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":87,"title":["Emotional Speech-Driven Animation with Content-Emotion Disentanglement"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1651-030X","authenticated-orcid":false,"given":"Radek","family":"Dan\u011b\u010dek","sequence":"first","affiliation":[{"name":"Max Planck Institute for Intelligent Systems, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7414-845X","authenticated-orcid":false,"given":"Kiran","family":"Chhatre","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0776-5124","authenticated-orcid":false,"given":"Shashank","family":"Tripathi","sequence":"additional","affiliation":[{"name":"Max Planck Institute for Intelligent Systems, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6330-7438","authenticated-orcid":false,"given":"Yandong","family":"Wen","sequence":"additional","affiliation":[{"name":"Max Planck Institute for Intelligent Systems, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6077-4540","authenticated-orcid":false,"given":"Michael","family":"Black","sequence":"additional","affiliation":[{"name":"Max Planck Institute for Intelligent Systems, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3829-3924","authenticated-orcid":false,"given":"Timo","family":"Bolkart","sequence":"additional","affiliation":[{"name":"Max Planck Institute for Intelligent Systems, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,12,11]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2889052"},{"key":"e_1_3_2_2_2_1","volume-title":"LRS3-TED: a large-scale dataset for visual speech recognition. CoRR abs\/1809.00496","author":"Afouras Triantafyllos","year":"2018","unstructured":"Triantafyllos Afouras, Joon\u00a0Son Chung, and Andrew Zisserman. 2018. LRS3-TED: a large-scale dataset for visual speech recognition. CoRR abs\/1809.00496 (2018). arXiv:1809.00496http:\/\/arxiv.org\/abs\/1809.00496"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3130800.3130818"},{"key":"e_1_3_2_2_4_1","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in Neural Information Processing Systems 33 (2020), 12449\u201312460.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1208"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.116"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2014.2336244"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/1095878.1095881"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.14641"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00802"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3550469.3555399"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00821"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"e_1_3_2_2_14_1","volume-title":"AVSP 2001","author":"Cohen M.","year":"2001","unstructured":"Michael\u00a0M. Cohen, Rashid Clark, and Dominic\u00a0W. Massaro. 2001. Animated speech: research progress and applications. In Auditory-Visual Speech Processing, AVSP 2001, Aalborg, Denmark, September 7-9, 2001, Dominic\u00a0W. Massaro, Joanna Light, and Kristin Geraci (Eds.). ISCA, 200. http:\/\/www.isca-speech.org\/archive_open\/avsp01\/av01_200c.html"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01034"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01967"},{"key":"e_1_3_2_2_17_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 568\u2013577","author":"Apolito Stefano","year":"2021","unstructured":"Stefano d\u2019Apolito, Danda\u00a0Pani Paudel, Zhiwu Huang, Andres Romero, and Luc Van\u00a0Gool. 2021. GANmut: Learning Interpretable Conditional Space for Gamut of Emotions. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 568\u2013577."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2019.00038"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/n19-1423"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12277"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/2897824.2925984"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3388767.3407339"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3395208"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01821"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2010.2052239"},{"key":"e_1_3_2_2_26_1","article-title":"Learning an Animatable Detailed 3D Face Model from In-the-Wild Images","volume":"40","author":"Feng Yao","year":"2021","unstructured":"Yao Feng, Haiwen Feng, Michael\u00a0J. Black, and Timo Bolkart. 2021. Learning an Animatable Detailed 3D Face Model from In-the-Wild Images. Transactions on Graphics, (Proc. SIGGRAPH) 40, 4 (2021), 88:1\u201388:13.","journal-title":"Transactions on Graphics, (Proc. SIGGRAPH)"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Panagiotis\u00a0P. Filntisis George Retsinas Foivos Paraperas-Papantoniou Athanasios Katsamanis Anastasios Roussos and Petros Maragos. 2022. Visual Speech-Aware Perceptual 3D Facial Expression Reconstruction from Videos.","DOI":"10.1109\/CVPRW59228.2023.00609"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00874"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01386"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073658"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3355089.3356500"},{"key":"e_1_3_2_2_33_1","volume-title":"Kingma and Jimmy Ba","author":"P.","year":"2015","unstructured":"Diederik\u00a0P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings, Yoshua Bengio and Yann LeCun (Eds.). http:\/\/arxiv.org\/abs\/1412.6980"},{"key":"e_1_3_2_2_34_1","volume-title":"Kingma and Max Welling","author":"P.","year":"2014","unstructured":"Diederik\u00a0P. Kingma and Max Welling. 2014. Auto-Encoding Variational Bayes. (2014). http:\/\/arxiv.org\/abs\/1312.6114"},{"key":"e_1_3_2_2_35_1","article-title":"Learning a model of facial shape and expression from 4D scans","volume":"36","author":"Li Tianye","year":"2017","unstructured":"Tianye Li, Timo Bolkart, Michael.\u00a0J. Black, Hao Li, and Javier Romero. 2017. Learning a model of facial shape and expression from 4D scans. Transactions on Graphics, (Proc. SIGGRAPH Asia) 36, 6 (2017), 194:1\u2013194:17.","journal-title":"Transactions on Graphics, (Proc. SIGGRAPH Asia)"},{"key":"e_1_3_2_2_36_1","unstructured":"Alexandra Lindt Pablo V.\u00a0A. Barros Henrique Siqueira and Stefan Wermter. 2020. Facial Expression Editing with Continuous Emotion Labels. CoRR abs\/2006.12210. arXiv:2006.12210https:\/\/arxiv.org\/abs\/2006.12210"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391"},{"key":"e_1_3_2_2_38_1","volume-title":"MediaPipe: A Framework for Building Perception Pipelines. CoRR abs\/1906.08172","author":"Lugaresi Camillo","year":"2019","unstructured":"Camillo Lugaresi, Jiuqiang Tang, Hadon Nash, Chris McClanahan, Esha Uboweja, Michael Hays, Fan Zhang, Chuo-Ling Chang, Ming\u00a0Guang Yong, Juhyun Lee, Wan-Teh Chang, Wei Hua, Manfred Georg, and Matthias Grundmann. 2019. MediaPipe: A Framework for Building Perception Pipelines. CoRR abs\/1906.08172 (2019). arXiv:1906.08172http:\/\/arxiv.org\/abs\/1906.08172"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2017.2740923"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","unstructured":"Arsha Nagrani Joon\u00a0Son Chung and Andrew Zisserman. 2017. VoxCeleb: A Large-Scale Speaker Identification Dataset. (2017) 2616\u20132620. https:\/\/doi.org\/10.21437\/Interspeech.2017-950","DOI":"10.21437\/Interspeech.2017-950"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01975"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01822"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2303.11089"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.287"},{"key":"e_1_3_2_2_45_1","volume-title":"End-to-end Learning for 3D Facial Animation from Raw Waveforms of Speech. CoRR abs\/1710.00920","author":"Pham Hai\u00a0Xuan","year":"2017","unstructured":"Hai\u00a0Xuan Pham, Yuting Wang, and Vladimir Pavlovic. 2017b. End-to-end Learning for 3D Facial Animation from Raw Waveforms of Speech. CoRR abs\/1710.00920 (2017). arXiv:1710.00920http:\/\/arxiv.org\/abs\/1710.00920"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/p19-1050"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413532"},{"key":"e_1_3_2_2_48_1","volume-title":"Accelerating 3D Deep Learning with PyTorch3D. CoRR abs\/2007.08501","author":"Ravi Nikhila","year":"2020","unstructured":"Nikhila Ravi, Jeremy Reizenstein, David Novotn\u00fd, Taylor Gordon, Wan-Yen Lo, Justin Johnson, and Georgia Gkioxari. 2020. Accelerating 3D Deep Learning with PyTorch3D. CoRR abs\/2007.08501 (2020). arXiv:2007.08501https:\/\/arxiv.org\/abs\/2007.08501"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00121"},{"key":"e_1_3_2_2_50_1","volume-title":"FaceForensics: A Large-scale Video Dataset for Forgery Detection in Human Faces. CoRR abs\/1803.09179","author":"R\u00f6ssler Andreas","year":"2018","unstructured":"Andreas R\u00f6ssler, Davide Cozzolino, Luisa Verdoliva, Christian Riess, Justus Thies, and Matthias Nie\u00dfner. 2018. FaceForensics: A Large-scale Video Dataset for Forgery Detection in Human Faces. CoRR abs\/1803.09179 (2018). arXiv:1803.09179http:\/\/arxiv.org\/abs\/1803.09179"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.12603"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00795"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58555-6_4"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073640"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073699"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.2312\/SCA\/SCA12\/275-284"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01107"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00270"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.401"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2301.00023"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58517-4_42"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","unstructured":"Soumya Tripathy Juho Kannala and Esa Rahtu. 2020. ICface: Interpretable and Controllable Face Reenactment Using GANs. (2020) 3374\u20133383. https:\/\/doi.org\/10.1109\/WACV45572.2020.9093474","DOI":"10.1109\/WACV45572.2020.9093474"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00137"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58589-1_42"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2207.11243"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","unstructured":"Jinbo Xing Menghan Xia Yuechen Zhang Xiaodong Cun Jue Wang and Tien-Tsin Wong. 2023. CodeTalker: Speech-Driven 3D Facial Animation with Discrete Motion Prior. (2023) 12780\u201312790. https:\/\/doi.org\/10.1109\/CVPR52729.2023.01229","DOI":"10.1109\/CVPR52729.2023.01229"},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/2522628.2522904"},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00068"},{"key":"e_1_3_2_2_69_1","volume-title":"MOSI: Multimodal Corpus of Sentiment Intensity and Subjectivity Analysis in Online Opinion Videos. CoRR abs\/1606.06259","author":"Zadeh Amir","year":"2016","unstructured":"Amir Zadeh, Rowan Zellers, Eli Pincus, and Louis-Philippe Morency. 2016. MOSI: Multimodal Corpus of Sentiment Intensity and Subjectivity Analysis in Online Opinion Videos. CoRR abs\/1606.06259 (2016). arXiv:1606.06259http:\/\/arxiv.org\/abs\/1606.06259"},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00955"},{"key":"e_1_3_2_2_71_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2019.02.025"},{"key":"e_1_3_2_2_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201292"},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20071-7_38"},{"key":"e_1_3_2_2_74_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19778-9_15"},{"key":"e_1_3_2_2_75_1","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.13382"}],"event":{"name":"SA '23: SIGGRAPH Asia 2023","location":"Sydney NSW Australia","acronym":"SA '23","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["SIGGRAPH Asia 2023 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610548.3618183","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3610548.3618183","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T09:34:01Z","timestamp":1755768841000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610548.3618183"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,10]]},"references-count":75,"alternative-id":["10.1145\/3610548.3618183","10.1145\/3610548"],"URL":"https:\/\/doi.org\/10.1145\/3610548.3618183","relation":{},"subject":[],"published":{"date-parts":[[2023,12,10]]},"assertion":[{"value":"2023-12-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}