{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T16:19:01Z","timestamp":1783095541628,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":58,"publisher":"ACM","funder":[{"name":"National Science Foundation of China","award":["62472285"],"award-info":[{"award-number":["62472285"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754933","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"7538-7547","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["MEDTalk: Multimodal Controlled 3D Facial Animation with Dynamic Emotions by Disentangled Embedding"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-1379-2583","authenticated-orcid":false,"given":"Chang","family":"Liu","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0355-989X","authenticated-orcid":false,"given":"Ye","family":"Pan","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6079-0688","authenticated-orcid":false,"given":"Chenyang","family":"Ding","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0831-6934","authenticated-orcid":false,"given":"Susanto","family":"Rahardja","sequence":"additional","affiliation":[{"name":"Singapore Institute of Technology, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4029-3322","authenticated-orcid":false,"given":"Xiaokang","family":"Yang","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","article-title":"A Review of Virtual Reality Applications in an Educational Domain","volume":"15","author":"Farsi Ghaliya Al","year":"2021","unstructured":"Ghaliya Al Farsi, Azmi bin Mohd Yusof, Awanis Romli, Ragad M Tawafak, Sohail Iqbal Malik, Jasiya Jabbar, and Mohd Ezanee Bin Rsuli. 2021. A Review of Virtual Reality Applications in an Educational Domain. International Journal of Interactive Mobile Technologies, Vol. 15, 22 (2021).","journal-title":"International Journal of Interactive Mobile Technologies"},{"key":"e_1_3_2_1_2_1","volume-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems, Vol. 33 (2020), 12449-12460."},{"key":"e_1_3_2_1_3_1","volume-title":"Emova: Empowering language models to see, hear and speak with vivid emotions. arXiv preprint arXiv:2409.18042","author":"Chen Kai","year":"2024","unstructured":"Kai Chen, Yunhao Gou, Runhui Huang, Zhili Liu, Daxin Tan, Jing Xu, Chunwei Wang, Yi Zhu, Yihan Zeng, Kuo Yang, et al., 2024. Emova: Empowering language models to see, hear and speak with vivid emotions. arXiv preprint arXiv:2409.18042 (2024)."},{"key":"e_1_3_2_1_4_1","volume-title":"Diffusiontalker: Personalization and acceleration for speech-driven 3d face diffuser. arXiv preprint arXiv:2311.16565","author":"Chen Peng","year":"2023","unstructured":"Peng Chen, Xiaobao Wei, Ming Lu, Yitong Zhu, Naiming Yao, Xingyu Xiao, and Hui Chen. 2023. Diffusiontalker: Personalization and acceleration for speech-driven 3d face diffuser. arXiv preprint arXiv:2311.16565 (2023)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"e_1_3_2_1_6_1","volume-title":"International conference on machine learning. PMLR, 1779-1788","author":"Cheng Pengyu","year":"2020","unstructured":"Pengyu Cheng, Weituo Hao, Shuyang Dai, Jiachang Liu, Zhe Gan, and Lawrence Carin. 2020. Club: A contrastive log-ratio upper bound of mutual information. In International conference on machine learning. PMLR, 1779-1788."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01034"},{"key":"e_1_3_2_1_8_1","volume-title":"Revisiting pre-trained models for Chinese natural language processing. arXiv preprint arXiv:2004.13922","author":"Cui Yiming","year":"2020","unstructured":"Yiming Cui, Wanxiang Che, Ting Liu, Bing Qin, Shijin Wang, and Guoping Hu. 2020. Revisiting pre-trained models for Chinese natural language processing. arXiv preprint arXiv:2004.13922 (2020)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618183"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2897824.2925984"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01821"},{"key":"e_1_3_2_1_12_1","unstructured":"Chaoyou Fu Haojia Lin Xiong Wang Yi-Fan Zhang Yunhang Shen Xiaoyu Liu Haoyu Cao Zuwei Long Heting Gao Ke Li et al. 2025. Vita-1.5: Towards gpt-4o level real-time vision and speech interaction. arXiv preprint arXiv:2501.01957 (2025)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02069"},{"key":"e_1_3_2_1_14_1","volume-title":"International conference on machine learning. PMLR, 1180-1189","author":"Ganin Yaroslav","year":"2015","unstructured":"Yaroslav Ganin and Victor Lempitsky. 2015. Unsupervised domain adaptation by backpropagation. In International conference on machine learning. PMLR, 1180-1189."},{"key":"e_1_3_2_1_15_1","volume-title":"Zero-shot synthesis with group-supervised learning. arXiv preprint arXiv:2009.06586","author":"Ge Yunhao","year":"2020","unstructured":"Yunhao Ge, Sami Abu-El-Haija, Gan Xin, and Laurent Itti. 2020. Zero-shot synthesis with group-supervised learning. arXiv preprint arXiv:2009.06586 (2020)."},{"key":"e_1_3_2_1_16_1","unstructured":"Google. 2025. Gemini AI. https:\/\/gemini.google.com\/app Accessed: 2025-03-19."},{"key":"e_1_3_2_1_17_1","unstructured":"Tianshun Han Shengnan Gui Yiqing Huang Baihui Li Lijian Liu Benjia Zhou Ning Jiang Quan Lu Ruicong Zhi Yanyan Liang et al. 2024. PMMTalk : Speech-Driven 3D Facial Animation from Complementary Pseudo Multi-modal Features. IEEE Transactions on Multimedia (2024)."},{"key":"e_1_3_2_1_18_1","volume-title":"Denoising diffusion probabilistic models. Advances in neural information processing systems","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems, Vol. 33 (2020), 6840-6851."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01386"},{"key":"e_1_3_2_1_20_1","volume-title":"Geon Kim, and Youngjae Yu.","author":"Kim Jisoo","year":"2024","unstructured":"Jisoo Kim, Jungbin Cho, Joonho Park, Soonmin Hwang, Da Eun Kim, Geon Kim, and Youngjae Yu. 2024. DEEPTalk: Dynamic Emotion Embedding for Probabilistic Speech-Driven 3D Face Animation. arXiv preprint arXiv:2408.06010 (2024)."},{"key":"e_1_3_2_1_21_1","volume-title":"A survey on applications of digital human avatars toward virtual co-presence. arXiv preprint arXiv:2201.04168","author":"Korban Matthew","year":"2022","unstructured":"Matthew Korban and Xin Li. 2022. A survey on applications of digital human avatars toward virtual co-presence. arXiv preprint arXiv:2201.04168 (2022)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/VR58804.2024.00060"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391"},{"key":"e_1_3_2_1_24_1","volume-title":"Juhyun Lee, et al.","author":"Lugaresi Camillo","year":"2019","unstructured":"Camillo Lugaresi, Jiuqiang Tang, Hadon Nash, Chris McClanahan, Esha Uboweja, Michael Hays, Fan Zhang, Chuo-Ling Chang, Ming Guang Yong, Juhyun Lee, et al., 2019. Mediapipe: A framework for building perception pipelines. arXiv preprint arXiv:1906.08172 (2019)."},{"key":"e_1_3_2_1_25_1","volume-title":"Talkclip: Talking head generation with text-guided expressive speaking styles. arXiv preprint arXiv:2304.00334","author":"Ma Yifeng","year":"2023","unstructured":"Yifeng Ma, Suzhen Wang, Yu Ding, Bowen Ma, Tangjie Lv, Changjie Fan, Zhipeng Hu, Zhidong Deng, and Xin Yu. 2023a. Talkclip: Talking head generation with text-guided expressive speaking styles. arXiv preprint arXiv:2304.00334 (2023)."},{"key":"e_1_3_2_1_26_1","volume-title":"emotion2vec: Self-supervised pre-training for speech emotion representation. arXiv preprint arXiv:2312.15185","author":"Ma Ziyang","year":"2023","unstructured":"Ziyang Ma, Zhisheng Zheng, Jiaxin Ye, Jinchao Li, Zhifu Gao, Shiliang Zhang, and Xie Chen. 2023b. emotion2vec: Self-supervised pre-training for speech emotion representation. arXiv preprint arXiv:2312.15185 (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.5555\/1324818"},{"key":"e_1_3_2_1_28_1","volume-title":"EmoVOCA: Speech-Driven Emotional 3D Talking Heads. arXiv preprint arXiv:2403.12886","author":"Nocentini Federico","year":"2024","unstructured":"Federico Nocentini, Claudio Ferrari, and Stefano Berretti. 2024. EmoVOCA: Speech-Driven Emotional 3D Talking Heads. arXiv preprint arXiv:2403.12886 (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"VASA-Rig: Audio-Driven 3D Facial Animation with 'Live'Mood Dynamics in Virtual Reality","author":"Pan Ye","year":"2025","unstructured":"Ye Pan, Chang Liu, Sicheng Xu, Shuai Tan, and Jiaolong Yang. 2025. VASA-Rig: Audio-Driven 3D Facial Animation with 'Live'Mood Dynamics in Virtual Reality. IEEE Transactions on Visualization and Computer Graphics (2025)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2023.3247101"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611734"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01891"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.287"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3242969.3243017"},{"key":"e_1_3_2_1_35_1","volume-title":"International conference on machine learning. PmLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PmLR, 8748-8763."},{"key":"e_1_3_2_1_36_1","volume-title":"International conference on machine learning. PMLR, 28492-28518","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International conference on machine learning. PMLR, 28492-28518."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00121"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681359"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3623264.3624447"},{"key":"e_1_3_2_1_40_1","volume-title":"AVI-Talking: Learning Audio-Visual Instructions for Expressive 3D Talking Face Generation","author":"Sun Yasheng","year":"2024","unstructured":"Yasheng Sun, Wenqing Chu, Hang Zhou, Kaisiyuan Wang, and Hideki Koike. 2024a. AVI-Talking: Learning Audio-Visual Instructions for Expressive 3D Talking Face Generation. IEEE Access (2024)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3658221"},{"key":"e_1_3_2_1_42_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision.","author":"Tan Shuai","year":"2025","unstructured":"Shuai Tan, Bill Gong, Bin Ji, and Ye Pan. 2025a. FixTalk: Taming Identity Leakage for High-Quality Talking Head Generation in Extreme Cases. In Proceedings of the IEEE\/CVF International Conference on Computer Vision."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72658-3_23"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i5.28314"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02024"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02486"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i5.28313"},{"key":"e_1_3_2_1_48_1","volume-title":"Salmonn: Towards generic hearing abilities for large language models. arXiv preprint arXiv:2310.13289","author":"Tang Changli","year":"2023","unstructured":"Changli Tang, Wenyi Yu, Guangzhi Sun, Xianzhao Chen, Tian Tan, Wei Li, Lu Lu, Zejun Ma, and Chao Zhang. 2023. Salmonn: Towards generic hearing abilities for large language models. arXiv preprint arXiv:2310.13289 (2023)."},{"key":"e_1_3_2_1_49_1","volume-title":"Ryan Burnell, Libin Bai, Anmol Gulati, Garrett Tanzer, Damien Vincent, Zhufeng Pan, Shibo Wang, et al.","author":"Team Gemini","year":"2024","unstructured":"Gemini Team, Petko Georgiev, Ving Ian Lei, Ryan Burnell, Libin Bai, Anmol Gulati, Garrett Tanzer, Damien Vincent, Zhufeng Pan, Shibo Wang, et al., 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv preprint arXiv:2403.05530 (2024)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413543"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681366"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01229"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00639"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"SiCheng Yang Methawee Tantrawenith Haolin Zhuang Zhiyong Wu Aolan Sun Jianzong Wang Ning Cheng Huaizhen Tang Xintao Zhao Jie Wang et al. 2022. Speech representation disentanglement with adversarial mutual information learning for one-shot voice conversion. arXiv preprint arXiv:2208.08757 (2022).","DOI":"10.21437\/Interspeech.2022-571"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657413"},{"key":"e_1_3_2_1_56_1","volume-title":"Breathing Life into Faces: Speech-driven 3D Facial Animation with Natural Head Pose and Detailed Shape. arXiv preprint arXiv:2310.20240","author":"Zhao Wei","year":"2023","unstructured":"Wei Zhao, Yijun Wang, Tianyu He, Lianying Yin, Jianxin Lin, and Xin Jin. 2023. Breathing Life into Faces: Speech-driven 3D Facial Animation with Natural Head Pose and Detailed Shape. arXiv preprint arXiv:2310.20240 (2023)."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i7.28594"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3197517.3201292","article-title":"Visemenet: Audio-driven animator-centric speech animation","volume":"37","author":"Zhou Yang","year":"2018","unstructured":"Yang Zhou, Zhan Xu, Chris Landreth, Evangelos Kalogerakis, Subhransu Maji, and Karan Singh. 2018. Visemenet: Audio-driven animator-centric speech animation. ACM Transactions on Graphics (ToG), Vol. 37, 4 (2018), 1-10.","journal-title":"ACM Transactions on Graphics (ToG)"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754933","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:16:25Z","timestamp":1765340185000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754933"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":58,"alternative-id":["10.1145\/3746027.3754933","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754933","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}