{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:20:24Z","timestamp":1778048424105,"version":"3.51.4"},"reference-count":43,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00442","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"4546-4555","source":"Crossref","is-referenced-by-count":0,"title":["SynchroRaMa : Lip-Synchronized and Emotion-Aware Talking Face Generation via Multi-Modal Emotion Embedding"],"prefix":"10.1109","author":[{"given":"Phyo Thet","family":"Yee","sequence":"first","affiliation":[{"name":"IIT Ropar,India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dimitrios","family":"Kollias","sequence":"additional","affiliation":[{"name":"Queen Mary University of London,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sudeepta","family":"Mishra","sequence":"additional","affiliation":[{"name":"IIT Ropar,India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Abhinav","family":"Dhall","sequence":"additional","affiliation":[{"name":"Monash University,Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i3.32241"},{"key":"ref2","article-title":"Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms","author":"Cheng","year":"2024"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.01964"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2019.00038"},{"key":"ref5","first-page":"8780","article-title":"Diffusion models beat gans on image synthesis","volume":"34","author":"Dhariwal","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547838"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00812"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3422622"},{"key":"ref9","volume-title":"Emotion english distilroberta-base","author":"Hartmann","year":"2022"},{"key":"ref10","author":"He","year":"2024","journal-title":"Gaia: Zero-shot talking avatar generation"},{"key":"ref11","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref12","first-page":"8153","article-title":"Animate anyone: Consistent and controllable image-to-video synthesis for character animation","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Hu"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448489"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351066"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681198"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02104"},{"key":"ref17","article-title":"Mediapipe: A framework for building perception pipelines","volume":"abs\/1906.08172","author":"Lugaresi","year":"2019","journal-title":"CoRR"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687587"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00521"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413532"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413532"},{"key":"ref22","author":"Radford","year":"2022","journal-title":"Robust speech recognition via large-scale weak supervision"},{"key":"ref23","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref24","first-page":"13963","article-title":"Portaspeech: Portable and high-quality generative text-to-speech","volume":"34","author":"Ren","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00197"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/3dv66043.2025.00071"},{"key":"ref28","first-page":"244","article-title":"Emo: Emote portrait alive generating expressive portrait videos with audio2video diffusion model under weak conditions","volume-title":"European Conference on Computer Vision","author":"Tian"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3263585"},{"key":"ref30","article-title":"V-express: Conditional dropout for progressive training of portrait video generation","author":"Wang","year":"2024"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58589-1_42"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2003.819861"},{"key":"ref33","article-title":"Aniportrait: Audio-driven synthesis of photorealistic portrait animation","author":"Wei","year":"2024"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW56347.2022.00081"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657459"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1055\/s-0033-1343591"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0021"},{"key":"ref38","article-title":"Geneface++: Generalized and stable real-time audio-driven 3d talking face generation","author":"Ye","year":"2023"},{"key":"ref39","article-title":"Geneface: Generalized and high-fidelity audio-driven 3d talking face synthesis","author":"Ye","year":"2023"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/FG59268.2024.10581924"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TBIOM.2025.3576111"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00366"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492133.pdf?arnumber=11492133","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:04:07Z","timestamp":1778047447000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492133\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":43,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00442","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}