{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:19:12Z","timestamp":1778048352587,"version":"3.51.4"},"reference-count":46,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00134","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"1314-1323","source":"Crossref","is-referenced-by-count":0,"title":["Narrating For You: Prompt-guided Audio-visual Narrating Face Generation Employing Multi-entangled Latent Space"],"prefix":"10.1109","author":[{"given":"Aashish Chandra","family":"K","sequence":"first","affiliation":[{"name":"BITS Pilani,Machine Intelligence Group,Department of CS&amp;IS,India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aashutosh A","family":"V","sequence":"additional","affiliation":[{"name":"BITS Pilani,Machine Intelligence Group,Department of CS&amp;IS,India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Abhijit","family":"Das","sequence":"additional","affiliation":[{"name":"BITS Pilani,Machine Intelligence Group,Department of CS&amp;IS,India"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.393"},{"key":"ref4","author":"Baevski","year":"2020","journal-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations"},{"key":"ref5","author":"Betker","year":"2022","journal-title":"Tortoise text-to-speech"},{"key":"ref6","author":"Casanova","year":"2023","journal-title":"Yourtts: Towards zero-shot multi-speaker tts and zeroshot voice conversion for everyone"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-2016"},{"key":"ref8","article-title":"Anifacediff: High-fidelity face reenactment via facial parametric conditioned diffusion models","author":"Chen","year":"2024"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3395208"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02069"},{"key":"ref11","author":"Ho","year":"2020","journal-title":"Denoising diffusion probabilistic models"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1016"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00842"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351066"},{"key":"ref15","author":"Khalid","year":"2022","journal-title":"Fakeavceleb: A novel audio-video multimodal deep-fake dataset"},{"key":"ref16","author":"Kim","year":"2021","journal-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1561\/2200000056"},{"key":"ref18","author":"Kingma","year":"2022","journal-title":"Auto-encoding variational bayes"},{"key":"ref19","author":"Kong","year":"2020","journal-title":"Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2019.101027"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413532"},{"key":"ref22","author":"Radford","year":"2022","journal-title":"Robust speech recognition via large-scale weak supervision"},{"key":"ref23","author":"Raina","year":"2022","journal-title":"Syncnet: Using causal convolutions and correlating objective for time delay estimation in audio signals"},{"key":"ref24","author":"Ren","year":"2019","journal-title":"Fastspeech: Fast, robust and controllable text to speech"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01350"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref29","author":"Siarohin","year":"2020","journal-title":"First order motion model for image animation"},{"key":"ref30","author":"Song","year":"2022","journal-title":"Denoising diffusion implicit models"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/wacv57701.2024.00502"},{"key":"ref32","author":"van den Oord","year":"2016","journal-title":"Wavenet: A generative model for raw audio"},{"key":"ref33","author":"Vaswani","year":"2023","journal-title":"Attention is all you need"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/152"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2024.3449075"},{"key":"ref37","article-title":"Text-to-video: a two-stage framework for zero-shot identity-agnostic talking-head generation","author":"Wang","year":"2023"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00129"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1055\/s-0033-1343591"},{"key":"ref40","author":"Xu","year":"2024","journal-title":"Vasa-1: Lifelike audio-driven talking faces generated in real time"},{"key":"ref41","author":"Zhang","year":"2023","journal-title":"Dream-talk: Diffusion-based realistic emotional audio-driven method for single image talking face generation"},{"key":"ref42","author":"Zhang","year":"2020","journal-title":"Mediapip. hands: On-device real-time hand tracking"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747380"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00836"},{"key":"ref45","author":"Zhang","year":"2019","journal-title":"One-shot face reenactment"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00366"},{"key":"ref47","author":"Zhu","year":"2022","journal-title":"Celebvhq: A large-scale video facial attributes dataset"},{"key":"ref48","author":"Zouhar","year":"2024","journal-title":"A formal perspective on byte-pair encoding"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492660.pdf?arnumber=11492660","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T05:58:57Z","timestamp":1778047137000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492660\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":46,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00134","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}