{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T16:48:53Z","timestamp":1742921333684,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":47,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819609161"},{"type":"electronic","value":"9789819609178"}],"license":[{"start":{"date-parts":[[2024,12,8]],"date-time":"2024-12-08T00:00:00Z","timestamp":1733616000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,8]],"date-time":"2024-12-08T00:00:00Z","timestamp":1733616000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-0917-8_8","type":"book-chapter","created":{"date-parts":[[2024,12,7]],"date-time":"2024-12-07T07:57:20Z","timestamp":1733558240000},"page":"131-147","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["EmoTalker: Audio Driven Emotion Aware Talking Head Generation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6284-520X","authenticated-orcid":false,"given":"Xiaoqian","family":"Shen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3645-603X","authenticated-orcid":false,"given":"Faizan Farooq","family":"Khan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9659-1551","authenticated-orcid":false,"given":"Mohamed","family":"Elhoseiny","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,8]]},"reference":[{"issue":"6","key":"8_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3550454.3555435","volume":"41","author":"T Ao","year":"2022","unstructured":"Ao, T., Gao, Q., Lou, Y., Chen, B., Liu, L.: Rhythmic gesticulator: Rhythm-aware co-speech gesture synthesis with hierarchical neural embeddings. ACM Transactions on Graphics (TOG) 41(6), 1\u201319 (2022)","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"8_CR2","doi-asserted-by":"crossref","unstructured":"Chen, L., Cui, G., Liu, C., Li, Z., Kou, Z., Xu, Y., Xu, C.: Talking-head generation with rhythmic head motion. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IX. pp. 35\u201351. Springer (2020)","DOI":"10.1007\/978-3-030-58545-7_3"},{"key":"8_CR3","doi-asserted-by":"crossref","unstructured":"Chen, L., Li, Z., Maddox, R.K., Duan, Z., Xu, C.: Lip movements generation at a glance. In: Proceedings of the European conference on computer vision (ECCV). pp. 520\u2013535 (2018)","DOI":"10.1007\/978-3-030-01234-2_32"},{"key":"8_CR4","doi-asserted-by":"crossref","unstructured":"Chen, L., Maddox, R.K., Duan, Z., Xu, C.: Hierarchical cross-modal talking face generation with dynamic pixel-wise loss. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp. 7832\u20137841 (2019)","DOI":"10.1109\/CVPR.2019.00802"},{"key":"8_CR5","doi-asserted-by":"crossref","unstructured":"Cheng, K., Cun, X., Zhang, Y., Xia, M., Yin, F., Zhu, M., Wang, X., Wang, J., Wang, N.: Videoretalking: Audio-based lip synchronization for talking head video editing in the wild. In: SIGGRAPH Asia 2022 Conference Papers. pp.\u00a01\u20139 (2022)","DOI":"10.1145\/3550469.3555399"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Chung, J.S., Zisserman, A.: Out of time: automated lip sync in the wild. In: Computer Vision\u2013ACCV 2016 Workshops: ACCV 2016 International Workshops, Taipei, Taiwan, November 20-24, 2016, Revised Selected Papers, Part II 13. pp. 251\u2013263. Springer (2017)","DOI":"10.1007\/978-3-319-54427-4_19"},{"key":"8_CR7","doi-asserted-by":"crossref","unstructured":"Deng, Y., Yang, J., Xu, S., Chen, D., Jia, Y., Tong, X.: Accurate 3d face reconstruction with weakly-supervised learning: From single image to image set. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops. pp.\u00a00\u20130 (2019)","DOI":"10.1109\/CVPRW.2019.00038"},{"key":"8_CR8","doi-asserted-by":"crossref","unstructured":"Esser, P., Rombach, R., Ommer, B.: Taming transformers for high-resolution image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp. 12873\u201312883 (2021)","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Fan, Y., Lin, Z., Saito, J., Wang, W., Komura, T.: Faceformer: Speech-driven 3d facial animation with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 18770\u201318780 (2022)","DOI":"10.1109\/CVPR52688.2022.01821"},{"key":"8_CR10","doi-asserted-by":"crossref","unstructured":"Guo, Y., Chen, K., Liang, S., Liu, Y.J., Bao, H., Zhang, J.: Ad-nerf: Audio driven neural radiance fields for talking head synthesis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 5784\u20135794 (2021)","DOI":"10.1109\/ICCV48922.2021.00573"},{"key":"8_CR11","doi-asserted-by":"crossref","unstructured":"Gururani, S., Mallya, A., Wang, T.C., Valle, R., Liu, M.Y.: Spacex: Speech-driven portrait animation with controllable expression. arXiv preprint arXiv:2211.09809 (2022)","DOI":"10.1109\/ICCV51070.2023.01912"},{"key":"8_CR12","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"WN Hsu","year":"2021","unstructured":"Hsu, W.N., Bolte, B., Tsai, Y.H.H., Lakhotia, K., Salakhutdinov, R., Mohamed, A.: Hubert: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Transactions on Audio, Speech, and Language Processing 29, 3451\u20133460 (2021)","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"8_CR13","doi-asserted-by":"crossref","unstructured":"Ji, X., Zhou, H., Wang, K., Wu, Q., Wu, W., Xu, F., Cao, X.: Eamm: One-shot emotional talking face via audio-based emotion-aware motion model. In: ACM SIGGRAPH 2022 Conference Proceedings. pp. 1\u201310 (2022)","DOI":"10.1145\/3528233.3530745"},{"key":"8_CR14","doi-asserted-by":"crossref","unstructured":"Ji, X., Zhou, H., Wang, K., Wu, W., Loy, C.C., Cao, X., Xu, F.: Audio-driven emotional video portraits. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp. 14080\u201314089 (2021)","DOI":"10.1109\/CVPR46437.2021.01386"},{"key":"8_CR15","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"8_CR16","unstructured":"KR, P., Mukhopadhyay, R., Philip, J., Jha, A., Namboodiri, V., Jawahar, C.: Towards automatic face-to-face translation. In: Proceedings of the 27th ACM international conference on multimedia. pp. 1428\u20131436 (2019)"},{"key":"8_CR17","doi-asserted-by":"crossref","unstructured":"Kreuk, F., Polyak, A., Copet, J., Kharitonov, E., Nguyen, T.A., Rivi\u00e8re, M., Hsu, W.N., Mohamed, A., Dupoux, E., Adi, Y.: Textless speech emotion conversion using discrete and decomposed representations. arXiv preprint arXiv:2111.07402 (2021)","DOI":"10.18653\/v1\/2022.emnlp-main.769"},{"key":"8_CR18","first-page":"1336","volume":"9","author":"K Lakhotia","year":"2021","unstructured":"Lakhotia, K., Kharitonov, E., Hsu, W.N., Adi, Y., Polyak, A., Bolte, B., Nguyen, T.A., Copet, J., Baevski, A., Mohamed, A., et al.: On generative spoken language modeling from raw audio. Transactions of the Association for Computational Linguistics 9, 1336\u20131354 (2021)","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"8_CR19","doi-asserted-by":"crossref","unstructured":"Liang, B., Pan, Y., Guo, Z., Zhou, H., Hong, Z., Han, X., Han, J., Liu, J., Ding, E., Wang, J.: Expressive talking head generation with granular audio-visual control. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 3387\u20133396 (2022)","DOI":"10.1109\/CVPR52688.2022.00338"},{"key":"8_CR20","unstructured":"Liu, X., Wu, Q., Zhou, H., Du, Y., Wu, W., Lin, D., Liu, Z.: Audio-driven co-speech gesture video generation. arXiv preprint arXiv:2212.02350 (2022)"},{"key":"8_CR21","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"8_CR22","doi-asserted-by":"crossref","unstructured":"Ma, Y., Wang, S., Hu, Z., Fan, C., Lv, T., Ding, Y., Deng, Z., Yu, X.: Styletalk: One-shot talking head generation with controllable speaking styles. arXiv preprint arXiv:2301.01081 (2023)","DOI":"10.1609\/aaai.v37i2.25280"},{"key":"8_CR23","unstructured":"Van\u00a0der Maaten, L., Hinton, G.: Visualizing data using t-sne. Journal of machine learning research 9(11) (2008)"},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Ng, E., Joo, H., Hu, L., Li, H., Darrell, T., Kanazawa, A., Ginosar, S.: Learning to listen: Modeling non-deterministic dyadic facial motion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 20395\u201320405 (2022)","DOI":"10.1109\/CVPR52688.2022.01975"},{"key":"8_CR25","doi-asserted-by":"crossref","unstructured":"Paysan, P., Knothe, R., Amberg, B., Romdhani, S., Vetter, T.: A 3d face model for pose and illumination invariant face recognition. In: 2009 sixth IEEE international conference on advanced video and signal based surveillance. pp. 296\u2013301. Ieee (2009)","DOI":"10.1109\/AVSS.2009.58"},{"key":"8_CR26","doi-asserted-by":"crossref","unstructured":"Polyak, A., Adi, Y., Copet, J., Kharitonov, E., Lakhotia, K., Hsu, W.N., Mohamed, A., Dupoux, E.: Speech resynthesis from discrete disentangled self-supervised representations. arXiv preprint arXiv:2104.00355 (2021)","DOI":"10.21437\/Interspeech.2021-475"},{"key":"8_CR27","doi-asserted-by":"crossref","unstructured":"Prajwal, K., Mukhopadhyay, R., Namboodiri, V.P., Jawahar, C.: A lip sync expert is all you need for speech to lip generation in the wild. In: Proceedings of the 28th ACM International Conference on Multimedia. pp. 484\u2013492 (2020)","DOI":"10.1145\/3394171.3413532"},{"key":"8_CR28","unstructured":"Press, O., Smith, N.A., Lewis, M.: Train short, test long: Attention with linear biases enables input length extrapolation. arXiv preprint arXiv:2108.12409 (2021)"},{"key":"8_CR29","unstructured":"Ramesh, A., Pavlov, M., Goh, G., Gray, S., Voss, C., Radford, A., Chen, M., Sutskever, I.: Zero-shot text-to-image generation. In: International Conference on Machine Learning. pp. 8821\u20138831. PMLR (2021)"},{"key":"8_CR30","unstructured":"Razavi, A., Van\u00a0den Oord, A., Vinyals, O.: Generating diverse high-fidelity images with vq-vae-2. Advances in neural information processing systems 32 (2019)"},{"key":"8_CR31","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"8_CR32","doi-asserted-by":"crossref","unstructured":"Siqueira, H., Magg, S., Wermter, S.: Efficient facial feature learning with wide ensemble-based convolutional neural networks. In: Proceedings of the AAAI conference on artificial intelligence. vol.\u00a034, pp. 5800\u20135809 (2020)","DOI":"10.1609\/aaai.v34i04.6037"},{"key":"8_CR33","unstructured":"Van Den\u00a0Oord, A., Vinyals, O., et\u00a0al.: Neural discrete representation learning. Advances in neural information processing systems 30 (2017)"},{"key":"8_CR34","doi-asserted-by":"publisher","first-page":"1398","DOI":"10.1007\/s11263-019-01251-8","volume":"128","author":"K Vougioukas","year":"2020","unstructured":"Vougioukas, K., Petridis, S., Pantic, M.: Realistic speech-driven facial animation with gans. Int. J. Comput. Vision 128, 1398\u20131413 (2020)","journal-title":"Int. J. Comput. Vision"},{"key":"8_CR35","doi-asserted-by":"crossref","unstructured":"Wang, K., Wu, Q., Song, L., Yang, Z., Wu, W., Qian, C., He, R., Qiao, Y., Loy, C.C.: Mead: A large-scale audio-visual dataset for emotional talking-face generation. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXI. pp. 700\u2013717. Springer (2020)","DOI":"10.1007\/978-3-030-58589-1_42"},{"key":"8_CR36","doi-asserted-by":"crossref","unstructured":"Wang, S., Li, L., Ding, Y., Fan, C., Yu, X.: Audio2head: Audio-driven one-shot talking-head generation with natural head motion. arXiv preprint arXiv:2107.09293 (2021)","DOI":"10.24963\/ijcai.2021\/152"},{"key":"8_CR37","doi-asserted-by":"crossref","unstructured":"Wiles, O., Koepke, A., Zisserman, A.: X2face: A network for controlling face generation using images, audio, and pose codes. In: Proceedings of the European conference on computer vision (ECCV). pp. 670\u2013686 (2018)","DOI":"10.1007\/978-3-030-01261-8_41"},{"key":"8_CR38","first-page":"4524","volume":"33","author":"W Williams","year":"2020","unstructured":"Williams, W., Ringer, S., Ash, T., MacLeod, D., Dougherty, J., Hughes, J.: Hierarchical quantized autoencoders. Adv. Neural. Inf. Process. Syst. 33, 4524\u20134535 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"8_CR39","doi-asserted-by":"crossref","unstructured":"Wu, H., Jia, J., Wang, H., Dou, Y., Duan, C., Deng, Q.: Imitating arbitrary talking style for realistic audio-driven talking face synthesis. In: Proceedings of the 29th ACM International Conference on Multimedia. pp. 1478\u20131486 (2021)","DOI":"10.1145\/3474085.3475280"},{"key":"8_CR40","doi-asserted-by":"crossref","unstructured":"Xing, J., Xia, M., Zhang, Y., Cun, X., Wang, J., Wong, T.T.: Codetalker: Speech-driven 3d facial animation with discrete motion prior. arXiv preprint arXiv:2301.02379 (2023)","DOI":"10.1109\/CVPR52729.2023.01229"},{"key":"8_CR41","doi-asserted-by":"crossref","unstructured":"Yang, S.w., Chi, P.H., Chuang, Y.S., Lai, C.I.J., Lakhotia, K., Lin, Y.Y., Liu, A.T., Shi, J., Chang, X., Lin, G.T., et\u00a0al.: Superb: Speech processing universal performance benchmark. arXiv preprint arXiv:2105.01051 (2021)","DOI":"10.21437\/Interspeech.2021-1775"},{"key":"8_CR42","unstructured":"Ye, Z., Jiang, Z., Ren, Y., Liu, J., He, J., Zhao, Z.: Geneface: Generalized and high-fidelity audio-driven 3d talking face synthesis. arXiv preprint arXiv:2301.13430 (2023)"},{"key":"8_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, J., Zhang, Y., Cun, X., Huang, S., Zhang, Y., Zhao, H., Lu, H., Shen, X.: T2m-gpt: Generating human motion from textual descriptions with discrete representations. arXiv preprint arXiv:2301.06052 (2023)","DOI":"10.1109\/CVPR52729.2023.01415"},{"key":"8_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, W., Cun, X., Wang, X., Zhang, Y., Shen, X., Guo, Y., Shan, Y., Wang, F.: Sadtalker: Learning realistic 3d motion coefficients for stylized audio-driven single image talking face animation. arXiv preprint arXiv:2211.12194 (2022)","DOI":"10.1109\/CVPR52729.2023.00836"},{"key":"8_CR45","doi-asserted-by":"crossref","unstructured":"Zhou, H., Liu, Y., Liu, Z., Luo, P., Wang, X.: Talking face generation by adversarially disentangled audio-visual representation. In: Proceedings of the AAAI conference on artificial intelligence. vol.\u00a033, pp. 9299\u20139306 (2019)","DOI":"10.1609\/aaai.v33i01.33019299"},{"key":"8_CR46","doi-asserted-by":"crossref","unstructured":"Zhou, H., Sun, Y., Wu, W., Loy, C.C., Wang, X., Liu, Z.: Pose-controllable talking face generation by implicitly modularized audio-visual representation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp. 4176\u20134186 (2021)","DOI":"10.1109\/CVPR46437.2021.00416"},{"issue":"6","key":"8_CR47","first-page":"1","volume":"39","author":"Y Zhou","year":"2020","unstructured":"Zhou, Y., Han, X., Shechtman, E., Echevarria, J., Kalogerakis, E., Li, D.: Makelttalk: speaker-aware talking-head animation. ACM Transactions On Graphics (TOG) 39(6), 1\u201315 (2020)","journal-title":"ACM Transactions On Graphics (TOG)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ACCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-0917-8_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,7]],"date-time":"2024-12-07T08:24:22Z","timestamp":1733559862000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-0917-8_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,8]]},"ISBN":["9789819609161","9789819609178"],"references-count":47,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-0917-8_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,12,8]]},"assertion":[{"value":"8 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ACCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Asian Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hanoi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Vietnam","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"accv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}