{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T19:40:37Z","timestamp":1784058037192,"version":"3.55.0"},"publisher-location":"Cham","reference-count":49,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729720","type":"print"},{"value":"9783031729737","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72973-7_2","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T14:03:04Z","timestamp":1730383384000},"page":"20-36","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":34,"title":["Face-Adapter for\u00a0Pre-trained Diffusion Models with\u00a0Fine-Grained ID and\u00a0Attribute Control"],"prefix":"10.1007","author":[{"given":"Yue","family":"Han","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junwei","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Keke","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xu","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanhao","family":"Ge","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangtai","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiangning","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chengjie","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"2_CR1","doi-asserted-by":"crossref","unstructured":"Agarwal, M., Mukhopadhyay, R., Namboodiri, V.P., Jawahar, C.: Audio-visual face reenactment. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 5178\u20135187 (2023)","DOI":"10.1109\/WACV56688.2023.00515"},{"key":"2_CR2","doi-asserted-by":"crossref","unstructured":"Bounareli, S., Tzelepis, C., Argyriou, V., Patras, I., Tzimiropoulos, G.: Hyperreenact: one-shot reenactment via jointly learning to refine and retarget faces. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7149\u20137159 (2023)","DOI":"10.1109\/ICCV51070.2023.00657"},{"key":"2_CR3","doi-asserted-by":"crossref","unstructured":"Chen, R., Chen, X., Ni, B., Ge, Y.: Simswap: an efficient framework for high fidelity face swapping. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 2003\u20132011 (2020)","DOI":"10.1145\/3394171.3413630"},{"key":"2_CR4","unstructured":"Choi, J., Choi, Y., Kim, Y., Kim, J., Yoon, S.: Custom-edit: Text-guided image editing with customized diffusion models. arXiv preprint arXiv:2305.15779 (2023)"},{"key":"2_CR5","doi-asserted-by":"crossref","unstructured":"Chung, J.S., Nagrani, A., Zisserman, A.: Voxceleb2: Deep speaker recognition. arXiv preprint arXiv:1806.05622 (2018)","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"2_CR6","doi-asserted-by":"crossref","unstructured":"Deng, J., Guo, J., Xue, N., Zafeiriou, S.: Arcface: additive angular margin loss for deep face recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4690\u20134699 (2019)","DOI":"10.1109\/CVPR.2019.00482"},{"key":"2_CR7","doi-asserted-by":"crossref","unstructured":"Deng, Y., Yang, J., Xu, S., Chen, D., Jia, Y., Tong, X.: Accurate 3d face reconstruction with weakly-supervised learning: From single image to image set. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops (2019)","DOI":"10.1109\/CVPRW.2019.00038"},{"key":"2_CR8","unstructured":"Face, H.: Runwayml stable diffusion v1.5. https:\/\/huggingface.co\/runwayml\/stable-diffusion-v1-5, Accessed on: yyyy-mm-dd"},{"key":"2_CR9","unstructured":"Foundations, M.: Openclip: Open-source implementation of clip (2022). https:\/\/github.com\/mlfoundations\/open_clip, Accessed on: yyyy-mm-dd"},{"key":"2_CR10","unstructured":"Gal, R., et al.: An image is worth one word: Personalizing text-to-image generation using textual inversion. arXiv preprint arXiv:2208.01618 (2022)"},{"key":"2_CR11","doi-asserted-by":"crossref","unstructured":"Gao, G., Huang, H., Fu, C., Li, Z., He, R.: Information bottleneck disentanglement for identity swapping. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3404\u20133413 (2021)","DOI":"10.1109\/CVPR46437.2021.00341"},{"key":"2_CR12","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: Gans trained by a two time-scale update rule converge to a local nash equilibrium. Adv. Neural Inform. Process. Syst. 30 (2017)"},{"key":"2_CR13","doi-asserted-by":"crossref","unstructured":"Hong, F.T., Xu, D.: Implicit identity representation conditioned memory compensation network for talking head video generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 23062\u201323072 (2023)","DOI":"10.1109\/ICCV51070.2023.02108"},{"key":"2_CR14","doi-asserted-by":"crossref","unstructured":"Hong, F.T., Zhang, L., Shen, L., Xu, D.: Depth-aware generative adversarial network for talking head video generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3397\u20133406 (2022)","DOI":"10.1109\/CVPR52688.2022.00339"},{"key":"2_CR15","doi-asserted-by":"crossref","unstructured":"Hsu, G.S., Tsai, C.H., Wu, H.Y.: Dual-generator face reenactment. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 642\u2013650 (2022)","DOI":"10.1109\/CVPR52688.2022.00072"},{"key":"2_CR16","unstructured":"Hu, L., Gao, X., Zhang, P., Sun, K., Zhang, B., Bo, L.: Animate anyone: Consistent and controllable image-to-video synthesis for character animation. arXiv preprint arXiv:2311.17117 (2023)"},{"key":"2_CR17","doi-asserted-by":"crossref","unstructured":"Huang, Y., et al.: Curricularface: adaptive curriculum learning loss for deep face recognition. In: proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5901\u20135910 (2020)","DOI":"10.1109\/CVPR42600.2020.00594"},{"key":"2_CR18","unstructured":"Li, L., Bao, J., Yang, H., Chen, D., Wen, F.: Faceshifter: Towards high fidelity and occlusion aware face swapping. arXiv preprint arXiv:1912.13457 (2019)"},{"key":"2_CR19","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Fine-grained face swapping via regional gan inversion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8578\u20138587 (2023)","DOI":"10.1109\/CVPR52729.2023.00829"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Chung, J.S., Zisserman, A.: Voxceleb: a large-scale speaker identification dataset. arXiv preprint arXiv:1706.08612 (2017)","DOI":"10.21437\/Interspeech.2017-950"},{"key":"2_CR21","doi-asserted-by":"crossref","unstructured":"Nirkin, Y., Keller, Y., Hassner, T.: Fsgan: subject agnostic face swapping and reenactment. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 7184\u20137193 (2019)","DOI":"10.1109\/ICCV.2019.00728"},{"key":"2_CR22","doi-asserted-by":"crossref","unstructured":"Peng, X., et al.: Portraitbooth: A versatile portrait model for fast identity-preserved personalization. arXiv preprint arXiv:2312.06354 (2023)","DOI":"10.1109\/CVPR52733.2024.02557"},{"key":"2_CR23","doi-asserted-by":"crossref","unstructured":"Ren, Y., Li, G., Chen, Y., Li, T.H., Liu, S.: Pirenderer: controllable portrait image generation via semantic neural rendering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13759\u201313768 (2021)","DOI":"10.1109\/ICCV48922.2021.01350"},{"key":"2_CR24","doi-asserted-by":"crossref","unstructured":"Rossler, A., Cozzolino, D., Verdoliva, L., Riess, C., Thies, J., Nie\u00dfner, M.: Faceforensics++: learning to detect manipulated facial images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1\u201311 (2019)","DOI":"10.1109\/ICCV.2019.00009"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: Dreambooth: fine tuning text-to-image diffusion models for subject-driven generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22500\u201322510 (2023)","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"2_CR26","doi-asserted-by":"crossref","unstructured":"Shiohara, K., Yang, X., Taketomi, T.: Blendface: re-designing identity encoders for face-swapping. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7634\u20137644 (2023)","DOI":"10.1109\/ICCV51070.2023.00702"},{"key":"2_CR27","doi-asserted-by":"crossref","unstructured":"Siarohin, A., Lathuili\u00e8re, S., Tulyakov, S., Ricci, E., Sebe, N.: Animating arbitrary objects via deep motion transfer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2377\u20132386 (2019)","DOI":"10.1109\/CVPR.2019.00248"},{"key":"2_CR28","unstructured":"Siarohin, A., Lathuili\u00e8re, S., Tulyakov, S., Ricci, E., Sebe, N.: First order motion model for image animation. Adv. Neural Inform. Process. Syst. 32 (2019)"},{"key":"2_CR29","doi-asserted-by":"crossref","unstructured":"Tao, J., et al.: Structure-aware motion transfer with deformable anchor model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3637\u20133646 (2022)","DOI":"10.1109\/CVPR52688.2022.00362"},{"key":"2_CR30","unstructured":"Wang, Q., Bai, X., Wang, H., Qin, Z., Chen, A.: Instantid: Zero-shot identity-preserving generation in seconds. arXiv preprint arXiv:2401.07519 (2024)"},{"key":"2_CR31","doi-asserted-by":"crossref","unstructured":"Wang, T.C., Mallya, A., Liu, M.Y.: One-shot free-view neural talking-head synthesis for video conferencing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10039\u201310049 (2021)","DOI":"10.1109\/CVPR46437.2021.00991"},{"key":"2_CR32","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Hififace: 3d shape and semantic prior guided high fidelity face swapping. arXiv preprint arXiv:2106.09965 (2021)","DOI":"10.24963\/ijcai.2021\/157"},{"key":"2_CR33","doi-asserted-by":"crossref","unstructured":"Xiao, G., Yin, T., Freeman, W.T., Durand, F., Han, S.: Fastcomposer: Tuning-free multi-subject image generation with localized attention. arXiv preprint arXiv:2305.10431 (2023)","DOI":"10.1007\/s11263-024-02227-z"},{"key":"2_CR34","doi-asserted-by":"publisher","unstructured":"Xu, C., et al.: Designing one unified framework for high-fidelity face reenactment and swapping. In: European Conference on Computer Vision, pp. 54\u201371. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19784-0_4","DOI":"10.1007\/978-3-031-19784-0_4"},{"key":"2_CR35","doi-asserted-by":"crossref","unstructured":"Xu, C., Zhang, J., Hua, M., He, Q., Yi, Z., Liu, Y.: Region-aware face swapping. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7632\u20137641 (2022)","DOI":"10.1109\/CVPR52688.2022.00749"},{"key":"2_CR36","doi-asserted-by":"crossref","unstructured":"Xu, Z., Hong, Z., Ding, C., Zhu, Z., Han, J., Liu, J., Ding, E.: Mobilefaceswap: a lightweight framework for video face swapping. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 2973\u20132981 (2022)","DOI":"10.1609\/aaai.v36i3.20203"},{"key":"2_CR37","doi-asserted-by":"publisher","unstructured":"Yang, K., Chen, K., Guo, D., Zhang, S.H., Guo, Y.C., Zhang, W.: Face2face $$\\rho $$: Real-time high-resolution one-shot face reenactment. In: European Conference on Computer Vision, pp. 55\u201371. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19778-9_4","DOI":"10.1007\/978-3-031-19778-9_4"},{"key":"2_CR38","unstructured":"Ye, H., Zhang, J., Liu, S., Han, X., Yang, W.: Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:2308.06721 (2023)"},{"key":"2_CR39","doi-asserted-by":"publisher","unstructured":"Yin, F., et al.: Styleheat: one-shot high-resolution editable talking face generation via pre-trained stylegan. In: European Conference on Computer Vision, pp. 85\u2013101. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19790-1_6","DOI":"10.1007\/978-3-031-19790-1_6"},{"key":"2_CR40","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"334","DOI":"10.1007\/978-3-030-01261-8_20","volume-title":"Computer Vision \u2013 ECCV 2018","author":"C Yu","year":"2018","unstructured":"Yu, C., Wang, J., Peng, C., Gao, C., Yu, G., Sang, N.: BiSeNet: bilateral segmentation network for real-time semantic segmentation. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11217, pp. 334\u2013349. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01261-8_20"},{"key":"2_CR41","doi-asserted-by":"crossref","unstructured":"Zeng, B., Liu, X., Gao, S., Liu, B., Li, H., Liu, J., Zhang, B.: Face animation with an attribute-guided diffusion model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 628\u2013637 (2023)","DOI":"10.1109\/CVPRW59228.2023.00070"},{"key":"2_CR42","doi-asserted-by":"crossref","unstructured":"Zeng, X., Pan, Y., Wang, M., Zhang, J., Liu, Y.: Realistic face reenactment via self-supervised disentangling of identity and pose. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 12757\u201312764 (2020)","DOI":"10.1609\/aaai.v34i07.6970"},{"key":"2_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, B., et al.: Metaportrait: identity-preserving talking head generation with fast personalized adaptation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22096\u201322105 (2023)","DOI":"10.1109\/CVPR52729.2023.02116"},{"key":"2_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, J., et al.: Freenet: Multi-identity face reenactment. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5326\u20135335 (2020)","DOI":"10.1109\/CVPR42600.2020.00537"},{"key":"2_CR45","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., Agrawala, M.: Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3836\u20133847 (2023)","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"2_CR46","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A., Shechtman, E., Wang, O.: The unreasonable effectiveness of deep features as a perceptual metric. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 586\u2013595 (2018)","DOI":"10.1109\/CVPR.2018.00068"},{"key":"2_CR47","doi-asserted-by":"crossref","unstructured":"Zhao, J., Zhang, H.: Thin-plate spline motion model for image animation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3657\u20133666 (2022)","DOI":"10.1109\/CVPR52688.2022.00364"},{"key":"2_CR48","doi-asserted-by":"crossref","unstructured":"Zhao, W., Rao, Y., Shi, W., Liu, Z., Zhou, J., Lu, J.: Diffswap: high-fidelity and controllable face swapping via 3d-aware masked diffusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8568\u20138577 (2023)","DOI":"10.1109\/CVPR52729.2023.00828"},{"key":"2_CR49","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Li, Q., Wang, J., Xu, C.Z., Sun, Z.: One shot face swapping on megapixels. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4834\u20134844 (2021)","DOI":"10.1109\/CVPR46437.2021.00480"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72973-7_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T14:06:20Z","timestamp":1730383580000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72973-7_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9783031729720","9783031729737"],"references-count":49,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72973-7_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}