{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T04:55:13Z","timestamp":1781585713364,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":93,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,13]],"date-time":"2024-07-13T00:00:00Z","timestamp":1720828800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-sa\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,13]]},"DOI":"10.1145\/3641519.3657430","type":"proceedings-article","created":{"date-parts":[[2024,7,12]],"date-time":"2024-07-12T10:39:28Z","timestamp":1720780768000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":25,"title":["The Chosen One: Consistent Characters in Text-to-Image Diffusion Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7628-7525","authenticated-orcid":false,"given":"Omri","family":"Avrahami","sequence":"first","affiliation":[{"name":"Hebrew University of Jerusalem, Israel"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3037-3556","authenticated-orcid":false,"given":"Amir","family":"Hertz","sequence":"additional","affiliation":[{"name":"Google, Israel"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4402-7267","authenticated-orcid":false,"given":"Yael","family":"Vinker","sequence":"additional","affiliation":[{"name":"Tel Aviv University, Israel"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8423-3538","authenticated-orcid":false,"given":"Moab","family":"Arar","sequence":"additional","affiliation":[{"name":"Tel Aviv University, Israel"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1797-2143","authenticated-orcid":false,"given":"Shlomi","family":"Fruchter","sequence":"additional","affiliation":[{"name":"Google, Israel"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7109-4006","authenticated-orcid":false,"given":"Ohad","family":"Fried","sequence":"additional","affiliation":[{"name":"Reichman University, Israel"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6777-7445","authenticated-orcid":false,"given":"Daniel","family":"Cohen-Or","sequence":"additional","affiliation":[{"name":"Tel Aviv University, Israel"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6191-0361","authenticated-orcid":false,"given":"Dani","family":"Lischinski","sequence":"additional","affiliation":[{"name":"Hebrew University of Jerusalem, Israel"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,7,13]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"DreamStyler: Paint by Style Inversion with Text-to-Image Diffusion Models. ArXiv abs\/2309.06933","author":"Ahn Namhyuk","year":"2023","unstructured":"Namhyuk Ahn, Junsoo Lee, Chunggi Lee, Kunhee Kim, Daesik Kim, Seung-Hun Nam, and Kibeom Hong. 2023. DreamStyler: Paint by Style Inversion with Text-to-Image Diffusion Models. ArXiv abs\/2309.06933 (2023). https:\/\/api.semanticscholar.org\/CorpusID:261706081"},{"key":"e_1_3_2_2_2_1","volume-title":"A Neural Space-Time Representation for Text-to-Image Personalization. ArXiv abs\/2305.15391","author":"Alaluf Yuval","year":"2023","unstructured":"Yuval Alaluf, Elad Richardson, Gal Metzer, and Daniel Cohen-Or. 2023. A Neural Space-Time Representation for Text-to-Image Personalization. ArXiv abs\/2305.15391 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258866047"},{"key":"e_1_3_2_2_3_1","unstructured":"Amazon. 2023. Amazon Mechanical Turk. https:\/\/www.mturk.com\/."},{"key":"e_1_3_2_2_4_1","volume-title":"Domain-agnostic tuning-encoder for fast personalization of text-to-image models. arXiv preprint arXiv:2307.06925","author":"Arar Moab","year":"2023","unstructured":"Moab Arar, Rinon Gal, Yuval Atzmon, Gal Chechik, Daniel Cohen-Or, Ariel Shamir, and Amit\u00a0H Bermano. 2023. Domain-agnostic tuning-encoder for fast personalization of text-to-image models. arXiv preprint arXiv:2307.06925 (2023)."},{"key":"e_1_3_2_2_5_1","volume-title":"ACM-SIAM Symposium on Discrete Algorithms. https:\/\/api.semanticscholar.org\/CorpusID:1782131","author":"Arthur David","year":"2007","unstructured":"David Arthur and Sergei Vassilvitskii. 2007. k-means++: the advantages of careful seeding. In ACM-SIAM Symposium on Discrete Algorithms. https:\/\/api.semanticscholar.org\/CorpusID:1782131"},{"key":"e_1_3_2_2_6_1","volume-title":"Break-A-Scene: Extracting Multiple Concepts from a Single Image. ArXiv abs\/2305.16311","author":"Avrahami Omri","year":"2023","unstructured":"Omri Avrahami, Kfir Aberman, Ohad Fried, Daniel Cohen-Or, and Dani Lischinski. 2023a. Break-A-Scene: Extracting Multiple Concepts from a Single Image. ArXiv abs\/2305.16311 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258888228"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592450"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01762"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"e_1_3_2_2_10_1","volume-title":"eDiff-I: Text-to-Image Diffusion Models with an Ensemble of Expert Denoisers. ArXiv abs\/2211.01324","author":"Balaji Yogesh","year":"2022","unstructured":"Yogesh Balaji, Seungjun Nah, Xun Huang, Arash Vahdat, Jiaming Song, Qinsheng Zhang, Karsten Kreis, Miika Aittala, Timo Aila, Samuli Laine, Bryan Catanzaro, Tero Karras, and Ming-Yu Liu. 2022. eDiff-I: Text-to-Image Diffusion Models with an Ensemble of Expert Denoisers. ArXiv abs\/2211.01324 (2022). https:\/\/api.semanticscholar.org\/CorpusID:253254800"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19784-0_41"},{"key":"e_1_3_2_2_12_1","volume-title":"Volumetric Disentanglement for 3D Scene Manipulation. ArXiv abs\/2206.02776","author":"Benaim Sagie","year":"2022","unstructured":"Sagie Benaim, Frederik Warburg, Peter\u00a0Ebert Christensen, and Serge\u00a0J. Belongie. 2022. Volumetric Disentanglement for 3D Scene Manipulation. ArXiv abs\/2206.02776 (2022). https:\/\/api.semanticscholar.org\/CorpusID:249394623"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02062"},{"key":"e_1_3_2_2_14_1","volume-title":"Emerging Properties in Self-Supervised Vision Transformers. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 9630\u20139640","author":"Caron Mathilde","year":"2021","unstructured":"Mathilde Caron, Hugo Touvron, Ishan Misra, Herv\u00e9 Jegou, Julien Mairal, Piotr Bojanowski, and Armand Joulin. 2021. Emerging Properties in Self-Supervised Vision Transformers. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 9630\u20139640."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592116"},{"key":"e_1_3_2_2_16_1","volume-title":"Subject-driven Text-to-Image Generation via Apprenticeship Learning. ArXiv abs\/2304.00186","author":"Chen Wenhu","year":"2023","unstructured":"Wenhu Chen, Hexiang Hu, Yandong Li, Nataniel Rui, Xuhui Jia, Ming-Wei Chang, and William\u00a0W. Cohen. 2023a. Subject-driven Text-to-Image Generation via Apprenticeship Learning. ArXiv abs\/2304.00186 (2023)."},{"key":"e_1_3_2_2_17_1","volume-title":"AnyDoor: Zero-shot Object-level Image Customization. ArXiv abs\/2307.09481","author":"Chen Xi","year":"2023","unstructured":"Xi Chen, Lianghua Huang, Yu Liu, Yujun Shen, Deli Zhao, and Hengshuang Zhao. 2023b. AnyDoor: Zero-shot Object-level Image Customization. ArXiv abs\/2307.09481 (2023). https:\/\/api.semanticscholar.org\/CorpusID:259951373"},{"key":"e_1_3_2_2_18_1","volume-title":"Zero-shot spatial layout conditioning for text-to-image diffusion models. ArXiv abs\/2306.13754","author":"Couairon Guillaume","year":"2023","unstructured":"Guillaume Couairon, Marlene Careil, Matthieu Cord, St\u00e9phane Lathuili\u00e8re, and Jakob Verbeek. 2023. Zero-shot spatial layout conditioning for text-to-image diffusion models. ArXiv abs\/2306.13754 (2023). https:\/\/api.semanticscholar.org\/CorpusID:259252153"},{"key":"e_1_3_2_2_19_1","unstructured":"AI Foundations. 2023. How to Create Consistent Characters in Midjourney. https:\/\/www.youtube.com\/watch?v=Z7_ta3RHijQ."},{"key":"e_1_3_2_2_20_1","volume-title":"SceneScape: Text-Driven Consistent Scene Generation. ArXiv abs\/2302.01133","author":"Fridman Rafail","year":"2023","unstructured":"Rafail Fridman, Amit Abecasis, Yoni Kasten, and Tali Dekel. 2023. SceneScape: Text-Driven Consistent Scene Generation. ArXiv abs\/2302.01133 (2023). https:\/\/api.semanticscholar.org\/CorpusID:256503775"},{"key":"e_1_3_2_2_21_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Gal Rinon","year":"2022","unstructured":"Rinon Gal, Yuval Alaluf, Yuval Atzmon, Or Patashnik, Amit\u00a0Haim Bermano, Gal Chechik, and Daniel Cohen-or. 2022. An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592133"},{"key":"e_1_3_2_2_23_1","volume-title":"Expressive Text-to-Image Generation with Rich Text. ArXiv abs\/2304.06720","author":"Ge Songwei","year":"2023","unstructured":"Songwei Ge, Taesung Park, Jun-Yan Zhu, and Jia-Bin Huang. 2023. Expressive Text-to-Image Generation with Rich Text. ArXiv abs\/2304.06720 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258108187"},{"key":"e_1_3_2_2_24_1","volume-title":"Tokenflow: Consistent diffusion features for consistent video editing. arXiv preprint arXiv:2307.10373","author":"Geyer Michal","year":"2023","unstructured":"Michal Geyer, Omer Bar-Tal, Shai Bagon, and Tali Dekel. 2023. Tokenflow: Consistent diffusion features for consistent video editing. arXiv preprint arXiv:2307.10373 (2023)."},{"key":"e_1_3_2_2_25_1","volume-title":"TaleCrafter: Interactive Story Visualization with Multiple Characters. ArXiv abs\/2305.18247","author":"Gong Yuan","year":"2023","unstructured":"Yuan Gong, Youxin Pang, Xiaodong Cun, Menghan Xia, Haoxin Chen, Longyue Wang, Yong Zhang, Xintao Wang, Ying Shan, and Yujiu Yang. 2023. TaleCrafter: Interactive Story Visualization with Multiple Characters. ArXiv abs\/2305.18247 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258960665"},{"key":"e_1_3_2_2_26_1","volume-title":"Blended-NeRF: Zero-Shot Object Generation and Blending in Existing Neural Radiance Fields. ArXiv abs\/2306.12760","author":"Gordon Ori","year":"2023","unstructured":"Ori Gordon, Omri Avrahami, and Dani Lischinski. 2023. Blended-NeRF: Zero-Shot Object Generation and Blending in Existing Neural Radiance Fields. ArXiv abs\/2306.12760 (2023). https:\/\/api.semanticscholar.org\/CorpusID:259224726"},{"key":"e_1_3_2_2_27_1","volume-title":"SVDiff: Compact Parameter Space for Diffusion Fine-Tuning. ArXiv abs\/2303.11305","author":"Han Ligong","year":"2023","unstructured":"Ligong Han, Yinxiao Li, Han Zhang, Peyman Milanfar, Dimitris\u00a0N. Metaxas, and Feng Yang. 2023. SVDiff: Compact Parameter Space for Diffusion Fine-Tuning. ArXiv abs\/2303.11305 (2023)."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00221"},{"key":"e_1_3_2_2_29_1","volume-title":"Prompt-to-prompt image editing with cross attention control. arXiv preprint arXiv:2208.01626","author":"Hertz Amir","year":"2022","unstructured":"Amir Hertz, Ron Mokady, Jay Tenenbaum, Kfir Aberman, Yael Pritch, and Daniel Cohen-Or. 2022. Prompt-to-prompt image editing with cross attention control. arXiv preprint arXiv:2208.01626 (2022)."},{"key":"e_1_3_2_2_30_1","unstructured":"Geoffrey\u00a0E. Hinton and Sam\u00a0T. Roweis. 2002. Stochastic Neighbor Embedding. In NIPS. https:\/\/api.semanticscholar.org\/CorpusID:20240"},{"key":"e_1_3_2_2_31_1","volume-title":"Proc.\u00a0NeurIPS.","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising Diffusion Probabilistic Models. In Proc.\u00a0NeurIPS."},{"key":"e_1_3_2_2_32_1","volume-title":"Text2Room: Extracting Textured 3D Meshes from 2D Text-to-Image Models. ArXiv abs\/2303.11989","author":"H\u00f6llein Lukas","year":"2023","unstructured":"Lukas H\u00f6llein, Ang Cao, Andrew Owens, Justin Johnson, and Matthias Nie\u00dfner. 2023. Text2Room: Extracting Textured 3D Meshes from 2D Text-to-Image Models. ArXiv abs\/2303.11989 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257636653"},{"key":"e_1_3_2_2_33_1","volume-title":"Conffusion: Confidence Intervals for Diffusion Models. ArXiv abs\/2211.09795","author":"Horwitz Eliahu","year":"2022","unstructured":"Eliahu Horwitz and Yedid Hoshen. 2022. Conffusion: Confidence Intervals for Diffusion Models. ArXiv abs\/2211.09795 (2022)."},{"key":"e_1_3_2_2_34_1","volume-title":"LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations.","author":"Hu J","year":"2021","unstructured":"Edward\u00a0J Hu, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, Weizhu Chen, 2021. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","unstructured":"Gabriel Ilharco Mitchell Wortsman Ross Wightman Cade Gordon Nicholas Carlini Rohan Taori Achal Dave Vaishaal Shankar Hongseok Namkoong John Miller Hannaneh Hajishirzi Ali Farhadi and Ludwig Schmidt. 2021. OpenCLIP. https:\/\/doi.org\/10.5281\/zenodo.5143773","DOI":"10.5281\/zenodo.5143773"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592123"},{"key":"e_1_3_2_2_37_1","volume-title":"Zero-shot Generation of Coherent Storybook from Plain Text Story using Diffusion Models. ArXiv abs\/2302.03900","author":"Jeong Hyeonho","year":"2023","unstructured":"Hyeonho Jeong, Gihyun Kwon, and Jong-Chul Ye. 2023. Zero-shot Generation of Coherent Storybook from Plain Text Story using Diffusion Models. ArXiv abs\/2302.03900 (2023). https:\/\/api.semanticscholar.org\/CorpusID:256662241"},{"key":"e_1_3_2_2_38_1","volume-title":"Taming Encoder for Zero Fine-tuning Image Customization with Text-to-Image Diffusion Models. ArXiv abs\/2304.02642","author":"Jia Xuhui","year":"2023","unstructured":"Xuhui Jia, Yang Zhao, Kelvin C.\u00a0K. Chan, Yandong Li, Han-Ying Zhang, Boqing Gong, Tingbo Hou, H. Wang, and Yu-Chuan Su. 2023. Taming Encoder for Zero Fine-tuning Image Customization with Text-to-Image Diffusion Models. ArXiv abs\/2304.02642 (2023)."},{"key":"e_1_3_2_2_39_1","unstructured":"JoshGreat. 2023. 8 ways to generate consistent characters (for comics storyboards books etc) : StableDiffusion. https:\/\/www.reddit.com\/r\/StableDiffusion\/comments\/10yxz3m\/8_ways_to_generate_consistent_characters_for\/."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00582"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"e_1_3_2_2_42_1","volume-title":"\u00a0H. Hoi","author":"Li Dongxu","year":"2023","unstructured":"Dongxu Li, Junnan Li, and Steven C.\u00a0H. Hoi. 2023. BLIP-Diffusion: Pre-trained Subject Representation for Controllable Text-to-Image Generation and Editing. ArXiv abs\/2305.14720 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258865473"},{"key":"e_1_3_2_2_43_1","volume-title":"StoryGAN: A Sequential Conditional GAN for Story Visualization. CVPR","author":"Li Yitong","year":"2019","unstructured":"Yitong Li, Zhe Gan, Yelong Shen, Jingjing Liu, Yu Cheng, Yuexin Wu, Lawrence Carin, David Carlson, and Jianfeng Gao. 2019. StoryGAN: A Sequential Conditional GAN for Story Visualization. CVPR (2019)."},{"key":"e_1_3_2_2_44_1","volume-title":"Video-p2p: Video editing with cross-attention control. arXiv preprint arXiv:2303.04761","author":"Liu Shaoteng","year":"2023","unstructured":"Shaoteng Liu, Yuechen Zhang, Wenbo Li, Zhe Lin, and Jiaya Jia. 2023a. Video-p2p: Video editing with cross-attention control. arXiv preprint arXiv:2303.04761 (2023)."},{"key":"e_1_3_2_2_45_1","volume-title":"Video-P2P: Video Editing with Cross-attention Control. ArXiv abs\/2303.04761","author":"Liu Shaoteng","year":"2023","unstructured":"Shaoteng Liu, Yuecheng Zhang, Wenbo Li, Zhe Lin, and Jiaya Jia. 2023b. Video-P2P: Video Editing with Cross-attention Control. ArXiv abs\/2303.04761 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257405406"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19836-6_5"},{"key":"e_1_3_2_2_47_1","volume-title":"SDEdit: Guided Image Synthesis and Editing with Stochastic Differential Equations. In International Conference on Learning Representations.","author":"Meng Chenlin","year":"2021","unstructured":"Chenlin Meng, Yutong He, Yang Song, Jiaming Song, Jiajun Wu, Jun-Yan Zhu, and Stefano Ermon. 2021. SDEdit: Guided Image Synthesis and Editing with Stochastic Differential Equations. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01218"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00585"},{"key":"e_1_3_2_2_50_1","volume-title":"Dreamix: Video Diffusion Models are General Video Editors","author":"Molad Eyal","year":"2023","unstructured":"Eyal Molad, Eliahu Horwitz, Dani Valevski, Alex\u00a0Rav Acha, Y. Matias, Yael Pritch, Yaniv Leviathan, and Yedid Hoshen. 2023. Dreamix: Video Diffusion Models are General Video Editors. ArXiv abs\/2302.01329 (2023)."},{"key":"e_1_3_2_2_51_1","volume-title":"T2i-adapter: Learning adapters to dig out more controllable ability for text-to-image diffusion models. arXiv preprint arXiv:2302.08453","author":"Mou Chong","year":"2023","unstructured":"Chong Mou, Xintao Wang, Liangbin Xie, Yanze Wu, Jian Zhang, Zhongang Qi, Ying Shan, and Xiaohu Qie. 2023. T2i-adapter: Learning adapters to dig out more controllable ability for text-to-image diffusion models. arXiv preprint arXiv:2302.08453 (2023)."},{"key":"e_1_3_2_2_52_1","volume-title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. In International Conference on Machine Learning. https:\/\/api.semanticscholar.org\/CorpusID:245335086","author":"Nichol Alex","year":"2021","unstructured":"Alex Nichol, Prafulla Dhariwal, Aditya Ramesh, Pranav Shyam, Pamela Mishkin, Bob McGrew, Ilya Sutskever, and Mark Chen. 2021. GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. In International Conference on Machine Learning. https:\/\/api.semanticscholar.org\/CorpusID:245335086"},{"key":"e_1_3_2_2_53_1","unstructured":"OpenAI. 2022. ChatGPT. https:\/\/chat.openai.com\/. Accessed: 2023-10-15."},{"key":"e_1_3_2_2_54_1","volume-title":"DINOv2: Learning Robust Visual Features without Supervision. ArXiv abs\/2304.07193","author":"Oquab Maxime","year":"2023","unstructured":"Maxime Oquab, Timoth\u00e9e Darcet, Th\u00e9o Moutakanni, Huy\u00a0Q. Vo, Marc Szafraniec, Vasil Khalidov, Pierre Fernandez, Daniel Haziza, Francisco Massa, Alaaeldin El-Nouby, Mahmoud Assran, Nicolas Ballas, Wojciech Galuba, Russ Howes, Po-Yao\u00a0(Bernie) Huang, Shang-Wen Li, Ishan Misra, Michael\u00a0G. Rabbat, Vasu Sharma, Gabriel Synnaeve, Huijiao Xu, Herv\u00e9 J\u00e9gou, Julien Mairal, Patrick Labatut, Armand Joulin, and Piotr Bojanowski. 2023. DINOv2: Learning Robust Visual Features without Supervision. ArXiv abs\/2304.07193 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258170077"},{"key":"e_1_3_2_2_55_1","volume-title":"Localizing Object-level Shape Variations with Text-to-Image Diffusion Models. ArXiv abs\/2303.11306","author":"Patashnik Or","year":"2023","unstructured":"Or Patashnik, Daniel Garibi, Idan Azuri, Hadar Averbuch-Elor, and Daniel Cohen-Or. 2023. Localizing Object-level Shape Variations with Text-to-Image Diffusion Models. ArXiv abs\/2303.11306 (2023)."},{"key":"e_1_3_2_2_56_1","volume-title":"State of the Art on Diffusion Models for Visual Computing. ArXiv abs\/2310.07204","author":"Po Ryan","year":"2023","unstructured":"Ryan Po, Wang Yifan, Vladislav Golyanik, Kfir Aberman, Jonathan\u00a0T. Barron, Amit\u00a0H. Bermano, Eric\u00a0Ryan Chan, Tali Dekel, Aleksander Holynski, Angjoo Kanazawa, C.\u00a0Karen Liu, Lingjie Liu, Ben Mildenhall, Matthias Nie\u00dfner, Bjorn Ommer, Christian Theobalt, Peter Wonka, and Gordon Wetzstein. 2023. State of the Art on Diffusion Models for Visual Computing. ArXiv abs\/2310.07204 (2023). https:\/\/api.semanticscholar.org\/CorpusID:263835355"},{"key":"e_1_3_2_2_57_1","volume-title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. ArXiv abs\/2307.01952","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, A. Blattmann, Tim Dockhorn, Jonas Muller, Joe Penna, and Robin Rombach. 2023. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. ArXiv abs\/2307.01952 (2023). https:\/\/api.semanticscholar.org\/CorpusID:259341735"},{"key":"e_1_3_2_2_58_1","volume-title":"Dreamfusion: Text-to-3d using 2d diffusion. arXiv preprint arXiv:2209.14988","author":"Poole Ben","year":"2022","unstructured":"Ben Poole, Ajay Jain, Jonathan\u00a0T Barron, and Ben Mildenhall. 2022. Dreamfusion: Text-to-3d using 2d diffusion. arXiv preprint arXiv:2209.14988 (2022)."},{"key":"e_1_3_2_2_59_1","volume-title":"Fatezero: Fusing attentions for zero-shot text-based video editing. arXiv preprint arXiv:2303.09535","author":"Qi Chenyang","year":"2023","unstructured":"Chenyang Qi, Xiaodong Cun, Yong Zhang, Chenyang Lei, Xintao Wang, Ying Shan, and Qifeng Chen. 2023. Fatezero: Fusing attentions for zero-shot text-based video editing. arXiv preprint arXiv:2303.09535 (2023)."},{"key":"e_1_3_2_2_60_1","volume-title":"Single Motion Diffusion. ArXiv abs\/2302.05905","author":"Raab Sigal","year":"2023","unstructured":"Sigal Raab, Inbal Leibovitch, Guy Tevet, Moab Arar, Amit\u00a0H. Bermano, and Daniel Cohen-Or. 2023. Single Motion Diffusion. ArXiv abs\/2302.05905 (2023). https:\/\/api.semanticscholar.org\/CorpusID:256827051"},{"key":"e_1_3_2_2_61_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning."},{"key":"e_1_3_2_2_62_1","volume-title":"Make-A-Story: Visual Memory Conditioned Consistent Story Generation. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Rahman Tanzila","year":"2022","unstructured":"Tanzila Rahman, Hsin-Ying Lee, Jian Ren, S. Tulyakov, Shweta Mahajan, and Leonid Sigal. 2022. Make-A-Story: Visual Memory Conditioned Consistent Story Generation. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022), 2493\u20132502. https:\/\/api.semanticscholar.org\/CorpusID:254017562"},{"key":"e_1_3_2_2_63_1","volume-title":"Hierarchical text-conditional image generation with CLIP latents. arXiv preprint arXiv:2204.06125","author":"Ramesh Aditya","year":"2022","unstructured":"Aditya Ramesh, Prafulla Dhariwal, Alex Nichol, Casey Chu, and Mark Chen. 2022. Hierarchical text-conditional image generation with CLIP latents. arXiv preprint arXiv:2204.06125 (2022)."},{"key":"e_1_3_2_2_64_1","volume-title":"ConceptLab: Creative Generation using Diffusion Prior Constraints. arXiv preprint arXiv:2308.02669","author":"Richardson Elad","year":"2023","unstructured":"Elad Richardson, Kfir Goldberg, Yuval Alaluf, and Daniel Cohen-Or. 2023a. ConceptLab: Creative Generation using Diffusion Prior Constraints. arXiv preprint arXiv:2308.02669 (2023)."},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591503"},{"key":"e_1_3_2_2_66_1","volume-title":"High-Resolution Image Synthesis with Latent Diffusion Models. 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Rombach Robin","year":"2021","unstructured":"Robin Rombach, A. Blattmann, Dominik Lorenz, Patrick Esser, and Bj\u00f6rn Ommer. 2021. High-Resolution Image Synthesis with Latent Diffusion Models. 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2021), 10674\u201310685."},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_2_2_68_1","unstructured":"Simo Ryu. 2022. Low-rank Adaptation for Fast Text-to-Image Diffusion Fine-tuning. https:\/\/github.com\/cloneofsimo\/lora."},{"key":"e_1_3_2_2_69_1","first-page":"36479","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume":"35","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily\u00a0L Denton, Kamyar Ghasemipour, Raphael Gontijo\u00a0Lopes, Burcu Karagol\u00a0Ayan, Tim Salimans, 2022. Photorealistic text-to-image diffusion models with deep language understanding. Advances in Neural Information Processing Systems 35 (2022), 36479\u201336494.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_70_1","volume-title":"Vox-E: Text-guided Voxel Editing of 3D Objects. ArXiv abs\/2303.12048","author":"Sella Etai","year":"2023","unstructured":"Etai Sella, Gal Fiebelman, Peter Hedman, and Hadar Averbuch-Elor. 2023. Vox-E: Text-guided Voxel Editing of 3D Objects. ArXiv abs\/2303.12048 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257636627"},{"key":"e_1_3_2_2_71_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Sheynin Shelly","year":"2022","unstructured":"Shelly Sheynin, Oron Ashual, Adam Polyak, Uriel Singer, Oran Gafni, Eliya Nachmani, and Yaniv Taigman. 2022. kNN-Diffusion: Image Generation via Large-Scale Retrieval. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_2_72_1","volume-title":"InstantBooth: Personalized Text-to-Image Generation without Test-Time Finetuning. ArXiv abs\/2304.03411","author":"Shi Jing","year":"2023","unstructured":"Jing Shi, Wei Xiong, Zhe\u00a0L. Lin, and Hyun\u00a0Joon Jung. 2023. InstantBooth: Personalized Text-to-Image Generation without Test-Time Finetuning. ArXiv abs\/2304.03411 (2023)."},{"key":"e_1_3_2_2_73_1","volume-title":"International Conference on Machine Learning. PMLR, 2256\u20132265","author":"Sohl-Dickstein Jascha","year":"2015","unstructured":"Jascha Sohl-Dickstein, Eric Weiss, Niru Maheswaranathan, and Surya Ganguli. 2015. Deep unsupervised learning using nonequilibrium thermodynamics. In International Conference on Machine Learning. PMLR, 2256\u20132265."},{"key":"e_1_3_2_2_74_1","volume-title":"StyleDrop: Text-to-Image Generation in Any Style. ArXiv abs\/2306.00983","author":"Sohn Kihyuk","year":"2023","unstructured":"Kihyuk Sohn, Nataniel Ruiz, Kimin Lee, Daniel\u00a0Castro Chin, Irina Blok, Huiwen Chang, Jarred Barber, Lu Jiang, Glenn Entis, Yuanzhen Li, Yuan Hao, Irfan Essa, Michael Rubinstein, and Dilip Krishnan. 2023. StyleDrop: Text-to-Image Generation in Any Style. ArXiv abs\/2306.00983 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258999204"},{"key":"e_1_3_2_2_75_1","volume-title":"Denoising Diffusion Implicit Models. In International Conference on Learning Representations.","author":"Song Jiaming","year":"2020","unstructured":"Jiaming Song, Chenlin Meng, and Stefano Ermon. 2020. Denoising Diffusion Implicit Models. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_76_1","volume-title":"Generative modeling by estimating gradients of the data distribution. Advances in Neural Information Processing Systems 32","author":"Song Yang","year":"2019","unstructured":"Yang Song and Stefano Ermon. 2019. Generative modeling by estimating gradients of the data distribution. Advances in Neural Information Processing Systems 32 (2019)."},{"key":"e_1_3_2_2_77_1","unstructured":"stassius. 2023. How to create consistent character faces without training (info in the comments) : StableDiffusion. https:\/\/www.reddit.com\/r\/StableDiffusion\/comments\/12djxvz\/how_to_create_consistent_character_faces_without\/."},{"key":"e_1_3_2_2_78_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-09037-0_23"},{"key":"e_1_3_2_2_79_1","volume-title":"Human Motion Diffusion Model. ArXiv abs\/2209.14916","author":"Tevet Guy","year":"2022","unstructured":"Guy Tevet, Sigal Raab, Brian Gordon, Yonatan Shafir, Daniel Cohen-Or, and Amit\u00a0H. Bermano. 2022. Human Motion Diffusion Model. ArXiv abs\/2209.14916 (2022). https:\/\/api.semanticscholar.org\/CorpusID:252595883"},{"key":"e_1_3_2_2_80_1","volume-title":"Key-Locked Rank One Editing for Text-to-Image Personalization. ACM SIGGRAPH 2023 Conference Proceedings","author":"Tewel Yoad","year":"2023","unstructured":"Yoad Tewel, Rinon Gal, Gal Chechik, and Yuval Atzmon. 2023. Key-Locked Rank One Editing for Text-to-Image Personalization. ACM SIGGRAPH 2023 Conference Proceedings (2023). https:\/\/api.semanticscholar.org\/CorpusID:258436985"},{"key":"e_1_3_2_2_81_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00191"},{"key":"e_1_3_2_2_82_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618249"},{"key":"e_1_3_2_2_83_1","volume-title":"Concept Decomposition for Visual Exploration and Inspiration. ArXiv abs\/2305.18203","author":"Vinker Yael","year":"2023","unstructured":"Yael Vinker, Andrey Voynov, Daniel Cohen-Or, and Ariel Shamir. 2023. Concept Decomposition for Visual Exploration and Inspiration. ArXiv abs\/2305.18203 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258959472"},{"key":"e_1_3_2_2_84_1","volume-title":"Sketch-Guided Text-to-Image Diffusion Models. arXiv preprint arXiv:2211.13752","author":"Voynov Andrey","year":"2022","unstructured":"Andrey Voynov, Kfir Aberman, and Daniel Cohen-Or. 2022. Sketch-Guided Text-to-Image Diffusion Models. arXiv preprint arXiv:2211.13752 (2022)."},{"key":"e_1_3_2_2_85_1","volume-title":"Extended Textual Conditioning in Text-to-Image Generation. ArXiv abs\/2303.09522","author":"Voynov Andrey","year":"2023","unstructured":"Andrey Voynov, Q. Chu, Daniel Cohen-Or, and Kfir Aberman. 2023. P+: Extended Textual Conditioning in Text-to-Image Generation. ArXiv abs\/2303.09522 (2023)."},{"key":"e_1_3_2_2_86_1","unstructured":"Yuxiang Wei. 2023. Official Implementation of ELITE. https:\/\/github.com\/csyxwei\/ELITE. Accessed: 2023-05-01."},{"key":"e_1_3_2_2_87_1","volume-title":"ELITE: Encoding Visual Concepts into Textual Embeddings for Customized Text-to-Image Generation. ArXiv abs\/2302.13848","author":"Wei Yuxiang","year":"2023","unstructured":"Yuxiang Wei, Yabo Zhang, Zhilong Ji, Jinfeng Bai, Lei Zhang, and Wangmeng Zuo. 2023. ELITE: Encoding Visual Concepts into Textual Embeddings for Customized Text-to-Image Generation. ArXiv abs\/2302.13848 (2023)."},{"key":"e_1_3_2_2_88_1","volume-title":"Rerender A Video: Zero-Shot Text-Guided Video-to-Video Translation. ArXiv abs\/2306.07954","author":"Yang Shuai","year":"2023","unstructured":"Shuai Yang, Yifan Zhou, Ziwei Liu, and Chen\u00a0Change Loy. 2023. Rerender A Video: Zero-Shot Text-Guided Video-to-Video Translation. ArXiv abs\/2306.07954 (2023). https:\/\/api.semanticscholar.org\/CorpusID:259144797"},{"key":"e_1_3_2_2_89_1","volume-title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models. ArXiv abs\/2308.06721","author":"Ye Hu","year":"2023","unstructured":"Hu Ye, Jun Zhang, Siyi Liu, Xiao Han, and Wei Yang. 2023. IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models. ArXiv abs\/2308.06721 (2023). https:\/\/api.semanticscholar.org\/CorpusID:260886966"},{"key":"e_1_3_2_2_90_1","volume-title":"Scaling Autoregressive Models for Content-Rich Text-to-Image Generation. arXiv preprint arXiv:2206.10789","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Yuanzhong Xu, Jing\u00a0Yu Koh, Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu\u00a0Karagol Ayan, 2022. Scaling Autoregressive Models for Content-Rich Text-to-Image Generation. arXiv preprint arXiv:2206.10789 (2022)."},{"key":"e_1_3_2_2_91_1","volume-title":"Text-to-image Diffusion Models in Generative AI: A Survey. ArXiv abs\/2303.07909","author":"Zhang Chenshuang","year":"2023","unstructured":"Chenshuang Zhang, Chaoning Zhang, Mengchun Zhang, and In-So Kweon. 2023b. Text-to-image Diffusion Models in Generative AI: A Survey. ArXiv abs\/2303.07909 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257505012"},{"key":"e_1_3_2_2_92_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_2_93_1","volume-title":"DreamEditor","author":"Zhuang Jingyu","year":"2023","unstructured":"Jingyu Zhuang, Chen Wang, Lingjie Liu, Liang Lin, and Guanbin Li. 2023. DreamEditor: Text-Driven 3D Scene Editing with Neural Fields. ArXiv abs\/2306.13455 (2023). https:\/\/api.semanticscholar.org\/CorpusID:259243782"}],"event":{"name":"SIGGRAPH '24: Special Interest Group on Computer Graphics and Interactive Techniques Conference","location":"Denver CO USA","acronym":"SIGGRAPH '24","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641519.3657430","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3641519.3657430","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:09:36Z","timestamp":1750295376000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641519.3657430"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,13]]},"references-count":93,"alternative-id":["10.1145\/3641519.3657430","10.1145\/3641519"],"URL":"https:\/\/doi.org\/10.1145\/3641519.3657430","relation":{},"subject":[],"published":{"date-parts":[[2024,7,13]]},"assertion":[{"value":"2024-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}