{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:34:33Z","timestamp":1777656873007,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763867","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:30:41Z","timestamp":1765211441000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Object-level Visual Prompts for Compositional Image Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8640-2243","authenticated-orcid":false,"given":"Gaurav","family":"Parmar","sequence":"first","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7757-6137","authenticated-orcid":false,"given":"Or","family":"Patashnik","sequence":"additional","affiliation":[{"name":"Tel Aviv University, Tel Aviv, Israel"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6785-8146","authenticated-orcid":false,"given":"Kuan-Chieh","family":"Wang","sequence":"additional","affiliation":[{"name":"Snap Inc., Palo Alto, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7259-7399","authenticated-orcid":false,"given":"Daniil","family":"Ostashev","sequence":"additional","affiliation":[{"name":"Snap Inc., London, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0389-1921","authenticated-orcid":false,"given":"Srinivasa","family":"Narasimhan","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8504-3410","authenticated-orcid":false,"given":"Jun-Yan","family":"Zhu","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6777-7445","authenticated-orcid":false,"given":"Daniel","family":"Cohen-Or","sequence":"additional","affiliation":[{"name":"Tel Aviv University, Tel Aviv, Israel"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4958-601X","authenticated-orcid":false,"given":"Kfir","family":"Aberman","sequence":"additional","affiliation":[{"name":"Snap Inc., Palo Alto, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657423"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"crossref","unstructured":"Yuval Alaluf Elad Richardson Gal Metzer and Daniel Cohen-Or. 2023. A neural space-time representation for text-to-image personalization. ACM Transactions on Graphics (TOG) 42 6 (2023) 1\u201310.","DOI":"10.1145\/3618322"},{"key":"e_1_3_3_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618173"},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618154"},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01762"},{"key":"e_1_3_3_2_7_1","unstructured":"Jimmy\u00a0Lei Ba. 2016. Layer normalization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1607.06450 (2016)."},{"key":"e_1_3_3_2_8_1","volume-title":"ICLR","author":"Bahdanau Dzmitry","year":"2015","unstructured":"Dzmitry Bahdanau, Kyunghyun Cho, and Yoshua Bengio. 2015. Neural Machine Translation by Jointly Learning to Align and Translate. In ICLR."},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657527"},{"key":"e_1_3_3_2_10_1","unstructured":"Minwoo Byeon Beomhee Park Haecheon Kim Sungjun Lee Woonhyuk Baek and Saehoon Kim. 2022. COYO-700M: Image-Text Pair Dataset. https:\/\/github.com\/kakaobrain\/coyo-dataset."},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02062"},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"crossref","unstructured":"Hila Chefer Yuval Alaluf Yael Vinker Lior Wolf and Daniel Cohen-Or. 2023. Attend-and-excite: Attention-based semantic guidance for text-to-image diffusion models. ACM Transactions on Graphics (TOG) 42 4 (2023) 1\u201310.","DOI":"10.1145\/3592116"},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00526"},{"key":"e_1_3_3_2_14_1","volume-title":"Thirty-seventh Conference on Neural Information Processing Systems","author":"Chen Wenhu","year":"2023","unstructured":"Wenhu Chen, Hexiang Hu, YANDONG LI, Nataniel Ruiz, Xuhui Jia, Ming-Wei Chang, and William\u00a0W. Cohen. 2023. Subject-driven Text-to-Image Generation via Apprenticeship Learning. In Thirty-seventh Conference on Neural Information Processing Systems. https:\/\/openreview.net\/forum?id=wv3bHyQbX7"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.00630"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01166"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72630-9_25"},{"key":"e_1_3_3_2_18_1","volume-title":"Advances in Neural Information Processing Systems","author":"Dhariwal Prafulla","year":"2021","unstructured":"Prafulla Dhariwal and Alexander\u00a0Quinn Nichol. 2021. Diffusion Models Beat GANs on Image Synthesis. In Advances in Neural Information Processing Systems, A.\u00a0Beygelzimer, Y.\u00a0Dauphin, P.\u00a0Liang, and J.\u00a0Wortman Vaughan (Eds.). https:\/\/openreview.net\/forum?id=AAWuCvzaVt"},{"key":"e_1_3_3_2_19_1","volume-title":"The Eleventh International Conference on Learning Representations","author":"Gal Rinon","year":"2022","unstructured":"Rinon Gal, Yuval Alaluf, Yuval Atzmon, Or Patashnik, Amit\u00a0Haim Bermano, Gal Chechik, and Daniel Cohen-or. 2022. An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"crossref","unstructured":"Rinon Gal Moab Arar Yuval Atzmon Amit\u00a0H Bermano Gal Chechik and Daniel Cohen-Or. 2023. Encoder-based domain tuning for fast personalization of text-to-image models. ACM Transactions on Graphics (TOG) 42 4 (2023) 1\u201313.","DOI":"10.1145\/3592133"},{"key":"e_1_3_3_2_21_1","unstructured":"Yuchao Gu Xintao Wang Jay\u00a0Zhangjie Wu Yujun Shi Yunpeng Chen Zihan Fan Wuyou Xiao Rui Zhao Shuning Chang Weijia Wu et\u00a0al. 2024. Mix-of-show: Decentralized low-rank adaptation for multi-concept customization of diffusion models. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_2_22_1","unstructured":"Yutong He Ruslan Salakhutdinov and J\u00a0Zico Kolter. 2023. Localized text-to-image generation for free via cross attention control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.14636 (2023)."},{"key":"e_1_3_3_2_23_1","volume-title":"ICLR","author":"Hertz Amir","year":"2023","unstructured":"Amir Hertz, Ron Mokady, Jay Tenenbaum, Kfir Aberman, Yael Pritch, and Daniel Cohen-Or. 2023. Prompt-to-prompt image editing with cross attention control. In ICLR."},{"key":"e_1_3_3_2_24_1","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems 33 (2020) 6840\u20136851."},{"key":"e_1_3_3_2_25_1","volume-title":"NeurIPS 2021 Workshop on Deep Generative Models and Downstream Applications","author":"Ho Jonathan","year":"2021","unstructured":"Jonathan Ho and Tim Salimans. 2021. Classifier-Free Diffusion Guidance. In NeurIPS 2021 Workshop on Deep Generative Models and Downstream Applications."},{"key":"e_1_3_3_2_26_1","unstructured":"HuggingFace. 2024. Diffusers. https:\/\/huggingface.co\/docs\/diffusers\/en\/using-diffusers\/ip_adapter."},{"key":"e_1_3_3_2_27_1","first-page":"4651","volume-title":"International conference on machine learning","author":"Jaegle Andrew","year":"2021","unstructured":"Andrew Jaegle, Felix Gimeno, Andy Brock, Oriol Vinyals, Andrew Zisserman, and Joao Carreira. 2021. Perceiver: General perception with iterative attention. In International conference on machine learning. PMLR, 4651\u20134664."},{"key":"e_1_3_3_2_28_1","unstructured":"Xuhui Jia Yang Zhao Kelvin\u00a0CK Chan Yandong Li Han Zhang Boqing Gong Tingbo Hou Huisheng Wang and Yu-Chuan Su. 2023. Taming encoder for zero fine-tuning image customization with text-to-image diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.02642 (2023)."},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00976"},{"key":"e_1_3_3_2_30_1","unstructured":"Chanran Kim Jeongin Lee Shichang Joung Bongmo Kim and Yeul-Min Baek. 2024. InstantFamily: Masked Attention for Zero-shot Multi-ID Image Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.19427 (2024)."},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00708"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"crossref","unstructured":"Harold\u00a0W Kuhn. 1955. The Hungarian method for the assignment problem. Naval research logistics quarterly 2 1-2 (1955) 83\u201397.","DOI":"10.1002\/nav.3800020109"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"e_1_3_3_2_34_1","unstructured":"Black\u00a0Forest Labs. 2024. FLUX. https:\/\/github.com\/black-forest-labs\/flux."},{"key":"e_1_3_3_2_35_1","unstructured":"Dongxu Li Junnan Li and Steven Hoi. 2024. Blip-diffusion: Pre-trained subject representation for controllable text-to-image generation and editing. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01611"},{"key":"e_1_3_3_2_39_1","unstructured":"Maxime Oquab Timoth\u00e9e Darcet Th\u00e9o Moutakanni Huy\u00a0V Vo Marc Szafraniec Vasil Khalidov Pierre Fernandez Daniel HAZIZA Francisco Massa Alaaeldin El-Nouby et\u00a0al. 2023. DINOv2: Learning Robust Visual Features without Supervision. Transactions on Machine Learning Research (2023)."},{"key":"e_1_3_3_2_40_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Pan Xichen","year":"2024","unstructured":"Xichen Pan, Li Dong, Shaohan Huang, Zhiliang Peng, Wenhu Chen, and Furu Wei. 2024. Kosmos-G: Generating Images in Context with Multimodal Large Language Models. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_3_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591513"},{"key":"e_1_3_3_2_42_1","unstructured":"Gaurav Parmar Taesung Park Srinivasa Narasimhan and Jun-Yan Zhu. 2024. One-step image translation with text-to-image models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.12036 (2024)."},{"key":"e_1_3_3_2_43_1","unstructured":"Maitreya Patel Sangmin Jung Chitta Baral and Yezhou Yang. 2024. \u03bb -ECLIPSE: Multi-Concept Personalized Text-to-Image Diffusion Models by Leveraging CLIP Latent Space. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.05195 (2024)."},{"key":"e_1_3_3_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.00758"},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00758"},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00761"},{"key":"e_1_3_3_2_47_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2023. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_3_2_48_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Podell Dustin","year":"2024","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2024. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=di52zR8xgf"},{"key":"e_1_3_3_2_49_1","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_50_1","unstructured":"Aditya Ramesh Prafulla Dhariwal Alex Nichol Casey Chu and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.06125 1 2 (2022) 3."},{"key":"e_1_3_3_2_51_1","unstructured":"Nikhila Ravi Valentin Gabeur Yuan-Ting Hu Ronghang Hu Chaitanya Ryali Tengyu Ma Haitham Khedr Roman R\u00e4dle Chloe Rolland Laura Gustafson Eric Mintun Junting Pan Kalyan\u00a0Vasudev Alwala Nicolas Carion Chao-Yuan Wu Ross Girshick Piotr Doll\u00e1r and Christoph Feichtenhofer. 2024. SAM 2: Segment Anything in Images and Videos. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.00714 (2024). https:\/\/arxiv.org\/abs\/2408.00714"},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_3_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00624"},{"key":"e_1_3_3_2_55_1","unstructured":"Chitwan Saharia William Chan Saurabh Saxena Lala Li Jay Whang Emily\u00a0L Denton Kamyar Ghasemipour Raphael Gontijo\u00a0Lopes Burcu Karagol\u00a0Ayan Tim Salimans et\u00a0al. 2022. Photorealistic text-to-image diffusion models with deep language understanding. Advances in neural information processing systems 35 (2022) 36479\u201336494."},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00816"},{"key":"e_1_3_3_2_57_1","first-page":"2256","volume-title":"International conference on machine learning","author":"Sohl-Dickstein Jascha","year":"2015","unstructured":"Jascha Sohl-Dickstein, Eric Weiss, Niru Maheswaranathan, and Surya Ganguli. 2015. Deep unsupervised learning using nonequilibrium thermodynamics. In International conference on machine learning. PMLR, 2256\u20132265."},{"key":"e_1_3_3_2_58_1","volume-title":"International Conference on Learning Representations","author":"Song Yang","year":"2021","unstructured":"Yang Song, Jascha Sohl-Dickstein, Diederik\u00a0P Kingma, Abhishek Kumar, Stefano Ermon, and Ben Poole. 2021. Score-Based Generative Modeling through Stochastic Differential Equations. In International Conference on Learning Representations."},{"key":"e_1_3_3_2_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591506"},{"key":"e_1_3_3_2_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687662"},{"key":"e_1_3_3_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00596"},{"key":"e_1_3_3_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01461"},{"key":"e_1_3_3_2_63_1","unstructured":"Guangxuan Xiao Tianwei Yin William\u00a0T Freeman Fr\u00e9do Durand and Song Han. 2024. Fastcomposer: Tuning-free multi-subject image generation with localized attention. International Journal of Computer Vision (2024) 1\u201320."},{"key":"e_1_3_3_2_64_1","unstructured":"Hu Ye Jun Zhang Sibo Liu Xiao Han and Wei Yang. 2023. Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.06721 (2023)."},{"key":"e_1_3_3_2_65_1","unstructured":"Lu Yuan Dongdong Chen Yi-Ling Chen Noel Codella Xiyang Dai Jianfeng Gao Houdong Hu Xuedong Huang Boxin Li Chunyuan Li et\u00a0al. 2021. Florence: A new foundation model for computer vision. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2111.11432 (2021)."},{"key":"e_1_3_3_2_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00648"},{"key":"e_1_3_3_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_3_2_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"e_1_3_3_2_69_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Zhao Shengyu","year":"2021","unstructured":"Shengyu Zhao, Jonathan Cui, Yilun Sheng, Yue Dong, Xiao Liang, Eric\u00a0I Chang, and Yan Xu. 2021. Large Scale Image Completion via Co-Modulated Generative Adversarial Networks. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_2_70_1","first-page":"465","volume-title":"Advances in Neural Information Processing Systems 30","author":"Zhu Jun-Yan","year":"2017","unstructured":"Jun-Yan Zhu, Richard Zhang, Deepak Pathak, Trevor Darrell, Alexei\u00a0A Efros, Oliver Wang, and Eli Shechtman. 2017. Toward Multimodal Image-to-Image Translation. In Advances in Neural Information Processing Systems 30, I.\u00a0Guyon, U.\u00a0V. Luxburg, S.\u00a0Bengio, H.\u00a0Wallach, R.\u00a0Fergus, S.\u00a0Vishwanathan, and R.\u00a0Garnett (Eds.). Curran Associates, Inc., 465\u2013476. http:\/\/papers.nips.cc\/paper\/6650-toward-multimodal-image-to-image-translation.pdf"}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763867","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:34:14Z","timestamp":1765251254000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763867"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":69,"alternative-id":["10.1145\/3757377.3763867","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763867","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}