{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T19:06:21Z","timestamp":1784228781471,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,19]]},"DOI":"10.1145\/3799902.3811219","type":"proceedings-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:15:27Z","timestamp":1784218527000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Canvas-to-Image: Compositional Image Generation with Multimodal Controls"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8402-8291","authenticated-orcid":false,"given":"Yusuf","family":"Dalva","sequence":"first","affiliation":[{"name":"Computer Science, Virginia Tech University, Blacksburg, Virginia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2935-8570","authenticated-orcid":false,"given":"Gordon Guocheng","family":"Qian","sequence":"additional","affiliation":[{"name":"Snap, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-8367-7876","authenticated-orcid":false,"given":"Maya","family":"Goldenberg","sequence":"additional","affiliation":[{"name":"Snap, Tel Aviv, Israel"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8085-0042","authenticated-orcid":false,"given":"Tsai-Shien","family":"Chen","sequence":"additional","affiliation":[{"name":"Snap, Santa Monica, California, USA and Computer Science, University of California Merced, Merced, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4958-601X","authenticated-orcid":false,"given":"Kfir","family":"Aberman","sequence":"additional","affiliation":[{"name":"Snap, Palo Alto, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3465-1592","authenticated-orcid":false,"given":"Sergey","family":"Tulyakov","sequence":"additional","affiliation":[{"name":"Snap, Santa Monica, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3452-7417","authenticated-orcid":false,"given":"Pinar","family":"Yanardag","sequence":"additional","affiliation":[{"name":"Computer Science, Virginia Tech University, Blacksburg, Virginia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6785-8146","authenticated-orcid":false,"given":"Kuan-Chieh Jackson","family":"Wang","sequence":"additional","affiliation":[{"name":"Snap, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618154"},{"key":"e_1_3_3_2_3_1","unstructured":"Black Forest Labs. 2024. Flux. https:\/\/github.com\/black-forest-labs\/flux."},{"key":"e_1_3_3_2_4_1","unstructured":"Black Forest Labs. 2025. FLUX. 1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.15742 (2025)."},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"crossref","unstructured":"Zhe Cao Gines Hidalgo Tomas Simon Shih-En Wei and Yaser Sheikh. 2019. Openpose: Realtime multi-person 2d pose estimation using part affinity fields. IEEE transactions on pattern analysis and machine intelligence 43 1 (2019) 172\u2013186.","DOI":"10.1109\/TPAMI.2019.2929257"},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00572"},{"key":"e_1_3_3_2_7_1","unstructured":"Yusuf Dalva Hidir Yesiltepe and Pinar Yanardag. 2025. LoRAShop: Training-Free Multi-Concept Image Generation and Editing with Rectified Flow Transformers. CoRR abs\/2505.23758 (2025)."},{"key":"e_1_3_3_2_8_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","volume":"2505","author":"Deng Chaorui","year":"2025","unstructured":"Chaorui Deng, Deyao Zhu, Kunchang Li, Chenhui Gou, Feng Li, Zeyu Wang, Shu Zhong, Weihao Yu, Xiaonan Nie, Ziang Song, Shi Guang, and Haoqi Fan. 2025. Emerging Properties in Unified Multimodal Pretraining. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , Vol.\u00a0abs\/2505.14683."},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"crossref","unstructured":"Jiankang Deng Jia Guo Jing Yang Niannan Xue Irene Kotsia and Stefanos Zafeiriou. 2022. ArcFace: Additive Angular Margin Loss for Deep Face Recognition. IEEE Trans. Pattern Anal. Mach. Intell. 44 10 (2022) 5962\u20135979.","DOI":"10.1109\/TPAMI.2021.3087709"},{"key":"e_1_3_3_2_10_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML)","author":"Esser Patrick","year":"2024","unstructured":"Patrick Esser, Sumith Kulal, Andreas Blattmann, Rahim Entezari, Jonas M\u00fcller, Harry Saini, Yam Levi, Dominik Lorenz, Axel Sauer, Frederic Boesel, Dustin Podell, Tim Dockhorn, Zion English, and Robin Rombach. 2024. Scaling Rectified Flow Transformers for High-Resolution Image Synthesis. In Proceedings of the International Conference on Machine Learning (ICML). OpenReview.net."},{"key":"e_1_3_3_2_11_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Gal Rinon","year":"2023","unstructured":"Rinon Gal, Yuval Alaluf, Yuval Atzmon, Or Patashnik, Amit\u00a0Haim Bermano, Gal Chechik, and Daniel Cohen-Or. 2023. An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion. In International Conference on Learning Representations (ICLR). OpenReview.net."},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"crossref","unstructured":"Daniel Garibi Shahar Yadin Roni Paiss Omer Tov Shiran Zada Ariel Ephrat Tomer Michaeli Inbar Mosseri and Tali Dekel. 2025. Tokenverse: Versatile multi-concept personalization in token modulation space. ACM Transactions on Graphics (TOG) 44 4 (2025) 1\u201311.","DOI":"10.1145\/3730843"},{"key":"e_1_3_3_2_13_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Goyal Anujraaj\u00a0Argo","year":"2025","unstructured":"Anujraaj\u00a0Argo Goyal, Guocheng\u00a0Gordon Qian, Huseyin Coskun, Aarush Gupta, Himmy Tam, Daniil Ostashev, Ju Hu, Dhritiman Sagar, Sergey Tulyakov, Kfir Aberman, and Kuan-Chieh\u00a0Jackson Wang. 2025. Preventing Shortcuts in Adapter Training via Providing the Shortcuts. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1159"},{"key":"e_1_3_3_2_15_1","unstructured":"Yucheng Han Rui Wang Chi Zhang Juntao Hu Pei Cheng Bin Fu and Hanwang Zhang. 2024. EMMA: Your Text-to-Image Diffusion Model Can Secretly Accept Multi-Modal Prompts. CoRR abs\/2406.09162 (2024)."},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01336"},{"key":"e_1_3_3_2_17_1","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. NeurIPS 33 (2020) 6840\u20136851."},{"key":"e_1_3_3_2_18_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Hu Edward\u00a0J.","year":"2022","unstructured":"Edward\u00a0J. Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations (ICLR). OpenReview.net."},{"key":"e_1_3_3_2_19_1","first-page":"13753","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Huang Lianghua","year":"2023","unstructured":"Lianghua Huang, Di Chen, Yu Liu, Yujun Shen, Deli Zhao, and Jingren Zhou. 2023. Composer: creative and controllable image synthesis with composable conditions. In Proceedings of the 40th International Conference on Machine Learning. 13753\u201313773."},{"key":"e_1_3_3_2_20_1","unstructured":"ilkerzgi and gokaygokay. 2025. Overlay-Kontext-Dev-LoRA. https:\/\/huggingface.co\/ilkerzgi\/Overlay-Kontext-Dev-LoRA. LoRA fine-tune of FLUX.1-Kontext-dev for image overlay tasks."},{"key":"e_1_3_3_2_21_1","first-page":"253","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV)","author":"Kong Zhe","year":"2024","unstructured":"Zhe Kong, Yong Zhang, Tianyu Yang, Tao Wang, Kaihao Zhang, Bizhu Wu, Guanying Chen, Wei Liu, and Wenhan Luo. 2024. Omg: Occlusion-friendly personalized multi-concept generation in diffusion models. In Proceedings of the European Conference on Computer Vision (ECCV). Springer, 253\u2013270."},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"e_1_3_3_2_24_1","first-page":"366","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV)","author":"Lin Zhiqiu","year":"2024","unstructured":"Zhiqiu Lin, Deepak Pathak, Baiqi Li, Jiayao Li, Xide Xia, Graham Neubig, Pengchuan Zhang, and Deva Ramanan. 2024. Evaluating text-to-visual generation with image-to-text generation. In Proceedings of the European Conference on Computer Vision (ECCV). Springer, 366\u2013384."},{"key":"e_1_3_3_2_25_1","volume-title":"ICLR","author":"Loshchilov Ilya","year":"2019","unstructured":"Ilya Loshchilov and Frank Hutter. 2019. Decoupled Weight Decay Regularization. In ICLR."},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01400"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3757377.3763956"},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3721238.3730634"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00761"},{"key":"e_1_3_3_2_32_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Podell Dustin","year":"2024","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2024. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. In International Conference on Learning Representations (ICLR). OpenReview.net."},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00821"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3757377.3763984"},{"key":"e_1_3_3_2_35_1","unstructured":"Guocheng\u00a0Gordon Qian Ruihang Zhang Tsai-Shien Chen Yusuf Dalva Anujraaj Goyal Willi Menapace Ivan Skorokhodov Daniil Ostashev Meng Dong Arpit Sahni Ju Hu Sergey Tulyakov and Kuan-Chieh\u00a0Jackson Wang. 2025c. LayerComposer: Multi-Human Personalized Generation via Layered Canvas. arXiv (2025)."},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"crossref","unstructured":"Can Qin Shu Zhang Ning Yu Yihao Feng Xinyi Yang Yingbo Zhou Huan Wang Juan\u00a0Carlos Niebles Caiming Xiong Silvio Savarese et\u00a0al. 2023. UniControl: A Unified Diffusion Model for Controllable Visual Generation In the Wild. Advances in Neural Information Processing Systems 36 (2023) 42961\u201342992.","DOI":"10.52202\/075280-1862"},{"key":"e_1_3_3_2_37_1","unstructured":"Aditya Ramesh Prafulla Dhariwal Alex Nichol Casey Chu and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.06125 (2022)."},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2643"},{"key":"e_1_3_3_2_41_1","unstructured":"Jiaming Song Chenlin Meng and Stefano Ermon. 2020. Denoising diffusion implicit models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.02502 (2020)."},{"key":"e_1_3_3_2_42_1","unstructured":"Gemini Team. 2025. Gemini 2.5: Pushing the Frontier with Advanced Reasoning Multimodality Long Context and Next Generation Agentic Capabilities. CoRR abs\/2507.06261 (2025)."},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687662"},{"key":"e_1_3_3_2_44_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Yang Fan Kai Dang Mengfei Du Xuancheng Ren Rui Men Dayiheng Liu Chang Zhou Jingren Zhou and Junyang Lin. 2024a. Qwen2-VL: Enhancing Vision-Language Model\u2019s Perception of the World at Any Resolution. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.12191 (2024)."},{"key":"e_1_3_3_2_45_1","unstructured":"Qixun Wang Xu Bai Haofan Wang Zekui Qin Anthony Chen Huaxia Li Xu Tang and Yao Hu. 2024b. Instantid: Zero-shot identity-preserving generation in seconds. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.07519 (2024)."},{"key":"e_1_3_3_2_46_1","unstructured":"Chenfei Wu Jiahao Li Jingren Zhou Junyang Lin Kaiyuan Gao Kun Yan Shengming Yin Shuai Bai Xiao Xu Yilei Chen Yuxiang Chen Zecheng Tang Zekai Zhang Zhengyi Wang An Yang Bowen Yu Chen Cheng Dayiheng Liu Deqing Li Hang Zhang Hao Meng Hu Wei Jingyuan Ni Kai Chen Kuan Cao Liang Peng Lin Qu Minggang Wu Peng Wang Shuting Yu Tingkun Wen Wensen Feng Xiaoxiao Xu Yi Wang Yichang Zhang Yongqiang Zhu Yujia Wu Yuxuan Cai and Zenan Liu. 2025b. Qwen-Image Technical Report. CoRR abs\/2508.02324 (2025)."},{"key":"e_1_3_3_2_47_1","unstructured":"Chenyuan Wu Pengfei Zheng Ruiran Yan Shitao Xiao Xin Luo Yueze Wang Wanli Li Xiyan Jiang Yexin Liu Junjie Zhou Ze Liu Ziyi Xia Chaofan Li Haoge Deng Jiahao Wang Kun Luo Bo Zhang Defu Lian Xinlong Wang Zhongyuan Wang Tiejun Huang and Zheng Liu. 2025c. OmniGen2: Exploration to Advanced Multimodal Generation. CoRR abs\/2506.18871 (2025)."},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01736"},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"crossref","unstructured":"Guangxuan Xiao Tianwei Yin William\u00a0T Freeman Fr\u00e9do Durand and Song Han. 2025. Fastcomposer: Tuning-free multi-subject image generation with localized attention. International Journal of Computer Vision 133 3 (2025) 1175\u20131194.","DOI":"10.1007\/s11263-024-02227-z"},{"key":"e_1_3_3_2_50_1","unstructured":"Hu Ye Jun Zhang Sibo Liu Xiao Han and Wei Yang. 2023. Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.06721 (2023)."},{"key":"e_1_3_3_2_51_1","unstructured":"Hui Zhang Dexiang Hong Maoke Yang Yutao Cheng Zhao Zhang Jie Shao Xinglong Wu Zuxuan Wu and Yu-Gang Jiang. 2025a. Creatidesign: A unified multi-conditional diffusion transformer for creative graphic design. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.19114 (2025)."},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_3_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00284"},{"key":"e_1_3_3_2_54_1","doi-asserted-by":"crossref","unstructured":"Shihao Zhao Dongdong Chen Yen-Chun Chen Jianmin Bao Shaozhe Hao Lu Yuan and Kwan-Yee\u00a0K. Wong. 2023. Uni-ControlNet: All-in-One Control to Text-to-Image Diffusion Models. Advances in Neural Information Processing Systems (2023).","DOI":"10.52202\/075280-0491"},{"key":"e_1_3_3_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02154"},{"key":"e_1_3_3_2_56_1","unstructured":"Zhengguang Zhou Jing Li Huaxia Li Nemo Chen and Xu Tang. 2024. Storymaker: Towards holistic consistent characters in text-to-image generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.12576 (2024)."}],"event":{"name":"SIGGRAPH Conference Papers '26: Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers","location":"Los Angeles CA USA","acronym":"SIGGRAPH Conference Papers '26","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T18:23:22Z","timestamp":1784226202000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3799902.3811219"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":55,"alternative-id":["10.1145\/3799902.3811219","10.1145\/3799902"],"URL":"https:\/\/doi.org\/10.1145\/3799902.3811219","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}