{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:07:49Z","timestamp":1784268469398,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":57,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,3]]},"DOI":"10.1145\/3680528.3687662","type":"proceedings-article","created":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T08:14:37Z","timestamp":1733213677000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":13,"title":["MoA: Mixture-of-Attention for Subject-Context Disentanglement in Personalized Image Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6785-8146","authenticated-orcid":false,"given":"Kuan-Chieh","family":"Wang","sequence":"first","affiliation":[{"name":"Snap Inc., Mountain View, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7259-7399","authenticated-orcid":false,"given":"Daniil","family":"Ostashev","sequence":"additional","affiliation":[{"name":"Snap Inc., London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8879-0178","authenticated-orcid":false,"given":"Yuwei","family":"Fang","sequence":"additional","affiliation":[{"name":"Snap Inc., Seattle, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3465-1592","authenticated-orcid":false,"given":"Sergey","family":"Tulyakov","sequence":"additional","affiliation":[{"name":"Snap Inc., Los Angeles, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4958-601X","authenticated-orcid":false,"given":"Kfir","family":"Aberman","sequence":"additional","affiliation":[{"name":"Snap Inc., Palo Alto, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,12,3]]},"reference":[{"key":"e_1_3_3_2_2_1","unstructured":"2022. Low-rank Adaptation for Fast Text-to-Image Diffusion Fine-tuning. https:\/\/github.com\/cloneofsimo\/lora."},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"crossref","unstructured":"Yuval Alaluf Elad Richardson Gal Metzer and Daniel Cohen-Or. 2023. A Neural Space-Time Representation for Text-to-Image Personalization. ACM Transactions on Graphics (TOG) 42 6 (2023) 1\u201310.","DOI":"10.1145\/3618322"},{"key":"e_1_3_3_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618173"},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"crossref","unstructured":"Omri Avrahami Kfir Aberman Ohad Fried Daniel Cohen-Or and Dani Lischinski. 2023. Break-A-Scene: Extracting Multiple Concepts from a Single Image. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.16311 (2023).","DOI":"10.1145\/3610548.3618154"},{"key":"e_1_3_3_2_6_1","unstructured":"Omer Bar-Tal Lior Yariv Yaron Lipman and Tali Dekel. 2023. MultiDiffusion: Fusing Diffusion Paths for Controlled Image Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.08113 (2023)."},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"e_1_3_3_2_8_1","unstructured":"Prafulla Dhariwal and Alexander Nichol. 2021. Diffusion models beat gans on image synthesis. Advances in neural information processing systems 34 (2021) 8780\u20138794."},{"key":"e_1_3_3_2_9_1","unstructured":"William Fedus Barret Zoph and Noam Shazeer. 2022. Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity. The Journal of Machine Learning Research 23 1 (2022) 5232\u20135270."},{"key":"e_1_3_3_2_10_1","unstructured":"Rinon Gal Yuval Alaluf Yuval Atzmon Or Patashnik Amit\u00a0H Bermano Gal Chechik and Daniel Cohen-Or. 2022. An image is worth one word: Personalizing text-to-image generation using textual inversion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2208.01618 (2022)."},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"crossref","unstructured":"Rinon Gal Moab Arar Yuval Atzmon Amit\u00a0H Bermano Gal Chechik and Daniel Cohen-Or. 2023. Encoder-based domain tuning for fast personalization of text-to-image models. ACM Transactions on Graphics (TOG) 42 4 (2023) 1\u201313.","DOI":"10.1145\/3592133"},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"crossref","unstructured":"Rinon Gal Or Lichter Elad Richardson Or Patashnik Amit\u00a0H Bermano Gal Chechik and Daniel Cohen-Or. 2024. LCM-Lookahead for Encoder-based Text-to-Image Personalization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.03620 (2024).","DOI":"10.1007\/978-3-031-72630-9_19"},{"key":"e_1_3_3_2_13_1","unstructured":"Yuchao Gu Xintao Wang Jay\u00a0Zhangjie Wu Yujun Shi Yunpeng Chen Zihan Fan Wuyou Xiao Rui Zhao Shuning Chang Weijia Wu et\u00a0al. 2024. Mix-of-show: Decentralized low-rank adaptation for multi-concept customization of diffusion models. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_2_14_1","unstructured":"Ligong Han Yinxiao Li Han Zhang Peyman Milanfar Dimitris Metaxas and Feng Yang. 2023. Svdiff: Compact parameter space for diffusion fine-tuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.11305 (2023)."},{"key":"e_1_3_3_2_15_1","unstructured":"Amir Hertz Ron Mokady Jay Tenenbaum Kfir Aberman Yael Pritch and Daniel Cohen-Or. 2023. Prompt-to-Prompt Image Editing with Cross Attention Control. ICLR (2023)."},{"key":"e_1_3_3_2_16_1","unstructured":"Martin Heusel Hubert Ramsauer Thomas Unterthiner Bernhard Nessler and Sepp Hochreiter. 2017. Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_2_17_1","unstructured":"Jonathan Ho. 2022. Classifier-Free Diffusion Guidance. ArXiv abs\/2207.12598 (2022). https:\/\/api.semanticscholar.org\/CorpusID:249145348"},{"key":"e_1_3_3_2_18_1","unstructured":"Jonathan Ho Ajay Jain and P. Abbeel. 2020. Denoising Diffusion Probabilistic Models. ArXiv abs\/2006.11239 (2020). https:\/\/api.semanticscholar.org\/CorpusID:219955663"},{"key":"e_1_3_3_2_19_1","volume-title":"ICLR","author":"Hu Edward\u00a0J","year":"2022","unstructured":"Edward\u00a0J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In ICLR."},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"crossref","unstructured":"Robert\u00a0A Jacobs Michael\u00a0I Jordan Steven\u00a0J Nowlan and Geoffrey\u00a0E Hinton. 1991. Adaptive mixtures of local experts. Neural computation 3 1 (1991) 79\u201387.","DOI":"10.1162\/neco.1991.3.1.79"},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00453"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01639"},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"e_1_3_3_2_25_1","unstructured":"Dongxu Li Junnan Li and Steven Hoi. 2024. Blip-diffusion: Pre-trained subject representation for controllable text-to-image generation and editing. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_2_26_1","unstructured":"Junnan Li Dongxu Li Silvio Savarese and Steven Hoi. 2023b. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.12597 (2023)."},{"key":"e_1_3_3_2_27_1","unstructured":"Zhen Li Mingdeng Cao Xintao Wang Zhongang Qi Ming-Ming Cheng and Ying Shan. 2023a. Photomaker: Customizing realistic human photos via stacked id embedding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.04461 (2023)."},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.425"},{"key":"e_1_3_3_2_29_1","unstructured":"Zhiheng Liu Yifei Zhang Yujun Shen Kecheng Zheng Kai Zhu Ruili Feng Yu Liu Deli Zhao Jingren Zhou and Yang Cao. 2023a. Cones 2: Customizable image synthesis with multiple subjects. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.19327 (2023)."},{"key":"e_1_3_3_2_30_1","unstructured":"Zhiheng Liu Yifei Zhang Yujun Shen Kecheng Zheng Kai Zhu Ruili Feng Yu Liu Deli Zhao Jingren Zhou and Yang Cao. 2023b. Cones 2: Customizable image synthesis with multiple subjects. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.19327 (2023)."},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00585"},{"key":"e_1_3_3_2_32_1","first-page":"8162","volume-title":"International Conference on Machine Learning","author":"Nichol Alexander\u00a0Quinn","year":"2021","unstructured":"Alexander\u00a0Quinn Nichol and Prafulla Dhariwal. 2021. Improved denoising diffusion probabilistic models. In International Conference on Machine Learning. PMLR, 8162\u20138171."},{"key":"e_1_3_3_2_33_1","unstructured":"Kushagra Pandey Avideep Mukherjee Piyush Rai and Abhishek Kumar. 2022. DiffuseVAE: Efficient Controllable and High-Fidelity Generation from Low-Dimensional Latents. Trans. Mach. Learn. Res. 2022 (2022). https:\/\/api.semanticscholar.org\/CorpusID:245650542"},{"key":"e_1_3_3_2_34_1","unstructured":"Ryan Po Guandao Yang Kfir Aberman and Gordon Wetzstein. 2023. Orthogonal adaptation for modular customization of diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.02432 (2023)."},{"key":"e_1_3_3_2_35_1","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_36_1","unstructured":"Stephen Roller Sainbayar Sukhbaatar Jason Weston et\u00a0al. 2021. Hash layers for large sparse models. Advances in Neural Information Processing Systems 34 (2021) 17555\u201317566."},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"crossref","unstructured":"Robin Rombach A. Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2021. High-Resolution Image Synthesis with Latent Diffusion Models. 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2021) 10674\u201310685. https:\/\/api.semanticscholar.org\/CorpusID:245335280","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_3_2_40_1","unstructured":"Nataniel Ruiz Yuanzhen Li Varun Jampani Wei Wei Tingbo Hou Yael Pritch Neal Wadhwa Michael Rubinstein and Kfir Aberman. 2023b. Hyperdreambooth: Hypernetworks for fast personalization of text-to-image models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.06949 (2023)."},{"key":"e_1_3_3_2_41_1","doi-asserted-by":"crossref","unstructured":"Michael\u00a0S Ryoo Chia-Chih Chen JK Aggarwal and Amit Roy-Chowdhury. 2010. An overview of contest on semantic description of human activities (sdha) 2010. Recognizing Patterns in Signals Speech Images and Videos: ICPR 2010 Contests Istanbul Turkey August 23-26 2010 Contest Reports (2010) 270\u2013285.","DOI":"10.1007\/978-3-642-17711-8_28"},{"key":"e_1_3_3_2_42_1","first-page":"36479","volume-title":"NeurIPS","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily\u00a0L Denton, Kamyar Ghasemipour, Raphael Gontijo\u00a0Lopes, Burcu Karagol\u00a0Ayan, Tim Salimans, et\u00a0al. 2022. Photorealistic text-to-image diffusion models with deep language understanding. In NeurIPS. 36479\u201336494."},{"key":"e_1_3_3_2_43_1","volume-title":"International Conference on Learning Representations","author":"Shazeer Noam","year":"2017","unstructured":"Noam Shazeer, *Azalia Mirhoseini, *Krzysztof Maziarz, Andy Davis, Quoc Le, Geoffrey Hinton, and Jeff Dean. 2017. Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B1ckMDqlg"},{"key":"e_1_3_3_2_44_1","unstructured":"Jing Shi Wei Xiong Zhe Lin and Hyun\u00a0Joon Jung. 2023. Instantbooth: Personalized text-to-image generation without test-time finetuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.03411 (2023)."},{"key":"e_1_3_3_2_45_1","unstructured":"Jiaming Song Chenlin Meng and Stefano Ermon. 2020a. Denoising Diffusion Implicit Models. ArXiv abs\/2010.02502 (2020). https:\/\/api.semanticscholar.org\/CorpusID:222140788"},{"key":"e_1_3_3_2_46_1","unstructured":"Jiaming Song Chenlin Meng and Stefano Ermon. 2020b. Denoising diffusion implicit models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.02502 (2020)."},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591506"},{"key":"e_1_3_3_2_48_1","unstructured":"Andrey Voynov Qinghao Chu Daniel Cohen-Or and Kfir Aberman. 2023. P + : Extended Textual Conditioning in Text-to-Image Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.09522 (2023)."},{"key":"e_1_3_3_2_49_1","unstructured":"Qixun Wang Xu Bai Haofan Wang Zekui Qin and Anthony Chen. 2024. Instantid: Zero-shot identity-preserving generation in seconds. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.07519 (2024)."},{"key":"e_1_3_3_2_50_1","unstructured":"Yibin Wang Weizhong Zhang Jianwei Zheng and Cheng Jin. 2023. High-fidelity Person-centric Subject-to-Image Synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.10329 (2023)."},{"key":"e_1_3_3_2_51_1","unstructured":"Yuxiang Wei Yabo Zhang Zhilong Ji Jinfeng Bai Lei Zhang and Wangmeng Zuo. 2023. Elite: Encoding visual concepts into textual embeddings for customized text-to-image generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.13848 (2023)."},{"key":"e_1_3_3_2_52_1","unstructured":"Xiaoshi Wu Yiming Hao Keqiang Sun Yixiong Chen Feng Zhu Rui Zhao and Hongsheng Li. 2023. Human preference score v2: A solid benchmark for evaluating human preferences of text-to-image synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.09341 (2023)."},{"key":"e_1_3_3_2_53_1","unstructured":"Guangxuan Xiao Tianwei Yin William\u00a0T Freeman Fr\u00e9do Durand and Song Han. 2023. FastComposer: Tuning-Free Multi-Subject Image Generation with Localized Attention. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.10431 (2023)."},{"key":"e_1_3_3_2_54_1","unstructured":"Hu Ye Jun Zhang Sibo Liu Xiao Han and Wei Yang. 2023. IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models. (2023)."},{"key":"e_1_3_3_2_55_1","unstructured":"Jiahui Yu Yuanzhong Xu Jing\u00a0Yu Koh Thang Luong Gunjan Baid Zirui Wang Vijay Vasudevan Alexander Ku Yinfei Yang Burcu\u00a0Karagol Ayan et\u00a0al. 2022. Scaling autoregressive models for content-rich text-to-image generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2206.10789 2 3 (2022) 5."},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_3_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"e_1_3_3_2_58_1","doi-asserted-by":"crossref","unstructured":"Yuxin Zhang Weiming Dong Fan Tang Nisha Huang Haibin Huang Chongyang Ma Tong-Yee Lee Oliver Deussen and Changsheng Xu. 2023a. Prospect: Prompt spectrum for attribute-aware personalization of diffusion models. ACM Transactions on Graphics (TOG) 42 6 (2023) 1\u201314.","DOI":"10.1145\/3618342"}],"event":{"name":"SA '24: SIGGRAPH Asia 2024 Conference Papers","location":"Tokyo Japan","acronym":"SA '24","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["SIGGRAPH Asia 2024 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3680528.3687662","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3680528.3687662","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:20Z","timestamp":1750295900000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3680528.3687662"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"references-count":57,"alternative-id":["10.1145\/3680528.3687662","10.1145\/3680528"],"URL":"https:\/\/doi.org\/10.1145\/3680528.3687662","relation":{},"subject":[],"published":{"date-parts":[[2024,12,3]]},"assertion":[{"value":"2024-12-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}