{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:20:45Z","timestamp":1765340445953,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372014, 62525201, 62132001, 62432001"],"award-info":[{"award-number":["62372014, 62525201, 62132001, 62432001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005090","name":"Beijing Nova Program","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100005090","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Beijing Natural Science Foundation","award":["4252040, L247006"],"award-info":[{"award-number":["4252040, L247006"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754911","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"9500-9508","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Interact-Custom: Customized Human Object Interaction Image Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5801-3734","authenticated-orcid":false,"given":"Zhu","family":"Xu","sequence":"first","affiliation":[{"name":"Wangxuan Institute of Computer Technology, Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-0180-760X","authenticated-orcid":false,"given":"Zhaowen","family":"Wang","sequence":"additional","affiliation":[{"name":"Adobe Research, San Jose, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7658-3845","authenticated-orcid":false,"given":"Yuxin","family":"Peng","sequence":"additional","affiliation":[{"name":"Wangxuan Institute of Computer Technology, Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4259-3882","authenticated-orcid":false,"given":"Yang","family":"Liu","sequence":"additional","affiliation":[{"name":"Wangxuan Institute of Computer Technology, Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.122"},{"key":"e_1_3_2_1_2_1","volume-title":"DisenBooth: Disentangled Parameter-Efficient Tuning for Subject-Driven Text-to-Image Generation. arXiv:2305.03374","author":"Chen Hong","year":"2023","unstructured":"Hong Chen, Yipeng Zhang, Xin Wang, Xuguang Duan, Yuwei Zhou, and Wenwu Zhu. 2023b. DisenBooth: Disentangled Parameter-Efficient Tuning for Subject-Driven Text-to-Image Generation. arXiv:2305.03374 (2023)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Xi Chen Lianghua Huang Yu Liu Yujun Shen Deli Zhao and Hengshuang Zhao. 2024. AnyDoor: Zero-shot Object-level Image Customization. arXiv:2307.09481 [cs.CV] https:\/\/arxiv.org\/abs\/2307.09481","DOI":"10.1109\/CVPR52733.2024.00630"},{"key":"e_1_3_2_1_4_1","volume-title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks. arXiv preprint arXiv:2312.14238","author":"Chen Zhe","year":"2023","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, Bin Li, Ping Luo, Tong Lu, Yu Qiao, and Jifeng Dai. 2023a. InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks. arXiv preprint arXiv:2312.14238 (2023)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3463944.3469097"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4471-0353-0_3"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Ganggui Ding Canyu Zhao Wen Wang Zhen Yang Zide Liu Hao Chen and Chunhua Shen. 2024. FreeCustom: Tuning-Free Customized Image Generation for Multi-Concept Composition. arXiv:2405.13870 [cs.CV] https:\/\/arxiv.org\/abs\/2405.13870","DOI":"10.1109\/CVPR52733.2024.00868"},{"key":"e_1_3_2_1_8_1","unstructured":"Rinon Gal Yuval Alaluf Yuval Atzmon Or Patashnik Amit H Bermano Gal Chechik and Daniel Cohen-Or. 2023. An image is worth one word: Personalizing text-to-image generation using textual inversion. In ICLR."},{"key":"e_1_3_2_1_9_1","volume-title":"ConMo: Controllable Motion Disentanglement and Recomposition for Zero-Shot Motion Transfer. arXiv preprint arXiv:2504.02451","author":"Gao Jiayi","year":"2025","unstructured":"Jiayi Gao, Zijin Yin, Changcheng Hua, Yuxin Peng, Kongming Liang, Zhanyu Ma, Jun Guo, and Yang Liu. 2025. ConMo: Controllable Motion Disentanglement and Recomposition for Zero-Shot Motion Transfer. arXiv preprint arXiv:2504.02451 (2025)."},{"key":"e_1_3_2_1_10_1","volume-title":"Yujun Shi, Yunpeng Chen, Zihan Fan, Wuyou Xiao, Rui Zhao, Shuning Chang, Weijia Wu, et al.","author":"Gu Yuchao","year":"2023","unstructured":"Yuchao Gu, Xintao Wang, Jay Zhangjie Wu, Yujun Shi, Yunpeng Chen, Zihan Fan, Wuyou Xiao, Rui Zhao, Shuning Chang, Weijia Wu, et al., 2023. Mix-of-Show: Decentralized Low-Rank Adaptation for Multi-Concept Customization of Diffusion Models. In NeurIPS."},{"key":"e_1_3_2_1_11_1","volume-title":"Visual Semantic Role Labeling. arXiv preprint arXiv:1505.04474","author":"Gupta Saurabh","year":"2015","unstructured":"Saurabh Gupta and Jitendra Malik. 2015. Visual Semantic Role Labeling. arXiv preprint arXiv:1505.04474 (2015)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00591"},{"key":"e_1_3_2_1_13_1","unstructured":"Qihan Huang Siming Fu Jinlong Liu Hao Jiang Yipeng Yu and Jie Song. 2024. Resolving Multi-Condition Confusion for Finetuning-Free Personalized Image Generation. arXiv:2409.17920 [cs.CV] https:\/\/arxiv.org\/abs\/2409.17920"},{"key":"e_1_3_2_1_14_1","volume-title":"Action Genome: Actions as Composition of Spatio-temporal Scene Graphs. arXiv:1912.06992 [cs.CV] https:\/\/arxiv.org\/abs\/1912.06992","author":"Ji Jingwei","year":"2019","unstructured":"Jingwei Ji, Ranjay Krishna, Li Fei-Fei, and Juan Carlos Niebles. 2019. Action Genome: Actions as Composition of Spatio-temporal Scene Graphs. arXiv:1912.06992 [cs.CV] https:\/\/arxiv.org\/abs\/1912.06992"},{"key":"e_1_3_2_1_15_1","volume-title":"Design of an image edge detection filter using the Sobel operator. JSSC","author":"Kanopoulos Nick","year":"1988","unstructured":"Nick Kanopoulos, Nagesh Vasanthavada, and Robert L Baker. 1988. Design of an image edge detection filter using the Sobel operator. JSSC (1988)."},{"key":"e_1_3_2_1_16_1","volume-title":"Adam: A method for stochastic optimization. arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Alexander Kirillov Eric Mintun Nikhila Ravi Hanzi Mao Chloe Rolland Laura Gustafson Tete Xiao Spencer Whitehead Alexander C Berg Wan-Yen Lo et al. 2023. Segment anything. In ICCV.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01576"},{"key":"e_1_3_2_1_19_1","volume-title":"European Conference on Computer Vision. Springer, 1-19","author":"Lei Ting","year":"2024","unstructured":"Ting Lei, Shaofeng Yin, Yuxin Peng, and Yang Liu. 2024b. Exploring conditional multi-modal prompts for zero-shot hoi detection. In European Conference on Computer Vision. Springer, 1-19."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1049\/cje.2020.00.088"},{"key":"e_1_3_2_1_21_1","volume-title":"GLIGEN: Open-Set Grounded Text-to-Image Generation. CVPR","author":"Li Yuheng","year":"2023","unstructured":"Yuheng Li, Haotian Liu, Qingyang Wu, Fangzhou Mu, Jianwei Yang, Jianfeng Gao, Chunyuan Li, and Yong Jae Lee. 2023b. GLIGEN: Open-Set Grounded Text-to-Image Generation. CVPR (2023)."},{"key":"e_1_3_2_1_22_1","unstructured":"Zhen Li Mingdeng Cao Xintao Wang Zhongang Qi Ming-Ming Cheng and Ying Shan. 2023a. PhotoMaker: Customizing Realistic Human Photos via Stacked ID Embedding. arXiv:2312.04461 [cs.CV] https:\/\/arxiv.org\/abs\/2312.04461"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.23919\/cje.2022.00.173"},{"volume-title":"Balancing Preservation and Modification: A Region and Semantic Aware Metric for Instruction-Based Image Editing. In Forty-second International Conference on Machine Learning.","author":"Li Zhuoying","key":"e_1_3_2_1_24_1","unstructured":"Zhuoying Li, Zhu Xu, Yuxin Peng, and Yang Liu. [n.d.]. Balancing Preservation and Modification: A Region and Semantic Aware Metric for Instruction-Based Image Editing. In Forty-second International Conference on Machine Learning."},{"key":"e_1_3_2_1_25_1","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong Jae Lee. 2023. Visual Instruction Tuning. arXiv:2304.08485 [cs.CV] https:\/\/arxiv.org\/abs\/2304.08485"},{"key":"e_1_3_2_1_26_1","unstructured":"Jian Ma Junhao Liang Chen Chen and Haonan Lu. 2024. Subject-Diffusion:Open Domain Personalized Text-to-Image Generation without Test-time Fine-tuning. arXiv:2307.11410 [cs.CV] https:\/\/arxiv.org\/abs\/2307.11410"},{"key":"e_1_3_2_1_27_1","volume-title":"FGAHOI: Fine-Grained Anchors forHuman-Object Interaction Detection.","author":"Ma Shuailei","year":"2023","unstructured":"Shuailei Ma, Yuefeng Wang, Shanze Wang, and Ying Wei. 2023. FGAHOI: Fine-Grained Anchors forHuman-Object Interaction Detection."},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV).","author":"Yuxin Peng Minghang Zheng Qingchao Chen","year":"2024","unstructured":"Qingchao Chen Yuxin Peng Minghang Zheng, Xinhao Cai and Yang Liu. 2024. Training Free Video Temporal Grounding using Large-scale Pre-trained Models. In Proceedings of the European Conference on Computer Vision (ECCV)."},{"key":"e_1_3_2_1_29_1","unstructured":"Maxime Oquab Timoth\u00e9e Darcet Th\u00e9o Moutakanni Huy Vo Marc Szafraniec Vasil Khalidov Pierre Fernandez Daniel Haziza Francisco Massa Alaaeldin El-Nouby et al. 2024. Dinov2: Learning robust visual features without supervision. TMLR (2024)."},{"key":"e_1_3_2_1_30_1","unstructured":"Yiming Qin Zhu Xu and Yang Liu. 2025. Apply Hierarchical-Chain-of-Generation to Complex Attributes Text-to-3D Generation. https:\/\/api.semanticscholar.org\/CorpusID:278481349"},{"key":"e_1_3_2_1_31_1","unstructured":"Tianhe Ren Shilong Liu Ailing Zeng Jing Lin Kunchang Li He Cao Jiayu Chen Xinyu Huang Yukang Chen Feng Yan Zhaoyang Zeng Hao Zhang Feng Li Jie Yang Hongyang Li Qing Jiang and Lei Zhang. 2024. Grounded SAM: Assembling Open-World Models for Diverse Visual Tasks. arXiv:2401.14159 [cs.CV]"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2022. High-resolution image synthesis with latent diffusion models. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_33_1","volume-title":"Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation. In CVPR.","author":"Ruiz Nataniel","year":"2023","unstructured":"Nataniel Ruiz, Yuanzhen Li, Varun Jampani, Yael Pritch, Michael Rubinstein, and Kfir Aberman. 2023. Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation. In CVPR."},{"key":"e_1_3_2_1_34_1","volume-title":"He Zhang, Wei Xiong, and Daniel Aliaga.","author":"Song Yizhi","year":"2024","unstructured":"Yizhi Song, Zhifei Zhang, Zhe Lin, Scott Cohen, Brian Price, Jianming Zhang, Soo Ye Kim, He Zhang, Wei Xiong, and Daniel Aliaga. 2024. IMPRINT: Generative Object Compositing by Learning Identity-Preserving Representation. arXiv:2403.10701 [cs.CV] https:\/\/arxiv.org\/abs\/2403.10701"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Yoad Tewel Rinon Gal Gal Chechik and Yuval Atzmon. 2024. Key-Locked Rank One Editing for Text-to-Image Personalization. arXiv:2305.01644 [cs.CV] https:\/\/arxiv.org\/abs\/2305.01644","DOI":"10.1145\/3588432.3591506"},{"key":"e_1_3_2_1_36_1","unstructured":"Qixun Wang Xu Bai Haofan Wang Zekui Qin Anthony Chen Huaxia Li Xu Tang and Yao Hu. 2024. InstantID: Zero-shot Identity-Preserving Generation in Seconds. arXiv:2401.07519 [cs.CV] https:\/\/arxiv.org\/abs\/2401.07519"},{"key":"e_1_3_2_1_37_1","volume-title":"FastComposer: Tuning-Free Multi-Subject Image Generation with Localized Attention. arXiv:2305.10431","author":"Xiao Guangxuan","year":"2023","unstructured":"Guangxuan Xiao, Tianwei Yin, William T Freeman, Fr\u00e9do Durand, and Song Han. 2023. FastComposer: Tuning-Free Multi-Subject Image Generation with Localized Attention. arXiv:2305.10431 (2023)."},{"key":"e_1_3_2_1_38_1","volume-title":"Semantic-Aware Human Object Interaction Image Generation. In Forty-first International Conference on Machine Learning.","author":"Xu Zhu","year":"2024","unstructured":"Zhu Xu, Qingchao Chen, Yuxin Peng, and Yang Liu. 2024. Semantic-Aware Human Object Interaction Image Generation. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01979"},{"key":"e_1_3_2_1_40_1","volume-title":"ControlCom: Controllable Image Composition using Diffusion Model. arXiv preprint arXiv:2308.10040","author":"Zhang Bo","year":"2023","unstructured":"Bo Zhang, Yuxuan Duan, Jun Lan, Yan Hong, Huijia Zhu, Weiqiang Wang, and Li Niu. 2023. ControlCom: Controllable Image Composition using Diffusion Model. arXiv preprint arXiv:2308.10040 (2023)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1049\/cje.2021.00.455"},{"key":"e_1_3_2_1_42_1","unstructured":"Chenyang Zhu Kai Li Yue Ma Chunming He and Li Xiu. 2024. MultiBooth: Towards Generating All Your Concepts in an Image from Text. arXiv:2404.14239 [cs.CV] https:\/\/arxiv.org\/abs\/2404.14239"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754911","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:17:55Z","timestamp":1765340275000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754911"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":42,"alternative-id":["10.1145\/3746027.3754911","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754911","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}