{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T19:05:01Z","timestamp":1784228701317,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":39,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Innovation and Technology Fund (ITF) of the Innovation and Technology Commission (ITC) of the Hong Kong Special Administrative Region (HKSAR) Government","award":["Project No. ITS\/269\/24FP"],"award-info":[{"award-number":["Project No. ITS\/269\/24FP"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,19]]},"DOI":"10.1145\/3799902.3811169","type":"proceedings-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:15:27Z","timestamp":1784218527000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["BoxCtrl: 3D-Aware Visual Prompting for Geometric Image Editing"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5611-3886","authenticated-orcid":false,"given":"Feifei","family":"Wang","sequence":"first","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8213-5803","authenticated-orcid":false,"given":"Shiyuan","family":"Yang","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2588-1687","authenticated-orcid":false,"given":"Xiaoyu","family":"Li","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7014-5377","authenticated-orcid":false,"given":"Jing","family":"Liao","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"crossref","unstructured":"Hadi Alzayer Zhihao Xia Xuaner\u00a0(Cecilia) Zhang Eli Shechtman Jia-Bin Huang and Michael Gharbi. 2025. Magic Fixup: Streamlining Photo Editing by Watching Dynamic Videos. ACM Transactions on Graphics 44 5 1\u201325.","DOI":"10.1145\/3750722"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657525"},{"key":"e_1_3_3_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"e_1_3_3_2_5_1","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.17450","author":"Chen Jiacheng","year":"2025","unstructured":"Jiacheng Chen, Ramin Mehran, Xuhui Jia, Saining Xie, and Sanghyun Woo. 2025. BlenderFusion: 3D-Grounded Visual Editing and Generative Compositing. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.17450."},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00630"},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3721238.3730695"},{"key":"e_1_3_3_2_8_1","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.06261","author":"Comanici Gheorghe","year":"2025","unstructured":"Gheorghe Comanici, Eric Bieber, Mike Schaekermann, Ice Pasupat, Noveen Sachdeva, Inderjit Dhillon, Marcel Blistein, Ori Ram, Dan Zhang, Evan Rosen, et\u00a0al. 2025. Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.06261."},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9811809"},{"key":"e_1_3_3_2_10_1","first-page":"1","volume-title":"International Conference on Learning Representations","author":"Eldesokey Abdelrahman","year":"2025","unstructured":"Abdelrahman Eldesokey and Peter Wonka. 2025. Build-A-Scene: Interactive 3D Layout Control for Diffusion-Based Image Generation. In International Conference on Learning Representations. 1\u201318."},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00373"},{"key":"e_1_3_3_2_12_1","first-page":"1","volume-title":"International Conference on Learning Representations","author":"Hu Edward\u00a0J","year":"2022","unstructured":"Edward\u00a0J Hu, yelong shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations. 1\u201313."},{"key":"e_1_3_3_2_13_1","first-page":"1","volume-title":"International Conference on Learning Representations","author":"Ju Xuan","year":"2024","unstructured":"Xuan Ju, Ailing Zeng, Yuxuan Bian, Shaoteng Liu, and Qiang Xu. 2024. PnP Inversion: Boosting Diffusion-based Editing with 3 Lines of Code. In International Conference on Learning Representations. 1\u201328."},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01648"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687564"},{"key":"e_1_3_3_2_16_1","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.15742","author":"Labs Black\u00a0Forest","year":"2025","unstructured":"Black\u00a0Forest Labs, Stephen Batifol, Andreas Blattmann, Frederic Boesel, Saksham Consul, Cyril Diagne, Tim Dockhorn, Jack English, Zion English, Patrick Esser, Sumith Kulal, Kyle Lacey, Yam Levi, Cheng Li, Dominik Lorenz, Jonas M\u00fcller, Dustin Podell, Robin Rombach, Harry Saini, Axel Sauer, and Luke Smith. 2025. FLUX. 1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.15742."},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3757377.3763897"},{"key":"e_1_3_3_2_18_1","first-page":"1","volume-title":"International Conference on Learning Representations","author":"Lipman Yaron","year":"2023","unstructured":"Yaron Lipman, Ricky T.\u00a0Q. Chen, Heli Ben-Hamu, Maximilian Nickel, and Matthew Le. 2023. Flow Matching for Generative Modeling. In International Conference on Learning Representations. 1\u201328."},{"key":"e_1_3_3_2_19_1","first-page":"38","volume-title":"European conference on computer vision","author":"Liu Shilong","year":"2024","unstructured":"Shilong Liu, Zhaoyang Zeng, Tianhe Ren, Feng Li, Hao Zhang, Jie Yang, Qing Jiang, Chunyuan Li, Jianwei Yang, Hang Su, Jun Zhu, and Lei Zhang. 2024. Grounding DINO: Marrying DINO with Grounded Pre-training for Open-Set Object Detection. In European conference on computer vision. 38\u201355."},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0155"},{"key":"e_1_3_3_2_21_1","first-page":"151684","volume-title":"Advances in Neural Information Processing Systems","author":"Min Yunhong","year":"2025","unstructured":"Yunhong Min, Daehyeon Choi, Kyeongmin Yeo, Jihyun Lee, and Minhyuk Sung. 2025. ORIGEN: Zero-Shot 3D Orientation Grounding in Text-to-Image Generation. In Advances in Neural Information Processing Systems. 151684\u2013151717."},{"key":"e_1_3_3_2_22_1","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.21446","author":"Omran Mohamed","year":"2025","unstructured":"Mohamed Omran, Dimitris Kalatzis, Jens Petersen, Amirhossein Habibian, and Auke Wiggers. 2025. Controllable 3D Placement of Objects with Scene-Aware Diffusion Models. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.21446."},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00735"},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00266"},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01461"},{"key":"e_1_3_3_2_26_1","first-page":"8748","volume-title":"International Conference on Machine Learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. 8748\u20138763."},{"key":"e_1_3_3_2_27_1","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.14159","author":"Ren Tianhe","year":"2024","unstructured":"Tianhe Ren, Shilong Liu, Ailing Zeng, Jing Lin, Kunchang Li, He Cao, Jiayu Chen, Xinyu Huang, Yukang Chen, Feng Yan, Zhaoyang Zeng, Hao Zhang, Feng Li, Jie Yang, Hongyang Li, Qing Jiang, and Lei Zhang. 2024. Grounded SAM: Assembling Open-World Models for Diverse Visual Tasks. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.14159."},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00056"},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01386"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"crossref","unstructured":"Zhou Wang A.C. Bovik H.R. Sheikh and E.P. Simoncelli. 2004. Image quality assessment: from error visibility to structural similarity. IEEE Transactions on Image Processing 13 4 (2004) 600\u2013612.","DOI":"10.1109\/TIP.2003.819861"},{"key":"e_1_3_3_2_31_1","first-page":"52622","volume-title":"Advances in Neural Information Processing Systems","author":"Wang Zehan","year":"2025","unstructured":"Zehan Wang, Ziang Zhang, Jiayang Xu, Jialei Wang, Tianyu Pang, Chao Du, Hengshuang Zhao, and Zhou Zhao. 2025. Orient Anything V2: Unifying Orientation and Rotation Understanding. In Advances in Neural Information Processing Systems. 52622\u201352645."},{"key":"e_1_3_3_2_32_1","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.02324","author":"Wu Chenfei","year":"2025","unstructured":"Chenfei Wu, Jiahao Li, Jingren Zhou, Junyang Lin, Kaiyuan Gao, Kun Yan, Sheng ming Yin, Shuai Bai, Xiao Xu, Yilei Chen, Yuxiang Chen, Zecheng Tang, Zekai Zhang, Zhengyi Wang, An Yang, Bowen Yu, Chen Cheng, Dayiheng Liu, Deqing Li, Hang Zhang, Hao Meng, Hu Wei, Jingyuan Ni, Kai Chen, Kuan Cao, Liang Peng, Lin Qu, Minggang Wu, Peng Wang, Shuting Yu, Tingkun Wen, Wensen Feng, Xiaoxiao Xu, Yi Wang, Yichang Zhang, Yongqiang Zhu, Yujia Wu, Yuxuan Cai, and Zenan Liu. 2025. Qwen-Image Technical Report. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.02324."},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2429"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00406"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01647"},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.00480"},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01241"},{"key":"e_1_3_3_2_39_1","first-page":"1","volume-title":"International Conference on Learning Representations","author":"Zheng Kaiwen","year":"2026","unstructured":"Kaiwen Zheng, Huayu Chen, Haotian Ye, Haoxiang Wang, Qinsheng Zhang, Kai Jiang, Hang Su, Stefano Ermon, Jun Zhu, and Ming-Yu Liu. 2026. DiffusionNFT: Online Diffusion Reinforcement with Forward Process. In International Conference on Learning Representations. 1\u201322."},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01778"}],"event":{"name":"SIGGRAPH Conference Papers '26: Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers","location":"Los Angeles CA USA","acronym":"SIGGRAPH Conference Papers '26","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T18:14:35Z","timestamp":1784225675000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3799902.3811169"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":39,"alternative-id":["10.1145\/3799902.3811169","10.1145\/3799902"],"URL":"https:\/\/doi.org\/10.1145\/3799902.3811169","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}