{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:20:39Z","timestamp":1778080839387,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"National Research Foundation of Korea, NRF","award":["RS-2024-00405857"],"award-info":[{"award-number":["RS-2024-00405857"]}]},{"name":"IITP grant funded by the Korea government, MSIT","award":["RS-2021-II211343"],"award-info":[{"award-number":["RS-2021-II211343"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,3]]},"DOI":"10.1145\/3680528.3687668","type":"proceedings-article","created":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T08:14:37Z","timestamp":1733213677000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["InstantDrag: Improving Interactivity in Drag-based Image Editing"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3780-3897","authenticated-orcid":false,"given":"Joonghyuk","family":"Shin","sequence":"first","affiliation":[{"name":"Seoul National University, Seoul, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1438-4949","authenticated-orcid":false,"given":"Daehyeon","family":"Choi","sequence":"additional","affiliation":[{"name":"POSTECH, Pohang, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5541-409X","authenticated-orcid":false,"given":"Jaesik","family":"Park","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,12,3]]},"reference":[{"key":"e_1_3_3_3_2_1","unstructured":"Hadi Alzayer Zhihao Xia Xuaner Zhang Eli Shechtman Jia-Bin Huang and Michael Gharbi. 2024. Magic Fixup: Streamlining Photo Editing by Watching Dynamic Videos. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.13044 (2024)."},{"key":"e_1_3_3_3_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"e_1_3_3_3_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33783-3_44"},{"key":"e_1_3_3_3_5_1","unstructured":"Yutao Cui Xiaotong Zhao Guozhen Zhang Shengming Cao Kai Ma and Limin Wang. 2024. StableDrag: Stable Dragging for Point-based Image Editing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.04437 (2024)."},{"key":"e_1_3_3_3_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.316"},{"key":"e_1_3_3_3_7_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Geng Daniel","year":"2023","unstructured":"Daniel Geng and Andrew Owens. 2023. Motion Guidance: Diffusion-Based Image Editing with Differentiable Motion Estimators. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_3_8_1","unstructured":"Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2014. Generative adversarial nets."},{"key":"e_1_3_3_3_9_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Hertz Amir","year":"2022","unstructured":"Amir Hertz, Ron Mokady, Jay Tenenbaum, Kfir Aberman, Yael Pritch, and Daniel Cohen-or. 2022. Prompt-to-Prompt Image Editing with Cross-Attention Control. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_3_10_1","volume-title":"Conference on Neural Information Processing Systems (NeurIPS)","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. In Conference on Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_3_11_1","volume-title":"Conference on Neural Information Processing Systems (NeurIPS) Workshop","author":"Ho Jonathan","year":"2021","unstructured":"Jonathan Ho and Tim Salimans. 2021. Classifier-Free Diffusion Guidance. In Conference on Neural Information Processing Systems (NeurIPS) Workshop."},{"key":"e_1_3_3_3_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00803"},{"key":"e_1_3_3_3_13_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Hu Edward\u00a0J","year":"2021","unstructured":"Edward\u00a0J Hu, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, Weizhu Chen, et\u00a0al. 2021. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_3_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19790-1_40"},{"key":"e_1_3_3_3_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.632"},{"key":"e_1_3_3_3_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00813"},{"key":"e_1_3_3_3_17_1","unstructured":"Ruining Li Chuanxia Zheng Christian Rupprecht and Andrea Vedaldi. 2024. DragAPart: Learning a Part-Level Motion Prior for Articulated Objects. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.15382 (2024)."},{"key":"e_1_3_3_3_18_1","unstructured":"Pengyang Ling Lin Chen Pan Zhang Huaian Chen and Yi Jin. 2023. Freedrag: Point tracking is not you need for interactive point-based image editing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.04684 (2023)."},{"key":"e_1_3_3_3_19_1","volume-title":"Conference on Neural Information Processing Systems (NeurIPS)","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong\u00a0Jae Lee. 2024a. Visual instruction tuning. In Conference on Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_3_20_1","unstructured":"Haofeng Liu Chenshu Xu Yifei Yang Lihua Zeng and Shengfeng He. 2024b. Drag Your Noise: Interactive Point-based Editing via Diffusion Semantic Propagation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.01050 (2024)."},{"key":"e_1_3_3_3_21_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Liu Xingchao","year":"2022","unstructured":"Xingchao Liu, Chengyue Gong, et\u00a0al. 2022. Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_3_22_1","volume-title":"Conference on Neural Information Processing Systems (NeurIPS)","author":"Lu Cheng","year":"2022","unstructured":"Cheng Lu, Yuhao Zhou, Fan Bao, Jianfei Chen, Chongxuan Li, and Jun Zhu. 2022a. Dpm-solver: A fast ode solver for diffusion probabilistic model sampling in around 10 steps. In Conference on Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_3_23_1","unstructured":"Cheng Lu Yuhao Zhou Fan Bao Jianfei Chen Chongxuan Li and Jun Zhu. 2022b. Dpm-solver++: Fast solver for guided sampling of diffusion probabilistic models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2211.01095 (2022)."},{"key":"e_1_3_3_3_24_1","doi-asserted-by":"crossref","unstructured":"Grace Luo Trevor Darrell Oliver Wang Dan\u00a0B Goldman and Aleksander Holynski. 2024. Readout Guidance: Learning Control from Diffusion Features. IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.00785"},{"key":"e_1_3_3_3_25_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Meng Chenlin","year":"2021","unstructured":"Chenlin Meng, Yutong He, Yang Song, Jiaming Song, Jiajun Wu, Jun-Yan Zhu, and Stefano Ermon. 2021. SDEdit: Guided Image Synthesis and Editing with Stochastic Differential Equations. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_3_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00585"},{"key":"e_1_3_3_3_27_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Mou Chong","year":"2023","unstructured":"Chong Mou, Xintao Wang, Jiechong Song, Ying Shan, and Jian Zhang. 2023. DragonDiffusion: Enabling Drag-style Manipulation on Diffusion Models. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_3_28_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Nie Shen","year":"2023","unstructured":"Shen Nie, Hanzhong\u00a0Allan Guo, Cheng Lu, Yuhao Zhou, Chenyu Zheng, and Chongxuan Li. 2023. The Blessing of Randomness: SDE Beats ODE in General Diffusion-based Image Editing. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_3_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591500"},{"key":"e_1_3_3_3_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00735"},{"key":"e_1_3_3_3_31_1","unstructured":"Gaurav Parmar Taesung Park Srinivasa Narasimhan and Jun-Yan Zhu. 2024. One-Step Image Translation with Text-to-Image Models. arXiv e-prints (2024) arXiv\u20132403."},{"key":"e_1_3_3_3_32_1","volume-title":"International Conference on Machine Learning (ICML)","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning (ICML)."},{"key":"e_1_3_3_3_33_1","unstructured":"Aditya Ramesh Prafulla Dhariwal Alex Nichol Casey Chu and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.06125 1 2 (2022) 3."},{"key":"e_1_3_3_3_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"e_1_3_3_3_35_1","doi-asserted-by":"crossref","unstructured":"Daniel Roich Ron Mokady Amit\u00a0H Bermano and Daniel Cohen-Or. 2022. Pivotal tuning for latent-based editing of real images. ACM Transactions on Graphics (TOG) 42 1 (2022) 1\u201313.","DOI":"10.1145\/3544777"},{"key":"e_1_3_3_3_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_3_37_1","unstructured":"Axel Sauer Dominik Lorenz Andreas Blattmann and Robin Rombach. 2023. Adversarial Diffusion Distillation. arXiv e-prints (2023) arXiv\u20132311."},{"key":"e_1_3_3_3_38_1","volume-title":"Conference on Neural Information Processing Systems (NeurIPS)","author":"Schuhmann Christoph","year":"2022","unstructured":"Christoph Schuhmann, Romain Beaumont, Richard Vencu, Cade Gordon, Ross Wightman, Mehdi Cherti, Theo Coombes, Aarush Katta, Clayton Mullis, Mitchell Wortsman, et\u00a0al. 2022. Laion-5b: An open large-scale dataset for training next generation image-text models. In Conference on Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_3_39_1","unstructured":"Yujun Shi Chuhui Xue Jiachun Pan Wenqing Zhang Vincent\u00a0YF Tan and Song Bai. 2023. DragDiffusion: Harnessing Diffusion Models for Interactive Point-based Image Editing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.14435 (2023)."},{"key":"e_1_3_3_3_40_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Song Jiaming","year":"2020","unstructured":"Jiaming Song, Chenlin Meng, and Stefano Ermon. 2020. Denoising Diffusion Implicit Models. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_3_41_1","volume-title":"International Conference on Machine Learning (ICML)","author":"Song Yang","year":"2023","unstructured":"Yang Song, Prafulla Dhariwal, Mark Chen, and Ilya Sutskever. 2023. Consistency Models. In International Conference on Machine Learning (ICML)."},{"key":"e_1_3_3_3_42_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Song Yang","year":"2021","unstructured":"Yang Song, Jascha Sohl-Dickstein, Diederik\u00a0P Kingma, Abhishek Kumar, Stefano Ermon, and Ben Poole. 2021. Score-Based Generative Modeling through Stochastic Differential Equations. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_3_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00991"},{"key":"e_1_3_3_3_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01422"},{"key":"e_1_3_3_3_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"e_1_3_3_3_46_1","unstructured":"Zewei Zhang Huan Liu Jun Chen and Xiangyu Xu. 2024. GoodDrag: Towards Good Practices for Drag Editing with Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.07206 (2024)."},{"key":"e_1_3_3_3_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.244"}],"event":{"name":"SA '24: SIGGRAPH Asia 2024 Conference Papers","location":"Tokyo Japan","acronym":"SA '24","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["SIGGRAPH Asia 2024 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3680528.3687668","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3680528.3687668","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:20Z","timestamp":1750295900000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3680528.3687668"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"references-count":46,"alternative-id":["10.1145\/3680528.3687668","10.1145\/3680528"],"URL":"https:\/\/doi.org\/10.1145\/3680528.3687668","relation":{},"subject":[],"published":{"date-parts":[[2024,12,3]]},"assertion":[{"value":"2024-12-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}