{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T12:43:22Z","timestamp":1777034602467,"version":"3.51.4"},"reference-count":46,"publisher":"Association for Computing Machinery (ACM)","issue":"3","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62572458"],"award-info":[{"award-number":["62572458"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100012401","name":"Beijing Science and Technology Plan Project","doi-asserted-by":"crossref","award":["Z251100008125009"],"award-info":[{"award-number":["Z251100008125009"]}],"id":[{"id":"10.13039\/501100012401","id-type":"DOI","asserted-by":"crossref"}]},{"name":"National Science and Technology Council, Taiwan","award":["113-2221-E006-161-MY3"],"award-info":[{"award-number":["113-2221-E006-161-MY3"]}]},{"DOI":"10.13039\/501100001659","name":"German Research Foundation","doi-asserted-by":"crossref","award":["508324734"],"award-info":[{"award-number":["508324734"]}],"id":[{"id":"10.13039\/501100001659","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62572458"],"award-info":[{"award-number":["62572458"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Beijing Science and Technology Plan","award":["Z251100008125009"],"award-info":[{"award-number":["Z251100008125009"]}]},{"name":"National Science and Technology Council, Taiwan","award":["113-2221-E006-161-MY3"],"award-info":[{"award-number":["113-2221-E006-161-MY3"]}]},{"DOI":"10.13039\/501100001659","name":"German Research Foundation","doi-asserted-by":"crossref","award":["508324734"],"award-info":[{"award-number":["508324734"]}],"id":[{"id":"10.13039\/501100001659","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["ACM Trans. Graph."],"published-print":{"date-parts":[[2026,6,30]]},"abstract":"<jats:p>\n                    Diffusion Transformers (DiTs) have exhibited robust capabilities in image generation tasks. However, accurate text-guided image editing for multimodal DiTs (MM-DiTs) still poses a significant challenge. Unlike UNet-based structures that could utilize self\/cross-attention maps for semantic editing, MM-DiTs inherently lack support for explicit and consistent incorporated text guidance, resulting in semantic misalignment between the edited results and texts. In this study, we disclose the sensitivity of different attention heads to different image semantics within MM-DiTs and introduce\n                    <jats:italic toggle=\"yes\">HeadRouter<\/jats:italic>\n                    , a training-free image editing framework that edits the source image by adaptively routing the text guidance to different attention heads in MM-DiTs. Furthermore, we propose a dual-token refinement module to refine text\/image token representations for precise semantic guidance and accurate region expression. Experiments on multiple benchmarks demonstrate HeadRouter\u2019s performance in terms of editing fidelity and image quality. The code is available at\n                    <jats:ext-link xmlns:xlink=\"http:\/\/www.w3.org\/1999\/xlink\" xlink:href=\"https:\/\/github.com\/ICTMCG\/HeadRouter\">https:\/\/github.com\/ICTMCG\/HeadRouter<\/jats:ext-link>\n                    .\n                  <\/jats:p>","DOI":"10.1145\/3797956","type":"journal-article","created":{"date-parts":[[2026,3,2]],"date-time":"2026-03-02T09:33:13Z","timestamp":1772443993000},"page":"1-14","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["HeadRouter: A Training-free Image Editing Framework for MM-DiTs by Adaptively Routing Attention Heads"],"prefix":"10.1145","volume":"45","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3459-5455","authenticated-orcid":false,"given":"Yu","family":"Xu","sequence":"first","affiliation":[{"name":"Institute of Computing Technology Chinese Academy of Sciences","place":["Beijing, China"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3975-2483","authenticated-orcid":false,"given":"Fan","family":"Tang","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology Chinese Academy of Sciences","place":["Beijing, China"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7857-1546","authenticated-orcid":false,"given":"Juan","family":"Cao","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology Chinese Academy of Sciences","place":["Beijing, China"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1386-9394","authenticated-orcid":false,"given":"Xiaoyu","family":"Kong","sequence":"additional","affiliation":[{"name":"Beihang University","place":["Beijing, China"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6433-2678","authenticated-orcid":false,"given":"Yuxin","family":"Zhang","sequence":"additional","affiliation":[{"name":"NLPR, Chinese Academy of Sciences Institute of Automation","place":["Beijing, China"]},{"name":"School of Artificial Intelligence, University of the Chinese Academy of Sciences","place":["Beijing, China"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4597-8534","authenticated-orcid":false,"given":"Jintao","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology Chinese Academy of Sciences","place":["Beijing, China"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5803-2185","authenticated-orcid":false,"given":"Oliver","family":"Deussen","sequence":"additional","affiliation":[{"name":"Universit\u00e4t Konstanz","place":["Konstanz, Germany"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6699-2944","authenticated-orcid":false,"given":"Tong-Yee","family":"Lee","sequence":"additional","affiliation":[{"name":"National Cheng Kung University","place":["Tainan City, Taiwan"]}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,4,24]]},"reference":[{"key":"e_1_3_2_2_1","volume-title":"The Eleventh International Conference on Learning Representations","author":"Albergo Michael Samuel","year":"2023","unstructured":"Michael Samuel Albergo and Eric Vanden-Eijnden. 2023. Building normalizing flows with stochastic interpolants. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_3_1","first-page":"7877","volume-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","author":"Avrahami Omri","year":"2025","unstructured":"Omri Avrahami, Or Patashnik, Ohad Fried, Egor Nemchinov, Kfir Aberman, Dani Lischinski, and Daniel Cohen-Or. 2025. Stable flow: Vital layers for training-free image editing. In Proceedings of the Computer Vision and Pattern Recognition Conference. 7877\u20137888."},{"key":"e_1_3_2_4_1","unstructured":"blackforestlabs.ai. 2024. FLUX offering state-of-the-art performance image generation. Retrieved from https:\/\/blackforestlabs.ai\/. Accessed: 2024-10-07."},{"key":"e_1_3_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00846"},{"key":"e_1_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"e_1_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02062"},{"key":"e_1_3_2_8_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Chen Junsong","year":"2024","unstructured":"Junsong Chen, Jincheng YU, Chongjian GE, Lewei Yao, Enze Xie, Zhongdao Wang, James Kwok, Ping Luo, Huchuan Lu, and Zhenguo Li. 2024. PixArt-$\\alpha$: Fast training of diffusion transformer for photorealistic text-to-image synthesis. In The Twelfth International Conference on Learning Representations. Retrieved from https:\/\/openreview.net\/forum?id=eAKmQPe3m1"},{"key":"e_1_3_2_9_1","unstructured":"Yusuf Dalva Kavana Venkatesh and Pinar Yanardag. 2024. Fluxspace: Disentangled semantic editing in rectified flow transformers. arXiv:2412.09611. Retrieved from https:\/\/arxiv.org\/abs\/2412.09611"},{"key":"e_1_3_2_10_1","unstructured":"Yingying Deng Xiangyu He Changwang Mei Peisong Wang and Fan Tang. 2024. FireFlow: Fast inversion of rectified flow for image semantic editing. arXiv:2412.07517. Retrieved from https:\/\/arxiv.org\/abs\/2412.07517"},{"key":"e_1_3_2_11_1","volume-title":"International Conference on Learning Representations","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, et\u00a0al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. In International Conference on Learning Representations."},{"key":"e_1_3_2_12_1","doi-asserted-by":"publisher","DOI":"10.5555\/3692070.3692573"},{"key":"e_1_3_2_13_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Gandelsman Yossi","year":"2024","unstructured":"Yossi Gandelsman, Alexei A. Efros, and Jacob Steinhardt. 2024. Interpreting CLIP\u2019s image representation via text-based decomposition. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_14_1","article-title":"ReNoise: Real image inversion through iterative noising","author":"Garibi Daniel","year":"2024","unstructured":"Daniel Garibi, Or Patashnik, Andrey Voynov, Hadar Averbuch-Elor, and Daniel Cohen-Or. 2024. ReNoise: Real image inversion through iterative noising. arXiv preprint arXiv:2403.14602 (2024).","journal-title":"arXiv preprint arXiv:2403.14602"},{"key":"e_1_3_2_15_1","volume-title":"The Eleventh International Conference on Learning Representations","author":"Hertz Amir","year":"2023","unstructured":"Amir Hertz, Ron Mokady, Jay Tenenbaum, Kfir Aberman, Yael Pritch, and Daniel Cohen-or. 2023. Prompt-to-prompt image editing with cross-attention control. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657435"},{"key":"e_1_3_2_17_1","unstructured":"Yi Huang Jiancheng Huang Yifan Liu Mingfu Yan Jiaxi Lv Jianzhuang Liu Wei Xiong He Zhang Shifeng Chen and Liangliang Cao. 2024a. Diffusion model-based image editing: A survey. arXiv:2402.17525. Retrieved from https:\/\/arxiv.org\/abs\/2402.17525"},{"key":"e_1_3_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01185"},{"key":"e_1_3_2_19_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.5143773"},{"key":"e_1_3_2_20_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Ju Xuan","year":"2024","unstructured":"Xuan Ju, Ailing Zeng, Yuxuan Bian, Shaoteng Liu, and Qiang Xu. 2024. Pnp inversion: Boosting diffusion-based editing with 3 lines of code. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00582"},{"key":"e_1_3_2_22_1","first-page":"19721","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Kulikov Vladimir","year":"2025","unstructured":"Vladimir Kulikov, Matan Kleiner, Inbar Huberman-Spiegelglas, and Tomer Michaeli. 2025. Flowedit: Inversion-free text-based editing using pre-trained flow models. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 19721\u201319730."},{"key":"e_1_3_2_23_1","volume-title":"The Eleventh International Conference on Learning Representations","author":"Lipman Yaron","year":"2023","unstructured":"Yaron Lipman, Ricky T. Q. Chen, Heli Ben-Hamu, Maximilian Nickel, and Matthew Le. 2023. Flow Matching for Generative Modeling. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00747"},{"key":"e_1_3_2_25_1","volume-title":"The Eleventh International Conference on Learning Representations","author":"Liu Xingchao","year":"2023","unstructured":"Xingchao Liu, Chengyue Gong, et\u00a0al. 2023. Flow straight and fast: Learning to generate and transfer data with rectified flow. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657469"},{"key":"e_1_3_2_27_1","volume-title":"International Conference on Learning Representations","author":"Meng Chenlin","year":"2021","unstructured":"Chenlin Meng, Yutong He, Yang Song, Jiaming Song, Jiajun Wu, Jun-Yan Zhu, and Stefano Ermon. 2021. SDEdit: Guided image synthesis and editing with stochastic differential equations. In International Conference on Learning Representations."},{"key":"e_1_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00585"},{"key":"e_1_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00878"},{"key":"e_1_3_2_30_1","unstructured":"OpenAI. 2024. Sora: Creating Video from Text. Retrieved from https:\/\/openai.com\/sora"},{"key":"e_1_3_2_31_1","article-title":"DINOv2: Learning robust visual features without supervision","author":"Oquab Maxime","year":"2023","unstructured":"Maxime Oquab, Timoth\u00e9e Darcet, Th\u00e9o Moutakanni, Huy V. Vo, Marc Szafraniec, Vasil Khalidov, Pierre Fernandez, Daniel HAZIZA, Francisco Massa, Alaaeldin El-Nouby, et\u00a0al. 2023. DINOv2: Learning robust visual features without supervision. Transactions on Machine Learning Research (2023).","journal-title":"Transactions on Machine Learning Research"},{"key":"e_1_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591513"},{"key":"e_1_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_2_34_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Podell Dustin","year":"2024","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2024. SDXL: Improving latent diffusion models for high-resolution image synthesis. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_35_1","first-page":"8748","volume-title":"International Conference on Machine Learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748\u20138763."},{"key":"e_1_3_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"e_1_3_2_38_1","unstructured":"Litu Rout Yujia Chen Nataniel Ruiz Constantine Caramanis Sanjay Shakkottai and Wen-Sheng Chu. 2024. Semantic image inversion and editing using rectified stochastic differential equations. International Conference on Learning Representations."},{"key":"e_1_3_2_39_1","first-page":"36479","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume":"35","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily L. Denton, Kamyar Ghasemipour, Raphael Gontijo Lopes, Burcu Karagol Ayan, Tim Salimans, et\u00a0al. 2022. Photorealistic text-to-image diffusion models with deep language understanding. Advances in Neural Information Processing Systems 35 (2022), 36479\u201336494.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_40_1","volume-title":"International Conference on Learning Representations","author":"Song Jiaming","year":"2021","unstructured":"Jiaming Song, Chenlin Meng, and Stefano Ermon. 2021. Denoising diffusion implicit models. In International Conference on Learning Representations."},{"key":"e_1_3_2_41_1","doi-asserted-by":"crossref","unstructured":"Vadim Titov Madina Khalmatova Alexandra Ivanova Dmitry Vetrov and Aibek Alanov. 2024. Guide-and-rescale: Self-guidance mechanism for effective tuning-free real image editing. arXiv:2409.01322. Retrieved from https:\/\/arxiv.org\/abs\/2409.01322","DOI":"10.1007\/978-3-031-73209-6_14"},{"key":"e_1_3_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00191"},{"key":"e_1_3_2_43_1","article-title":"Attention is all you need","author":"Vaswani A","year":"2017","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems 30 (2017), 5998\u20136008.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_44_1","unstructured":"Jiangshan Wang Junfu Pu Zhongang Qi Jiayi Guo Yue Ma Nisha Huang Yuxin Chen Xiu Li and Ying Shan. 2024. Taming rectified flow for inversion and editing. arXiv:2411.04746. Retrieved from https:\/\/arxiv.org\/abs\/2411.04746"},{"key":"e_1_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00903"},{"key":"e_1_3_2_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657445"},{"key":"e_1_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"}],"container-title":["ACM Transactions on Graphics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3797956","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T11:47:44Z","timestamp":1777031264000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797956"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,24]]},"references-count":46,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,6,30]]}},"alternative-id":["10.1145\/3797956"],"URL":"https:\/\/doi.org\/10.1145\/3797956","relation":{},"ISSN":["0730-0301","1557-7368"],"issn-type":[{"value":"0730-0301","type":"print"},{"value":"1557-7368","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,24]]},"assertion":[{"value":"2025-05-13","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2026-01-26","order":2,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2026-04-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}