{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,17]],"date-time":"2025-05-17T04:04:24Z","timestamp":1747454664247,"version":"3.40.5"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T00:00:00Z","timestamp":1737072000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T00:00:00Z","timestamp":1737072000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100003819","name":"Natural Science Foundation of Hubei Province","doi-asserted-by":"publisher","award":["2024AFB283"],"award-info":[{"award-number":["2024AFB283"]}],"id":[{"id":"10.13039\/501100003819","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Science Foundation of China Three Gorges University","award":["2023RCKJ0022"],"award-info":[{"award-number":["2023RCKJ0022"]}]},{"name":"Science Foundation of Hubei Key Laboratory of Intelligent Vision Based Monitoring for Hydroelectric Engineering","award":["2024SDSJ08"],"award-info":[{"award-number":["2024SDSJ08"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,6]]},"DOI":"10.1007\/s00371-024-03781-w","type":"journal-article","created":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T20:28:19Z","timestamp":1737145699000},"page":"6187-6199","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["HOIEdit: Human\u2013object interaction editing with text-to-image diffusion model"],"prefix":"10.1007","volume":"41","author":[{"given":"Tang","family":"Xu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4394-0145","authenticated-orcid":false,"given":"Wenbin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alin","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,17]]},"reference":[{"key":"3781_CR1","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"3781_CR2","unstructured":"Nichol, A.Q., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., Mcgrew, B., Sutskever, I., Chen, M.: Glide: Towards photorealistic image generation and editing with text-guided diffusion models. In: Proceedings of the international conference on machine learning (ICML), pp. 16784\u201316804 (2022)"},{"key":"3781_CR3","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 (2022)"},{"key":"3781_CR4","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., Cohen-Or, D.: Prompt-to-prompt image editing with cross attention control. In: Proceedings of the international conference on learning representations (ICLR). (2023)"},{"key":"3781_CR5","unstructured":"Couairon, G., Verbeek, J., Schwenk, H., Cord, M.: Diffedit: Diffusion-based semantic image editing with mask guidance. In: Proceedings of the international conference on learning representations (ICLR), (2023)"},{"key":"3781_CR6","doi-asserted-by":"crossref","unstructured":"Cao, M., Wang, X., Qi, Z., Shan, Y., Qie, X., Zheng, Y.: Masactrl: Tuning-free mutual self-attention control for consistent image synthesis and editing. In: Proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp. 22560\u201322570. (2023)","DOI":"10.1109\/ICCV51070.2023.02062"},{"key":"3781_CR7","doi-asserted-by":"crossref","unstructured":"Lu, C., Krishna, R., Bernstein, M., Fei-Fei, L.: Visual relationship detection with language priors. In: Proceedings of European conference on computer vision (ECCV), vol. 9905, pp. 852\u2013869. Springer (2016)","DOI":"10.1007\/978-3-319-46448-0_51"},{"key":"3781_CR8","doi-asserted-by":"crossref","unstructured":"Xu, D., Zhu, Y., Choy, C.B., Fei-Fei, L.: Scene graph generation by iterative message passing. In: Proceedings of the IEEE Conference on computer vision and pattern recognition (CVPR), pp. 5410\u20135419 (2017)","DOI":"10.1109\/CVPR.2017.330"},{"key":"3781_CR9","doi-asserted-by":"crossref","unstructured":"Zellers, R., Yatskar, M., Thomson, S., Choi, Y.: Neural motifs: Scene graph parsing with global context. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 5831\u20135840 (2018)","DOI":"10.1109\/CVPR.2018.00611"},{"key":"3781_CR10","doi-asserted-by":"crossref","unstructured":"Tang, K., Niu, Y., Huang, J., Shi, J., Zhang, H.: Unbiased scene graph generation from biased training. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 3716\u20133725 (2020)","DOI":"10.1109\/CVPR42600.2020.00377"},{"key":"3781_CR11","doi-asserted-by":"crossref","unstructured":"Wang, W., Wang, R., Shan, S., Chen, X.: Exploring context and visual pattern of relationship for scene graph generation. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 8188\u20138197 (2019)","DOI":"10.1109\/CVPR.2019.00838"},{"key":"3781_CR12","doi-asserted-by":"crossref","unstructured":"Wang, W., Wang, R., Shan, S., Chen, X.: Sketching image gist: Human-mimetic hierarchical scene graph generation. In: Proceedings of European conference on computer vision (ECCV), vol. 12358, pp. 222\u2013239. Springer (2020)","DOI":"10.1007\/978-3-030-58601-0_14"},{"issue":"10","key":"3781_CR13","doi-asserted-by":"publisher","first-page":"2489","DOI":"10.1007\/s11263-023-01817-7","volume":"131","author":"W Wang","year":"2023","unstructured":"Wang, W., Wang, R., Shan, S., Chen, X.: Importance first: Generating scene graph of human interest. Int. J. Comput. Vis. (IJCV) 131(10), 2489\u20132515 (2023)","journal-title":"Int. J. Comput. Vis. (IJCV)"},{"key":"3781_CR14","doi-asserted-by":"crossref","unstructured":"Li, R., Zhang, S., He, X.: Sgtr: End-to-end scene graph generation with transformer. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 19486\u201319496 (2022)","DOI":"10.1109\/CVPR52688.2022.01888"},{"key":"3781_CR15","doi-asserted-by":"crossref","unstructured":"Dong, X., Gan, T., Song, X., Wu, J., Cheng, Y., Nie, L.: Stacked hybrid-attention and group collaborative learning for unbiased scene graph generation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 19427\u201319436. (2022)","DOI":"10.1109\/CVPR52688.2022.01882"},{"key":"3781_CR16","doi-asserted-by":"crossref","unstructured":"Jung, D., Kim, S., Kim, W.H., Cho, M.: Devil\u2019s on the edges: Selective quad attention for scene graph generation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 18664\u201318674. (2023)","DOI":"10.1109\/CVPR52729.2023.01790"},{"key":"3781_CR17","doi-asserted-by":"crossref","unstructured":"Kundu, S., Aakur, S.N.: Is-ggt: Iterative scene graph generation with generative transformers. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 6292\u20136301. (2023)","DOI":"10.1109\/CVPR52729.2023.00609"},{"key":"3781_CR18","unstructured":"Li, L., Wei, J., Wang, W., Yang, Y.: Neural-logic human-object interaction detection. In: Advances in neural information processing systems (NeurIPS), vol.\u00a036, pp. 21158\u201321171 (2023)"},{"key":"3781_CR19","unstructured":"Li, L., Wang, W., Yang, Y.: Human-object interaction detection collaborated with large relation-driven diffusion models. In: Advances in neural information processing systems (NeurIPS), (2024)"},{"key":"3781_CR20","doi-asserted-by":"crossref","unstructured":"Qi, S., Wang, W., Jia, B., Shen, J., Zhu, S.C.: Learning human-object interactions by graph parsing neural networks. In: Proceedings of European conference on computer vision (ECCV), vol. 11213, pp. 407\u2013423. Springer (2018)","DOI":"10.1007\/978-3-030-01240-3_25"},{"key":"3781_CR21","doi-asserted-by":"crossref","unstructured":"Dhamo, H., Farshad, A., Laina, I., Navab, N., Hager, G.D., Tombari, F., Rupprecht, C.: Semantic image manipulation using scene graphs. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 5213\u20135222. (2020)","DOI":"10.1109\/CVPR42600.2020.00526"},{"key":"3781_CR22","first-page":"2672","volume":"27","author":"I Goodfellow","year":"2014","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., Bengio, Y.: Generative adversarial nets. Adv. Neural Inf. Process. Syst. (NIPS) 27, 2672\u20132680 (2014)","journal-title":"Adv. Neural Inf. Process. Syst. (NIPS)"},{"key":"3781_CR23","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., Agrawala, M.: Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International conference on computer vision (ICCV), pp. 3836\u20133847 (2023)","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"3781_CR24","doi-asserted-by":"crossref","unstructured":"Avrahami, O., Hayes, T., Gafni, O., Gupta, S., Taigman, Y., Parikh, D., Lischinski, D., Fried, O., Yin, X.: Spatext: Spatio-textual representation for controllable image generation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 18370\u201318380. (2023)","DOI":"10.1109\/CVPR52729.2023.01762"},{"key":"3781_CR25","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Huang, Z., Liao, J.: Continuous layout editing of single images with diffusion models. arXiv preprint arXiv:2306.13078 (2023)","DOI":"10.1111\/cgf.14966"},{"key":"3781_CR26","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 22500\u201322510 (2023)","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"3781_CR27","doi-asserted-by":"crossref","unstructured":"Huang, Z., Wu, T., Jiang, Y., Chan, K.C., Liu, Z.: Reversion: Diffusion-based relation inversion from images. arXiv preprint arXiv:2303.13495. (2023)","DOI":"10.1145\/3680528.3687658"},{"key":"3781_CR28","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.Y., Dollar, P., Girshick, R.: Segment anything. In: Proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp. 4015\u20134026. (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"3781_CR29","doi-asserted-by":"crossref","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Li, C., Yang, J., Su, H., Zhu, J., et\u00a0al.: Grounding dino: Marrying dino with grounded pre-training for open-set object detection. arXiv preprint arXiv:2303.05499 (2023)","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"3781_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, H., Xu, T., Li, H., Zhang, S., Wang, X., Huang, X., Metaxas, D.N.: Stackgan: Text to photo-realistic image synthesis with stacked generative adversarial networks. In: Proceedings of the IEEE international conference on computer vision (ICCV), pp. 5907\u20135915 (2017)","DOI":"10.1109\/ICCV.2017.629"},{"issue":"8","key":"3781_CR31","doi-asserted-by":"publisher","first-page":"1947","DOI":"10.1109\/TPAMI.2018.2856256","volume":"41","author":"H Zhang","year":"2018","unstructured":"Zhang, H., Xu, T., Li, H., Zhang, S., Wang, X., Huang, X., Metaxas, D.N.: Stackgan++: Realistic image synthesis with stacked generative adversarial networks. IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI) 41(8), 1947\u20131962 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI)"},{"key":"3781_CR32","doi-asserted-by":"crossref","unstructured":"Xu, T., Zhang, P., Huang, Q., Zhang, H., Gan, Z., Huang, X., He, X.: Attngan: Fine-grained text to image generation with attentional generative adversarial networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 1316\u20131324 (2018)","DOI":"10.1109\/CVPR.2018.00143"},{"key":"3781_CR33","unstructured":"Li, B., Qi, X., Lukasiewicz, T., Torr, P.: Controllable text-to-image generation. In: Advances in neural information processing systems (NeurIPS), vol.\u00a032 (2019)"},{"key":"3781_CR34","unstructured":"Ding, M., Yang, Z., Hong, W., Zheng, W., Zhou, C., Yin, D., Lin, J., Zou, X., Shao, Z., Yang, H., et\u00a0al.: Cogview: Mastering text-to-image generation via transformers. In: Advances in neural information processing systems (NeurIPS), vol.\u00a034, pp. 19822\u201319835. (2021)"},{"key":"3781_CR35","doi-asserted-by":"crossref","unstructured":"Esser, P., Rombach, R., Ommer, B.: Taming transformers for high-resolution image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 12873\u201312883. (2021)","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"3781_CR36","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: Advances in neural information processing systems (NeurIPS), pp. 6840\u20136851. (2020)"},{"key":"3781_CR37","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. In: Proceedings of the international conference on learning representations (ICLR) (2022)"},{"key":"3781_CR38","doi-asserted-by":"crossref","unstructured":"Gao, C., Liu, S., Zhu, D., Liu, Q., Cao, J., He, H., He, R., Yan, S.: Interactgan: Learning to generate human-object interaction. In: Proceedings of the ACM international conference on multimedia (ACM-MM), pp. 165\u2013173. (2020)","DOI":"10.1145\/3394171.3413854"},{"key":"3781_CR39","doi-asserted-by":"crossref","unstructured":"Hoe, J.T., Jiang, X., Chan, C.S., Tan, Y.P., Hu, W.: Interactdiffusion: Interaction control in text-to-image diffusion models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 6180\u20136189. (2024)","DOI":"10.1109\/CVPR52733.2024.00591"},{"issue":"4","key":"3781_CR40","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3592450","volume":"42","author":"O Avrahami","year":"2023","unstructured":"Avrahami, O., Fried, O., Lischinski, D.: Blended latent diffusion. ACM Trans. Gr. (TOG) 42(4), 1\u201311 (2023)","journal-title":"ACM Trans. Gr. (TOG)"},{"key":"3781_CR41","doi-asserted-by":"crossref","unstructured":"Kim, G., Kwon, T., Ye, J.C.: Diffusionclip: Text-guided diffusion models for robust image manipulation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 2426\u20132435. (2022)","DOI":"10.1109\/CVPR52688.2022.00246"},{"key":"3781_CR42","doi-asserted-by":"crossref","unstructured":"Tamura, M., Ohashi, H., Yoshinaga, T.: Qpic: Query-based pairwise human-object interaction detection with image-wide contextual information. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 10410\u201310419 (2021)","DOI":"10.1109\/CVPR46437.2021.01027"},{"key":"3781_CR43","doi-asserted-by":"crossref","unstructured":"Kim, B., Lee, J., Kang, J., Kim, E.S., Kim, H.J.: Hotr: End-to-end human-object interaction detection with transformers. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 74\u201383. (2021)","DOI":"10.1109\/CVPR46437.2021.00014"},{"key":"3781_CR44","doi-asserted-by":"crossref","unstructured":"Park, J., Park, J.W., Lee, J.S.: Viplo: Vision transformer based pose-conditioned self-loop graph for human-object interaction detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 17152\u201317162 (2023)","DOI":"10.1109\/CVPR52729.2023.01645"},{"issue":"6","key":"3781_CR45","doi-asserted-by":"publisher","first-page":"2827","DOI":"10.1109\/TPAMI.2021.3049156","volume":"44","author":"T Zhou","year":"2021","unstructured":"Zhou, T., Qi, S., Wang, W., Shen, J., Zhu, S.C.: Cascaded parsing of human-object interaction recognition. IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI) 44(6), 2827\u20132840 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI)"},{"key":"3781_CR46","doi-asserted-by":"crossref","unstructured":"Wei, J., Zhou, T., Yang, Y., Wang, W.: Nonverbal interaction detection. In: Proceedings of European conference on computer vision (ECCV), vol. 15080, pp. 277\u2013295. Springer (2024)","DOI":"10.1007\/978-3-031-72670-5_16"},{"key":"3781_CR47","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: Convolutional networks for biomedical image segmentation. In: Medical image computing and computer-assisted intervention (MICCAI), pp. 234\u2013241 (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"3781_CR48","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 770\u2013778. (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"3781_CR49","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. In: Advances in neural information processing systems (NIPS), pp. 5998\u20136008 (2017)"},{"key":"3781_CR50","doi-asserted-by":"crossref","unstructured":"Mokady, R., Hertz, A., Aberman, K., Pritch, Y., Cohen-Or, D.: Null-text inversion for editing real images using guided diffusion models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 6038\u20136047 (2023)","DOI":"10.1109\/CVPR52729.2023.00585"},{"key":"3781_CR51","doi-asserted-by":"crossref","unstructured":"Chao, Y.W., Liu, Y., Liu, X., Zeng, H., Deng, J.: Learning to detect human-object interactions. In: Proceedings of the IEEE winter conference on applications of computer vision (WACV), pp. 381\u2013389. (2018)","DOI":"10.1109\/WACV.2018.00048"},{"key":"3781_CR52","doi-asserted-by":"crossref","unstructured":"Tumanyan, N., Geyer, M., Bagon, S., Dekel, T.: Plug-and-play diffusion features for text-driven image-to-image translation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 1921\u20131930 (2023)","DOI":"10.1109\/CVPR52729.2023.00191"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-024-03781-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-024-03781-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-024-03781-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,16]],"date-time":"2025-05-16T08:53:09Z","timestamp":1747385589000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-024-03781-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,17]]},"references-count":52,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2025,6]]}},"alternative-id":["3781"],"URL":"https:\/\/doi.org\/10.1007\/s00371-024-03781-w","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"type":"print","value":"0178-2789"},{"type":"electronic","value":"1432-2315"}],"subject":[],"published":{"date-parts":[[2025,1,17]]},"assertion":[{"value":"26 December 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 January 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}