{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T04:38:17Z","timestamp":1764995897687,"version":"3.46.0"},"reference-count":72,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2025,10,16]],"date-time":"2025-10-16T00:00:00Z","timestamp":1760572800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,16]],"date-time":"2025-10-16T00:00:00Z","timestamp":1760572800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"crossref","award":["No.2023YFB4502804"],"award-info":[{"award-number":["No.2023YFB4502804"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100014219","name":"National Science Fund for Distinguished Young Scholars","doi-asserted-by":"publisher","award":["No.62025603"],"award-info":[{"award-number":["No.62025603"]}],"id":[{"id":"10.13039\/501100014219","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. U22B2051, No. U21B2037, No. 62072389, No. 62302411"],"award-info":[{"award-number":["No. U22B2051, No. U21B2037, No. 62072389, No. 62302411"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Natural Science Foundation of Fujian Province of China","award":["No.2021J06003"],"award-info":[{"award-number":["No.2021J06003"]}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["No. 2023M732948"],"award-info":[{"award-number":["No. 2023M732948"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1007\/s11263-025-02573-6","type":"journal-article","created":{"date-parts":[[2025,10,16]],"date-time":"2025-10-16T12:04:44Z","timestamp":1760616284000},"page":"8570-8588","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["TraDiffusion: Trajectory-Based Training-Free Image Generation"],"prefix":"10.1007","volume":"133","author":[{"given":"Mingrui","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Oucheng","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9956-6308","authenticated-orcid":false,"given":"Jiayi","family":"Ji","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianzhuang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoshuai","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liujuan","family":"Cao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9163-2932","authenticated-orcid":false,"given":"Rongrong","family":"Ji","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,10,16]]},"reference":[{"key":"2573_CR1","unstructured":"Atzmon, Y., Bala, M., Balaji, Y., Cai, T., Cui, Y., Fan, J., Ge, Y., Gururani, S., Huffman, J., Isaac, R. & others (2024). Edify image: High-quality image generation with pixel space laplacian diffusion models. arXiv preprint arXiv:2411.07126"},{"key":"2573_CR2","doi-asserted-by":"crossref","unstructured":"Avrahami, O., Lischinski, D., & Fried, O. (2022). Blended diffusion for text-driven editing of natural images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 18208\u201318218","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"2573_CR3","doi-asserted-by":"crossref","unstructured":"Avrahami, O., Hayes, T., Gafni, O., Gupta, S., Taigman, Y., Parikh, D., Lischinski, D., Fried, O., & Yin, X. (2023). Spatext: Spatio-textual representation for controllable image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 18370\u201318380.","DOI":"10.1109\/CVPR52729.2023.01762"},{"key":"2573_CR4","doi-asserted-by":"crossref","unstructured":"Balaji, Y., Min, M. R., Bai, B., Chellappa, R., & Graf, H. P. (2019). Conditional gan with discriminative filter generation for text-to-video synthesis. In: IJCAI, p\u00a02.","DOI":"10.24963\/ijcai.2019\/276"},{"key":"2573_CR5","unstructured":"Balaji, Y., Nah, S., Huang, X., Vahdat, A., Song, J., Zhang, Q., Kreis, K., Aittala, M., Aila, T., Laine, S. & others (2022) ediff-i: Text-to-image diffusion models with an ensemble of expert denoisers. arXiv preprint arXiv:2211.01324."},{"key":"2573_CR6","unstructured":"Bar-Tal, O., Yariv, L., Lipman, Y., & Dekel, T. (2023). Multidiffusion: Fusing diffusion paths for controlled image generation."},{"issue":"11","key":"2573_CR7","first-page":"120","volume":"25","author":"G Bradski","year":"2000","unstructured":"Bradski, G. (2000). The opencv library. Dr Dobb\u2019s Journal: Software Tools for the Professional Programmer, 25(11), 120\u2013123.","journal-title":"Dr Dobb\u2019s Journal: Software Tools for the Professional Programmer"},{"key":"2573_CR8","doi-asserted-by":"crossref","unstructured":"Chen, M., Laina, I., & Vedaldi, A. (2024). Training-free layout control with cross-attention guidance. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp 5343\u20135353.","DOI":"10.1109\/WACV57701.2024.00526"},{"key":"2573_CR9","first-page":"8780","volume":"34","author":"P Dhariwal","year":"2021","unstructured":"Dhariwal, P., & Nichol, A. (2021). Diffusion models beat gans on image synthesis. Advances in Neural Information Processing Systems, 34, 8780\u20138794.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2573_CR10","unstructured":"Esser, P., Kulal, S., Blattmann, A., Entezari, R., M\u00fcller, J., Saini, H., Levi, Y., Lorenz, D., Sauer, A., Boesel, F. & others (2024). Scaling rectified flow transformers for high-resolution image synthesis. In: Forty-first international conference on machine learning."},{"key":"2573_CR11","unstructured":"Feng, W., He, X., Fu, T. J., Jampani, V., Akula, A., Narayana, P., Basu, S., Wang, X. E., & Wang, W. Y. (2022). Training-free structured diffusion guidance for compositional text-to-image synthesis. arXiv preprint arXiv:2212.05032."},{"key":"2573_CR12","unstructured":"Feng, W., Zhu, W., Fu, T. J., Jampani, V., Akula, A., He, X., Basu, S., Wang, X. E., & Wang, W. Y. (2024). Layoutgpt: Compositional visual planning and generation with large language models. Advances in Neural Information Processing Systems 36."},{"key":"2573_CR13","doi-asserted-by":"crossref","unstructured":"Gafni, O., Polyak, A., Ashual, O., Sheynin, S., Parikh, D., & Taigman, Y. (2022). Make-a-scene: Scene-based text-to-image generation with human priors. In: European Conference on Computer Vision, Springer, pp 89\u2013106.","DOI":"10.1007\/978-3-031-19784-0_6"},{"issue":"11","key":"2573_CR14","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., & Bengio, Y. (2020). Generative adversarial networks. Communications of the ACM, 63(11), 139\u2013144.","journal-title":"Communications of the ACM"},{"key":"2573_CR15","doi-asserted-by":"crossref","unstructured":"Guo, X., Liu, J., Cui, M., Li, J., Yang, H., & Huang, D. (2024). Initno: Boosting text-to-image diffusion models via initial noise optimization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 9380\u20139389.","DOI":"10.1109\/CVPR52733.2024.00896"},{"key":"2573_CR16","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., & Cohen-Or, D. (2022). Prompt-to-prompt image editing with cross attention control. arXiv preprint arXiv:2208.01626."},{"key":"2573_CR17","unstructured":"Ho, J., & Salimans, T. (2022). Classifier-free diffusion guidance. arXiv preprint arXiv:2207.12598"},{"key":"2573_CR18","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models. Advances in Neural Information Processing Systems, 33, 6840\u20136851.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2573_CR19","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., & Sun, G. (2018). Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7132\u20137141.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"2573_CR20","unstructured":"Huang, L., Chen, D., Liu, Y., Shen, Y., Zhao, D., & Zhou, J. (2023). Composer: Creative and controllable image synthesis with composable conditions. arXiv preprint arXiv:2302.09778."},{"key":"2573_CR21","doi-asserted-by":"crossref","unstructured":"Huang, X., Mallya, A., Wang, T. C., & Liu, M. Y. (2022). Multimodal conditional image synthesis with product-of-experts gans. In: European Conference on Computer Vision, Springer, pp 91\u2013109.","DOI":"10.1007\/978-3-031-19787-1_6"},{"key":"2573_CR22","unstructured":"Huang, Y., Huang, J., Liu, Y., Yan,, M., Lv, J., Liu, J., Xiong, W., Zhang, H., Chen, S., & Cao, L. (2024). Diffusion model-based image editing: A survey. arXiv preprint arXiv:2402.17525."},{"key":"2573_CR23","doi-asserted-by":"crossref","unstructured":"Isola, P., Zhu, J. Y., Zhou, T., & Efros, A. A. (2017). Image-to-image translation with conditional adversarial networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 1125\u20131134.","DOI":"10.1109\/CVPR.2017.632"},{"key":"2573_CR24","unstructured":"Jocher, G., Chaurasia, A., & Qiu, J. (2023). Ultralytics yolov8. https:\/\/github.com\/ultralytics\/ultralytics."},{"key":"2573_CR25","doi-asserted-by":"crossref","unstructured":"Johnson, J., Gupta, A., & Fei-Fei, L. (2018). Image generation from scene graphs. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1219\u20131228.","DOI":"10.1109\/CVPR.2018.00133"},{"key":"2573_CR26","doi-asserted-by":"crossref","unstructured":"Kim, Y., Lee, J., Kim, J. H., Ha, J. W., & Zhu, J. Y. (2023). Dense text-to-image generation with attention modulation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 7701\u20137711.","DOI":"10.1109\/ICCV51070.2023.00708"},{"key":"2573_CR27","unstructured":"Kingma, D. P., & Welling, M. (2013). Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114"},{"key":"2573_CR28","doi-asserted-by":"publisher","first-page":"3416","DOI":"10.1109\/TMM.2021.3097900","volume":"24","author":"C Li","year":"2021","unstructured":"Li, C., Zhang, P., & Wang, C. (2021). Harmonious textual layout generation over natural images via deep aesthetics learning. IEEE Transactions on Multimedia, 24, 3416\u20133428.","journal-title":"IEEE Transactions on Multimedia"},{"key":"2573_CR29","doi-asserted-by":"crossref","unstructured":"Li, Y., Cheng, Y., Gan, Z., Yu, L., Wang, L., & Liu, J. (2020). Bachgan: High-resolution image synthesis from salient object layout. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 8365\u20138374.","DOI":"10.1109\/CVPR42600.2020.00839"},{"key":"2573_CR30","doi-asserted-by":"crossref","unstructured":"Li, Y., Liu, H., Wu, Q., Mu, F., Yang, J., Gao, J., Li, C., & Lee, Y. J. (2023). Gligen: Open-set grounded text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 22511\u201322521.","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"2573_CR31","doi-asserted-by":"crossref","unstructured":"Lin, T. Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., & Zitnick, C. L. (2014). Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, Springer, pp 740\u2013755.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2573_CR32","doi-asserted-by":"crossref","unstructured":"Liu, M. Y., Breuel, T., & Kautz, J. (2017). Unsupervised image-to-image translation networks. Advances in Neural Information Processing Systems 30.","DOI":"10.1007\/978-3-319-70139-4"},{"key":"2573_CR33","doi-asserted-by":"crossref","unstructured":"Liu, N., Li, S., Du, Y., Torralba, A., & Tenenbaum, J. B. (2022). Compositional visual generation with composable diffusion models. In: European Conference on Computer Vision, Springer, pp 423\u2013439.","DOI":"10.1007\/978-3-031-19790-1_26"},{"key":"2573_CR34","doi-asserted-by":"crossref","unstructured":"Mou, C., Wang, X., Xie, L., Wu, Y., Zhang, J., Qi, Z., & Shan, Y. (2024). T2i-adapter: Learning adapters to dig out more controllable ability for text-to-image diffusion models. In: Proceedings of the AAAI conference on artificial intelligence, pp 4296\u20134304.","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"2573_CR35","unstructured":"Nichol, A., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., McGrew, B., Sutskever, I., & Chen, M. (2021). Glide: Towards photorealistic image generation and editing with text-guided diffusion models. arXiv preprint arXiv:2112.10741."},{"key":"2573_CR36","unstructured":"Oktay, O., Schlemper, J., Folgoc, L. L., Lee, M., Heinrich, M., Misawa, K., Mori, K., McDonagh, S., Hammerla, N. Y., Kainz, B. & others (2018). Attention u-net: Learning where to look for the pancreas. arXiv preprint arXiv:1804.03999."},{"key":"2573_CR37","doi-asserted-by":"crossref","unstructured":"Park, T., Liu, M. Y., Wang, T. C., & Zhu, J. Y. (2019). Semantic image synthesis with spatially-adaptive normalization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 2337\u20132346.","DOI":"10.1109\/CVPR.2019.00244"},{"key":"2573_CR38","unstructured":"Podell, D., English, Z., Lacey, K., Blattmann, A., Dockhorn, T., M\u00fcller, J., Penna, J., & Rombach, R. (2023). Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952."},{"key":"2573_CR39","doi-asserted-by":"crossref","unstructured":"Pont-Tuset, J., Uijlings, J., Changpinyo, S., Soricut, R., & Ferrari, V. (2020). Connecting vision and language with localized narratives. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part V 16, Springer, pp 647\u2013664.","DOI":"10.1007\/978-3-030-58558-7_38"},{"key":"2573_CR40","doi-asserted-by":"crossref","unstructured":"Qin, Z., Zhong, W., Hu, F., Yang, X., Ye, L., & Zhang, Q. (2021). Layout structure assisted indoor image generation. In: 2021 IEEE 4th International Conference on Multimedia Information Processing and Retrieval (MIPR), IEEE, pp 323\u2013329.","DOI":"10.1109\/MIPR51284.2021.00061"},{"key":"2573_CR41","doi-asserted-by":"crossref","unstructured":"Qu, L., Wu, S., Fei, H., Nie, L., & Chua, T. S. (2023). Layoutllm-t2i: Eliciting layout guidance from llm for text-to-image generation. In: Proceedings of the 31st ACM International Conference on Multimedia, pp 643\u2013654.","DOI":"10.1145\/3581783.3612012"},{"key":"2573_CR42","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin P., Clark J, & others (2021) Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, PMLR, pp 8748\u20138763."},{"key":"2573_CR43","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., & Chen, M. (2022). Hierarchical text-conditional image generation with clip latents. 1(2), 3. arXiv preprint http:\/\/arxiv.org\/abs\/2204.06125"},{"key":"2573_CR44","doi-asserted-by":"crossref","unstructured":"Ren, J., Xu, M., Wu, J. C., Liu, Z., Xiang, T., Toisoul, A. (2024). Move anything with layered scene diffusion. 2404.07178.","DOI":"10.1109\/CVPR52733.2024.00610"},{"key":"2573_CR45","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10684\u201310695.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2573_CR46","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). U-net: Convolutional networks for biomedical image segmentation. In: Medical image computing and computer-assisted intervention\u2013MICCAI 2015: 18th international conference, Munich, Germany, October 5-9, 2015, proceedings, part III 18, Springer, pp 234\u2013241.","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"2573_CR47","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., & Aberman, K. (2023). Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 22500\u201322510.","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"2573_CR48","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E. L., Ghasemipour, K., Gontijo Lopes, R., Karagol Ayan, B., Salimans, T., et al. (2022). Photorealistic text-to-image diffusion models with deep language understanding. Advances in Neural Information Processing Systems, 35, 36479\u201336494.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2573_CR49","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., & Ganguli, S. (2015). Deep unsupervised learning using nonequilibrium thermodynamics. In: International Conference on Machine Learning, PMLR, pp 2256\u20132265."},{"key":"2573_CR50","unstructured":"Song, J., Meng, C., & Ermon, S. (2020a). Denoising diffusion implicit models. arXiv preprint arXiv:2010.02502."},{"key":"2573_CR51","unstructured":"Song, Y., Sohl-Dickstein, J., Kingma, D. P., Kumar, A., Ermon, S., & Poole, B. (2020b). Score-based generative modeling through stochastic differential equations. arXiv preprint arXiv:2011.13456."},{"key":"2573_CR52","doi-asserted-by":"crossref","unstructured":"Sun, W., & Wu, T. (2019). Image synthesis from reconfigurable layout and style. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 10531\u201310540.","DOI":"10.1109\/ICCV.2019.01063"},{"key":"2573_CR53","doi-asserted-by":"crossref","unstructured":"Sylvain, T., Zhang, P., Bengio, Y., Hjelm, R. D., & Sharma, S. (2021). Object-centric image generation from layouts. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 2647\u20132655.","DOI":"10.1609\/aaai.v35i3.16368"},{"key":"2573_CR54","doi-asserted-by":"crossref","unstructured":"Tan, H., Yin, B., Wei, K., Liu, X., & Li, X. (2023). Alr-gan: Adaptive layout refinement for text-to-image synthesis. IEEE Transactions on Multimedia.","DOI":"10.1109\/TMM.2023.3238554"},{"key":"2573_CR55","unstructured":"Van Den\u00a0Oord, A., Vinyals, O., & others (2017). Neural discrete representation learning. Advances in Neural Information Processing Systems 30."},{"key":"2573_CR56","doi-asserted-by":"crossref","unstructured":"Wang, T. C., Liu, M. Y., Zhu, J. Y., Tao, A., Kautz, J., & Catanzaro, B. (2018). High-resolution image synthesis and semantic manipulation with conditional gans. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 8798\u20138807.","DOI":"10.1109\/CVPR.2018.00917"},{"key":"2573_CR57","doi-asserted-by":"crossref","unstructured":"Wang, X., Darrell, T., Rambhatla, S. S., Girdhar, R., Misra, I. (2024). Instancediffusion: Instance-level control for image generation. arXiv preprint arXiv:2402.03290.","DOI":"10.1109\/CVPR52733.2024.00596"},{"key":"2573_CR58","doi-asserted-by":"crossref","unstructured":"Wu, S., Tang, H., Jing, X. Y., Zhao, H., Qian, J., Sebe, N., & Yan, Y. (2022). Cross-view panorama image synthesis. IEEE Transactions on Multimedia.","DOI":"10.1016\/j.patcog.2022.108884"},{"key":"2573_CR59","doi-asserted-by":"crossref","unstructured":"Xie, J., Li, Y., Huang, Y., Liu, H., Zhang, W., Zheng, Y., & Shou, M. Z. (2023). Boxdiff: Text-to-image synthesis with training-free box-constrained diffusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 7452\u20137461.","DOI":"10.1109\/ICCV51070.2023.00685"},{"key":"2573_CR60","doi-asserted-by":"crossref","unstructured":"Xu, J., Zhou, X., Yan, S., Gu, X., Arnab, A., Sun, C., Wang, X., & Schmid, C. (2023). Pixel aligned language models. arXiv preprint arXiv:2312.09237","DOI":"10.1109\/CVPR52733.2024.01238"},{"key":"2573_CR61","unstructured":"Xu, K., Ba, J., Kiros, R., Cho, K., Courville, A., Salakhudinov, R., Zemel, R., & Bengio, Y. (2015). Show, attend and tell: Neural image caption generation with visual attention. In: International Conference on Machine Learning, PMLR, pp 2048\u20132057."},{"key":"2573_CR62","doi-asserted-by":"crossref","unstructured":"Xu, T., Zhang, P., Huang, Q., Zhang, H., Gan, Z., Huang, X., & He, X. (2018). Attngan: Fine-grained text to image generation with attentional generative adversarial networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1316\u20131324.","DOI":"10.1109\/CVPR.2018.00143"},{"key":"2573_CR63","doi-asserted-by":"crossref","unstructured":"Yang, Z., Liu, D., Wang, C., Yang, J., & Tao, D. (2022). Modeling image composition for complex scene generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 7764\u20137773.","DOI":"10.1109\/CVPR52688.2022.00761"},{"key":"2573_CR64","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, J., Gan, Z., Li, L., Lin, K., Wu, C., Duan, N., Liu, Z., Liu, C., Zeng, M. & others (2023). Reco: Region-controlled text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 14246\u201314255.","DOI":"10.1109\/CVPR52729.2023.01369"},{"issue":"18","key":"2573_CR65","doi-asserted-by":"publisher","first-page":"27423","DOI":"10.1007\/s11042-021-11038-0","volume":"80","author":"J Zakraoui","year":"2021","unstructured":"Zakraoui, J., Saleh, M., Al-Maadeed, S., & Jaam, J. M. (2021). Improving text-to-image generation with object layout guidance. Multimedia Tools and Applications, 80(18), 27423\u201327443.","journal-title":"Multimedia Tools and Applications"},{"key":"2573_CR66","doi-asserted-by":"crossref","unstructured":"Zeiler, M. D., & Fergus, R. (2014). Visualizing and understanding convolutional networks. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part I 13, Springer, pp 818\u2013833.","DOI":"10.1007\/978-3-319-10590-1_53"},{"key":"2573_CR67","unstructured":"Zhang, H., Goodfellow, I., Metaxas, D., & Odena, A. (2019). Self-attention generative adversarial networks. In: International Conference on Machine Learning, PMLR, pp 7354\u20137363."},{"key":"2573_CR68","doi-asserted-by":"crossref","unstructured":"Zhang, L., Chen, Q., Hu, B., & Jiang, S. (2020). Text-guided neural image inpainting. In: Proceedings of the 28th ACM International Conference on Multimedia, pp 1302\u20131310.","DOI":"10.1145\/3394171.3414017"},{"key":"2573_CR69","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., & Agrawala, M. (2023). Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 3836\u20133847.","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"2573_CR70","first-page":"27196","volume":"34","author":"Z Zhang","year":"2021","unstructured":"Zhang, Z., Ma, J., Zhou, C., Men, R., Li, Z., Ding, M., Tang, J., Zhou, J., & Yang, H. (2021). Ufc-bert: Unifying multi-modal controls for conditional image synthesis. Advances in Neural Information Processing Systems, 34, 27196\u201327208.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2573_CR71","doi-asserted-by":"crossref","unstructured":"Zhao, B., Meng, L., Yin, W., & Sigal, L. (2019). Image generation from layout. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 8584\u20138593.","DOI":"10.1109\/CVPR.2019.00878"},{"key":"2573_CR72","doi-asserted-by":"crossref","unstructured":"Zhu, J. Y., Park, T., Isola, P., & Efros, A. A. (2017). Unpaired image-to-image translation using cycle-consistent adversarial networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp 2223\u20132232.","DOI":"10.1109\/ICCV.2017.244"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02573-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02573-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02573-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T04:04:06Z","timestamp":1764993846000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02573-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,16]]},"references-count":72,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2025,12]]}},"alternative-id":["2573"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02573-6","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"type":"print","value":"0920-5691"},{"type":"electronic","value":"1573-1405"}],"subject":[],"published":{"date-parts":[[2025,10,16]]},"assertion":[{"value":"18 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 August 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 October 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The authors have no relevant ethics approval to disclose.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"All authors agreed to publish the work.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}]}}