{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T14:26:25Z","timestamp":1785335185537,"version":"3.55.0"},"reference-count":49,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T00:00:00Z","timestamp":1777420800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T00:00:00Z","timestamp":1777420800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1007\/s11263-026-02862-8","type":"journal-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T14:25:06Z","timestamp":1777472706000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Advancing Aesthetic Image Generation via Composition Transfer"],"prefix":"10.1007","volume":"134","author":[{"given":"Kai","family":"Zou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiwei","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3977-8800","authenticated-orcid":false,"given":"Bin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nenghai","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,29]]},"reference":[{"key":"2862_CR1","doi-asserted-by":"crossref","unstructured":"Achanta, R., Hemami, S., Estrada, F., & Susstrunk, S. (2009). Frequency-tuned salient region detection. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 1597\u20131604. IEEE.","DOI":"10.1109\/CVPR.2009.5206596"},{"issue":"11","key":"2862_CR2","doi-asserted-by":"publisher","first-page":"2274","DOI":"10.1109\/TPAMI.2012.120","volume":"34","author":"R Achanta","year":"2012","unstructured":"Achanta, R., Shaji, A., Smith, K., Lucchi, A., Fua, P., & S\u00fcsstrunk, S. (2012). Slic superpixels compared to state-of-the-art superpixel methods. IEEE transactions on pattern analysis and machine intelligence, 34(11), 2274\u20132282.","journal-title":"IEEE transactions on pattern analysis and machine intelligence"},{"key":"2862_CR3","unstructured":"Betker, J., Goh, G., Jing, L., Brooks, T., Wang, J., Li, L., Ouyang, L., Zhuang, J., Lee, J., & Guo, Y., et al. (2023). Improving image generation with better captions. Computer Science. https:\/\/cdn.openai.com\/papers\/dall-e-3.pdf2(3), 8."},{"key":"2862_CR4","unstructured":"Black Forest Labs: Black Forest Labs; Frontier AI Lab (2024). https:\/\/blackforestlabs.ai\/."},{"key":"2862_CR5","doi-asserted-by":"crossref","unstructured":"Chen, J., Ge, C., Xie, E., Wu, Y., Yao, L., Ren, X., Wang, Z., Luo, P., Lu, H., & Li, Z. (2024). Pixart-$$\\backslash $$sigma: Weak-to-strong training of diffusion transformer for 4k text-to-image generation. arXiv preprint arXiv:2403.04692.","DOI":"10.1007\/978-3-031-73411-3_5"},{"key":"2862_CR6","unstructured":"Chen, S., Lai, J., Gao, J., Ye, T., Chen, H., Shi, H., Shao, S., Lin, Y., Fei, S., & Xing, Z., et al. (2025). Postercraft: Rethinking high-quality aesthetic poster generation in a unified framework. arXiv preprint arXiv:2506.10741."},{"key":"2862_CR7","doi-asserted-by":"crossref","unstructured":"Chen, H., Xu, X., Li, W., Ren, J., Ye, T., Liu, S., Chen, Y.-C., Zhu, L., & Wang, X. (2025). Posta: A go-to framework for customized artistic poster generation. In Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 28694\u201328704.","DOI":"10.1109\/CVPR52734.2025.02672"},{"key":"2862_CR8","doi-asserted-by":"crossref","unstructured":"Cohen-Or, D., Sorkine, O., Gal, R., Leyvand, T., & Xu, Y.-Q. (2006). Color harmonization. In: ACM SIGGRAPH 2006 Papers, pp. 624\u2013630.","DOI":"10.1145\/1179352.1141933"},{"key":"2862_CR9","unstructured":"Dai, X., Hou, J., Ma, C.-Y., Tsai, S., Wang, J., Wang, R., Zhang, P., Vandenhende, S., Wang, X., & Dubey, A., et al. (2023). Emu: Enhancing image generation models using photogenic needles in a haystack. arXiv preprint arXiv:2309.15807"},{"issue":"4","key":"2862_CR10","doi-asserted-by":"publisher","first-page":"636","DOI":"10.1109\/83.841940","volume":"9","author":"N Damera-Venkata","year":"2000","unstructured":"Damera-Venkata, N., Kite, T. D., Geisler, W. S., Evans, B. L., & Bovik, A. C. (2000). Image quality assessment based on a degradation model. IEEE transactions on image processing, 9(4), 636\u2013650.","journal-title":"IEEE transactions on image processing"},{"key":"2862_CR11","doi-asserted-by":"crossref","unstructured":"Datta, R., Joshi, D., Li, J., & Wang, J.Z. (2006). Studying aesthetics in photographic images using a computational approach. In: Computer Vision\u2013ECCV 2006: 9th European Conference on Computer Vision, Graz, Austria, May 7-13, 2006, Proceedings, Part III 9, pp. 288\u2013301. Springer.","DOI":"10.1007\/11744078_23"},{"key":"2862_CR12","doi-asserted-by":"crossref","unstructured":"Feng, Y., Gong, B., Chen, D., Shen, Y., Liu, Y., & Zhou, J. (2024). Ranni: Taming text-to-image diffusion for accurate instruction following. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4744\u20134753.","DOI":"10.1109\/CVPR52733.2024.00454"},{"key":"2862_CR13","doi-asserted-by":"crossref","unstructured":"He, S., Ming, A., Li, Y., Sun, J., Zheng, S., & Ma, H. (2023). Thinking image color aesthetics assessment: Models, datasets and benchmarks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 21838\u201321847.","DOI":"10.1109\/ICCV51070.2023.01996"},{"key":"2862_CR14","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., & Hochreiter, S. (2017). Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems 30."},{"key":"2862_CR15","unstructured":"Hu, E.J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2021). Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685."},{"key":"2862_CR16","doi-asserted-by":"crossref","unstructured":"Kong, S., Shen, X., Lin, Z., Mech, R., & Fowlkes, C. (2016). Photo aesthetics ranking network with attributes and content adaptation. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part I 14, pp. 662\u2013679. Springer.","DOI":"10.1007\/978-3-319-46448-0_40"},{"key":"2862_CR17","doi-asserted-by":"publisher","unstructured":"Li, Y., Liu, H., Wu, Q., Mu, F., Yang, J., Gao, J., Li, C., & Lee, Y.J. (2023). GLIGEN: open-set grounded text-to-image generation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2023, Vancouver, BC, Canada, June 17-24, 2023, pp. 22511\u201322521 . https:\/\/doi.org\/10.1109\/CVPR52729.2023.02156.","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"2862_CR18","doi-asserted-by":"crossref","unstructured":"Li, M., Yang, T., Kuang, H., Wu, J., Wang, Z., Xiao, X., & Chen, C. (2025). Controlnet++: Improving conditional controls with efficient consistency feedback. In: European Conference on Computer Vision, pp. 129\u2013147. Springer.","DOI":"10.1007\/978-3-031-72667-5_8"},{"key":"2862_CR19","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., & Zitnick, C.L. (2014). Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, pp. 740\u2013755. Springer.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2862_CR20","doi-asserted-by":"crossref","unstructured":"Liu, L., Chen, R., Wolf, L., & Cohen-Or, D. (2010). Optimizing photo composition. Computer Graphics Forum,29, 469\u2013478. Wiley Online Library","DOI":"10.1111\/j.1467-8659.2009.01616.x"},{"key":"2862_CR21","unstructured":"Liu, H., Li, C., Li, Y., Li, B., Zhang, Y., Shen, S., & Lee, Y.J. (2024). Llava-next: Improved reasoning, ocr, and world knowledge."},{"key":"2862_CR22","unstructured":"Liu, H., Li, C., Wu, Q., & Lee, Y.J. (2024). Visual instruction tuning. Advances in neural information processing systems 36."},{"key":"2862_CR23","unstructured":"Liu, Z., Ning, M., Zhang, Q., Yang, S., Wang, Z., Yang, Y., Xu, X., Song, Y., Chen, W., & Wang, F., et al. (2025). Cot-lized diffusion: Let\u2019s reinforce t2i generation step-by-step. arXiv preprint arXiv:2507.04451."},{"key":"2862_CR24","unstructured":"Liu, Z., Wang, Z., Yao, Y., Zhang, L., & Shao, L. (2018). Deep active learning with contaminated tags for image aesthetics assessment. IEEE Transactions on Image Processing."},{"key":"2862_CR25","unstructured":"Liu, M., Zhang, L., Tian, Y., Qu, X., Liu, L., & Liu, T. (2024). Draw like an artist: Complex scene generation with diffusion model via composition, painting, and retouching. arXiv preprint arXiv:2408.13858."},{"key":"2862_CR26","doi-asserted-by":"crossref","unstructured":"Martin, F.D. (1983). The Power of the Center: A Study of Composition in the Visual Arts. JSTOR.","DOI":"10.2307\/429879"},{"key":"2862_CR27","doi-asserted-by":"crossref","unstructured":"Obrador, P., Schmidt-Hackenberg, L., & Oliver, N. (2010). The role of image composition in image aesthetics. In: 2010 IEEE International Conference on Image Processing, pp. 3185\u20133188. IEEE.","DOI":"10.1109\/ICIP.2010.5654231"},{"key":"2862_CR28","unstructured":"Phung, Q., Ge, S., & Huang, J. (2023). Grounded text-to-image synthesis with attention refocusing. ArXiv preprint abs\/2306.05427."},{"key":"2862_CR29","unstructured":"Podell, D., English, Z., Lacey, K., Blattmann, A., Dockhorn, T., M\u00fcller, J., Penna, J., & Rombach, R. (2023). Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952"},{"key":"2862_CR30","unstructured":"ProGamerGov: Synthetic Dataset 1M DALLE3 High Quality Captions. https:\/\/huggingface.co\/datasets\/ProGamerGov\/synthetic-dataset-1m-dalle3-high-quality-captions. Accessed: 2024-10-01 (2024)."},{"key":"2862_CR31","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., & Clark, J., et al. (2021). Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR."},{"key":"2862_CR32","doi-asserted-by":"publisher","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2022, New Orleans, LA, USA, June 18-24, 2022, pp. 10674\u201310685 (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.01042 .","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2862_CR33","first-page":"25278","volume":"35","author":"C Schuhmann","year":"2022","unstructured":"Schuhmann, C., Beaumont, R., Vencu, R., Gordon, C., Wightman, R., Cherti, M., Coombes, T., Katta, A., Mullis, C., Wortsman, M., et al. (2022). Laion-5b: An open large-scale dataset for training next generation image-text models. Advances in Neural Information Processing Systems, 35, 25278\u201325294.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2862_CR34","unstructured":"Standard, C., & et al. (2007). Colorimetry-part 4: Cie 1976 l* a* b* colour space. International Standard, 2019\u201306."},{"key":"2862_CR35","first-page":"49659","volume":"36","author":"K Sun","year":"2023","unstructured":"Sun, K., Pan, J., Ge, Y., Li, H., Duan, H., Wu, X., Zhang, R., Zhou, A., Qin, Z., Wang, Y., et al. (2023). Journeydb: A benchmark for generative image understanding. Advances in neural information processing systems, 36, 49659\u201349678.","journal-title":"Advances in neural information processing systems"},{"issue":"8","key":"2862_CR36","doi-asserted-by":"publisher","first-page":"3998","DOI":"10.1109\/TIP.2018.2831899","volume":"27","author":"H Talebi","year":"2018","unstructured":"Talebi, H., & Milanfar, P. (2018). Nima: Neural image assessment. IEEE transactions on image processing, 27(8), 3998\u20134011.","journal-title":"IEEE transactions on image processing"},{"key":"2862_CR37","unstructured":"Wang, P., Bai, S., Tan, S., Wang, S., Fan, Z., Bai, J., Chen, K., Liu, X., Wang, J., & Ge, W., et al. (2024). Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution. arXiv preprint arXiv:2409.12191."},{"key":"2862_CR38","first-page":"24824","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Xia, F., Chi, E., Le, Q. V., Zhou, D., et al. (2022). Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, 35, 24824\u201324837.","journal-title":"Advances in neural information processing systems"},{"key":"2862_CR39","unstructured":"Wu, X., Hao, Y., Sun, K., Chen, Y., Zhu, F., Zhao, R., & Li, H. (2023). Human preference score v2: A solid benchmark for evaluating human preferences of text-to-image synthesis. arXiv preprint arXiv:2306.09341."},{"key":"2862_CR40","unstructured":"Xu, J., Liu, X., Wu, Y., Tong, Y., Li, Q., Ding, M., Tang, J., & Dong, Y. (2024). Imagereward: Learning and evaluating human preferences for text-to-image generation. Advances in Neural Information Processing Systems 36."},{"key":"2862_CR41","doi-asserted-by":"crossref","unstructured":"Yang, S., Wu, T., Shi, S., Lao, S., Gong, Y., Cao, M., Wang, J., & Yang, Y. (2022). Maniqa: Multi-dimension attention network for no-reference image quality assessment. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 1191\u20131200).","DOI":"10.1109\/CVPRW56347.2022.00126"},{"key":"2862_CR42","unstructured":"Yang, L., Yu, Z., Meng, C., Xu, M., Ermon, S., & Bin, C. (2024). Mastering text-to-image diffusion: Recaptioning, planning, and generating with multimodal llms. In: Forty-first International Conference on Machine Learning."},{"key":"2862_CR43","unstructured":"Ye, H., Zhang, J., Liu, S., Han, X., & Yang, W. (2023).(2023) Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:2308.06721."},{"key":"2862_CR44","unstructured":"Yu, J., Xu, Y., Koh, J.Y., Luong, T., Baid, G., Wang, Z., Vasudevan, V., Ku, A., Yang, Y., & Ayan, B.K., et al. (2025). Scaling autoregressive models for content-rich text-to-image generation. Transactions on Machine Learning Research"},{"key":"2862_CR45","doi-asserted-by":"publisher","first-page":"1548","DOI":"10.1109\/TIP.2019.2941778","volume":"29","author":"H Zeng","year":"2019","unstructured":"Zeng, H., Cao, Z., Zhang, L., & Bovik, A. C. (2019). A unified probabilistic formulation of image aesthetic assessment. IEEE Transactions on Image Processing, 29, 1548\u20131561.","journal-title":"IEEE Transactions on Image Processing"},{"key":"2862_CR46","doi-asserted-by":"crossref","unstructured":"Zhang, H., Duan, Z., Wang, X., Chen, Y., & Zhang, Y. (2025). Eligen: Entity-level controlled image generation with regional attention. arXiv preprint arXiv:2501.01097.","DOI":"10.1145\/3743093.3771013"},{"key":"2862_CR47","doi-asserted-by":"publisher","unstructured":"Zhang, L., Rao, A., & Agrawala, M. (2023). Adding conditional control to text-to-image diffusion models. In IEEE\/CVF International Conference on Computer Vision, ICCV 2023, Paris, France, October 1-6, 2023, pp. 3813\u20133824. https:\/\/doi.org\/10.1109\/ICCV51070.2023.00355.","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"2862_CR48","unstructured":"Zhang, X., Yang, L., Cai, Y., Yu, Z., Xie, J., Tian, Y., Xu, M., Tang, Y., Yang, Y., & Cui, B. (2024). Realcompo: Dynamic equilibrium between realism and compositionality improves text-to-image diffusion models. arXiv preprint arXiv:2402.12908."},{"key":"2862_CR49","doi-asserted-by":"crossref","unstructured":"Zhou, D., Li, Y., Ma, F., Zhang, X., & Yang, Y. (2024). Migc: Multi-instance generation controller for text-to-image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52733.2024.00651"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02862-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02862-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02862-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T10:29:02Z","timestamp":1780482542000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02862-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,29]]},"references-count":49,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2026,5]]}},"alternative-id":["2862"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02862-8","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,29]]},"assertion":[{"value":"22 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"252"}}