{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:54:00Z","timestamp":1784181240883,"version":"3.55.0"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2025,5,5]],"date-time":"2025-05-05T00:00:00Z","timestamp":1746403200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,5]],"date-time":"2025-05-05T00:00:00Z","timestamp":1746403200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s11263-025-02435-1","type":"journal-article","created":{"date-parts":[[2025,5,5]],"date-time":"2025-05-05T06:51:22Z","timestamp":1746427882000},"page":"5413-5434","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Paragraph-to-Image Generation with Information-Enriched Diffusion Model"],"prefix":"10.1007","volume":"133","author":[{"given":"Weijia","family":"Wu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhuang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yefei","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mike Zheng","family":"Shou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chunhua","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lele","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tingting","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Di","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,5]]},"reference":[{"key":"2435_CR1","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P. & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10\u00a0684\u201310\u00a0695.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2435_CR2","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C. & Chen, M. (2022). Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125."},{"issue":"36","key":"2435_CR3","first-page":"479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E. L., Ghasemipour, K., Gontijo Lopes, R., Karagol Ayan, B., Salimans, T., et al. (2022). Photorealistic text-to-image diffusion models with deep language understanding. Advances in Neural Information Processing Systems, 35(36), 479.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2435_CR4","unstructured":"Xue, Z., Song, G., Guo, Q., Liu, B., Zong, Z., Liu, Y. & Luo, P. (2023). Raphael: Text-to-image generation via large mixture of diffusion paths. arXiv preprint arXiv:2305.18295."},{"key":"2435_CR5","unstructured":"Dai, X., Hou, J., Ma, C.-Y., Tsai, S., Wang, J., Wang, R., Zhang, P. Vandenhende, S., Wang, X., Dubey, A. et\u00a0al. (2023). Emu: Enhancing image generation models using photogenic needles in a haystack. arXiv preprint arXiv:2309.15807."},{"key":"2435_CR6","unstructured":"Schuhmann, C., Beaumont, R., Vencu, R., Gordon, C., Wightman, R., Cherti, M., Coombes, T., Katta, A., Mullis, C., Wortsman, M. et\u00a0al. (2022). Laion-5b: An open large-scale dataset for training next generation image-text models. arXiv preprint arXiv:2210.08402."},{"key":"2435_CR7","unstructured":"Radford, A., Kim, J.\u00a0W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J. et\u00a0al. (2021). Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning. PMLR, pp 8748\u20138763."},{"key":"2435_CR8","unstructured":"Deepfloyd. (2023). Deepfloyd. https:\/\/www.deepfloyd.ai\/."},{"key":"2435_CR9","doi-asserted-by":"crossref","unstructured":"Chen, J., Yu, J., Ge, C., Yao, L., Xie, E., Wu, Y., Wang, Z., Kwok, J., Luo, P., Lu, H. et\u00a0al. (2023). Pixart-$$\\alpha $$: Fast training of diffusion transformer for photorealistic text-to-image synthesis. arXiv preprint arXiv:2310.00426.","DOI":"10.1007\/978-3-031-73411-3_5"},{"issue":"1","key":"2435_CR10","first-page":"5485","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., Shazeer, N., Roberts, A., Lee, K., Narang, S., Matena, M., Zhou, Y., Li, W., & Liu, P. J. (2020). Exploring the limits of transfer learning with a unified text-to-text transformer. The Journal of Machine Learning Research, 21(1), 5485\u20135551.","journal-title":"The Journal of Machine Learning Research"},{"key":"2435_CR11","unstructured":"WeihanWang, W.\u00a0Y., Qingsong, Lv., Wenyi\u00a0Hong, Y.\u00a0W., Qi, Ji, Junhui\u00a0Ji, Z.\u00a0Y., Lei\u00a0Zhao, X.\u00a0S., Jiazheng\u00a0Xu, X.\u00a0B., Juanzi\u00a0Li, Y.\u00a0D. & Ming\u00a0Dingz, J.\u00a0T. (2023). Cogvlm: Visual expert for large language models. arXiv preprint arXiv:5148899."},{"key":"2435_CR12","unstructured":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y., Bashlykov, N., Batra, S., Bhargava, P., Bhosale, S. et\u00a0al. (2023). Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288."},{"key":"2435_CR13","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models. Advances in Neural Information Processing Systems, 33, 6840\u20136851.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2435_CR14","unstructured":"Yu, J., Xu, Y., Koh, J.\u00a0Y., Luong, T., Baid, G., Wang, Z., Vasudevan, V., Ku, A., Yang, Y., Ayan, B.\u00a0K. et\u00a0al. (2022). Scaling autoregressive models for content-rich text-to-image generation. arXiv preprint arXiv:2206.10789."},{"key":"2435_CR15","unstructured":"Gu, Y., Wang, X., Wu, J.\u00a0Z., Shi, Y., Chen, Y., Fan, Z., Xiao, W., Zhao, R., Chang, S., Wu, W. et\u00a0al. (2023). Mix-of-show: Decentralized low-rank adaptation for multi-concept customization of diffusion models. arXiv preprint arXiv:2305.18292."},{"key":"2435_CR16","unstructured":"He, Y., Liu, L., Liu, J., Wu, W., Zhou, H. & Zhuang, B. (2023). Ptqd: Accurate post-training quantization for diffusion models. arXiv preprint arXiv:2305.10657."},{"key":"2435_CR17","unstructured":"Chang, H., Zhang, H., Barber, J., Maschinot, A., Lezama, J., Jiang, L., Yang, M.-H., Murphy, K., Freeman, W.\u00a0T., Rubinstein, M. et\u00a0al. (2023). Muse: Text-to-image generation via masked generative transformers. arXiv preprint arXiv:2301.00704."},{"key":"2435_CR18","doi-asserted-by":"crossref","unstructured":"Kawar, B., Zada, S., Lang, O., Tov, O., Chang, H., Dekel, T., Mosseri, I. & Irani, M. (2023). Imagic: Text-based real image editing with diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 6007\u20136017.","DOI":"10.1109\/CVPR52729.2023.00582"},{"key":"2435_CR19","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A. & Agrawala, M. (2023). Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 3836\u20133847.","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"2435_CR20","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M. & Aberman, K. (2023). Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 22\u00a0500\u201322\u00a0510.","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"2435_CR21","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y. & Cohen-Or, D. (2022). Prompt-to-prompt image editing with cross attention control. arXiv preprint arXiv:2208.01626."},{"key":"2435_CR22","doi-asserted-by":"crossref","unstructured":"Wu, W., Zhao, Y., Shou, M.\u00a0Z., Zhou, H. & Shen, C. (2023). Diffumask: Synthesizing images with pixel-level annotations for semantic segmentation using diffusion models. arXiv preprint arXiv:2303.11681.","DOI":"10.1109\/ICCV51070.2023.00117"},{"key":"2435_CR23","unstructured":"Wu, W., Zhao, Y., Chen, H., Gu, Y., Zhao, R., He, Y., Zhou, H., Shou, M.\u00a0Z. & Shen, C. (2024). Datasetdm: Synthesizing data with perception annotations using diffusion models. Advances in Neural Information Processing Systems, 36."},{"issue":"4","key":"2435_CR24","first-page":"4713","volume":"45","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Ho, J., Chan, W., Salimans, T., Fleet, D. J., & Norouzi, M. (2022). Image super-resolution via iterative refinement. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(4), 4713\u20134726.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2435_CR25","doi-asserted-by":"crossref","unstructured":"Feng, Z., Zhang, Z., Yu, X., Fang, Y., Li, L., Chen, X., Lu, Y., Liu, J., Yin, W., Feng, S. et\u00a0al. (2023). Ernie-vilg 2.0: Improving text-to-image diffusion model with knowledge-enhanced mixture-of-denoising-experts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10\u00a0135\u201310\u00a0145.","DOI":"10.1109\/CVPR52729.2023.00977"},{"key":"2435_CR26","unstructured":"Balaji, Y., Nah, S., Huang, X., Vahdat, A., Song, J., Kreis, K., Aittala, M., Aila, T., Laine, S., Catanzaro, B. et\u00a0al. (2022). Ediffi: Text-to-image diffusion models with an ensemble of expert denoisers. arXiv preprint arXiv:2211.01324."},{"key":"2435_CR27","doi-asserted-by":"crossref","unstructured":"Lewis, M., Liu, Y., Goyal, N., Ghazvininejad, M., Mohamed, A., Levy, O., Stoyanov, V. & Zettlemoyer, L. (2019). Bart: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461.","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"2435_CR28","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., Lachaux, M.-A., Lacroix, T., Rozi\u00e8re, B., Goyal, N., Hambro, E., Azhar, F. et\u00a0al. (2023). Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971."},{"issue":"27","key":"2435_CR29","first-page":"730","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang, L., Wu, J., Jiang, X., Almeida, D., Wainwright, C., Mishkin, P., Zhang, C., Agarwal, S., Slama, K., Ray, A., et al. (2022). Training language models to follow instructions with human feedback. Advances in Neural Information Processing Systems, 35(27), 730.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2435_CR30","doi-asserted-by":"crossref","unstructured":"Wang, Y., Kordi, Y., Mishra, S., Liu, A., Smith, N.\u00a0A., Khashabi, D. & Hajishirzi, H. (2022). Self-instruct: Aligning language model with self generated instructions. arXiv preprint arXiv:2212.10560.","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"2435_CR31","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R. & Constant, N. (2021). The power of scale for parameter-efficient prompt tuning. arXiv preprint arXiv:2104.08691.","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"2435_CR32","unstructured":"Hu, E.\u00a0J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L. & Chen, W. (2021). Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685."},{"key":"2435_CR33","unstructured":"Gani, H., Bhat, S.\u00a0F., Naseer, M., Khan, S. & Wonka, P. (2023). Llm blueprint: Enabling text-to-image generation with complex and detailed prompts. arXiv preprint arXiv:2310.10640."},{"key":"2435_CR34","unstructured":"Lian, L., Li, B., Yala, A. & Darrell, T. (2023). Llm-grounded diffusion: Enhancing prompt understanding of text-to-image diffusion models with large language models. arXiv preprint arXiv:2305.13655."},{"key":"2435_CR35","unstructured":"Feng, W., Zhu, W., Fu, T.-j., Jampani, V., Akula, A., He, X., Basu, S., Wang, X.\u00a0E. & Wang, W.\u00a0Y. (2024). Layoutgpt: Compositional visual planning and generation with large language models. Advances in Neural Information Processing Systems, vol.\u00a036."},{"key":"2435_CR36","doi-asserted-by":"crossref","unstructured":"Li, Y., Liu, H., Wu, Q., Mu, F., Yang, J., Gao, J., Li, C. & Lee, Y.\u00a0J. (2023). Gligen: Open-set grounded text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 22\u00a0511.","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"2435_CR37","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P. & Brox, T. (2015). U-net: Convolutional networks for biomedical image segmentation. In: Medical Image Computing and Computer-Assisted Intervention\u2013MICCAI 2015: 18th International Conference, Munich, Germany, October 5-9, 2015, Proceedings, Part III 18. Springer, pp 234\u2013241.","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"2435_CR38","unstructured":"Podell, D., English, Z., Lacey, K., Blattmann, A., Dockhorn, T., M\u00fcller, J., Penna, J. & Rombach, R. (2023). Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952."},{"key":"2435_CR39","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X. & Elhoseiny, M. (2023). Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592."},{"key":"2435_CR40","unstructured":"Liu, H., Li, C., Wu, Q. & Lee, Y.\u00a0J. (2023). Visual instruction tuning. arXiv preprint arXiv:2304.08485."},{"key":"2435_CR41","unstructured":"Laion. (2022). https:\/\/laion.ai\/blog\/laion-aesthetics\/. blog."},{"key":"2435_CR42","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.\u00a0C., Lo, W.-Y. et\u00a0al. (2023). Segment anything. arXiv preprint arXiv:2304.02643.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"2435_CR43","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P. & Zitnick, C.\u00a0L. (2014). Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13. Springer, pp 740\u2013755.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2435_CR44","doi-asserted-by":"crossref","unstructured":"Lai, X., Tian, Z., Chen, Y., Li, Y., Yuan, Y., Liu, S. & Jia, J. (2023). Lisa: Reasoning segmentation via large language model. arXiv preprint arXiv:2308.00692.","DOI":"10.1109\/CVPR52733.2024.00915"},{"key":"2435_CR45","unstructured":"Nichol, A., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., McGrew, B., Sutskever, I. & Chen, M. (2021). Glide: Towards photorealistic image generation and editing with text-guided diffusion models. arXiv preprint arXiv:2112.10741."},{"key":"2435_CR46","doi-asserted-by":"crossref","unstructured":"Kang, M., Zhu, J.-Y., Zhang, R., Park, J., Shechtman, E., Paris, S. & Park, T. (2023). Scaling up gans for text-to-image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10\u00a0124","DOI":"10.1109\/CVPR52729.2023.00976"},{"key":"2435_CR47","doi-asserted-by":"crossref","unstructured":"Yang, B., Gu, S., Zhang, B., Zhang, T., Chen, X., Sun, X., Chen, D. & Wen, F. (2023). Paint by example: Exemplar-based image editing with diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 18:381\u2013391","DOI":"10.1109\/CVPR52729.2023.01763"},{"key":"2435_CR48","doi-asserted-by":"crossref","unstructured":"Li, Y., Liu, H., Wu, Q., Mu, F., Yang, J., Gao, J., Li, C. & Lee, Y.\u00a0J. (2023). Gligen: Open-set grounded text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), June, pp 22\u00a0511.","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"2435_CR49","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y. & Lee, Y.\u00a0J. (2024). Improved baselines with visual instruction tuning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 26\u00a0296.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"2435_CR50","first-page":"5775","volume":"35","author":"C Lu","year":"2022","unstructured":"Lu, C., Zhou, Y., Bao, F., Chen, J., Li, C., & Zhu, J. (2022). Dpm-solver: A fast ode solver for diffusion probabilistic model sampling in around 10 steps. Advances in Neural Information Processing Systems, 35, 5775\u20135787.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2435_CR51","unstructured":"Song, Y., Dhariwal, P., Chen, M. & Sutskever, I. (2023). Consistency models"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02435-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02435-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02435-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T05:24:40Z","timestamp":1753334680000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02435-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,5]]},"references-count":51,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["2435"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02435-1","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,5]]},"assertion":[{"value":"6 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 March 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}