{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T16:08:45Z","timestamp":1783094925490,"version":"3.54.6"},"publisher-location":"Cham","reference-count":65,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729195","type":"print"},{"value":"9783031729201","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72920-1_10","type":"book-chapter","created":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T08:02:57Z","timestamp":1727683377000},"page":"167-184","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Idea2Img: Iterative Self-refinement with\u00a0GPT-4V for\u00a0Automatic Image Design and\u00a0Generation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5808-0889","authenticated-orcid":false,"given":"Zhengyuan","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3156-4429","authenticated-orcid":false,"given":"Jianfeng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linjie","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kevin","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6507-3657","authenticated-orcid":false,"given":"Chung-Ching","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5894-7828","authenticated-orcid":false,"given":"Zicheng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5705-876X","authenticated-orcid":false,"given":"Lijuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,1]]},"reference":[{"key":"10_CR1","unstructured":"ChatGPT can now see, hear, and speak (2023). https:\/\/openai.com\/blog\/chatgpt-can-now-see-hear-and-speak"},{"key":"10_CR2","doi-asserted-by":"crossref","unstructured":"Avrahami, O., Aberman, K., Fried, O., Cohen-Or, D., Lischinski, D.: Break-a-scene: extracting multiple concepts from a single image. arXiv preprint arXiv:2305.16311 (2023)","DOI":"10.1145\/3610548.3618154"},{"key":"10_CR3","doi-asserted-by":"crossref","unstructured":"Avrahami, O., Lischinski, D., Fried, O.: Blended diffusion for text-driven editing of natural images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18208\u201318218 (2022)","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"10_CR4","unstructured":"Betker, J., et al.: Improving image generation with better captions (2023)"},{"key":"10_CR5","unstructured":"Black, K., Janner, M., Du, Y., Kostrikov, I., Levine, S.: Training diffusion models with reinforcement learning. arXiv preprint arXiv:2305.13301 (2023)"},{"key":"10_CR6","doi-asserted-by":"crossref","unstructured":"Brooks, T., Holynski, A., Efros, A.A.: Instructpix2pix: learning to follow image editing instructions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18392\u201318402 (2023)","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"10_CR7","doi-asserted-by":"crossref","unstructured":"Chefer, H., Alaluf, Y., Vinker, Y., Wolf, L., Cohen-Or, D.: Attend-and-excite: attention-based semantic guidance for text-to-image diffusion models. arXiv preprint arXiv:2301.13826 (2023)","DOI":"10.1145\/3592116"},{"key":"10_CR8","unstructured":"Chen, W., et al.: Subject-driven text-to-image generation via apprenticeship learning. arXiv preprint arXiv:2304.00186 (2023)"},{"key":"10_CR9","unstructured":"Chen, X., Lin, M., Sch\u00e4rli, N., Zhou, D.: Teaching large language models to self-debug. arXiv preprint arXiv:2304.05128 (2023)"},{"key":"10_CR10","unstructured":"Fan, Y., et al.: DPOK: reinforcement learning for fine-tuning text-to-image diffusion models. arXiv preprint arXiv:2305.16381 (2023)"},{"key":"10_CR11","unstructured":"Fan, Y., et al.: Reinforcement learning for fine-tuning text-to-image diffusion models. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"10_CR12","unstructured":"Feng, W., et al.: Training-free structured diffusion guidance for compositional text-to-image synthesis. In: The Eleventh International Conference on Learning Representations (2022)"},{"key":"10_CR13","doi-asserted-by":"crossref","unstructured":"Gatys, L.A., Ecker, A.S., Bethge, M.: A neural algorithm of artistic style. arXiv preprint arXiv:1508.06576 (2015)","DOI":"10.1167\/16.12.326"},{"key":"10_CR14","unstructured":"Google: Bard (2023). https:\/\/bard.google.com. Accessed 17 July 2023"},{"key":"10_CR15","unstructured":"Guo, Y., Liang, Y., Wu, C., Wu, W., Zhao, D., Duan, N.: Learning to program with natural language. arXiv preprint arXiv:2304.10464 (2023)"},{"key":"10_CR16","doi-asserted-by":"crossref","unstructured":"Gupta, T., Kembhavi, A.: Visual programming: compositional visual reasoning without training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14953\u201314962 (2023)","DOI":"10.1109\/CVPR52729.2023.01436"},{"key":"10_CR17","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., Cohen-or, D.: Prompt-to-prompt image editing with cross-attention control. In: The Eleventh International Conference on Learning Representations (2022)"},{"key":"10_CR18","doi-asserted-by":"crossref","unstructured":"Kawar, B., et al.: Imagic: text-based real image editing with diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6007\u20136017 (2023)","DOI":"10.1109\/CVPR52729.2023.00582"},{"key":"10_CR19","doi-asserted-by":"crossref","unstructured":"Kumari, N., Zhang, B., Zhang, R., Shechtman, E., Zhu, J.Y.: Multi-concept customization of text-to-image diffusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1931\u20131941 (2023)","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"10_CR20","unstructured":"Lab, D.: Deepfloyd IF (2023). https:\/\/github.com\/deep-floyd\/IF"},{"key":"10_CR21","unstructured":"Lee, K., et al.: Aligning text-to-image models using human feedback. arXiv preprint arXiv:2302.12192 (2023)"},{"key":"10_CR22","doi-asserted-by":"crossref","unstructured":"Li, C., et al.: Multimodal foundation models: from specialists to general-purpose assistants. arXiv preprint arXiv:2309.10020 (2023)","DOI":"10.1561\/9781638283379"},{"key":"10_CR23","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: GLIGEN: open-set grounded text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22511\u201322521 (2023)","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"10_CR24","unstructured":"Lu, P., et al.: Chameleon: plug-and-play compositional reasoning with large language models. arXiv preprint arXiv:2304.09842 (2023)"},{"key":"10_CR25","unstructured":"Madaan, A., et\u00a0al.: Self-refine: iterative refinement with self-feedback. arXiv preprint arXiv:2303.17651 (2023)"},{"key":"10_CR26","unstructured":"Meng, C., et al.: SDEdit: guided image synthesis and editing with stochastic differential equations. arXiv preprint arXiv:2108.01073 (2021)"},{"key":"10_CR27","unstructured":"Nasiriany, S., et\u00a0al.: Pivot: iterative visual prompting elicits actionable knowledge for VLMs. arXiv preprint arXiv:2402.07872 (2024)"},{"key":"10_CR28","unstructured":"OpenAI: Dall$$\\cdot $$e 3 system card (2023). https:\/\/cdn.openai.com\/papers\/DALL_E_3_System_Card.pdf"},{"key":"10_CR29","unstructured":"OpenAI: GPT-4 technical report (2023)"},{"key":"10_CR30","unstructured":"OpenAI: GPT-4v(ision) system card (2023). https:\/\/cdn.openai.com\/papers\/GPTV_System_Card.pdf"},{"key":"10_CR31","unstructured":"OpenAI: GPT-4v(ision) technical work and authors (2023). https:\/\/cdn.openai.com\/contributions\/gpt-4v.pdf"},{"key":"10_CR32","doi-asserted-by":"crossref","unstructured":"Pan, L., Saxon, M., Xu, W., Nathani, D., Wang, X., Wang, W.Y.: Automatically correcting large language models: surveying the landscape of diverse self-correction strategies. arXiv preprint arXiv:2308.03188 (2023)","DOI":"10.1162\/tacl_a_00660"},{"key":"10_CR33","unstructured":"Paranjape, B., Lundberg, S., Singh, S., Hajishirzi, H., Zettlemoyer, L., Ribeiro, M.T.: Art: automatic multi-step reasoning and tool-use for large language models. arXiv preprint arXiv:2303.09014 (2023)"},{"key":"10_CR34","unstructured":"Podell, D., et al.: SDXL: improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952 (2023)"},{"key":"10_CR35","doi-asserted-by":"crossref","unstructured":"Pryzant, R., Iter, D., Li, J., Lee, Y.T., Zhu, C., Zeng, M.: Automatic prompt optimization with \u201cgradient descent\u201d and beam search. arXiv preprint arXiv:2305.03495 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.494"},{"key":"10_CR36","unstructured":"Qi, J., et\u00a0al.: CogCoM: train large vision-language models diving into details through chain of manipulations. arXiv preprint arXiv:2402.04236 (2024)"},{"key":"10_CR37","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 (2022)"},{"key":"10_CR38","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"10_CR39","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: DreamBooth: fine tuning text-to-image diffusion models for subject-driven generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22500\u201322510 (2023)","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"10_CR40","unstructured":"Saharia, C., et\u00a0al.: Photorealistic text-to-image diffusion models with deep language understanding. arXiv preprint arXiv:2205.11487 (2022)"},{"key":"10_CR41","unstructured":"Schick, T., et al.: ToolFormer: language models can teach themselves to use tools. arXiv preprint arXiv:2302.04761 (2023)"},{"key":"10_CR42","unstructured":"Shen, Y., Song, K., Tan, X., Li, D., Lu, W., Zhuang, Y.: HuggingGPT: solving AI tasks with ChatGPT and its friends in huggingface. arXiv preprint arXiv:2303.17580 (2023)"},{"key":"10_CR43","doi-asserted-by":"crossref","unstructured":"Shi, J., Xiong, W., Lin, Z., Jung, H.J.: InstantBooth: personalized text-to-image generation without test-time finetuning. arXiv preprint arXiv:2304.03411 (2023)","DOI":"10.1109\/CVPR52733.2024.00816"},{"key":"10_CR44","unstructured":"Shinn, N., Cassano, F., Labash, B., Gopinath, A., Narasimhan, K., Yao, S.: Reflexion: language agents with verbal reinforcement learning (2023)"},{"key":"10_CR45","unstructured":"Shridhar, M., Yuan, X., C\u00f4t\u00e9, M.A., Bisk, Y., Trischler, A., Hausknecht, M.: ALFWorld: aligning text and embodied environments for interactive learning. arXiv preprint arXiv:2010.03768 (2020)"},{"key":"10_CR46","unstructured":"Singer, U., et\u00a0al.: Make-a-video: text-to-video generation without text-video data. arXiv preprint arXiv:2209.14792 (2022)"},{"key":"10_CR47","doi-asserted-by":"crossref","unstructured":"Sur\u00eds, D., Menon, S., Vondrick, C.: ViperGPT: visual inference via python execution for reasoning. arXiv preprint arXiv:2303.08128 (2023)","DOI":"10.1109\/ICCV51070.2023.01092"},{"key":"10_CR48","unstructured":"Wang, J., et al.: GIT: a generative image-to-text transformer for vision and language. arXiv preprint arXiv:2205.14100 (2022)"},{"key":"10_CR49","doi-asserted-by":"crossref","unstructured":"Wang, Z.J., Montoya, E., Munechika, D., Yang, H., Hoover, B., Chau, D.H.: DiffusionDB: a large-scale prompt gallery dataset for text-to-image generative models. arXiv preprint arXiv:2210.14896 (2022)","DOI":"10.18653\/v1\/2023.acl-long.51"},{"key":"10_CR50","unstructured":"Wu, C., Yin, S., Qi, W., Wang, X., Tang, Z., Duan, N.: Visual ChatGPT: talking, drawing and editing with visual foundation models. arXiv preprint arXiv:2303.04671 (2023)"},{"key":"10_CR51","unstructured":"Wu, J., et al.: Grit: a generative region-to-text transformer for object understanding. arXiv preprint arXiv:2212.00280 (2022)"},{"key":"10_CR52","unstructured":"Wu, P., Xie, S.: V*: guided visual search as a core mechanism in multimodal LLMs. arXiv preprint arXiv:2312.14135, 17 (2023)"},{"key":"10_CR53","unstructured":"Yan, A., et\u00a0al.: GPT-4v in wonderland: large multimodal models for zero-shot smartphone GUI navigation. arXiv preprint arXiv:2311.07562 (2023)"},{"key":"10_CR54","unstructured":"Yang, C., et al..: Large language models as optimizers. arXiv preprint arXiv:2309.03409 (2023)"},{"key":"10_CR55","unstructured":"Yang, Z., et al.: The dawn of LMMs: preliminary explorations with GPT-4v (ision). arXiv preprint arXiv:2309.17421 (2023)"},{"key":"10_CR56","unstructured":"Yang, Z., et al.: MM-react: prompting ChatGPT for multimodal reasoning and action. arXiv preprint arXiv:2303.11381 (2023)"},{"key":"10_CR57","doi-asserted-by":"crossref","unstructured":"Yang, Z., et\u00a0al.: ReCo: region-controlled text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14246\u201314255 (2023)","DOI":"10.1109\/CVPR52729.2023.01369"},{"key":"10_CR58","doi-asserted-by":"crossref","unstructured":"Yang, Z., et al.: HotpotQA: a dataset for diverse, explainable multi-hop question answering. arXiv preprint arXiv:1809.09600 (2018)","DOI":"10.18653\/v1\/D18-1259"},{"key":"10_CR59","unstructured":"Yao, S., et al.: React: synergizing reasoning and acting in language models. arXiv preprint arXiv:2210.03629 (2022)"},{"key":"10_CR60","doi-asserted-by":"crossref","unstructured":"Yin, S., et\u00a0al.: NUWA-XL: diffusion over diffusion for extremely long video generation. arXiv preprint arXiv:2303.12346 (2023)","DOI":"10.18653\/v1\/2023.acl-long.73"},{"key":"10_CR61","unstructured":"Yu, J., et\u00a0al.: Scaling autoregressive models for content-rich text-to-image generation. Trans. Mach. Learn. Res. (2022)"},{"key":"10_CR62","unstructured":"Yu, W., et al.: MM-vet: evaluating large multimodal models for integrated capabilities. arXiv preprint arXiv:2308.02490 (2023)"},{"key":"10_CR63","doi-asserted-by":"crossref","unstructured":"Zhang, L., Agrawala, M.: Adding conditional control to text-to-image diffusion models. arXiv preprint arXiv:2302.05543 (2023)","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"10_CR64","doi-asserted-by":"crossref","unstructured":"Zhao, A., Huang, D., Xu, Q., Lin, M., Liu, Y.J., Huang, G.: Expel: LLM agents are experiential learners. arXiv preprint arXiv:2308.10144 (2023)","DOI":"10.1609\/aaai.v38i17.29936"},{"key":"10_CR65","doi-asserted-by":"crossref","unstructured":"Zhu, W., et al.: Collaborative generative AI: integrating GPT-K for efficient editing in text-to-image generation. arXiv preprint arXiv:2305.11317 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.685"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72920-1_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T08:09:57Z","timestamp":1727683797000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72920-1_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,1]]},"ISBN":["9783031729195","9783031729201"],"references-count":65,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72920-1_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,1]]},"assertion":[{"value":"1 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}