{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,4]],"date-time":"2025-06-04T04:10:11Z","timestamp":1749010211151,"version":"3.41.0"},"publisher-location":"Cham","reference-count":50,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031938634","type":"print"},{"value":"9783031938641","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-93864-1_7","type":"book-chapter","created":{"date-parts":[[2025,6,3]],"date-time":"2025-06-03T09:15:10Z","timestamp":1748942110000},"page":"86-101","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Aug-Creativity: Framework for\u00a0Human-Centered Creativity with\u00a0Vision Language Models"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1870-9598","authenticated-orcid":false,"given":"Dan","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7926-1593","authenticated-orcid":false,"given":"Lei","family":"Xia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ling","family":"Fan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,4]]},"reference":[{"key":"7_CR1","unstructured":"WikiArt. WikiArt.Org - Visual Art Encyclopedia. Accessed 19 Jan 2025. https:\/\/www.wikiart.org\/"},{"key":"7_CR2","unstructured":"The Metropolitan Museum of Art. The Metropolitan Museum of Art. Accessed 19 Jan 2025. https:\/\/www.metmuseum.org\/art\/collection"},{"issue":"2","key":"7_CR3","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1037\/0022-3514.45.2.357","volume":"45","author":"TM Amabile","year":"1983","unstructured":"Amabile, T.M.: The social psychology of creativity: a componential conceptualization. J. Pers. Soc. Psychol. 45(2), 357\u2013376 (1983). https:\/\/doi.org\/10.1037\/0022-3514.45.2.357","journal-title":"J. Pers. Soc. Psychol."},{"key":"7_CR4","volume-title":"The Nature of Human Intelligence","author":"JP Guilford","year":"1967","unstructured":"Guilford, J.P.: The Nature of Human Intelligence. McGraw-Hill, New York (1967)"},{"key":"7_CR5","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Proceedings of the 38th International Conference on Machine Learning (ICML), pp. 8748\u20138763. PMLR (2021). https:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"7_CR6","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. In: Proceedings of the 36th International Conference on Neural Information Processing Systems (NIPS 2022), pp. 23716\u201323736. Curran Associates Inc., Red Hook, NY, USA (2024)"},{"key":"7_CR7","doi-asserted-by":"publisher","unstructured":"Wang, J., et al.: GIT: a generative image-to-text transformer for vision and language. Trans. Mach. Learn. Res. (2022). https:\/\/doi.org\/10.48550\/arXiv.2205.14100","DOI":"10.48550\/arXiv.2205.14100"},{"key":"7_CR8","volume-title":"Creativity: Flow and the Psychology of Discovery and Invention","author":"M Csikszentmihalyi","year":"1996","unstructured":"Csikszentmihalyi, M.: Creativity: Flow and the Psychology of Discovery and Invention, 1st edn. HarperCollinsPublishers, New York (1996)","edition":"1"},{"key":"7_CR9","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1080\/10400419.2016.1162497","volume":"28","author":"G Goldschmidt","year":"2016","unstructured":"Goldschmidt, G.: Linkographic evidence for concurrent divergent and convergent thinking in creative design. Creat. Res. J. 28, 115\u2013122 (2016). https:\/\/doi.org\/10.1080\/10400419.2016.1162497","journal-title":"Creat. Res. J."},{"key":"7_CR10","doi-asserted-by":"publisher","unstructured":"Zhang, W., Sjoerds, Z., Hommel, B.: Metacontrol of human creativity: the neurocognitive mechanisms of convergent and divergent thinking. NeuroImage 116572 (2020). https:\/\/doi.org\/10.1016\/j.neuroimage.2020.116572","DOI":"10.1016\/j.neuroimage.2020.116572"},{"key":"7_CR11","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1016\/j.neuropsychologia.2018.02.014","volume":"118","author":"DL Zabelina","year":"2018","unstructured":"Zabelina, D.L., Ganis, G.: Creativity and cognitive control: behavioral and ERP evidence that divergent thinking, but not real-life creative achievement. Relat. Better Cogn. Control. Neuropsychologia 118, 20\u201328 (2018). https:\/\/doi.org\/10.1016\/j.neuropsychologia.2018.02.014","journal-title":"Relat. Better Cogn. Control. Neuropsychologia"},{"issue":"5","key":"7_CR12","doi-asserted-by":"publisher","first-page":"858","DOI":"10.1037\/aca0000513","volume":"18","author":"BG Wigert","year":"2024","unstructured":"Wigert, B.G., Murugavel, V.R., Reiter-Palmon, R.: The utility of divergent and convergent thinking in the problem construction processes during creative problem-solving. Psychol. Aesthet. Creat. Arts 18(5), 858\u2013868 (2024). https:\/\/doi.org\/10.1037\/aca0000513","journal-title":"Psychol. Aesthet. Creat. Arts"},{"key":"7_CR13","doi-asserted-by":"publisher","unstructured":"Cropley, A.: In praise of convergent thinking. Creat. Res. J. 18, 391\u2013404 (2006). https:\/\/doi.org\/10.1207\/s15326934crj1803_13","DOI":"10.1207\/s15326934crj1803_13"},{"key":"7_CR14","doi-asserted-by":"publisher","first-page":"262","DOI":"10.1080\/10400419.2015.1063877","volume":"27","author":"D Simonton","year":"2015","unstructured":"Simonton, D.: On praising convergent thinking: creativity as blind variation and selective retention. Creat. Res. J. 27, 262\u2013270 (2015). https:\/\/doi.org\/10.1080\/10400419.2015.1063877","journal-title":"Creat. Res. J."},{"key":"7_CR15","doi-asserted-by":"publisher","unstructured":"Shneiderman, B.: Creativity Support Tools: A Grand Challenge for HCI Researchers. In: Redondo, M., Bravo, C., Ortega, M. (eds.) Engineering the User Interface, pp. 1\u20139. Springer, London (2009). https:\/\/doi.org\/10.1007\/978-1-84800-136-7_1","DOI":"10.1007\/978-1-84800-136-7_1"},{"issue":"12","key":"7_CR16","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1145\/1323688.1323689","volume":"50","author":"B Shneiderman","year":"2007","unstructured":"Shneiderman, B.: Creativity support tools: accelerating discovery and innovation. Commun. ACM 50(12), 20\u201332 (2007). https:\/\/doi.org\/10.1145\/1323688.1323689","journal-title":"Commun. ACM"},{"key":"7_CR17","volume-title":"Visual Thinking","author":"R Arnheim","year":"1969","unstructured":"Arnheim, R.: Visual Thinking. University of California Press, Berkeley, CA, US (1969)"},{"key":"7_CR18","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/7722.001.0001","volume-title":"Creative Cognition: Theory, Research, and Applications","author":"RA Finke","year":"1992","unstructured":"Finke, R.A., Ward, T.B., Smith, S.M.: Creative Cognition: Theory, Research, and Applications. The MIT Press, Cambridge, MA, US (1992)"},{"key":"7_CR19","unstructured":"Alayrac, J.-B., et al.: Flamingo: a visual language model for few-shot learning. In: Proceedings of the 36th International Conference on Neural Information Processing Systems, 23716-36. NIPS 2022. Red Hook, NY, USA: Curran Associates Inc. (2024)"},{"key":"7_CR20","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Proceedings of the 38th International Conference on Machine Learning, pp. 8748-8763. PMLR (2021). https:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"7_CR21","unstructured":"Wang, P., et al.: OFA: unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In: Proceedings of the 39th International Conference on Machine Learning, pp. 23318-23340. PMLR (2022). https:\/\/proceedings.mlr.press\/v162\/wang22al.html"},{"key":"7_CR22","doi-asserted-by":"publisher","unstructured":"Kar, O.F., et al.: BRAVE: broadening the visual encoding of vision-language models. arXiv, 10 April 2024. https:\/\/doi.org\/10.48550\/arXiv.2404.07204","DOI":"10.48550\/arXiv.2404.07204"},{"key":"7_CR23","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of the 40th International Conference on Machine Learning, pp. 19730-19742. PMLR (2023). https:\/\/proceedings.mlr.press\/v202\/li23q.html"},{"key":"7_CR24","unstructured":"Chen, X., et al.: PaLI: a jointly-scaled multilingual language-image model (2023). https:\/\/arxiv.org\/abs\/2209.06794"},{"key":"7_CR25","doi-asserted-by":"publisher","unstructured":"Dai, W., et al.: InstructBLIP: towards general-purpose vision-language models with instruction tuning. arXiv, 15 June 2023. https:\/\/doi.org\/10.48550\/arXiv.2305.06500","DOI":"10.48550\/arXiv.2305.06500"},{"key":"7_CR26","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual Instruction Tuning. Adv. Neural Inf. Process. Syst. 36, 34892\u201334916 (2023)"},{"key":"7_CR27","doi-asserted-by":"publisher","unstructured":"Wang, J., et al.: GIT: a generative image-to-text transformer for vision and language. Trans. Mach. Learn. Res. 15 December 2022. https:\/\/doi.org\/10.48550\/arXiv.2205.14100","DOI":"10.48550\/arXiv.2205.14100"},{"key":"7_CR28","doi-asserted-by":"publisher","unstructured":"Yang, Z., et al.: MM-REACT: prompting ChatGPT for multimodal reasoning and action. arXiv, 20 March 2023. https:\/\/doi.org\/10.48550\/arXiv.2303.11381","DOI":"10.48550\/arXiv.2303.11381"},{"key":"7_CR29","doi-asserted-by":"publisher","unstructured":"Wu, C., et al.: Visual ChatGPT: talking, drawing and editing with visual foundation models. arXiv, 8 Mar 2023. https:\/\/doi.org\/10.48550\/arXiv.2303.04671","DOI":"10.48550\/arXiv.2303.04671"},{"key":"7_CR30","doi-asserted-by":"publisher","unstructured":"Sur\u00eds, D., Menon, S., Vondrick, C.: ViperGPT: visual inference via python execution for reasoning. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 11854\u201311864 (2023). https:\/\/doi.org\/10.1109\/ICCV51070.2023.01092","DOI":"10.1109\/ICCV51070.2023.01092"},{"key":"7_CR31","doi-asserted-by":"publisher","unstructured":"Wu, C., Yin, S., Qi, W., Wang, X., Tang, Z., Duan, N.: Visual ChatGPT: talking, drawing and editing with visual foundation models. arXiv (2023). https:\/\/doi.org\/10.48550\/arXiv.2303.04671.","DOI":"10.48550\/arXiv.2303.04671."},{"key":"7_CR32","doi-asserted-by":"publisher","unstructured":"Aslam, J.A., Yilmaz, E., Pavlu, V.: A geometric interpretation of r-precision and its correlation with average precision. In: Proceedings of the 28th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2005), pp. 573\u2013574. Association for Computing Machinery, New York, NY, USA (2005). https:\/\/doi.org\/10.1145\/1076034.1076134","DOI":"10.1145\/1076034.1076134"},{"key":"7_CR33","unstructured":"Banerjee, S., Lavie, A.: METEOR: An Automatic metric for MT evaluation with improved correlation with human judgments. In: Goldstein, J., Lavie, A., Lin, C.-Y., Voss, C. (eds.) Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization, pp. 65\u201372. Association for Computational Linguistics, Ann Arbor, Michigan, USA (2005). https:\/\/aclanthology.org\/W05-0909\/"},{"key":"7_CR34","doi-asserted-by":"publisher","unstructured":"Hessel, J., Holtzman, A., Forbes, M., Le Bras, R., Choi, Y.: CLIPScore: a reference-free evaluation metric for image captioning. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 7514\u20137528. Association for Computational Linguistics, Punta Cana, Dominican Republic (2021). https:\/\/doi.org\/10.18653\/v1\/2021.emnlp-main.595","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"7_CR35","doi-asserted-by":"publisher","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: GANs trained by a two time-scale update rule converge to a local nash equilibrium. arXiv (2018). https:\/\/doi.org\/10.48550\/arXiv.1706.08500","DOI":"10.48550\/arXiv.1706.08500"},{"key":"7_CR36","doi-asserted-by":"publisher","unstructured":"Kafle, K., Kanan, C.: An analysis of visual question answering algorithms. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp. 1983\u20131991. IEEE, Venice, Italy (2017). https:\/\/doi.org\/10.1109\/ICCV.2017.217","DOI":"10.1109\/ICCV.2017.217"},{"key":"7_CR37","unstructured":"Lin, C.-Y.: ROUGE: a package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp. 74\u201381. Association for Computational Linguistics, Barcelona, Spain (2004). https:\/\/aclanthology.org\/W04-1013\/"},{"key":"7_CR38","doi-asserted-by":"publisher","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J.: BLEU: a method for automatic evaluation of machine translation. In: Isabelle, P., Charniak, E., Lin, D. (eds.) Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311-318. Association for Computational Linguistics, Philadelphia, PA, USA (2002). https:\/\/doi.org\/10.3115\/1073083.1073135","DOI":"10.3115\/1073083.1073135"},{"key":"7_CR39","doi-asserted-by":"publisher","unstructured":"Patel, Y., Tolias, G., Matas, J.: Recall@k surrogate loss with large batches and similarity Mixup. arXiv (2022). https:\/\/doi.org\/10.48550\/arXiv.2108.11179","DOI":"10.48550\/arXiv.2108.11179"},{"key":"7_CR40","doi-asserted-by":"publisher","unstructured":"Zhang, T., Kishore, V., Wu, F., Weinberger, K.Q., Artzi, Y.: BERTScore: evaluating text generation with BERT. arXiv (2020). https:\/\/doi.org\/10.48550\/arXiv.1904.09675","DOI":"10.48550\/arXiv.1904.09675"},{"key":"7_CR41","doi-asserted-by":"publisher","unstructured":"Wilber, M.J., Fang, C., Jin, H., Hertzmann, A., Collomosse, J., Belongie, S.: BAM! the behance artistic media dataset for recognition beyond photography. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp. 1211\u20131220. IEEE, Venice, Italy (2017). https:\/\/doi.org\/10.1109\/ICCV.2017.136","DOI":"10.1109\/ICCV.2017.136"},{"key":"7_CR42","doi-asserted-by":"publisher","unstructured":"Liao, P., Li, X., Liu, X., Keutzer, K.: The ArtBench dataset: benchmarking generative models with artworks. arXiv (2022). https:\/\/doi.org\/10.48550\/arXiv.2206.11404","DOI":"10.48550\/arXiv.2206.11404"},{"issue":"6","key":"7_CR43","doi-asserted-by":"publisher","first-page":"1385","DOI":"10.1007\/s00138-014-0621-6","volume":"25","author":"FS Khan","year":"2014","unstructured":"Khan, F.S., Beigpour, S., van de Weijer, J., Felsberg, M.: Painting-91: a large scale database for computational painting categorization. Mach. Vis. Appl. 25(6), 1385\u20131397 (2014). https:\/\/doi.org\/10.1007\/s00138-014-0621-6","journal-title":"Mach. Vis. Appl."},{"key":"7_CR44","unstructured":"The Tate Collection. Accessed 3 Feb. 2025. https:\/\/www.kaggle.com\/datasets\/rtatman\/the-tate-collection"},{"key":"7_CR45","unstructured":"The Metropolitan Museum of Art Open Access. Accessed 3 Feb 2025. https:\/\/www.kaggle.com\/datasets\/metmuseum\/the-metropolitan-museum-of-art-open-access"},{"key":"7_CR46","unstructured":"Rijksmuseum. https:\/\/www.kaggle.com\/datasets\/lgmoneda\/rijksmuseum"},{"key":"7_CR47","unstructured":"Museum of Modern Art Collection. Accessed 3 Feb 2025. https:\/\/www.kaggle.com\/datasets\/momanyc\/museum-collection"},{"key":"7_CR48","doi-asserted-by":"publisher","unstructured":"Strezoski, G., Worring, M.: OmniArt: a large-scale artistic benchmark. ACM Trans. Multimedia Comput. Commun. Appl. 14(4), 88:1-88:21 (2018). https:\/\/doi.org\/10.1145\/3273022","DOI":"10.1145\/3273022"},{"key":"7_CR49","unstructured":"Hu, E.J., et al.: LoRA: low-rank adaptation of large language models (2021). https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"7_CR50","unstructured":"O*NET Online. Accessed 29 Jan 2025. https:\/\/www.onetonline.org\/"}],"container-title":["Lecture Notes in Computer Science","Human-Computer Interaction"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-93864-1_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,3]],"date-time":"2025-06-03T09:15:25Z","timestamp":1748942125000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-93864-1_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031938634","9783031938641"],"references-count":50,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-93864-1_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"4 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors\u00a0have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"HCII","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Human-Computer Interaction","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Gothenburg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Sweden","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 June 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 June 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"hcii2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2025.hci.international\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}