{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T01:42:50Z","timestamp":1784943770323,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":60,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,4,25]],"date-time":"2025-04-25T00:00:00Z","timestamp":1745539200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"National Research Foundation of Korea (NRF) grant funded by the Korean government (MSIT)","award":["2023R1A2C200520911"],"award-info":[{"award-number":["2023R1A2C200520911"]}]},{"name":"Institute of Information & commu- nications Technology Planning & Evaluation (IITP) grant funded by the Korean government (MSIT)","award":["RS-2021-II21134"],"award-info":[{"award-number":["RS-2021-II21134"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,4,26]]},"DOI":"10.1145\/3706598.3714213","type":"proceedings-article","created":{"date-parts":[[2025,4,24]],"date-time":"2025-04-24T03:33:32Z","timestamp":1745465612000},"page":"1-22","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":15,"title":["Leveraging Multimodal LLM for Inspirational User Interface Search"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-1685-4027","authenticated-orcid":false,"given":"Seokhyeon","family":"Park","sequence":"first","affiliation":[{"name":"Department of Computer Science and Engineering, Seoul National University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5277-4822","authenticated-orcid":false,"given":"Yumin","family":"Song","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, Seoul National University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3075-3981","authenticated-orcid":false,"given":"Soohyun","family":"Lee","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, Seoul National University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1868-7148","authenticated-orcid":false,"given":"Jaeyoung","family":"Kim","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, Seoul National University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7734-822X","authenticated-orcid":false,"given":"Jinwook","family":"Seo","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, Seoul National University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,4,25]]},"reference":[{"key":"e_1_3_3_3_2_2","unstructured":"Xavier Amatriain. 2024. Prompt Design and Engineering: Introduction and Advanced Methods. arxiv:https:\/\/arXiv.org\/abs\/2401.14423\u00a0[cs.SE]"},{"key":"e_1_3_3_3_3_2","unstructured":"Anthropic. 2023. Be clear direct and detailed - Anthropic \u2014 docs.anthropic.com. https:\/\/docs.anthropic.com\/en\/docs\/build-with-claude\/prompt-engineering\/be-clear-and-direct"},{"key":"e_1_3_3_3_4_2","unstructured":"Anthropic. 2023. Prompt Engineering \/ Use XML Tags. https:\/\/docs.anthropic.com\/en\/docs\/build-with-claude\/prompt-engineering\/use-xml-tags"},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2024\/339"},{"key":"e_1_3_3_3_6_2","unstructured":"Chongyang Bai Xiaoxue Zang Ying Xu Srinivas Sunkara Abhinav Rastogi Jindong Chen and Blaise\u00a0Aguera y Arcas. 2021. UIBert: Learning Generic Multimodal Representations for UI Understanding. arxiv:https:\/\/arXiv.org\/abs\/2107.13731\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2107.13731"},{"key":"e_1_3_3_3_7_2","unstructured":"Romain Beaumont. 2022. Clip Retrieval: Easily compute clip embeddings and build a clip retrieval system with them. https:\/\/github.com\/rom1504\/clip-retrieval."},{"key":"e_1_3_3_3_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/317561.317589"},{"key":"e_1_3_3_3_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445762"},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"publisher","unstructured":"Joel Chan Steven\u00a0P. Dow and Christian\u00a0D. Schunn. 2015. Do the best design ideas (really) come from conceptually distant sources of inspiration? Design Studies 36 (2015) 31\u201358. 10.1016\/j.destud.2014.08.001","DOI":"10.1016\/j.destud.2014.08.001"},{"key":"e_1_3_3_3_11_2","unstructured":"Kanzhi Cheng Qiushi Sun Yougang Chu Fangzhi Xu Yantao Li Jianbing Zhang and Zhiyong Wu. 2024. SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents. arxiv:https:\/\/arXiv.org\/abs\/2401.10935\u00a0[cs.HC] https:\/\/arxiv.org\/abs\/2401.10935"},{"key":"e_1_3_3_3_12_2","unstructured":"Design Council. [n. d.]. Double Diamond. https:\/\/www.designcouncil.org.uk\/our-resources\/the-double-diamond\/"},{"key":"e_1_3_3_3_13_2","doi-asserted-by":"publisher","unstructured":"Nigel Cross. 2004. Expertise in design: an overview. Design Studies 25 5 (2004) 427\u2013441. 10.1016\/j.destud.2004.06.002Expertise in Design.","DOI":"10.1016\/j.destud.2004.06.002"},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/3126594.3126651"},{"key":"e_1_3_3_3_15_2","unstructured":"Oluwole Fagbohun Rachel\u00a0M. Harrison and Anton Dereventsov. 2024. An Empirical Categorization of Prompting Techniques for Large Language Models: A Practitioner\u2019s Guide. arxiv:https:\/\/arXiv.org\/abs\/2402.14837\u00a0[cs.CL]"},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"crossref","unstructured":"Yuan Gao Kunyu Shi Pengkai Zhu Edouard Belval Oren Nuriel Srikar Appalaraju Shabnam Ghadar Vijay Mahadevan Zhuowen Tu and Stefano Soatto. 2024. Enhancing Vision-Language Pre-training with Rich Supervisions. arxiv:https:\/\/arXiv.org\/abs\/2403.03346\u00a0[cs.CV]","DOI":"10.1109\/CVPR52733.2024.01280"},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"publisher","unstructured":"Milene Gon\u00e7alves Carlos Cardoso and Petra Badke-Schaub. 2014. What inspires designers? Preferences on inspirational approaches during idea generation. Design Studies 35 1 (2014) 29\u201353. 10.1016\/j.destud.2013.09.001","DOI":"10.1016\/j.destud.2013.09.001"},{"key":"e_1_3_3_3_18_2","unstructured":"Google. 2024. Give clear and specific instructions. https:\/\/cloud.google.com\/vertex-ai\/generative-ai\/docs\/learn\/prompts\/clear-instructions"},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"publisher","unstructured":"Greg Guest Arwen Bunce and Laura Johnson. 2006. How Many Interviews Are Enough?: An Experiment with Data Saturation and Variability. Field Methods 18 1 (2006) 59\u201382. 10.1177\/1525822X05279903","DOI":"10.1177\/1525822X05279903"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/1518701.1518717"},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"publisher","unstructured":"Matthew Honnibal Ines Montani Sofie Van\u00a0Landeghem and Adriane Boyd. 2020. spaCy: Industrial-strength Natural Language Processing in Python. 10.5281\/zenodo.1212303","DOI":"10.5281\/zenodo.1212303"},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"publisher","unstructured":"Chih-Pei HU and Yan-Yi CHANG. 2017. John W. Creswell Research Design: Qualitative Quantitative and Mixed Methods Approaches. Journal of Social and Administrative Sciences 4 2 (Jun. 2017) 205\u2013207. 10.1453\/jsas.v4i2.1313","DOI":"10.1453\/jsas.v4i2.1313"},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300334"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491101.3504030"},{"key":"e_1_3_3_3_25_2","unstructured":"Jong\u00a0Wook Kim. 2021. CLIP\/Prompt_Engineering_for_ImageNet.ipynb at main \u00b7 openai\/CLIP \u2014 github.com. https:\/\/github.com\/openai\/CLIP\/blob\/main\/notebooks\/Prompt_Engineering_for_ImageNet.ipynb."},{"key":"e_1_3_3_3_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300863"},{"key":"e_1_3_3_3_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/1753326.1753667"},{"key":"e_1_3_3_3_28_2","series-title":"Proceedings of Machine Learning Research","first-page":"18893","volume-title":"Proceedings of the 40th International Conference on Machine Learning","volume":"202","author":"Lee Kenton","year":"2023","unstructured":"Kenton Lee, Mandar Joshi, Iulia\u00a0Raluca Turc, Hexiang Hu, Fangyu Liu, Julian\u00a0Martin Eisenschlos, Urvashi Khandelwal, Peter Shaw, Ming-Wei Chang, and Kristina Toutanova. 2023. Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding. In Proceedings of the 40th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0202), Andreas Krause, Emma Brunskill, Kyunghyun Cho, Barbara Engelhardt, Sivan Sabato, and Jonathan Scarlett (Eds.). PMLR, 18893\u201318912. https:\/\/proceedings.mlr.press\/v202\/lee23g.html"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/VIS55277.2024.00053"},{"key":"e_1_3_3_3_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/3406324.3410710"},{"key":"e_1_3_3_3_31_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3502042"},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445049"},{"key":"e_1_3_3_3_33_2","unstructured":"Yang Li Gang Li Luheng He Jingjie Zheng Hong Li and Zhiwei Guan. 2020. Widget Captioning: Generating Natural Language Description for Mobile User Interface Elements. arxiv:https:\/\/arXiv.org\/abs\/2010.04295\u00a0[cs.LG]"},{"key":"e_1_3_3_3_34_2","unstructured":"Junpeng Liu Tianyue Ou Yifan Song Yuxiao Qu Wai Lam Chenyan Xiong Wenhu Chen Graham Neubig and Xiang Yue. 2024. Harnessing Webpage UIs for Text-Rich Visual Understanding. (2024). arxiv:https:\/\/arXiv.org\/abs\/2410.13824\u00a0[cs.CV]"},{"key":"e_1_3_3_3_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3242587.3242650"},{"key":"e_1_3_3_3_36_2","unstructured":"Elya Livshitz. 2023. YAML vs. JSON: Which is more efficient for language models?https:\/\/betterprogramming.pub\/yaml-vs-json-which-is-more-efficient-for-language-models-5bc11dd0f6df"},{"key":"e_1_3_3_3_37_2","unstructured":"Yuwen Lu Ziang Tong Qinyi Zhao Chengzhi Zhang and Toby Jia-Jun Li. 2023. UI Layout Generation with LLMs Guided by UI Grammar. arxiv:https:\/\/arXiv.org\/abs\/2310.15455\u00a0[cs.HC]"},{"key":"e_1_3_3_3_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491101.3519809"},{"key":"e_1_3_3_3_39_2","unstructured":"Kate Moran. 2016. Tone-of-voice words. https:\/\/www.nngroup.com\/articles\/tone-voice-words\/"},{"key":"e_1_3_3_3_40_2","doi-asserted-by":"publisher","unstructured":"Janice\u00a0M. Morse. 2000. Determining Sample Size. 3-5\u00a0pages. 10.1177\/104973200129118183","DOI":"10.1177\/104973200129118183"},{"key":"e_1_3_3_3_41_2","volume-title":"Introducing vision to the fine-tuning API","year":"2024","unstructured":"OpenAI. 2024. Introducing vision to the fine-tuning API. https:\/\/openai.com\/index\/introducing-vision-to-the-fine-tuning-api\/"},{"key":"e_1_3_3_3_42_2","volume-title":"OpenAI GPT-4 API Documentation","year":"2024","unstructured":"OpenAI. 2024. OpenAI GPT-4 API Documentation. https:\/\/platform.openai.com\/docs\/models\/gpt-4o"},{"key":"e_1_3_3_3_43_2","unstructured":"OpenAI. 2024. Prompt engineering. https:\/\/platform.openai.com\/docs\/guides\/prompt-engineering"},{"key":"e_1_3_3_3_44_2","unstructured":"Seokhyeon Park Wonjae Kim Young-Ho Kim and Jinwook Seo. 2023. Computational Approaches for App-to-App Retrieval and Design Consistency Check. arxiv:https:\/\/arXiv.org\/abs\/2309.10328\u00a0[cs.HC]"},{"key":"e_1_3_3_3_45_2","series-title":"Proceedings of Machine Learning Research","first-page":"8748","volume-title":"Proceedings of the 38th International Conference on Machine Learning","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0139), Marina Meila and Tong Zhang (Eds.). PMLR, 8748\u20138763. https:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"e_1_3_3_3_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/2047196.2047216"},{"key":"e_1_3_3_3_47_2","volume-title":"Interaction Design: Beyond Human-Computer Interaction, 6th Edition","author":"Rogers Yvonne","year":"2023","unstructured":"Yvonne Rogers, Helen Sharp, and Jennifer Preece. 2023. Interaction Design: Beyond Human-Computer Interaction, 6th Edition. Wiley. https:\/\/www.wiley.com\/en-us\/Interaction+Design%3A+Beyond+Human+Computer+Interaction%2C+6th+Edition-p-00381113"},{"key":"e_1_3_3_3_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/1518701.1519064"},{"key":"e_1_3_3_3_49_2","doi-asserted-by":"publisher","DOI":"10.1145\/2757226.2757230"},{"key":"e_1_3_3_3_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3580895"},{"key":"e_1_3_3_3_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3472749.3474765"},{"key":"e_1_3_3_3_52_2","doi-asserted-by":"publisher","unstructured":"Jinge Wang Qing Ye Li Liu Nancy\u00a0Lan Guo and Gangqing Hu. 2024. Scientific figures interpreted by ChatGPT: strengths in plot recognition and limits in color perception. NPJ Precision Oncology 8 1 (2024) 84. 10.1038\/s41698-024-00576-z","DOI":"10.1038\/s41698-024-00576-z"},{"key":"e_1_3_3_3_53_2","doi-asserted-by":"publisher","unstructured":"Jialiang Wei Anne-Lise Courbis Thomas Lambolais Binbin Xu Pierre\u00a0Louis Bernard G\u00e9rard Dray and Walid Maalej. 2024. GUing: A Mobile GUI Search Engine using a Vision-Language Model. ACM Trans. Softw. Eng. Methodol. (Nov. 2024). 10.1145\/3702993Just Accepted.","DOI":"10.1145\/3702993"},{"key":"e_1_3_3_3_54_2","doi-asserted-by":"publisher","DOI":"10.1109\/BigData59044.2023.10386743"},{"key":"e_1_3_3_3_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676408"},{"key":"e_1_3_3_3_56_2","doi-asserted-by":"publisher","DOI":"10.1145\/3640543.3645176"},{"key":"e_1_3_3_3_57_2","doi-asserted-by":"publisher","DOI":"10.1145\/3447526.3472048"},{"key":"e_1_3_3_3_58_2","unstructured":"Benfeng Xu An Yang Junyang Lin Quan Wang Chang Zhou Yongdong Zhang and Zhendong Mao. 2023. ExpertPrompting: Instructing Large Language Models to be Distinguished Experts. arxiv:https:\/\/arXiv.org\/abs\/2305.14688\u00a0[cs.CL]"},{"key":"e_1_3_3_3_59_2","doi-asserted-by":"publisher","unstructured":"Shukang Yin Chaoyou Fu Sirui Zhao Ke Li Xing Sun Tong Xu and Enhong Chen. 2024. A Survey on Multimodal Large Language Models. National Science Review (11 2024) nwae403. 10.1093\/nsr\/nwae403","DOI":"10.1093\/nsr\/nwae403"},{"key":"e_1_3_3_3_60_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73039-9_14"},{"key":"e_1_3_3_3_61_2","unstructured":"Chi Zhang Zhao Yang Jiaxuan Liu Yucheng Han Xin Chen Zebiao Huang Bin Fu and Gang Yu. 2023. AppAgent: Multimodal Agents as Smartphone Users. arxiv:https:\/\/arXiv.org\/abs\/2312.13771\u00a0[cs.CV]"}],"event":{"name":"CHI 2025: CHI Conference on Human Factors in Computing Systems","location":"Yokohama Japan","acronym":"CHI '25","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2025 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3706598.3714213","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3706598.3714213","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,4]],"date-time":"2025-07-04T05:16:22Z","timestamp":1751606182000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3706598.3714213"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,25]]},"references-count":60,"alternative-id":["10.1145\/3706598.3714213","10.1145\/3706598"],"URL":"https:\/\/doi.org\/10.1145\/3706598.3714213","relation":{},"subject":[],"published":{"date-parts":[[2025,4,25]]},"assertion":[{"value":"2025-04-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}