{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:59:20Z","timestamp":1776931160599,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":67,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3772318.3791222","type":"proceedings-article","created":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T05:14:30Z","timestamp":1776057270000},"page":"1-17","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Data-Prompt Co-Evolution: Growing Test Sets to Refine LLM Behavior"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-0232-7486","authenticated-orcid":false,"given":"Minjae","family":"Lee","sequence":"first","affiliation":[{"name":"Department of Computer Science and Engineering, Yonsei University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0291-6026","authenticated-orcid":false,"given":"Minsuk","family":"Kahng","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, Yonsei University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,4,13]]},"reference":[{"key":"e_1_3_3_3_2_2","doi-asserted-by":"publisher","unstructured":"Saleema Amershi Maya Cakmak William\u00a0Bradley Knox and Todd Kulesza. 2014. Power to the people: The role of humans in interactive machine learning. AI Magazine 35 4 (2014) 105\u2013120. 10.1609\/aimag.v35i4.2513","DOI":"10.1609\/aimag.v35i4.2513"},{"key":"e_1_3_3_3_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642016"},{"key":"e_1_3_3_3_4_2","unstructured":"Yuntao Bai Saurav Kadavath Sandipan Kundu Amanda Askell Jackson Kernion Andy Jones Anna Chen Anna Goldie Azalia Mirhoseini Cameron McKinnon et\u00a0al. 2022. Constitutional AI: Harmlessness from AI feedback. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2212.08073 (2022). https:\/\/arxiv.org\/abs\/2212.08073"},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581268"},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"publisher","unstructured":"\u00c1ngel\u00a0Alexander Cabrera Adam Perer and Jason\u00a0I Hong. 2023. Improving human-AI collaboration with descriptions of AI behavior. Proceedings of the ACM on Human-Computer Interaction 7 CSCW1 Article 136 (2023) 21\u00a0pages. 10.1145\/3579612","DOI":"10.1145\/3579612"},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"publisher","unstructured":"\u00c1ngel\u00a0Alexander Cabrera Marco Tulio\u00a0Ribeiro Bongshin Lee Robert Deline Adam Perer and Steven\u00a0M. Drucker. 2023. What did my AI learn? How data scientists make sense of model behavior. ACM Transactions on Computer-Human Interaction 30 1 Article 1 (2023) 27\u00a0pages. 10.1145\/3542921","DOI":"10.1145\/3542921"},{"key":"e_1_3_3_3_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3025453.3026044"},{"key":"e_1_3_3_3_9_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.758"},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.176"},{"key":"e_1_3_3_3_11_2","doi-asserted-by":"publisher","DOI":"10.5555\/3294996.3295184"},{"key":"e_1_3_3_3_12_2","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et\u00a0al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.06261 (2025). https:\/\/arxiv.org\/abs\/2507.06261"},{"key":"e_1_3_3_3_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445775"},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"publisher","unstructured":"Dazhen Deng Chuhan Zhang Huawei Zheng Yuwen Pu Shouling Ji and Yingcai Wu. 2025. AdversaFlow: Visual red teaming for large language models with multi-level adversarial flow. IEEE Transactions on Visualization and Computer Graphics (VIS) 31 1 (2025) 492\u2013502. 10.1109\/TVCG.2024.3456150","DOI":"10.1109\/TVCG.2024.3456150"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"publisher","unstructured":"John\u00a0J Dudley and Per\u00a0Ola Kristensson. 2018. A review of user interface design for interactive machine learning. ACM Transactions on Interactive Intelligent Systems (TiiS) 8 2 Article 8 (2018) 37\u00a0pages. 10.1145\/3185517","DOI":"10.1145\/3185517"},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"publisher","DOI":"10.1609\/aies.v7i1.31647"},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581352"},{"key":"e_1_3_3_3_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3714319"},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376177"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300830"},{"key":"e_1_3_3_3_21_2","volume-title":"International Conference on Learning Representations (ICLR)","author":"Hu Edward\u00a0J","year":"2022","unstructured":"Edward\u00a0J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-rank adaptation of large language models. In International Conference on Learning Representations (ICLR). https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"publisher","unstructured":"Minsuk Kahng Ian Tenney Mahima Pushkarna Michael\u00a0Xieyang Liu James Wexler Emily Reif Krystal Kallarackal Minsuk Chang Michael Terry and Lucas Dixon. 2025. LLM Comparator: Interactive analysis of side-by-side evaluation of large language models. IEEE Transactions on Visualization and Computer Graphics (VIS) 31 1 (2025) 503\u2013513. 10.1109\/TVCG.2024.3456354","DOI":"10.1109\/TVCG.2024.3456354"},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.324"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581001"},{"key":"e_1_3_3_3_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642216"},{"key":"e_1_3_3_3_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/2556288.2557238"},{"key":"e_1_3_3_3_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/2678025.2701399"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"publisher","unstructured":"Tom Kwiatkowski Jennimaria Palomaki Olivia Redfield Michael Collins Ankur Parikh Chris Alberti Danielle Epstein Illia Polosukhin Jacob Devlin Kenton Lee Kristina Toutanova Llion Jones Matthew Kelcey Ming-Wei Chang Andrew\u00a0M. Dai Jakob Uszkoreit Quoc Le and Slav Petrov. 2019. Natural Questions: A benchmark for question answering research. Transactions of the Association for Computational Linguistics 7 (2019) 452\u2013466. 10.1162\/tacl_a_00276","DOI":"10.1162\/tacl_a_00276"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"publisher","DOI":"10.1145\/3746059.3747680"},{"key":"e_1_3_3_3_30_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"e_1_3_3_3_31_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376590"},{"key":"e_1_3_3_3_33_2","doi-asserted-by":"publisher","unstructured":"Qianou Ma Weirui Peng Chenyang Yang Hua Shen Ken Koedinger and Tongshuang Wu. 2025. What should we engineer in prompts? Training humans in requirement-driven LLM use. ACM Transactions on Computer-Human Interaction 32 4 Article 41 (2025) 27\u00a0pages. 10.1145\/3731756","DOI":"10.1145\/3731756"},{"key":"e_1_3_3_3_34_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.441"},{"key":"e_1_3_3_3_35_2","doi-asserted-by":"publisher","DOI":"10.5555\/3600270.3602281"},{"key":"e_1_3_3_3_36_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.225"},{"key":"e_1_3_3_3_37_2","doi-asserted-by":"publisher","DOI":"10.1145\/3640543.3645144"},{"key":"e_1_3_3_3_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706599.3719677"},{"key":"e_1_3_3_3_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3658913"},{"key":"e_1_3_3_3_40_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-2124"},{"key":"e_1_3_3_3_41_2","doi-asserted-by":"publisher","unstructured":"Charvi Rastogi Tian\u00a0Huey Teh Pushkar Mishra Roma Patel Zoe Ashwood Aida\u00a0Mostafazadeh Davani Mark Diaz Michela Paganini Alicia Parrish Ding Wang Vinodkumar Prabhakaran Lora Aroyo and Verena Rieser. 2024. Insights on disagreement patterns in multimodal safety perception across diverse rater groups. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.17032 (2024). 10.48550\/arXiv.2410.17032","DOI":"10.48550\/arXiv.2410.17032"},{"key":"e_1_3_3_3_42_2","doi-asserted-by":"publisher","unstructured":"Alexander Ratner Stephen\u00a0H Bach Henry Ehrenberg Jason Fries Sen Wu and Christopher R\u00e9. 2017. Snorkel: Rapid training data creation with weak supervision. Proceedings of the VLDB endowment (VLDB) 11 3 (2017) 269\u2013282. 10.14778\/3157794.3157797","DOI":"10.14778\/3157794.3157797"},{"key":"e_1_3_3_3_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3714051"},{"key":"e_1_3_3_3_44_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.230"},{"key":"e_1_3_3_3_45_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.442"},{"key":"e_1_3_3_3_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445518"},{"key":"e_1_3_3_3_47_2","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676450"},{"key":"e_1_3_3_3_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706599.3716291"},{"key":"e_1_3_3_3_49_2","unstructured":"Patrice\u00a0Y. Simard Saleema Amershi David\u00a0M. Chickering Alicia\u00a0Edelman Pelton Soroush Ghorashi Christopher Meek Gonzalo Ramos Jina Suh Johan Verwey Mo Wang and John Wernsing. 2017. Machine Teaching: A new paradigm for building machine learning systems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1707.06742 (2017). https:\/\/arxiv.org\/abs\/1707.06742"},{"key":"e_1_3_3_3_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581075"},{"key":"e_1_3_3_3_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713103"},{"key":"e_1_3_3_3_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713664"},{"key":"e_1_3_3_3_53_2","doi-asserted-by":"publisher","unstructured":"Hendrik Strobelt Albert Webson Victor Sanh Benjamin Hoover Johanna Beyer Hanspeter Pfister and Alexander\u00a0M. Rush. 2023. Interactive and visual prompt engineering for ad-hoc task adaptation with large language models. IEEE Transactions on Visualization and Computer Graphics (VIS) 29 1 (2023) 1146\u20131156. 10.1109\/TVCG.2022.3209479","DOI":"10.1109\/TVCG.2022.3209479"},{"key":"e_1_3_3_3_54_2","doi-asserted-by":"publisher","DOI":"10.1145\/3461778.3462012"},{"key":"e_1_3_3_3_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713166"},{"key":"e_1_3_3_3_56_2","doi-asserted-by":"publisher","unstructured":"Helena Vasconcelos Matthew J\u00f6rke Madeleine Grunde-McLaughlin Tobias Gerstenberg Michael\u00a0S Bernstein and Ranjay Krishna. 2023. Explanations can reduce overreliance on AI systems during decision-making. Proceedings of the ACM on Human-Computer Interaction 7 CSCW1 (2023) 1\u201338. 10.1145\/3579605","DOI":"10.1145\/3579605"},{"key":"e_1_3_3_3_57_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3641960"},{"key":"e_1_3_3_3_58_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642335"},{"key":"e_1_3_3_3_59_2","doi-asserted-by":"publisher","unstructured":"Steven\u00a0Euijong Whang Yuji Roh Hwanjun Song and Jae-Gil Lee. 2023. Data collection and quality challenges in deep learning: A data-centric AI perspective. The VLDB Journal 32 4 (2023) 791\u2013813. 10.1007\/s00778-022-00775-9","DOI":"10.1007\/s00778-022-00775-9"},{"key":"e_1_3_3_3_60_2","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3659002"},{"key":"e_1_3_3_3_61_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1073"},{"key":"e_1_3_3_3_62_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581641.3584059"},{"key":"e_1_3_3_3_63_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376451"},{"key":"e_1_3_3_3_64_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.1130"},{"key":"e_1_3_3_3_65_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713491"},{"key":"e_1_3_3_3_66_2","unstructured":"Yaoning Yu Ye Yu Kai Wei Haojing Luo and Haohan Wang. 2025. SIPDO: Closed-loop prompt optimization via synthetic data feedback. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.19514 (2025). https:\/\/arxiv.org\/abs\/2505.19514"},{"key":"e_1_3_3_3_67_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581388"},{"key":"e_1_3_3_3_68_2","doi-asserted-by":"publisher","DOI":"10.5555\/3666122.3668142"}],"event":{"name":"CHI 2026: CHI Conference on Human Factors in Computing Systems","location":"Barcelona Spain","acronym":"CHI '26","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3772318.3791222","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,21]],"date-time":"2026-04-21T20:11:51Z","timestamp":1776802311000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3772318.3791222"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,13]]},"references-count":67,"alternative-id":["10.1145\/3772318.3791222","10.1145\/3772318"],"URL":"https:\/\/doi.org\/10.1145\/3772318.3791222","relation":{},"subject":[],"published":{"date-parts":[[2026,4,13]]},"assertion":[{"value":"2026-04-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}