{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:48:23Z","timestamp":1782762503026,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":66,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T00:00:00Z","timestamp":1782345600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,25]]},"DOI":"10.1145\/3805689.3812353","type":"proceedings-article","created":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T16:20:39Z","timestamp":1782231639000},"page":"4733-4755","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Who Defines \u201cBest\u201d? Towards Interactive, User-Defined Evaluation of LLM Leaderboards"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-0010-1887","authenticated-orcid":false,"given":"Minji","family":"Jung","sequence":"first","affiliation":[{"name":"Yonsei University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0232-7486","authenticated-orcid":false,"given":"Minjae","family":"Lee","sequence":"additional","affiliation":[{"name":"Yonsei University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9173-4184","authenticated-orcid":false,"given":"Yejin","family":"Kim","sequence":"additional","affiliation":[{"name":"Yonsei University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8656-2069","authenticated-orcid":false,"given":"Sarang","family":"Choi","sequence":"additional","affiliation":[{"name":"Yonsei University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0291-6026","authenticated-orcid":false,"given":"Minsuk","family":"Kahng","sequence":"additional","affiliation":[{"name":"Yonsei University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,25]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1609\/aimag.v36i1.2564"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3461702.3462610"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3367"},{"key":"e_1_3_2_1_4_1","volume-title":"Conference on Fairness, Accountability and Transparency. PMLR, 77\u201391","author":"Buolamwini Joy","year":"2018","unstructured":"Joy Buolamwini and Timnit Gebru. 2018. Gender shades: Intersectional accuracy disparities in commercial gender classification. In Conference on Fairness, Accountability and Transparency. PMLR, 77\u201391."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/VAST47406.2019.8986948"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581268"},{"key":"e_1_3_2_1_7_1","volume-title":"Humans or llms as the judge? a study on judgement biases. arXiv preprint arXiv:2402.10669","author":"Chen Guiming Hardy","year":"2024","unstructured":"Guiming Hardy Chen, Shunian Chen, Ziche Liu, Feng Jiang, and Benyou Wang. 2024. Humans or llms as the judge? a study on judgement biases. arXiv preprint arXiv:2402.10669 (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"Forty-first International Conference on Machine Learning (ICML).","author":"Chiang Wei-Lin","year":"2024","unstructured":"Wei-Lin Chiang, Lianmin Zheng, Ying Sheng, Anastasios Nikolas Angelopoulos, Tianle Li, Dacheng Li, Banghua Zhu, Hao Zhang, Michael Jordan, Joseph E Gonzalez, et al. 2024. Chatbot arena: An open platform for evaluating llms by human preference. In Forty-first International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_9_1","first-page":"1","article-title":"Scaling instruction-finetuned language models","volume":"25","author":"Chung Hyung Won","year":"2024","unstructured":"Hyung Won Chung, Le Hou, Shayne Longpre, Barret Zoph, Yi Tay, William Fedus, Yunxuan Li, Xuezhi Wang, Mostafa Dehghani, Siddhartha Brahma, et al. 2024. Scaling instruction-finetuned language models. Journal of Machine Learning Research 25, 70 (2024), 1\u201353.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3531146.3533108"},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the AAAI\/ACM Conference on AI, Ethics, and Society","volume":"7","author":"Diaz Fernando","year":"2024","unstructured":"Fernando Diaz and Michael Madaio. 2024. Scaling laws do not scale. In Proceedings of the AAAI\/ACM Conference on AI, Ethics, and Society, Vol. 7. 341\u2013357."},{"key":"e_1_3_2_1_12_1","volume-title":"Length-controlled alpacaeval: A simple way to debias automatic evaluators. arXiv preprint arXiv:2404.04475","author":"Dubois Yann","year":"2024","unstructured":"Yann Dubois, Bal\u00e1zs Galambosi, Percy Liang, and Tatsunori B Hashimoto. 2024. Length-controlled alpacaeval: A simple way to debias automatic evaluators. arXiv preprint arXiv:2404.04475 (2024)."},{"key":"e_1_3_2_1_13_1","volume-title":"The Thirteenth International Conference on Learning Representation (ICLR)","author":"Feuer Benjamin","year":"2024","unstructured":"Benjamin Feuer, Micah Goldblum, Teresa Datta, Sanjana Nambiar, Raz Besaleli, Samuel Dooley, Max Cembalest, and John P Dickerson. 2024. Style outweighs substance: Failure modes of llm judges in alignment benchmarking. The Thirteenth International Conference on Learning Representation (ICLR) (2024)."},{"key":"e_1_3_2_1_14_1","unstructured":"Joseph L Fleiss Bruce Levin and Myunghee Cho Paik. 2013. Statistical methods for rates and proportions. john wiley & sons."},{"key":"e_1_3_2_1_15_1","volume-title":"Prompt-to-Leaderboard: Prompt-Adaptive LLM Evaluations. In Forty-second International Conference on Machine Learning (ICML).","author":"Frick Evan","year":"2025","unstructured":"Evan Frick, Connor Chen, Joseph Tennyson, Tianle Li, Wei-Lin Chiang, Anastasios Nikolas Angelopoulos, and Ion Stoica. 2025. Prompt-to-Leaderboard: Prompt-Adaptive LLM Evaluations. In Forty-second International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3502004"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2013.173"},{"key":"e_1_3_2_1_18_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Hendrycks Dan","year":"2021","unstructured":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, and Jacob Steinhardt. 2021. Measuring massive multitask language understanding. International Conference on Learning Representations (ICLR) (2021)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3658908"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.1395"},{"key":"e_1_3_2_1_21_1","volume-title":"Forty-second International Conference on Machine Learning (ICML)","author":"Huang Yangsibo","year":"2025","unstructured":"Yangsibo Huang, Milad Nasr, Anastasios Angelopoulos, Nicholas Carlini, Wei-Lin Chiang, Christopher A Choquette-Choo, Daphne Ippolito, Matthew Jagielski, Katherine Lee, Ken Ziyu Liu, et al. 2025. Exploring and mitigating adversarial manipulation of voting-based leaderboards. Forty-second International Conference on Machine Learning (ICML) (2025)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2024.3456354"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0681"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642216"},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the Neural Information Processing Systems (NeurIPS) Track on Datasets and Benchmarks 1","author":"Koch Bernard","year":"2021","unstructured":"Bernard Koch, Emily Denton, Alex Hanna, and Jacob G Foster. 2021. Reduced, Reused and Recycled: The Life of a Dataset in Machine Learning Research. Proceedings of the Neural Information Processing Systems (NeurIPS) Track on Datasets and Benchmarks 1 (2021)."},{"key":"e_1_3_2_1_26_1","volume-title":"Content analysis: An introduction to its methodology","author":"Krippendorff Klaus","unstructured":"Klaus Krippendorff. 2018. Content analysis: An introduction to its methodology. Sage publications."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.643"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642830"},{"key":"e_1_3_2_1_29_1","volume-title":"Evaluating agents using social choice theory. arXiv preprint arXiv:2312.03121","author":"Lanctot Marc","year":"2023","unstructured":"Marc Lanctot, Kate Larson, Yoram Bachrach, Luke Marris, Zun Li, Avishkar Bhoopchand, Thomas Anthony, Brian Tanner, and Anna Koop. 2023. Evaluating agents using social choice theory. arXiv preprint arXiv:2312.03121 (2023)."},{"key":"e_1_3_2_1_30_1","unstructured":"Tianle Li Wei-Lin Chiang and Lisa Dunlap. 2024. Introducing Hard Prompts Category in Chatbot Arena. https:\/\/lmsys.org\/blog\/2024-05- 17-category-hard\/."},{"key":"e_1_3_2_1_31_1","volume-title":"Forty-second International Conference on Machine Learning (ICML)","author":"Li Tianle","year":"2024","unstructured":"Tianle Li, Wei-Lin Chiang, Evan Frick, Lisa Dunlap, Tianhao Wu, Banghua Zhu, Joseph E Gonzalez, and Ion Stoica. 2024. From crowdsourced data to high-quality benchmarks: Arena-hard and benchbuilder pipeline. Forty-second International Conference on Machine Learning (ICML) (2024)."},{"key":"e_1_3_2_1_32_1","unstructured":"Percy Liang Rishi Bommasani Tony Lee Dimitris Tsipras Dilara Soylu Michihiro Yasunaga Yian Zhang Deepak Narayanan Yuhuai Wu Ananya Kumar et al. 2022. Holistic evaluation of language models. Transactions on Machine Learning Research (2022)."},{"key":"e_1_3_2_1_33_1","first-page":"10351","article-title":"Dynaboard: An evaluation-as-a-service platform for holistic next-generation benchmarking","volume":"34","author":"Ma Zhiyi","year":"2021","unstructured":"Zhiyi Ma, Kawin Ethayarajh, Tristan Thrush, Somya Jain, Ledell Wu, Robin Jia, Christopher Potts, Adina Williams, and Douwe Kiela. 2021. Dynaboard: An evaluation-as-a-service platform for holistic next-generation benchmarking. Advances in Neural Information Processing Systems (NeurIPS) 34 (2021), 10351\u201310367.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512899"},{"key":"e_1_3_2_1_35_1","volume-title":"Forty-second International Conference on Machine Learning (ICML)","author":"Min Rui","year":"2025","unstructured":"Rui Min, Tianyu Pang, Chao Du, Qian Liu, Minhao Cheng, and Min Lin. 2025. Improving your model ranking on chatbot arena by vote rigging. Forty-second International Conference on Machine Learning (ICML) (2025)."},{"key":"e_1_3_2_1_36_1","volume-title":"Machine learning: a probabilistic perspective","author":"Murphy Kevin P","unstructured":"Kevin P Murphy. 2012. Machine learning: a probabilistic perspective. MIT press."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCG.2006.70"},{"key":"e_1_3_2_1_38_1","volume-title":"Dissecting racial bias in an algorithm used to manage the health of populations. Science 366, 6464","author":"Obermeyer Ziad","year":"2019","unstructured":"Ziad Obermeyer, Brian Powers, Christine Vogeli, and Sendhil Mullainathan. 2019. Dissecting racial bias in an algorithm used to manage the health of populations. Science 366, 6464 (2019), 447\u2013453."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3659012"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2016.2598589"},{"key":"e_1_3_2_1_41_1","unstructured":"Stephen R Pfohl Natalie Harris Chirag Nagpal David Madras Vishwali Mhasawade Olawale Salaudeen Awa Dieng Shannon Sequeira Santiago Arciniegas Lillian Sung et al. 2025. Understanding challenges to the interpretation of disaggregated evaluations of algorithmic fairness. arXiv preprint arXiv:2506.04193 (2025)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.164"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i01.5385"},{"key":"e_1_3_2_1_44_1","volume-title":"Proceedings of the Neural Information Processing Systems (NeurIPS) Track on Datasets and Benchmarks","author":"Raji Inioluwa Deborah","year":"2021","unstructured":"Inioluwa Deborah Raji, Emily M Bender, Amandalynne Paullada, Emily Denton, and Alex Hanna. 2021. AI and the everything in the whole wide world benchmark. Proceedings of the Neural Information Processing Systems (NeurIPS) Track on Datasets and Benchmarks (2021)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.442"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.eacl-main.48"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1057\/palgrave.ivs.9500091"},{"key":"e_1_3_2_1_48_1","volume-title":"Irene Y Chen, and Marzyeh Ghassemi.","author":"Seyyed-Kalantari Laleh","year":"2021","unstructured":"Laleh Seyyed-Kalantari, Haoran Zhang, Matthew BA McDermott, Irene Y Chen, and Marzyeh Ghassemi. 2021. Underdiagnosis bias of artificial intelligence algorithms applied to chest radiographs in under-served patient populations. Nature medicine 27, 12 (2021), 2176\u20132182."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676450"},{"key":"e_1_3_2_1_50_1","unstructured":"Toby Shevlane Sebastian Farquhar Ben Garfinkel Mary Phuong Jess Whittlestone Jade Leung Daniel Kokotajlo Nahema Marchal Markus Anderljung Noam Kolt et al. 2023. Model evaluation for extreme risks. arXiv preprint arXiv:2305.15324 (2023)."},{"key":"e_1_3_2_1_51_1","volume-title":"Annual Conference on Neural Information Processing Systems (NeurIPS).","author":"Singh Shivalika","year":"2025","unstructured":"Shivalika Singh, Yiyang Nan, Alex Wang, Daniel D'Souza, Sayash Kapoor, Ahmet \u00dcst\u00fcn, Sanmi Koyejo, Yuntian Deng, Shayne Longpre, Noah A Smith, Beyza Ermis, Marzieh Fadaee, and Sara Hooker. 2025. The leaderboard illusion. In Annual Conference on Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713103"},{"key":"e_1_3_2_1_53_1","volume-title":"Abubakar Abid, Adam Fisch, Adam R Brown, Adam Santoro, Aditya Gupta, Adri\u00e0 Garriga-Alonso, et al.","author":"Srivastava Aarohi","year":"2023","unstructured":"Aarohi Srivastava, Abhinav Rastogi, Abhishek Rao, Abu Awal Md Shoeb, Abubakar Abid, Adam Fisch, Adam R Brown, Adam Santoro, Aditya Gupta, Adri\u00e0 Garriga-Alonso, et al. 2023. Beyond the imitation game: Quantifying and extrapolating the capabilities of language models. Transactions on Machine Learning Research (2023)."},{"key":"e_1_3_2_1_54_1","volume-title":"Aakanksha Chowdhery, Quoc Le, Ed Chi, Denny Zhou, et al.","author":"Suzgun Mirac","year":"2023","unstructured":"Mirac Suzgun, Nathan Scales, Nathanael Sch\u00e4rli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc Le, Ed Chi, Denny Zhou, et al. 2023. Challenging big-bench tasks and whether chain-of-thought can solve them. In Findings of the Association for Computational Linguistics (ACL). 13003\u201313051."},{"key":"e_1_3_2_1_55_1","volume-title":"Clio: Privacy-preserving insights into real-world ai use. arXiv preprint arXiv:2412.13678","author":"Tamkin Alex","year":"2024","unstructured":"Alex Tamkin, Miles McCain, Kunal Handa, Esin Durmus, Liane Lovitt, Ankur Rathi, Saffron Huang, Alfred Mountfield, Jerry Hong, Stuart Ritchie, et al. 2024. Clio: Privacy-preserving insights into real-world ai use. arXiv preprint arXiv:2412.13678 (2024)."},{"key":"e_1_3_2_1_56_1","volume-title":"Drawing Conclusions from Draws: Rethinking Preference Semantics in Arena-Style LLM Evaluation. arXiv preprint arXiv:2510.02306","author":"Tang Raphael","year":"2025","unstructured":"Raphael Tang, Crystina Zhang, Wenyan Li, Carmen Lai, Pontus Stenetorp, and Yao Lu. 2025. Drawing Conclusions from Draws: Rethinking Preference Semantics in Arena-Style LLM Evaluation. arXiv preprint arXiv:2510.02306 (2025)."},{"key":"e_1_3_2_1_57_1","volume-title":"Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/.","author":"Team Vicuna","year":"2023","unstructured":"Vicuna Team. 2023. Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2017.2745078"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2024.3357065"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.657"},{"key":"e_1_3_2_1_61_1","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics. 297\u2013312","author":"Wu Minghao","year":"2025","unstructured":"Minghao Wu and Alham Fikri Aji. 2025. Style over substance: Evaluation biases for large language models. In Proceedings of the 31st International Conference on Computational Linguistics. 297\u2013312."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1073"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.5555\/3692070.3694505"},{"key":"e_1_3_2_1_64_1","first-page":"842","article-title":"SliceTeller: A data slice-driven approach for machine learning model validation","volume":"29","author":"Zhang Xiaoyu","year":"2022","unstructured":"Xiaoyu Zhang, Jorge Piazentin Ono, Huan Song, Liang Gou, Kwan-Liu Ma, and Liu Ren. 2022. SliceTeller: A data slice-driven approach for machine learning model validation. IEEE Transactions on Visualization and Computer Graphics (VIS) 29, 1 (2022), 842\u2013852.","journal-title":"IEEE Transactions on Visualization and Computer Graphics (VIS)"},{"key":"e_1_3_2_1_65_1","volume-title":"The Twelfth International Conference on Learning Representations (ICLR).","author":"Zhao Wenting","year":"2024","unstructured":"Wenting Zhao, Xiang Ren, Jack Hessel, Claire Cardie, Yejin Choi, and Yuntian Deng. 2024. WildChat: 1M ChatGPT interaction logs in the wild. In The Twelfth International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2020"}],"event":{"name":"FAccT '26: The 2026 ACM Conference on Fairness, Accountability, and Transparency","location":"Montreal QC Canada","acronym":"FAccT '26","sponsor":["ACM\/SIG"]},"container-title":["Proceedings of the 2026 ACM Conference on Fairness, Accountability, and Transparency"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3805689.3812353","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:06:36Z","timestamp":1782759996000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805689.3812353"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,25]]},"references-count":66,"alternative-id":["10.1145\/3805689.3812353","10.1145\/3805689"],"URL":"https:\/\/doi.org\/10.1145\/3805689.3812353","relation":{},"subject":[],"published":{"date-parts":[[2026,6,25]]},"assertion":[{"value":"2026-06-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}