{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T21:17:52Z","timestamp":1783804672787,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T00:00:00Z","timestamp":1746662400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,8]]},"DOI":"10.1145\/3701716.3715265","type":"proceedings-article","created":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T16:12:56Z","timestamp":1748016776000},"page":"228-237","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Training an LLM-as-a-Judge Model: Pipeline, Insights, and Practical Lessons"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1094-6890","authenticated-orcid":false,"given":"Renjun","family":"Hu","sequence":"first","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3075-8413","authenticated-orcid":false,"given":"Yi","family":"Cheng","sequence":"additional","affiliation":[{"name":"Alibaba Cloud Computing, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5979-5234","authenticated-orcid":false,"given":"Libin","family":"Meng","sequence":"additional","affiliation":[{"name":"Alibaba Cloud Computing, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9682-7320","authenticated-orcid":false,"given":"Jiaxin","family":"Xia","sequence":"additional","affiliation":[{"name":"Alibaba Cloud Computing, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3093-5957","authenticated-orcid":false,"given":"Yi","family":"Zong","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-5946-3187","authenticated-orcid":false,"given":"Xing","family":"Shi","sequence":"additional","affiliation":[{"name":"Alibaba Cloud Computing, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3003-0150","authenticated-orcid":false,"given":"Wei","family":"Lin","sequence":"additional","affiliation":[{"name":"Alibaba Cloud Computing, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,5,23]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Introducing Meta Llama 3: The most capable openly available LLM to date. https:\/\/ai.meta.com\/blog\/meta-llama-3\/ Retrieved","author":"Meta AI.","year":"2024","unstructured":"Meta AI. 2024. Introducing Meta Llama 3: The most capable openly available LLM to date. https:\/\/ai.meta.com\/blog\/meta-llama-3\/ Retrieved July 23, 2024 from"},{"key":"e_1_3_2_2_2_1","unstructured":"Yuntao Bai Andy Jones Kamal Ndousse Amanda Askell Anna Chen Nova Dassarma Dawn Drain Stanislav Fort and et al. 2022. Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback. ArXiv Vol. abs\/2204.05862 (2022)."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Yoshua Bengio Geoffrey Hinton Andrew Yao Dawn Song Pieter Abbeel Trevor Darrell Yuval Noah Harari Ya-Qin Zhang Lan Xue Shai Shalev-Shwartz Gillian Hadfield Jeff Clune Tegan Maharaj Frank Hutter Sheila McIlraith Qiqi Gao Ashwin Acharya David Krueger Anca Dragan Philip Torr Stuart Russell Daniel Kahneman Jan Brauner and S\u00f6ren Mindermann. 2024. Managing extreme AI risks amid rapid progress. Science 384 6698 (2024) 842-845.","DOI":"10.1126\/science.adn0117"},{"key":"e_1_3_2_2_4_1","unstructured":"Pablo Biedma Xiaoyuan Yi Linus Huang Maosong Sun and Xing Xie. 2024. Beyond Human Norms: Unveiling Unique Values of Large Language Models through Interdisciplinary Approaches. arxiv: 2404.12744 [cs.CL]"},{"key":"e_1_3_2_2_5_1","unstructured":"Sebastian Bordt Harsha Nori and Rich Caruana. 2024. Elephants Never Forget: Testing Language Models for Memorization of Tabular Data. arxiv: 2403.06644 [cs.LG]"},{"key":"e_1_3_2_2_6_1","volume-title":"How Do Large Language Models Acquire Factual Knowledge During Pretraining? CoRR","author":"Chang Hoyeon","year":"1813","unstructured":"Hoyeon Chang, Jinho Park, Seonghyeon Ye, Sohee Yang, Youngkyung Seo, Du-Seong Chang, and Minjoon Seo. 2024. How Do Large Language Models Acquire Factual Knowledge During Pretraining? CoRR, Vol. abs\/2406.11813 (2024)."},{"key":"e_1_3_2_2_7_1","volume-title":"Jared Kaplan, Harri Edwards, Yuri Burda, and et al.","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, and et al. 2021. Evaluating Large Language Models Trained on Code. (2021). arxiv: 2107.03374 [cs.LG]"},{"key":"e_1_3_2_2_8_1","volume-title":"Tianle Li, Dacheng Li, Hao Zhang, Banghua Zhu, Michael Jordan, Joseph E. Gonzalez, and Ion Stoica.","author":"Chiang Wei-Lin","year":"2024","unstructured":"Wei-Lin Chiang, Lianmin Zheng, Ying Sheng, Anastasios Nikolas Angelopoulos, Tianle Li, Dacheng Li, Hao Zhang, Banghua Zhu, Michael Jordan, Joseph E. Gonzalez, and Ion Stoica. 2024. Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference. arxiv: 2403.04132 [cs.AI]"},{"key":"e_1_3_2_2_9_1","unstructured":"Guanting Dong Hongyi Yuan Keming Lu Chengpeng Li Mingfeng Xue Dayiheng Liu Wei Wang Zheng Yuan Chang Zhou and Jingren Zhou. 2024. How Abilities in Large Language Models are Affected by Supervised Fine-tuning Data Composition. arxiv: 2310.05492 [cs.CL]"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"Jessica Echterhoff Yao Liu Abeer Alessa Julian McAuley and Zexue He. 2024. Cognitive Bias in High-Stakes Decision-Making with LLMs. arxiv: 2403.00811 [cs.AI]","DOI":"10.18653\/v1\/2024.findings-emnlp.739"},{"key":"e_1_3_2_2_11_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR)","author":"Hendrycks Dan","year":"2021","unstructured":"Dan Hendrycks, Collin Burns, Steven Basart, Andrew Critch, Jerry Li, Dawn Song, and Jacob Steinhardt. 2021a. Aligning AI With Shared Human Values. Proceedings of the International Conference on Learning Representations (ICLR) (2021)."},{"key":"e_1_3_2_2_12_1","volume-title":"Measuring Massive Multitask Language Understanding. arxiv","author":"Hendrycks Dan","year":"2009","unstructured":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, and Jacob Steinhardt. 2021b. Measuring Massive Multitask Language Understanding. arxiv: 2009.03300 [cs.CY]"},{"key":"e_1_3_2_2_13_1","volume-title":"Measuring Mathematical Problem Solving With the MATH Dataset. In Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2).","author":"Hendrycks Dan","year":"2021","unstructured":"Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, and Jacob Steinhardt. 2021c. Measuring Mathematical Problem Solving With the MATH Dataset. In Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2)."},{"key":"e_1_3_2_2_14_1","unstructured":"Renjun Hu Yi Cheng Libin Meng Jiaxin Xia Yi Zong Xing Shi and Wei Lin. 2025. Training an LLM-as-a-Judge Model: Pipeline Insights and Practical Lessons. arxiv: 2502.02988 [cs.CL]"},{"key":"e_1_3_2_2_15_1","unstructured":"Yuzhen Huang Yuzhuo Bai Zhihao Zhu Junlei Zhang Jinghan Zhang Tangjun Su Junteng Liu Chuancheng Lv Yikai Zhang Jiayi Lei Yao Fu Maosong Sun and Junxian He. 2023. C-Eval: A Multi-Level Multi-Discipline Chinese Evaluation Suite for Foundation Models. arxiv: 2305.08322 [cs.CL]"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.2016299"},{"key":"e_1_3_2_2_17_1","unstructured":"Han Jiang Xiaoyuan Yi Zhihua Wei Shu Wang and Xing Xie. 2024. Raising the Bar: Investigating the Values of Large Language Models via Generative Evolving Testing. arxiv: 2406.14230 [cs.CL]"},{"key":"e_1_3_2_2_18_1","volume-title":"Jenny Liang, Jesse Dodge, Keisuke Sakaguchi, Maxwell Forbes, Jon Borchardt, Saadia Gabriel, Yulia Tsvetkov, Oren Etzioni, Maarten Sap, Regina Rini, and Yejin Choi.","author":"Jiang Liwei","year":"2022","unstructured":"Liwei Jiang, Jena D. Hwang, Chandra Bhagavatula, Ronan Le Bras, Jenny Liang, Jesse Dodge, Keisuke Sakaguchi, Maxwell Forbes, Jon Borchardt, Saadia Gabriel, Yulia Tsvetkov, Oren Etzioni, Maarten Sap, Regina Rini, and Yejin Choi. 2022. Can Machines Learn Morality? The Delphi Experiment. arxiv: 2110.07574 [cs.CL]"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","unstructured":"Pei Ke Bosi Wen Zhuoer Feng Xiao Liu Xuanyu Lei Jiale Cheng Shengyuan Wang Aohan Zeng Yuxiao Dong Hongning Wang Jie Tang and Minlie Huang. 2024. CritiqueLLM: Towards an Informative Critique Generation Model for Evaluation of Large Language Model Generation. arxiv: 2311.18702 [cs.CL]","DOI":"10.18653\/v1\/2024.acl-long.704"},{"key":"e_1_3_2_2_20_1","volume-title":"PertEval: Unveiling Real Knowledge Capacity of LLMs with Knowledge-Invariant Perturbations. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track.","author":"Li Jiatong","year":"2024","unstructured":"Jiatong Li, Renjun Hu, Kunzhe Huang, Yan Zhuang, Qi Liu, Mengxiao Zhu, Xing Shi, and Wei Lin. 2024b. PertEval: Unveiling Real Knowledge Capacity of LLMs with Knowledge-Invariant Perturbations. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_2_21_1","volume-title":"Generative Judge for Evaluating Alignment. arXiv preprint arXiv:2310.05470","author":"Li Junlong","year":"2023","unstructured":"Junlong Li, Shichao Sun, Weizhe Yuan, Run-Ze Fan, Hai Zhao, and Pengfei Liu. 2023. Generative Judge for Evaluating Alignment. arXiv preprint arXiv:2310.05470 (2023)."},{"key":"e_1_3_2_2_22_1","unstructured":"Ming Li Yong Zhang Zhitao Li Jiuhai Chen Lichang Chen Ning Cheng Jianzong Wang Tianyi Zhou and Jing Xiao. 2024c. From Quantity to Quality: Boosting LLM Performance with Self-Guided Data Selection for Instruction Tuning. arxiv: 2308.12032 [cs.CL]"},{"key":"e_1_3_2_2_23_1","unstructured":"Tianle Li Wei-Lin Chiang Evan Frick Lisa Dunlap Tianhao Wu Banghua Zhu Joseph E. Gonzalez and Ion Stoica. 2024a. From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline. arxiv: 2406.11939 [cs.LG]"},{"key":"e_1_3_2_2_24_1","volume-title":"Ronan Le Bras, and Yejin Choi","author":"Lin Bill Yuchen","year":"2024","unstructured":"Bill Yuchen Lin, Yuntian Deng, Khyathi Chandu, Faeze Brahman, Abhilasha Ravichander, Valentina Pyatkin, Nouha Dziri, Ronan Le Bras, and Yejin Choi. 2024a. WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild. arxiv: 2406.04770 [cs.CL]"},{"key":"e_1_3_2_2_25_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Lin Bill Yuchen","year":"2024","unstructured":"Bill Yuchen Lin, Abhilasha Ravichander, Ximing Lu, Nouha Dziri, Melanie Sclar, Khyathi Chandu, Chandra Bhagavatula, and Yejin Choi. 2024b. The Unlocking Spell on Base LLMs: Rethinking Alignment via In-Context Learning. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_2_26_1","volume-title":"ROUGE: A Package for Automatic Evaluation of Summaries. In Annual Meeting of the Association for Computational Linguistics.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. ROUGE: A Package for Automatic Evaluation of Summaries. In Annual Meeting of the Association for Computational Linguistics."},{"key":"e_1_3_2_2_27_1","volume-title":"Xiaohan Zhang, Lichao Sun, Hongning Wang, Jing Zhang, Minlie Huang, Yuxiao Dong, and Jie Tang.","author":"Liu Xiao","year":"2023","unstructured":"Xiao Liu, Xuanyu Lei, Shengyuan Wang, Yue Huang, Zhuoer Feng, Bosi Wen, Jiale Cheng, Pei Ke, Yifan Xu, Weng Lam Tam, Xiaohan Zhang, Lichao Sun, Hongning Wang, Jing Zhang, Minlie Huang, Yuxiao Dong, and Jie Tang. 2023. AlignBench: Benchmarking Chinese Alignment of Large Language Models. arxiv: 2311.18743 [cs.CL]"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"crossref","unstructured":"Seyed Mahed Mousavi Simone Alghisi and Giuseppe Riccardi. 2024. DyKnow:Dynamically Verifying Time-Sensitive Factual Knowledge in LLMs. arxiv: 2404.08700 [cs.CL]","DOI":"10.18653\/v1\/2024.findings-emnlp.471"},{"key":"e_1_3_2_2_29_1","volume-title":"Orca: Progressive Learning from Complex Explanation Traces of GPT-4. arxiv: 2306.02707 [cs.CL]","author":"Mukherjee Subhabrata","year":"2023","unstructured":"Subhabrata Mukherjee, Arindam Mitra, Ganesh Jawahar, Sahaj Agarwal, Hamid Palangi, and Ahmed Awadallah. 2023. Orca: Progressive Learning from Complex Explanation Traces of GPT-4. arxiv: 2306.02707 [cs.CL]"},{"key":"e_1_3_2_2_30_1","unstructured":"OpenAI. 2024. GPT-4 Technical Report. arxiv: 2303.08774 [cs.CL]"},{"key":"e_1_3_2_2_31_1","unstructured":"Long Ouyang Jeff Wu Xu Jiang Diogo Almeida Carroll L. Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray John Schulman Jacob Hilton Fraser Kelton Luke Miller Maddie Simens Amanda Askell Peter Welinder Paul Christiano Jan Leike and Ryan Lowe. 2022. Training language models to follow instructions with human feedback. arxiv: 2203.02155 [cs.CL]"},{"key":"e_1_3_2_2_32_1","volume-title":"Fine-Tuning or Retrieval? Comparing Knowledge Injection in LLMs. CoRR","author":"Ovadia Oded","year":"2023","unstructured":"Oded Ovadia, Menachem Brief, Moshik Mishaeli, and Oren Elisha. 2023. Fine-Tuning or Retrieval? Comparing Knowledge Injection in LLMs. CoRR, Vol. abs\/2312.05934 (2023)."},{"key":"e_1_3_2_2_33_1","volume-title":"Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, July 6--12","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a Method for Automatic Evaluation of Machine Translation. In Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, July 6--12, 2002, Philadelphia, PA, USA. ACL, 311--318."},{"key":"e_1_3_2_2_34_1","volume-title":"Mustafa Mamon Ahmed, William R. Hogan, Elizabeth A. Shenkman, Yi Guo, Jiang Bian, and Yonghui Wu.","author":"Peng C.A.I.","year":"2023","unstructured":"C.A.I. Peng, Xi Yang, Aokun Chen, Kaleb E. Smith, Nima M. Pournejatian, Anthony B Costa, Cheryl Martin, Mona G. Flores, Ying Zhang, Tanja Magoc, Gloria P. Lipori, Duane A. Mitchell, Naykky Singh Ospina, Mustafa Mamon Ahmed, William R. Hogan, Elizabeth A. Shenkman, Yi Guo, Jiang Bian, and Yonghui Wu. 2023. A study of generative large language model for medical research and healthcare. NPJ Digital Medicine, Vol. 6 (2023)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"e_1_3_2_2_36_1","volume-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023","author":"Schaeffer Rylan","year":"2023","unstructured":"Rylan Schaeffer, Brando Miranda, and Sanmi Koyejo. 2023. Are Emergent Abilities of Large Language Models a Mirage?. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023."},{"key":"e_1_3_2_2_37_1","volume-title":"Evaluating the Moral Beliefs Encoded in LLMs. In Thirty-seventh Conference on Neural Information Processing Systems.","author":"Scherrer Nino","year":"2023","unstructured":"Nino Scherrer, Claudia Shi, Amir Feder, and David Blei. 2023. Evaluating the Moral Beliefs Encoded in LLMs. In Thirty-seventh Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1093\/cid\/ciad633"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Yunfan Shao Linyang Li Zhaoye Fei Hang Yan Dahua Lin and Xipeng Qiu. 2024. Balanced Data Sampling for Language Model Training with Clustering. arxiv: 2402.14526 [cs.CL]","DOI":"10.18653\/v1\/2024.findings-acl.833"},{"key":"e_1_3_2_2_40_1","unstructured":"Lichao Sun Yue Huang Haoran Wang Siyuan Wu Qihui Zhang Yuan Li Chujie Gao Yixin Huang and et al. 2024. TrustLLM: Trustworthiness in Large Language Models. arxiv: 2401.05561 [cs.CL]"},{"key":"e_1_3_2_2_41_1","volume-title":"Hashimoto","author":"Taori Rohan","year":"2023","unstructured":"Rohan Taori, Ishaan Gulrajani, Tianyi Zhang, Yann Dubois, Xuechen Li, Carlos Guestrin, Percy Liang, and Tatsunori B. Hashimoto. 2023. Stanford Alpaca: An Instruction-following LLaMA model. https:\/\/github.com\/tatsu-lab\/stanford_alpaca."},{"key":"e_1_3_2_2_42_1","volume-title":"Gemini: A Family of Highly Capable Multimodal Models. arxiv: 2312.11805 [cs.CL]","author":"Team Gemini","year":"2024","unstructured":"Gemini Team. 2024a. Gemini: A Family of Highly Capable Multimodal Models. arxiv: 2312.11805 [cs.CL]"},{"key":"e_1_3_2_2_43_1","unstructured":"Qwen Team. 2024b. Qwen2 Technical Report. arxiv: 2407.10671 [cs.CL]"},{"key":"e_1_3_2_2_44_1","volume-title":"Foundational Autoraters: Taming Large Language Models for Better Automatic Evaluation. arxiv: 2407.10817 [cs.CL]","author":"Vu Tu","year":"2024","unstructured":"Tu Vu, Kalpesh Krishna, Salaheddin Alzubi, Chris Tar, Manaal Faruqui, and Yun-Hsuan Sung. 2024. Foundational Autoraters: Taming Large Language Models for Better Automatic Evaluation. arxiv: 2407.10817 [cs.CL]"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"Yizhong Wang Yeganeh Kordi Swaroop Mishra Alisa Liu Noah A. Smith Daniel Khashabi and Hannaneh Hajishirzi. 2023. Self-Instruct: Aligning Language Models with Self-Generated Instructions. arxiv: 2212.10560 [cs.CL]","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.165"},{"key":"e_1_3_2_2_47_1","unstructured":"Yu Yang Siddhartha Mishra Jeffrey N Chiang and Baharan Mirzasoleiman. 2024. SmallToLarge (S2L): Scalable Data Selection for Fine-tuning Large Language Models by Summarizing Training Trajectories of Small Models. arxiv: 2403.07384 [cs.CL]"},{"key":"e_1_3_2_2_48_1","unstructured":"Jiasheng Ye Peiju Liu Tianxiang Sun Yunhua Zhou Jun Zhan and Xipeng Qiu. 2024. Data Mixing Laws: Optimizing Data Mixtures by Predicting Language Modeling Performance. arxiv: 2403.16952 [cs.CL]"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1472"},{"key":"e_1_3_2_2_50_1","unstructured":"Yunpu Zhao Rui Zhang Wenyi Li Di Huang Jiaming Guo Shaohui Peng Yifan Hao Yuanbo Wen Xing Hu Zidong Du Qi Guo Ling Li and Yunji Chen. 2024. Assessing and Understanding Creativity in Large Language Models. arxiv: 2401.12491 [cs.CL]"},{"key":"e_1_3_2_2_51_1","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Tianle Li Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zhuohan Li Zi Lin Eric P. Xing Joseph E. Gonzalez Ion Stoica and Hao Zhang. 2024. LMSYS-Chat-1M: A Large-Scale Real-World LLM Conversation Dataset. arxiv: 2309.11998 [cs.CL]"},{"key":"e_1_3_2_2_52_1","volume-title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena. In Thirty-seventh Conference on Neural Information Processing Systems Datasets and Benchmarks Track.","author":"Zheng Lianmin","year":"2023","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi Lin, Zhuohan Li, Dacheng Li, Eric Xing, Hao Zhang, Joseph E. Gonzalez, and Ion Stoica. 2023. Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena. In Thirty-seventh Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"crossref","unstructured":"Wanjun Zhong Ruixiang Cui Yiduo Guo Yaobo Liang Shuai Lu Yanlin Wang Amin Saied Weizhu Chen and Nan Duan. 2023. AGIEval: A Human-Centric Benchmark for Evaluating Foundation Models. arxiv: 2304.06364 [cs.CL]","DOI":"10.18653\/v1\/2024.findings-naacl.149"},{"key":"e_1_3_2_2_54_1","volume-title":"Xu Chen, Yankai Lin, Ji-Rong Wen, and Jiawei Han.","author":"Zhou Kun","year":"2023","unstructured":"Kun Zhou, Yutao Zhu, Zhipeng Chen, Wentong Chen, Wayne Xin Zhao, Xu Chen, Yankai Lin, Ji-Rong Wen, and Jiawei Han. 2023. Don't Make Your LLM an Evaluation Benchmark Cheater. arxiv: 2311.01964 [cs.CL]"}],"event":{"name":"WWW '25: The ACM Web Conference 2025","location":"Sydney NSW Australia","acronym":"WWW '25","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Companion Proceedings of the ACM on Web Conference 2025"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3701716.3715265","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3701716.3715265","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,8]],"date-time":"2025-10-08T03:04:59Z","timestamp":1759892699000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3701716.3715265"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,8]]},"references-count":54,"alternative-id":["10.1145\/3701716.3715265","10.1145\/3701716"],"URL":"https:\/\/doi.org\/10.1145\/3701716.3715265","relation":{},"subject":[],"published":{"date-parts":[[2025,5,8]]},"assertion":[{"value":"2025-05-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}