{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T17:49:29Z","timestamp":1784915369977,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,2,22]]},"DOI":"10.1145\/3773966.3779397","type":"proceedings-article","created":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T17:50:01Z","timestamp":1771264201000},"page":"1130-1134","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["TRUE: A Reproducible Framework for LLM-Driven Relevance Judgment in Information Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-2350-7079","authenticated-orcid":false,"given":"Mouly","family":"Dewan","sequence":"first","affiliation":[{"name":"University of Washington, Seattle, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3643-2182","authenticated-orcid":false,"given":"Jiqun","family":"Liu","sequence":"additional","affiliation":[{"name":"The University of Oklahoma, Norman, OK, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3797-4293","authenticated-orcid":false,"given":"Chirag","family":"Shah","sequence":"additional","affiliation":[{"name":"University of Washington, Seattle, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,2,21]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Zahra Abbasiantaeb Chuan Meng Leif Azzopardi and Mohammad Aliannejadi. 2024. Can We Use Large Language Models to Fill Relevance Judgment Holes? http:\/\/arxiv.org\/abs\/2405.05600 arXiv:2405.05600."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3673791.3698431"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730305"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the ACM SIGIR 2008 Workshop on Beyond Binary Relevance: Preferences, Diversity, and Set-Level Judgments. http:\/\/research. microsoft.com\/ pauben\/bbr-workshop. Citeseer.","author":"Belkin Nicholas J","year":"2008","unstructured":"Nicholas J Belkin, Michael Cole, and Ralf Bierig. 2008. Is relevance the right criterion for evaluating interactive information retrieval. In Proceedings of the ACM SIGIR 2008 Workshop on Beyond Binary Relevance: Preferences, Diversity, and Set-Level Judgments. http:\/\/research. microsoft.com\/ pauben\/bbr-workshop. Citeseer."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3450127"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","unstructured":"Nuo Chen Jiqun Liu Xiaoyu Dong Qijiong Liu Tetsuya Sakai and Xiao-Ming Wu. 2024. AI Can Be Cognitively Biased: An Exploratory Study on Threshold Priming in LLM-Based Batch Relevance Assessment. https:\/\/doi.org\/10.48550\/arXiv.2409.16022 arXiv:2409.16022 [cs].","DOI":"10.48550\/arXiv.2409.16022"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1108\/eb050097"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1108\/eb049778"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.6028\/NIST.SP.1266.deep-overview"},{"key":"e_1_3_2_1_10_1","volume-title":"Overview of the TREC 2022 deep learning track. arXiv preprint arXiv:2507","author":"Craswell Nick","year":"2025","unstructured":"Nick Craswell, Bhaskar Mitra, Emine Yilmaz, Daniel Campos, Jimmy Lin, Ellen M Voorhees, and Ian Soboroff. 2025. Overview of the TREC 2022 deep learning track. arXiv preprint arXiv:2507.10865 (2025)."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval. 3055-3059","author":"Dewan Mouly","year":"2025","unstructured":"Mouly Dewan, Jiqun Liu, and Chirag Shah. 2025. LLM-driven usefulness labeling for IR evaluation. In Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval. 3055-3059."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731120.3744588"},{"key":"e_1_3_2_1_13_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Guglielmo Faggioli Laura Dietz Charles Clarke Gianluca Demartini Matthias Hagen Claudia Hauff Noriko Kando Evangelos Kanoulas Martin Potthast Benno Stein and Henning Wachsmuth. 2023. Perspectives on Large Language Models for Relevance Judgment. http:\/\/arxiv.org\/abs\/2304.09161 arXiv:2304.09161.","DOI":"10.1145\/3578337.3605136"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731120.3744591"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730317"},{"key":"e_1_3_2_1_17_1","unstructured":"Jiawei Gu Xuhui Jiang Zhichao Shi Hexiang Tan Xuehao Zhai Chengjin Xu Wei Li Yinghan Shen Shengjie Ma Honghao Liu et al. 2024. A Survey on LLM-as-a-Judge. arXiv preprint arXiv:2411.15594 (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1002\/asi.25020"},{"key":"e_1_3_2_1_19_1","unstructured":"Ying-Chun Lin Jennifer Neville Jack W Stokes Longqi Yang Tara Safavi Mengting Wan Scott Counts Siddharth Suri Reid Andersen Xiaofeng Xu et al. 2024. Interpretable user satisfaction estimation for conversational systems with large language models. arXiv preprint arXiv:2403.12388 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2022.103007"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343413.3377976"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3529372.3530926"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3592032"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080750"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10791-023-09426-1"},{"key":"e_1_3_2_1_26_1","volume-title":"Guglielmo Faggioli, Bhaskar Mitra, Paul Thomas, and Emine Yilmaz.","author":"Rahmani Hossein A","year":"2025","unstructured":"Hossein A Rahmani, Clemencia Siro, Mohammad Aliannejadi, Nick Craswell, Charles LA Clarke, Guglielmo Faggioli, Bhaskar Mitra, Paul Thomas, and Emine Yilmaz. 2025. Judging the judges: A collection of llm-generated relevance judgements. arXiv preprint arXiv:2502.13908 (2025)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657992"},{"key":"e_1_3_2_1_28_1","volume-title":"Mohammad Aliannejadi, Clemencia Siro, and Guglielmo Faggioli.","author":"Rahmani Hossein A","year":"2024","unstructured":"Hossein A Rahmani, Emine Yilmaz, Nick Craswell, Bhaskar Mitra, Paul Thomas, Charles LA Clarke, Mohammad Aliannejadi, Clemencia Siro, and Guglielmo Faggioli. 2024b. Llmjudge: Llms for relevance judgments. arXiv preprint arXiv:2408.08896 (2024)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1002\/asi.20682"},{"key":"e_1_3_2_1_30_1","volume-title":"Is ChatGPT good at search? investigating large language models as re-ranking agents. arXiv preprint arXiv:2304.09542","author":"Sun Weiwei","year":"2023","unstructured":"Weiwei Sun, Lingyong Yan, Xinyu Ma, Shuaiqiang Wang, Pengjie Ren, Zhumin Chen, Dawei Yin, and Zhaochun Ren. 2023. Is ChatGPT good at search? investigating large language models as re-ranking agents. arXiv preprint arXiv:2304.09542 (2023)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Paul Thomas Seth Spielman Nick Craswell and Bhaskar Mitra. 2024. Large language models can accurately predict searcher preferences. http:\/\/arxiv.org\/abs\/2309.10621 arXiv:2309.10621.","DOI":"10.1145\/3626772.3657707"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2411.08275"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2406.06519"},{"key":"e_1_3_2_1_34_1","volume-title":"Overview of the TREC 2019 Deep Learning Track.","author":"Voorhees Ellen M","year":"2020","unstructured":"Ellen M Voorhees, Nick Craswell, Bhaskar Mitra, Daniel Campos, and Emine Yilmaz. 2020. Overview of the TREC 2019 Deep Learning Track. (2020)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627508.3638344"},{"key":"e_1_3_2_1_36_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al., 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, Vol. 35 (2022), 24824-24837."},{"key":"e_1_3_2_1_37_1","first-page":"1","volume-title":"ACM SIGIR Forum","volume":"59","author":"Yu Ran","year":"2025","unstructured":"Ran Yu and Jiqun Liu. 2025. Chat as Learning in Interactive Information Retrieval and Generation. In ACM SIGIR Forum, Vol. 59. ACM New York, NY, USA, 1-19."},{"key":"e_1_3_2_1_38_1","unstructured":"Tianyang Zhong Zhengliang Liu Yi Pan Yutong Zhang Yifan Zhou Shizhe Liang Zihao Wu Yanjun Lyu Peng Shu Xiaowei Yu et al. 2024. Evaluation of openai o1: Opportunities and challenges of agi. arXiv preprint arXiv:2409.18486 (2024)."}],"event":{"name":"WSDM '26:The Nineteenth ACM International Conference on Web Search and Data Mining","location":"Boise ID USA","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGIR ACM Special Interest Group on Information Retrieval","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the Nineteenth ACM International Conference on Web Search and Data Mining"],"original-title":[],"deposited":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T17:54:49Z","timestamp":1771264489000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3773966.3779397"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,21]]},"references-count":38,"alternative-id":["10.1145\/3773966.3779397","10.1145\/3773966"],"URL":"https:\/\/doi.org\/10.1145\/3773966.3779397","relation":{},"subject":[],"published":{"date-parts":[[2026,2,21]]},"assertion":[{"value":"2026-02-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}