{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T17:04:42Z","timestamp":1784048682263,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T00:00:00Z","timestamp":1784851200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,25]]},"DOI":"10.1145\/3805713.3820429","type":"proceedings-article","created":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T16:08:31Z","timestamp":1784045311000},"page":"44-55","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["LLM-Driven Usefulness Judgment for Web Search Evaluation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-2350-7079","authenticated-orcid":false,"given":"Mouly","family":"Dewan","sequence":"first","affiliation":[{"name":"Information School, University of Washington, Seattle, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3643-2182","authenticated-orcid":false,"given":"Jiqun","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Oklahoma, Norman, OK, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7324-4079","authenticated-orcid":false,"given":"Aditya","family":"Gautam","sequence":"additional","affiliation":[{"name":"Facebook, Bellevue, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3797-4293","authenticated-orcid":false,"given":"Chirag","family":"Shah","sequence":"additional","affiliation":[{"name":"Information School, University of Washington, Seattle, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,24]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Zahra Abbasiantaeb Chuan Meng Leif Azzopardi and Mohammad Aliannejadi. 2024. Can We Use Large Language Models to Fill Relevance Judgment Holes? http:\/\/arxiv.org\/abs\/2405.05600 arXiv:2405.05600."},{"key":"e_1_3_2_1_2_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/2766462.2767854"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the ACM SIGIR 2008 Workshop on Beyond Binary Relevance: Preferences, Diversity, and Set-Level Judgments. http:\/\/research. microsoft.com\/ pauben\/bbr-workshop.","author":"Belkin Nicholas J","year":"2008","unstructured":"Nicholas J Belkin, Michael Cole, and Ralf Bierig. 2008. Is relevance the right criterion for evaluating interactive information retrieval. In Proceedings of the ACM SIGIR 2008 Workshop on Beyond Binary Relevance: Preferences, Diversity, and Set-Level Judgments. http:\/\/research. microsoft.com\/ pauben\/bbr-workshop."},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the SIGIR 2009 Workshop on the Future of IR Evaluation. 7-8.","author":"Belkin Nicholas J","year":"2009","unstructured":"Nicholas J Belkin, Michael Cole, and Jingjing Liu. 2009. A model for evaluation of interactive information retrieval. In Proceedings of the SIGIR 2009 Workshop on the Future of IR Evaluation. 7-8."},{"key":"e_1_3_2_1_6_1","volume-title":"Inpars: Data augmentation for information retrieval using large language models. arXiv preprint arXiv:2202.05144","author":"Bonifacio Luiz","year":"2022","unstructured":"Luiz Bonifacio, Hugo Abonizio, Marzieh Fadaee, and Rodrigo Nogueira. 2022. Inpars: Data augmentation for information retrieval using large language models. arXiv preprint arXiv:2202.05144 (2022)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1835449.1835683"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/1645953.1646033"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3450127"},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the Nineteenth ACM International Conference on Web Search and Data Mining. 1099-1104","author":"Chen Nuo","year":"2026","unstructured":"Nuo Chen, Hanpei Fang, Jiqun Liu, Wilson Wei, Tetsuya Sakai, and Xiao-Ming Wu. 2026. Mitigating the Threshold Priming Effect in Large Language Model-Based Relevance Judgments via Personality Simulation. In Proceedings of the Nineteenth ACM International Conference on Web Search and Data Mining. 1099-1104."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080804"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/2484028.2484071"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the third workshop on human-computer interaction and information retrieval Cambridge. HCIR, 1-4.","author":"Cole Michael","year":"2009","unstructured":"Michael Cole, Jingjing Liu, Nicholas J Belkin, Ralf Bierig, Jacek Gwizdka, Chang Liu, Jin Zhang, and Xiangmin Zhang. 2009. Usefulness as the criterion for evaluation of interactive information retrieval. In Proceedings of the third workshop on human-computer interaction and information retrieval Cambridge. HCIR, 1-4."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/290941.291009"},{"key":"e_1_3_2_1_15_1","volume-title":"Promptagator: Few-shot dense retrieval from 8 examples. arXiv preprint arXiv:2209.11755","author":"Dai Zhuyun","year":"2022","unstructured":"Zhuyun Dai, Vincent Y Zhao, Ji Ma, Yi Luan, Jianmo Ni, Jing Lu, Anton Bakalov, Kelvin Guu, Keith B Hall, and Ming-Wei Chang. 2022. Promptagator: Few-shot dense retrieval from 8 examples. arXiv preprint arXiv:2209.11755 (2022)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730223"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3773966.3779397"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731120.3744588"},{"key":"e_1_3_2_1_19_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Guglielmo Faggioli Laura Dietz Charles Clarke Gianluca Demartini Matthias Hagen Claudia Hauff Noriko Kando Evangelos Kanoulas Martin Potthast Benno Stein and Henning Wachsmuth. 2023. Perspectives on Large Language Models for Relevance Judgment. http:\/\/arxiv.org\/abs\/2304.09161 arXiv:2304.09161.","DOI":"10.1145\/3578337.3605136"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731120.3744591"},{"key":"e_1_3_2_1_22_1","unstructured":"Jiawei Gu Xuhui Jiang Zhichao Shi Hexiang Tan Xuehao Zhai Chengjin Xu Wei Li Yinghan Shen Shengjie Ma Honghao Liu et al. 2024. A Survey on LLM-as-a-Judge. arXiv preprint arXiv:2411.15594 (2024)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/1842890.1842906"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-2111"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2684822.2685319"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 2026 Conference on Human Information Interaction and Retrieval. 524-528","author":"Jiang Tianji","year":"2026","unstructured":"Tianji Jiang, Wenqi Li, and Jiqun Liu. 2026. Improving Data Reusability in Interactive IR: Insights from the Community. In Proceedings of the 2026 Conference on Human Information Interaction and Retrieval. 524-528."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3130332.3130334"},{"key":"e_1_3_2_1_28_1","unstructured":"Ying-Chun Lin Jennifer Neville Jack W Stokes Longqi Yang Tara Safavi Mengting Wan Scott Counts Siddharth Suri Reid Andersen Xiaofeng Xu et al. 2024. Interpretable user satisfaction estimation for conversational systems with large language models. arXiv preprint arXiv:2403.12388 (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"V58, 3","author":"Liu Jiqun","year":"2021","unstructured":"Jiqun Liu. 2021. Deconstructing search tasks in interactive information retrieval: A systematic review of task dimensions and predictors. Information Processing & Management, V58, 3 (2021), 102522."},{"key":"e_1_3_2_1_30_1","volume-title":"V59, 5","author":"Liu Jiqun","year":"2022","unstructured":"Jiqun Liu. 2022. Toward Cranfield-inspired reusability assessment in interactive information retrieval evaluation. Information Processing & Management, V59, 5 (2022), 103007."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343413.3377976"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3529372.3530926"},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the 30th ACM international conference on information & knowledge management. 3258-3262","author":"Liu Jiqun","year":"2021","unstructured":"Jiqun Liu and Ran Yu. 2021. State-aware meta-evaluation of evaluation metrics in interactive information retrieval. In Proceedings of the 30th ACM international conference on information & knowledge management. 3258-3262."},{"key":"e_1_3_2_1_34_1","volume-title":"V55, 9","author":"Liu Pengfei","year":"2023","unstructured":"Pengfei Liu, Weizhe Yuan, Jinlan Fu, Zhengbao Jiang, Hiroaki Hayashi, and Graham Neubig. 2023. Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing. ACM computing surveys, V55, 9 (2023), 1-35."},{"key":"e_1_3_2_1_35_1","volume-title":"https:\/\/guidelines.raterhub.com\/searchqualityevaluatorguidelines.pdf Accessed","author":"General Guidelines Google LLC.","year":"2025","unstructured":"Google LLC. 2022. General Guidelines. https:\/\/guidelines.raterhub.com\/searchqualityevaluatorguidelines.pdf Accessed: 20 January 2025."},{"key":"e_1_3_2_1_36_1","first-page":"129","volume-title":"Proceedings of the AAAI conference on human computation and crowdsourcing","volume":"4","author":"Maddalena Eddy","year":"2016","unstructured":"Eddy Maddalena, Marco Basaldella, Dario De Nart, Dante Degl'Innocenti, Stefano Mizzaro, and Gianluca Demartini. 2016. Crowdsourcing relevance assessments: The unexpected benefits of limiting the time to judge. In Proceedings of the AAAI conference on human computation and crowdsourcing, Vol. V4. 129-138."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080750"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080750"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911451.2911507"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911451.2911507"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080737"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657992"},{"key":"e_1_3_2_1_43_1","unstructured":"Hossein A. Rahmani Emine Yilmaz Nick Craswell Bhaskar Mitra Paul Thomas Charles L. A. Clarke Mohammad Aliannejadi Clemencia Siro and Guglielmo Faggioli. 2024b. LLMJudge: LLMs for Relevance Judgments. http:\/\/arxiv.org\/abs\/2408.08896 arXiv:2408.08896."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657957"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Paul Thomas Seth Spielman Nick Craswell and Bhaskar Mitra. 2024. Large language models can accurately predict searcher preferences. http:\/\/arxiv.org\/abs\/2309.10621 arXiv:2309.10621.","DOI":"10.1145\/3626772.3657707"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2411.08275"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2406.06519"},{"key":"e_1_3_2_1_48_1","volume-title":"V36, 1","author":"Voorhees Ellen M","year":"2000","unstructured":"Ellen M Voorhees and Donna Harman. 2000. Overview of the sixth text retrieval conference (TREC-6). Information Processing & Management, V36, 1 (2000), 3-35."},{"key":"e_1_3_2_1_49_1","volume-title":"Proceedings of the 34th ACM International Conference on Information and Knowledge Management. 3133-3143","author":"Wang Xingzhu","year":"2025","unstructured":"Xingzhu Wang, Erhan Zhang, Yiqun Chen, Jinghan Xuan, Yucheng Hou, Yitong Xu, Ying Nie, Shuaiqiang Wang, Dawei Yin, and Jiaxin Mao. 2025. CLUE: Using Large Language Models for Judging Document Usefulness in Web Search Evaluation. In Proceedings of the 34th ACM International Conference on Information and Knowledge Management. 3133-3143."},{"key":"e_1_3_2_1_50_1","volume-title":"Tatsunori Hashimoto, Oriol Vinyals, Percy Liang, Jeff Dean, and William Fedus.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Yi Tay, Rishi Bommasani, Colin Raffel, Barret Zoph, Sebastian Borgeaud, Dani Yogatama, Maarten Bosma, Denny Zhou, Donald Metzler, Ed H. Chi, Tatsunori Hashimoto, Oriol Vinyals, Percy Liang, Jeff Dean, and William Fedus. 2022a. Emergent Abilities of Large Language Models. ArXiv, Vabs\/2206.07682 (2022). https:\/\/api.semanticscholar.org\/CorpusID:249674500"},{"key":"e_1_3_2_1_51_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al., 2022b. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, V35 (2022), 24824-24837."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/2661829.2661953"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401162"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","unstructured":"Hengran Zhang Ruqing Zhang Jiafeng Guo Maarten de Rijke Yixing Fan and Xueqi Cheng. 2024. Are Large Language Models Good at Utility Judgments? https:\/\/doi.org\/10.48550\/arXiv.2403.19216 arXiv:2403.19216 [cs].","DOI":"10.48550\/arXiv.2403.19216"}],"event":{"name":"ICTIR '26: International ACM SIGIR Conference on Innovative Concepts and Theories in Information Retrieval (ICTIR)","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 2026 International ACM SIGIR Conference on Innovative Concepts and Theories in Information Retrieval (ICTIR)"],"original-title":[],"deposited":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T16:11:08Z","timestamp":1784045468000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805713.3820429"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,24]]},"references-count":54,"alternative-id":["10.1145\/3805713.3820429","10.1145\/3805713"],"URL":"https:\/\/doi.org\/10.1145\/3805713.3820429","relation":{},"subject":[],"published":{"date-parts":[[2026,7,24]]},"assertion":[{"value":"2026-07-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}