{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T17:05:47Z","timestamp":1784048747445,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":19,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T00:00:00Z","timestamp":1784851200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Australian Research Council","award":["CE200100005"],"award-info":[{"award-number":["CE200100005"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,25]]},"DOI":"10.1145\/3805713.3820403","type":"proceedings-article","created":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T16:08:31Z","timestamp":1784045311000},"page":"105-110","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Reasoning with Large Language Models for Relevance Judgements"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-7932-0206","authenticated-orcid":false,"given":"Louis","family":"Geiger","sequence":"first","affiliation":[{"name":"University of Freiburg, Freiburg, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3875-5727","authenticated-orcid":false,"given":"Danula","family":"Hettiachchi","sequence":"additional","affiliation":[{"name":"RMIT University, Melbourne, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9094-0810","authenticated-orcid":false,"given":"Falk","family":"Scholer","sequence":"additional","affiliation":[{"name":"RMIT University, Melbourne, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7801-0239","authenticated-orcid":false,"given":"Johanne R.","family":"Trippas","sequence":"additional","affiliation":[{"name":"RMIT University, Melbourne, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,24]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proc. ACM SIGIR-AP.","author":"Alaofi Marwah","year":"2024","unstructured":"Marwah Alaofi, Paul Thomas, Falk Scholer, and Mark Sanderson. 2024. LLMs can be Fooled into Labelling a Document as Relevant. In Proc. ACM SIGIR-AP."},{"key":"e_1_3_2_1_2_1","volume-title":"Proc. ACM SIGIR.","author":"Arabzadeh Negar","year":"2025","unstructured":"Negar Arabzadeh and Charles LA Clarke. 2025a. Benchmarking LLM-based relevance judgment methods. In Proc. ACM SIGIR."},{"key":"e_1_3_2_1_3_1","first-page":"2784","article-title":"A human-ai comparative analysis of prompt sensitivity in llm-based relevance judgment","author":"Arabzadeh Negar","year":"2025","unstructured":"Negar Arabzadeh and Charles LA Clarke. 2025b. A human-ai comparative analysis of prompt sensitivity in llm-based relevance judgment. In Proc. ACM SIGIR. 2784-2788.","journal-title":"Proc. ACM SIGIR."},{"key":"e_1_3_2_1_4_1","volume-title":"Proc. ACM SIGIR.","author":"Balog Krisztian","year":"2025","unstructured":"Krisztian Balog, Don Metzler, and Zhen Qin. 2025. Rankers, judges, and assistants: Towards understanding the interplay of llms in information retrieval evaluation. In Proc. ACM SIGIR."},{"key":"e_1_3_2_1_5_1","unstructured":"Yanda Chen Joe Benton Ansh Radhakrishnan Jonathan Uesato Carson Denison John Schulman Arushi Somani Peter Hase Misha Wagner Fabien Roger et al. 2025. Reasoning Models Don't Always Say What They Think. arXiv preprint arXiv:2505.05410 (2025)."},{"key":"e_1_3_2_1_6_1","volume-title":"Overview of the TREC 2023 deep learning track. arXiv preprint arXiv:2507","author":"Craswell Nick","year":"2025","unstructured":"Nick Craswell, Bhaskar Mitra, Emine Yilmaz, Hossein A. Rahmani, Daniel Campos, Jimmy Lin, Ellen M. Voorhees, and Ian Soboroff. 2025. Overview of the TREC 2023 deep learning track. arXiv preprint arXiv:2507.08890 (2025)."},{"key":"e_1_3_2_1_7_1","volume-title":"Towards reasoning in large language models: A survey. arXiv preprint arXiv:2212.10403","author":"Huang Jie","year":"2022","unstructured":"Jie Huang and Kevin Chen-Chuan Chang. 2022. Towards reasoning in large language models: A survey. arXiv preprint arXiv:2212.10403 (2022)."},{"key":"e_1_3_2_1_8_1","volume-title":"Proc. ACM ICTIR.","author":"Barbera David La","year":"2025","unstructured":"David La Barbera, Riccardo Lunardi, Mengdie Zhuang, and Kevin Roitero. 2025. Impersonating the Crowd: Evaluating LLMs' Ability to Replicate Human Judgment in Misinformation Assessment. In Proc. ACM ICTIR."},{"key":"e_1_3_2_1_9_1","first-page":"49","volume-title":"JASIST","volume":"70","author":"Losada David E","year":"2019","unstructured":"David E Losada, Javier Parapar, and Alvaro Barreiro. 2019. When to stop making relevance judgments? A study of stopping methods for building information retrieval test collections. JASIST, Vol. 70, 1 (2019), 49-60."},{"key":"e_1_3_2_1_10_1","first-page":"2335","article-title":"How deep is your learning: The DL-HARD annotated deep learning dataset","author":"Mackie Iain","year":"2021","unstructured":"Iain Mackie, Jeffrey Dalton, and Andrew Yates. 2021. How deep is your learning: The DL-HARD annotated deep learning dataset. In Proc. ACM SIGIR. 2335-2341.","journal-title":"Proc. ACM SIGIR."},{"key":"e_1_3_2_1_11_1","volume-title":"Benno Krojer, Xing Han L\u00f9, et al.","author":"Marjanovi\u0107 Sara Vera","year":"2025","unstructured":"Sara Vera Marjanovi\u0107, Arkil Patel, Vaibhav Adlakha, Milad Aghajohari, Parishad BehnamGhader, Mehar Bhatia, Aditi Khandelwal, Austin Kraft, Benno Krojer, Xing Han L\u00f9, et al., 2025. DeepSeek-R1 Thoughtology: Let's think about LLM Reasoning. arXiv preprint arXiv:2504.07128 (2025)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.1025"},{"key":"e_1_3_2_1_13_1","volume-title":"European Conference on Information Retrieval. Springer, 230-246","author":"Rathee Mandeep","year":"2025","unstructured":"Mandeep Rathee, Sean MacAvaney, and Avishek Anand. 2025. Guiding retrieval using llm-based listwise rankers. In European Conference on Information Retrieval. Springer, 230-246."},{"key":"e_1_3_2_1_14_1","volume-title":"Companion Proc. ACM WebConf.","author":"Schnabel Julian A","year":"2025","unstructured":"Julian A Schnabel, Johanne R Trippas, Falk Scholer, and Danula Hettiachchi. 2025. Multi-stage large language model pipelines can outperform gpt-4o in relevance assessment. In Companion Proc. ACM WebConf."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","first-page":"29","DOI":"10.54195\/irrj.19625","article-title":"Don't Use LLMs, to Make Relevance Judgments","volume":"1","author":"Soboroff Ian","year":"2025","unstructured":"Ian Soboroff. 2025. Don't Use LLMs, to Make Relevance Judgments. Information Retrieval Research, Vol. 1, 1 (2025), 29-46.","journal-title":"Information Retrieval Research"},{"key":"e_1_3_2_1_16_1","volume-title":"Proc. ACM SIGIR.","author":"Takehi Rikiya","year":"2025","unstructured":"Rikiya Takehi, Ellen M. Voorhees, Tetsuya Sakai, and Ian Soboroff. 2025. LLM-Assisted Relevance Assessments: When Should We Ask LLMs, for Help, ?. In Proc. ACM SIGIR."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Paul Thomas Seth Spielman Nick Craswell and Bhaskar Mitra. 2024. Large Language Models can Accurately Predict Searcher Preferences.","DOI":"10.1145\/3626772.3657707"},{"key":"e_1_3_2_1_18_1","volume-title":"Umbrela: Umbrela is the (open-source reproduction of the) bing relevance assessor. arXiv preprint arXiv:2406.06519","author":"Upadhyay Shivani","year":"2024","unstructured":"Shivani Upadhyay, Ronak Pradeep, Nandan Thakur, Nick Craswell, and Jimmy Lin. 2024. Umbrela: Umbrela is the (open-source reproduction of the) bing relevance assessor. arXiv preprint arXiv:2406.06519 (2024)."},{"key":"e_1_3_2_1_19_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al., 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, Vol. 35 (2022), 24824-24837."}],"event":{"name":"ICTIR '26: International ACM SIGIR Conference on Innovative Concepts and Theories in Information Retrieval (ICTIR)","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 2026 International ACM SIGIR Conference on Innovative Concepts and Theories in Information Retrieval (ICTIR)"],"original-title":[],"deposited":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T16:12:14Z","timestamp":1784045534000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805713.3820403"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,24]]},"references-count":19,"alternative-id":["10.1145\/3805713.3820403","10.1145\/3805713"],"URL":"https:\/\/doi.org\/10.1145\/3805713.3820403","relation":{},"subject":[],"published":{"date-parts":[[2026,7,24]]},"assertion":[{"value":"2026-07-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}