{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T02:12:48Z","timestamp":1774318368849,"version":"3.50.1"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032212887","type":"print"},{"value":"9783032212894","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21289-4_27","type":"book-chapter","created":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T01:04:00Z","timestamp":1774314240000},"page":"418-433","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Reducing Human Effort to\u00a0Validate LLM Relevance Judgements via\u00a0Stratified Sampling"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-8003-4795","authenticated-orcid":false,"given":"Simone","family":"Merlo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0362-5893","authenticated-orcid":false,"given":"Stefano","family":"Marchesin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5070-2049","authenticated-orcid":false,"given":"Guglielmo","family":"Faggioli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9219-6239","authenticated-orcid":false,"given":"Nicola","family":"Ferro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,25]]},"reference":[{"key":"27_CR1","unstructured":"Abbasiantaeb, Z., Meng, C., Azzopardi, L., Aliannejadi, M.: Can we use large language models to fill relevance judgment holes? In: Acharya, P., Clarke, C.L.A., et al. (eds.) Joint Proceedings of the 1st Workshop on Evaluation Methodologies,Testbeds and Community for Information Access Research (EMTCIR 2024) and the 1st Workshop on User Modelling in Conversational Information Retrieval (UM-CIR 2024) co-located with the 2nd International ACM SIGIR Conference on Information Retrieval in the Asia Pacific (SIGIR-AP 2024), Tokyo, Japan, 12 December 2024, CEUR Workshop Proceedings, vol. 3854. CEUR-WS.org (2024). https:\/\/ceur-ws.org\/Vol-3854\/emtcir-2.pdf"},{"key":"27_CR2","doi-asserted-by":"publisher","unstructured":"Alaofi, M., Thomas, P., Scholer, F., Sanderson, M.: LLMs can be fooled into labelling a document as relevant: best caf\u00e9 near me; this paper is perfectly relevant. In: Sakai, T., Ishita, E., Ohshima, H., Hasibi, F., Mao, J., Jose, J.M. (eds.) Proceedings of the 2024 Annual International ACM SIGIR Conference on Research and Development in Information Retrieval in the Asia Pacific Region, SIGIR-AP 2024, Tokyo, Japan, 9-12 December 2024, pp. 32\u201341. ACM (2024). https:\/\/doi.org\/10.1145\/3673791.3698431","DOI":"10.1145\/3673791.3698431"},{"key":"27_CR3","doi-asserted-by":"publisher","unstructured":"Chen, N., Liu, J., Dong, X., Liu, Q., Sakai, T., Wu, X.: AI can be cognitively biased: an exploratory study on threshold priming in LLM-based batch relevance assessment. In: Sakai, T., Ishita, E., Ohshima, H., Hasibi, F., Mao, J., Jose, J.M. (eds.) Proceedings of the 2024 Annual International ACM SIGIR Conference on Research and Development in Information Retrieval in the Asia Pacific Region, SIGIR-AP 2024, Tokyo, Japan, 9-12 December 2024, pp. 54\u201363. ACM (2024). https:\/\/doi.org\/10.1145\/3673791.3698420","DOI":"10.1145\/3673791.3698420"},{"key":"27_CR4","doi-asserted-by":"publisher","unstructured":"Chiang, D.C., Lee, H.: Can large language models be an alternative to human evaluations? In: Rogers, A., Boyd-Graber, J.L., Okazaki, N. (eds.) Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2023, Toronto, Canada, 9\u201314 July 2023, pp. 15607\u201315631. Association for Computational Linguistics (2023). https:\/\/doi.org\/10.18653\/V1\/2023.ACL-LONG.870","DOI":"10.18653\/V1\/2023.ACL-LONG.870"},{"key":"27_CR5","doi-asserted-by":"publisher","unstructured":"Chu, X., Ilyas, I.F., Krishnan, S., Wang, J.: Data cleaning: overview and emerging challenges. In: \u00d6zcan, F., Koutrika, G., Madden, S. (eds.) Proceedings of the 2016 International Conference on Management of Data, SIGMOD Conference 2016, San Francisco, CA, USA, June 26 - July 01, 2016, pp. 2201\u20132206. ACM (2016). https:\/\/doi.org\/10.1145\/2882903.2912574","DOI":"10.1145\/2882903.2912574"},{"key":"27_CR6","doi-asserted-by":"publisher","unstructured":"Clarke, C., Dietz, L.: LLM-based relevance assessment still can\u2019t replace human relevance assessment. In: Proceedings of the Eleventh International Workshop on Evaluating Information Access, EVIA 2025, a Satellite Workshop of the NTCIR-18 Conference, Tokyo, Japan, 10 June 2025, National Institute of Informatics (NII) (2025). https:\/\/doi.org\/10.20736\/0002002105","DOI":"10.20736\/0002002105"},{"key":"27_CR7","doi-asserted-by":"publisher","unstructured":"Cleverdon, C.W.: The Aslib Cranfield research project on the comparative efficiency of indexing systems. ASLIB Proc. 12(12), 421\u2013431 (1960). https:\/\/doi.org\/10.1108\/eb049778. ISSN 0001-253","DOI":"10.1108\/eb049778"},{"key":"27_CR8","volume-title":"Sampling Techniques","author":"WG Cochran","year":"1963","unstructured":"Cochran, W.G.: Sampling Techniques. John Wiley, Hoboken (1963)"},{"key":"27_CR9","doi-asserted-by":"crossref","unstructured":"Craswell, N., Mitra, B., Yilmaz, E., Campos, D.: Overview of the TREC 2020 deep learning track. In: Voorhees, E.M., Ellis, A. (eds.) Proceedings of the Twenty-Ninth Text REtrieval Conference, TREC 2020, Virtual Event [Gaithersburg, Maryland, USA], November 16-20, 2020, NIST Special Publication, vol. 1266, National Institute of Standards and Technology (NIST) (2020),https:\/\/trec.nist.gov\/pubs\/trec29\/papers\/OVERVIEW.DL.pdf","DOI":"10.6028\/NIST.SP.1266.deep-overview"},{"key":"27_CR10","doi-asserted-by":"crossref","unstructured":"Craswell, N., Mitra, B., Yilmaz, E., Campos, D., Voorhees, E.M.: Overview of the TREC 2019 deep learning track. CoRR abs\/2003.07820 (2020), https:\/\/arxiv.org\/abs\/2003.07820","DOI":"10.6028\/NIST.SP.1266.deep-overview"},{"key":"27_CR11","doi-asserted-by":"publisher","unstructured":"Faggioli, G., et al.: Perspectives on large language models for relevance judgment. In: Yoshioka, M., Kiseleva, J., Aliannejadi, M. (eds.) Proceedings of the 2023 ACM SIGIR International Conference on Theory of Information Retrieval, ICTIR 2023, Taipei, Taiwan, 23 July 2023, pp. 39\u201350. ACM (2023). https:\/\/doi.org\/10.1145\/3578337.3605136","DOI":"10.1145\/3578337.3605136"},{"key":"27_CR12","doi-asserted-by":"publisher","unstructured":"Faggioli, G., Ferro, N., Fuhr, N.: Detecting significant differences between information retrieval systems via generalized linear models. In: Hasan, M.A., Xiong, L. (eds.) Proceedings of the 31st ACM International Conference on Information & Knowledge Management, Atlanta, GA, USA, 17\u201321 October 2022, pp. 446\u2013456. ACM (2022). https:\/\/doi.org\/10.1145\/3511808.3557286","DOI":"10.1145\/3511808.3557286"},{"key":"27_CR13","doi-asserted-by":"publisher","unstructured":"Farzi, N., Dietz, L.: Does UMBRELA work on other LLMs? In: Ferro, N., Maistro, M., Pasi, G., Alonso, O., Trotman, A., Verberne, S. (eds.) Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2025, Padua, Italy, 13\u201318 July 2025, pp. 3214\u20133222. ACM (2025). https:\/\/doi.org\/10.1145\/3726302.3730317","DOI":"10.1145\/3726302.3730317"},{"key":"27_CR14","doi-asserted-by":"publisher","unstructured":"Fr\u00f6be, M., et al.: Large language model relevance assessors agree with one another more than with human assessors. In: Ferro, N., Maistro, M., Pasi, G., Alonso, O., Trotman, A., Verberne, S. (eds.) Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2025, Padua, Italy, 13\u201318 July 2025, pp. 2858\u20132863. ACM (2025). https:\/\/doi.org\/10.1145\/3726302.3730218","DOI":"10.1145\/3726302.3730218"},{"key":"27_CR15","doi-asserted-by":"publisher","unstructured":"Gilardi, F., Alizadeh, M., Kubli, M.: ChatGPT outperforms crowd-workers for text-annotation tasks. CoRR abs\/2303.15056 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2303.15056","DOI":"10.48550\/ARXIV.2303.15056"},{"key":"27_CR16","unstructured":"de\u00a0Jesus, G., Nunes, S.S.: Exploring large language models for relevance judgments in Tetun. In: Siro, C., et al. (eds.) Proceedings of The First Workshop on Large Language Models for Evaluation in Information Retrieval (LLM4Eval 2024) co-located with 10th International Conference on Online Publishing (SIGIR 2024), Washington D.C., USA, 18 July 2024, CEUR Workshop Proceedings, vol. 3752, pp. 19\u201330. CEUR-WS.org (2024). https:\/\/ceur-ws.org\/Vol-3752\/paper2.pdf"},{"key":"27_CR17","doi-asserted-by":"publisher","unstructured":"Marchesin, S., Silvello, G.: Efficient and reliable estimation of knowledge graph accuracy. Proc. VLDB Endow. 17(9), 2392\u20132404 (2024). https:\/\/doi.org\/10.14778\/3665844.3665865, https:\/\/www.vldb.org\/pvldb\/vol17\/p2392-marchesin.pdf","DOI":"10.14778\/3665844.3665865"},{"issue":"4","key":"27_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3736402","volume":"43","author":"C Meng","year":"2025","unstructured":"Meng, C., Arabzadeh, N., Askari, A., Aliannejadi, M., de Rijke, M.: Query performance prediction using relevance judgments generated by large language models. ACM Trans. Inf. Syst. 43(4), 1\u201335 (2025). https:\/\/doi.org\/10.1145\/3736402","journal-title":"ACM Trans. Inf. Syst."},{"key":"27_CR19","doi-asserted-by":"publisher","unstructured":"Merlo, S., Marchesin, S., Faggioli, G., Ferro, N.: A cost-effective framework to evaluate LLM-generated relevance judgements. In: Cha, M., et al. (eds.) Proceedings of the 34th ACM International Conference on Information and Knowledge Management, CIKM 2025, Seoul, Republic of Korea, 10\u201314 November 2025, pp. 2115\u20132126. ACM (2025). https:\/\/doi.org\/10.1145\/3746252.3761200","DOI":"10.1145\/3746252.3761200"},{"key":"27_CR20","doi-asserted-by":"publisher","unstructured":"Otero, D., Parapar, J., Barreiro, \u00c1.: Limitations of automatic relevance assessments with large language models for fair and reliable retrieval evaluation. In: Ferro, N., Maistro, M., Pasi, G., Alonso, O., Trotman, A., Verberne, S. (eds.) Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2025, Padua, Italy, 13\u201318 July 2025, pp. 2545\u20132549. ACM (2025).https:\/\/doi.org\/10.1145\/3726302.3730221","DOI":"10.1145\/3726302.3730221"},{"key":"27_CR21","doi-asserted-by":"publisher","unstructured":"Rahmani, H.A., et al.: LLM4Eval: large language model for evaluation in IR. In: Yang, G.H., Wang, H., Han, S., Hauff, C., Zuccon, G., Zhang, Y. (eds.) Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2024, Washington DC, USA, 14\u201318 July 2024, pp. 3040\u20133043. ACM (2024). https:\/\/doi.org\/10.1145\/3626772.3657992","DOI":"10.1145\/3626772.3657992"},{"key":"27_CR22","doi-asserted-by":"publisher","unstructured":"Rahmani, H.A., et al.: Judging the judges: a collection of LLM-generated relevance judgements. CoRR abs\/2502.13908 (2025). https:\/\/doi.org\/10.48550\/ARXIV.2502.13908","DOI":"10.48550\/ARXIV.2502.13908"},{"key":"27_CR23","doi-asserted-by":"crossref","unstructured":"Rahmani, H.A., et al.: LLMJudge: LLMs for relevance judgments. In: Siro, C., et al. (eds.) Proceedings of The First Workshop on Large Language Models for Evaluation in Information Retrieval (LLM4Eval 2024) co-located with 10th International Conference on Online Publishing (SIGIR 2024), Washington D.C., USA, 18 July 2024, CEUR Workshop Proceedings, vol. 3752, pp. 1\u20133. CEUR-WS.org (2024). https:\/\/ceur-ws.org\/Vol-3752\/paper8.pdf","DOI":"10.1145\/3722449.3722461"},{"key":"27_CR24","doi-asserted-by":"publisher","unstructured":"Siro, C., et al.: LLM4Eval: large language model for evaluation in IR. In: Ferro, N., Maistro, M., Pasi, G., Alonso, O., Trotman, A., Verberne, S. (eds.) Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2025, Padua, Italy, 13\u201318 July 2025, pp. 4188\u20134191. ACM (2025). https:\/\/doi.org\/10.1145\/3726302.3730367","DOI":"10.1145\/3726302.3730367"},{"key":"27_CR25","doi-asserted-by":"crossref","unstructured":"Soboroff, I.: Don\u2019t use LLMs to make relevance judgments. Inf. Retr. Res. J. 1(1), 29\u201346 (2025). https:\/\/doi.org\/10.54195\/IRRJ.19625","DOI":"10.54195\/irrj.19625"},{"key":"27_CR26","doi-asserted-by":"publisher","unstructured":"Stehman, S.V.: Estimating standard errors of accuracy assessment statistics under cluster sampling. Remote Sens. Environ. 60(3), 258\u2013269 (1997). https:\/\/doi.org\/10.1016\/S0034-4257(96)00176-9. ISSN 0034-4257","DOI":"10.1016\/S0034-4257(96)00176-9"},{"key":"27_CR27","unstructured":"Thakur, N., Pradeep, R., Upadhyay, S., Campos, D., Craswell, N., Lin, J.: Support evaluation for the TREC 2024 RAG track: comparing human versus LLM judges (2025). https:\/\/arxiv.org\/abs\/2504.15205"},{"key":"27_CR28","doi-asserted-by":"publisher","unstructured":"Thomas, P., Oard, D.W., Yang, E., Lawrie, D.J., Mayfield, J.: System comparison using automated generation of relevance judgements in multiple languages. In: Ferro, N., Maistro, M., Pasi, G., Alonso, O., Trotman, A., Verberne, S. (eds.) Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2025, Padua, Italy, 13\u201318 July 2025, pp. 2812\u20132816. ACM (2025). https:\/\/doi.org\/10.1145\/3726302.3730252","DOI":"10.1145\/3726302.3730252"},{"key":"27_CR29","doi-asserted-by":"publisher","unstructured":"Thomas, P., Spielman, S., Craswell, N., Mitra, B.: Large language models can accurately predict searcher preferences. In: Yang, G.H., Wang, H., Han, S., Hauff, C., Zuccon, G., Zhang, Y. (eds.) Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2024, Washington DC, USA, 14\u201318 July 2024, pp. 1930\u20131940. ACM (2024). https:\/\/doi.org\/10.1145\/3626772.3657707","DOI":"10.1145\/3626772.3657707"},{"key":"27_CR30","doi-asserted-by":"publisher","unstructured":"Upadhyay, S., Pradeep, R., Thakur, N., Craswell, N., Lin, J.: UMBRELA: UMbrela is the (open-source reproduction of the) Bing relevance assessor. CoRR abs\/2406.06519 (2024). https:\/\/doi.org\/10.48550\/ARXIV.2406.06519","DOI":"10.48550\/ARXIV.2406.06519"},{"key":"27_CR31","doi-asserted-by":"crossref","unstructured":"Voorhees, E.: Overview of the TREC 2004 robust retrieval track. In: TREC (2004)","DOI":"10.6028\/NIST.SP.500-261.robust-overview"},{"key":"27_CR32","doi-asserted-by":"publisher","unstructured":"Voorhees, E.M.: NIST TREC disks 4 and 5: retrieval test collections document set (1996). https:\/\/doi.org\/10.18434\/t47g6m","DOI":"10.18434\/t47g6m"},{"key":"27_CR33","doi-asserted-by":"publisher","unstructured":"Zhu, Y., Zhang, P., ul\u00a0Haq, E., Hui, P., Tyson, G.: Can ChatGPT reproduce human-generated labels? A study of social computing tasks. CoRR abs\/2304.10145 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2304.10145","DOI":"10.48550\/ARXIV.2304.10145"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21289-4_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T01:04:04Z","timestamp":1774314244000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21289-4_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032212887","9783032212894"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21289-4_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"25 March 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Delft","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 March 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"48","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2026.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}