{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T13:53:24Z","timestamp":1774360404417,"version":"3.50.1"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032212993","type":"print"},{"value":"9783032213006","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21300-6_34","type":"book-chapter","created":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T12:57:20Z","timestamp":1774357040000},"page":"436-443","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Revisiting Human-vs-LLM Judgments Using the\u00a0TREC Podcast Track"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9463-595X","authenticated-orcid":false,"given":"Watheq","family":"Mansour","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1902-9087","authenticated-orcid":false,"given":"J.","family":"Shane Culpepper","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7992-4633","authenticated-orcid":false,"given":"Joel","family":"Mackenzie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5970-880X","authenticated-orcid":false,"given":"Andrew","family":"Yates","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,25]]},"reference":[{"key":"34_CR1","unstructured":"Abbasiantaeb, Z., Meng, C., Azzopardi, L., Aliannejadi, M.: Can we use large language models to fill relevance judgment holes? In: Proc. EMTCIR (2024)"},{"key":"34_CR2","doi-asserted-by":"crossref","unstructured":"Alaofi, M., Thomas, P., Scholer, F., Sanderson, M.: LLMs can be fooled into labelling a document as relevant: best caf\u00e9 near me; this paper is perfectly relevant. In: Proc. SIGIR-AP, pp. 32\u201341 (2024)","DOI":"10.1145\/3673791.3698431"},{"key":"34_CR3","doi-asserted-by":"crossref","unstructured":"Arabzadeh, N., Clarke, C.L.A.: Benchmarking LLM-based relevance judgment methods. In: Proc. SIGIR, pp. 3194\u20133204 (2025)","DOI":"10.1145\/3726302.3730305"},{"key":"34_CR4","unstructured":"Bajaj, P., et al.: MS MARCO: A Human Generated MAchine Reading COmprehension Dataset. arXiv:1611.09268v3 (2018)"},{"key":"34_CR5","doi-asserted-by":"crossref","unstructured":"Balog, K., Metzler, D., Qin, Z.: Rankers, judges, and assistants: towards understanding the interplay of LLMs in information retrieval evaluation. In: Proc. SIGIR, pp. 3865\u20133875 (2025)","DOI":"10.1145\/3726302.3730348"},{"key":"34_CR6","unstructured":"Clarke, C.L., Dietz, L.: LLM-based relevance assessment still can\u2019t replace human relevance assessment. In: Proc. NTCIR (2025)"},{"key":"34_CR7","doi-asserted-by":"crossref","unstructured":"Clifton, A., et al.: 100,000 podcasts: a spoken English document corpus. In: Proc. COLING, pp. 5903\u20135917 (2020)","DOI":"10.18653\/v1\/2020.coling-main.519"},{"key":"34_CR8","doi-asserted-by":"crossref","unstructured":"Craswell, N., Mitra, B., Yilmaz, E., Campos, D.: Overview of the TREC 2020 deep learning track. In: Proc. TREC (2021)","DOI":"10.6028\/NIST.SP.1266.deep-overview"},{"key":"34_CR9","doi-asserted-by":"crossref","unstructured":"Craswell, N., Mitra, B., Yilmaz, E., Campos, D., Lin, J.: Overview of the TREC 2021 deep learning track. In: Proc. TREC (2021)","DOI":"10.6028\/NIST.SP.500-335.deep-overview"},{"key":"34_CR10","doi-asserted-by":"crossref","unstructured":"Craswell, N., Mitra, B., Yilmaz, E., Campos, D., Voorhees, E.M.: Overview of the TREC 2019 deep learning track. arXiv preprint arXiv:2003.07820 (2019)","DOI":"10.6028\/NIST.SP.1266.deep-overview"},{"key":"34_CR11","doi-asserted-by":"crossref","unstructured":"Faggioli, G., et\u00a0al.: Perspectives on large language models for relevance judgment. In: Proc. ICTIR, pp. 39\u201350 (2023)","DOI":"10.1145\/3578337.3605136"},{"issue":"4","key":"34_CR12","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1145\/3624730","volume":"67","author":"G Faggioli","year":"2024","unstructured":"Faggioli, G., et al.: Who determines what is relevant? Humans or AI? Why not both? Comm. ACM 67(4), 31\u201334 (2024)","journal-title":"Comm. ACM"},{"key":"34_CR13","unstructured":"Gemma Team: Gemma 2: Improving open language models at a practical size. arXiv:2408.00118 (2024)"},{"issue":"4","key":"34_CR14","doi-asserted-by":"publisher","first-page":"422","DOI":"10.1145\/582415.582418","volume":"20","author":"K J\u00e4rvelin","year":"2002","unstructured":"J\u00e4rvelin, K., Kek\u00e4l\u00e4inen, J.: Cumulated gain-based evaluation of IR techniques. ACM Trans. Inf. Sys. 20(4), 422\u2013446 (2002)","journal-title":"ACM Trans. Inf. Sys."},{"key":"34_CR15","unstructured":"Jiang, A.Q., Sablayrolles, A., et al.: Mistral 7b. arXiv:2310.06825 (2023)"},{"key":"34_CR16","doi-asserted-by":"crossref","unstructured":"Jones, R., et al.: TREC 2020 podcasts track overview. In: Proc. TREC (2020)","DOI":"10.6028\/NIST.SP.1266.podcast-overview"},{"key":"34_CR17","doi-asserted-by":"crossref","unstructured":"Karlgren, J., et al.: TREC 2021 podcasts track overview. In: Proc. TREC (2021)","DOI":"10.6028\/NIST.SP.500-335.podcast-overview"},{"issue":"1\/2","key":"34_CR18","doi-asserted-by":"publisher","first-page":"81","DOI":"10.2307\/2332226","volume":"30","author":"MG Kendall","year":"1938","unstructured":"Kendall, M.G.: A new measure of rank correlation. Biometrika 30(1\/2), 81\u201393 (1938)","journal-title":"Biometrika"},{"key":"34_CR19","unstructured":"Llama 3 team: The llama 3 herd of models. arXiv:2407.21783 (2024)"},{"key":"34_CR20","doi-asserted-by":"crossref","unstructured":"MacAvaney, S., Soldaini, L.: One-shot labeling for automatic relevance estimation. In: Proc. SIGIR, pp. 2230\u20132235 (2023)","DOI":"10.1145\/3539618.3592032"},{"key":"34_CR21","doi-asserted-by":"crossref","unstructured":"Mansour, W., Culpepper, J.S., Mackenzie, J.: Examining the impact of transcript variation on podcast search and re-ranking. In: Proc. ECIR, pp. 118\u2013127 (2025)","DOI":"10.1007\/978-3-031-88714-7_9"},{"key":"34_CR22","doi-asserted-by":"crossref","unstructured":"Moffat, A., Mackenzie, J., Mallia, A., Petri, M.: Rank-biased quality measurement for sets and rankings. In: Proc. SIGIR-AP, pp. 135\u2013144 (2024)","DOI":"10.1145\/3673791.3698405"},{"key":"34_CR23","doi-asserted-by":"crossref","unstructured":"Moffat, A., Zobel, J.: Rank-biased precision for measurement of retrieval effectiveness. ACM Trans. Inf. Syst. 27(1) (2008)","DOI":"10.1145\/1416950.1416952"},{"key":"34_CR24","doi-asserted-by":"crossref","unstructured":"Rahmani, H.A., et al.: Report on the 1st workshop on large language model for evaluation in information retrieval (LLM4Eval 2024) at SIGIR 2024. SIGIR Forum (2024)","DOI":"10.1145\/3722449.3722461"},{"key":"34_CR25","doi-asserted-by":"crossref","unstructured":"Rahmani, H.A., Yilmaz, E., Craswell, N., Mitra, B.: JudgeBlender: ensembling automatic relevance judgments. In: Proc. WWW, pp. 1268\u20131272 (2025)","DOI":"10.1145\/3701716.3715536"},{"key":"34_CR26","first-page":"29","volume":"1","author":"I Soboroff","year":"2025","unstructured":"Soboroff, I.: Don\u2019t use LLMs to make relevance judgments. Inf. Retr. Res. 1, 29\u201346 (2025)","journal-title":"Inf. Retr. Res."},{"key":"34_CR27","doi-asserted-by":"crossref","unstructured":"Sormunen, E.: Liberal relevance criteria of TREC: counting on negligible documents? In: Proc. SIGIR, pp. 324\u2013330 (2002)","DOI":"10.1145\/564376.564433"},{"key":"34_CR28","doi-asserted-by":"crossref","unstructured":"Tam, Z.R., Wu, C.K., Tsai, Y.L., Lin, C.Y., Lee, H.y., Chen, Y.N.: Let me speak freely? A study on the impact of format restrictions on performance of large language models. In: Proc. EMNLP (Industry Track) (2024)","DOI":"10.18653\/v1\/2024.emnlp-industry.91"},{"key":"34_CR29","doi-asserted-by":"crossref","unstructured":"Thomas, P., Spielman, S., Craswell, N., Mitra, B.: Large language models can accurately predict searcher preferences. In: Proc. SIGIR, pp. 1930\u20131940 (2024)","DOI":"10.1145\/3626772.3657707"},{"key":"34_CR30","doi-asserted-by":"crossref","unstructured":"Upadhyay, S., et al.: A large-scale study of relevance assessments with large language models using UMBRELA. In: Proc. ICTIR, pp. 358\u2013368 (2025)","DOI":"10.1145\/3731120.3744605"},{"key":"34_CR31","unstructured":"Upadhyay, S., Pradeep, R., Thakur, N., Craswell, N., Lin, J.: UMBRELA: UMbrela is the (Open-Source Reproduction of the) Bing RELevance Assessor. arXiv 2406.06519 (2024)"},{"key":"34_CR32","unstructured":"Yang, A., et al.: Qwen2.5 technical report. arXiv:2412.15115 (2025)"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21300-6_34","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T12:57:35Z","timestamp":1774357055000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21300-6_34"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032212993","9783032213006"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21300-6_34","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"25 March 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests of any sort.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Delft","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 March 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"48","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2026.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}