{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T13:59:12Z","timestamp":1774360752191,"version":"3.50.1"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032212993","type":"print"},{"value":"9783032213006","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21300-6_25","type":"book-chapter","created":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T13:07:25Z","timestamp":1774357645000},"page":"349-359","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Analyzing AI Evaluation Benchmarks Through Information Retrieval and\u00a0Network Science"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-3294-4506","authenticated-orcid":false,"given":"Gaia","family":"Simeoni","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7337-7592","authenticated-orcid":false,"given":"Michael","family":"Soprano","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-5550-317X","authenticated-orcid":false,"given":"Riccardo","family":"Lunardi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9191-3280","authenticated-orcid":false,"given":"Kevin","family":"Roitero","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2852-168X","authenticated-orcid":false,"given":"Stefano","family":"Mizzaro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,25]]},"reference":[{"key":"25_CR1","doi-asserted-by":"crossref","unstructured":"Baldelli, D., Jiang, J., Aizawa, A., Torroni, P.: TWOLAR: a TWO-step LLM-augmented distillation method for passage reranking. In: Advances in Information Retrieval: 46th European Conference on Information Retrieval, ECIR 2024, Glasgow, UK, 24\u201328 March 2024, Proceedings, Part I, pp. 470\u2013485. Springer, Heidelberg (2024)","DOI":"10.1007\/978-3-031-56027-9_29"},{"key":"25_CR2","unstructured":"Bhakthavatsalam, S., et al.: Think You Have Solved Direct-Answer Question Answering? Try ARC-DA, the Direct-Answer AI2 Reasoning Challenge. arXiv preprint arXiv:2102.03315 (2021)"},{"key":"25_CR3","doi-asserted-by":"crossref","unstructured":"Bowman, S.R., Dahl, G.: What will it take to fix benchmarking in natural language understanding? In: Toutanova, K., et al. (eds.) Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 4843\u20134855. Association for Computational Linguistics, Online (2021)","DOI":"10.18653\/v1\/2021.naacl-main.385"},{"key":"25_CR4","doi-asserted-by":"crossref","unstructured":"Buckley, C., Voorhees, E.M.: Evaluating evaluation measure stability. In: Proceedings of the 23rd Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2000), pp. 235\u2013242. ACM, Athens, Greece (2000)","DOI":"10.1145\/345508.345543"},{"key":"25_CR5","doi-asserted-by":"crossref","unstructured":"Buckley, C., Voorhees, E.M.: Retrieval evaluation with incomplete information. In: Proceedings of the 27th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2004). ACM, Sheffield, UK (2004)","DOI":"10.1145\/1008992.1009000"},{"key":"25_CR6","unstructured":"Chiang, W.L., et al.: Chatbot arena: an open platform for evaluating LLMs by human preference. In: Proceedings of the 41st International Conference on Machine Learning, ICML 2024. JMLR.org (2024)"},{"key":"25_CR7","unstructured":"Clark, P., et al.: Think You Have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge. arXiv preprint arXiv:1803.05457 (2018)"},{"key":"25_CR8","doi-asserted-by":"crossref","unstructured":"Fan, W., et al.: A survey on RAG meeting LLMs: towards retrieval-augmented large language models. In: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, KDD 2024, pp. 6491\u20136501. Association for Computing Machinery, New York, NY, USA (2024)","DOI":"10.1145\/3637528.3671470"},{"key":"25_CR9","doi-asserted-by":"crossref","unstructured":"Ferro, N., Peters, C.: From multilingual to multimodal: the evolution of CLEF over two decades. In: Ferro, N., Peters, C. (eds.) Information Retrieval Evaluation in a Changing World: Lessons Learned from 20 Years of CLEF, The Information Retrieval Series, vol.\u00a041, pp. 3\u201344. Springer, Cham (2019)","DOI":"10.1007\/978-3-030-22948-1_1"},{"key":"25_CR10","unstructured":"Frohberg, J., Binder, F.: CRASS: a novel data set and benchmark to test counterfactual reasoning of large language models. In: Proceedings of the Thirteenth Language Resources and Evaluation Conference (LREC 2022), pp. 2126\u20132140. European Language Resources Association, Marseille, France (2022)"},{"key":"25_CR11","unstructured":"Grattafiori, A., et al.: The Llama 3 Herd of Models. arXiv preprint arXiv:2407.21783 (2024)"},{"key":"25_CR12","unstructured":"Hendrycks, D., et al.: Measuring Massive Multitask Language Understanding. arXiv preprint arXiv:2009.03300 (2020)"},{"key":"25_CR13","unstructured":"Jiang, A.Q., et al.: Mistral 7B. arXiv preprint arXiv:2310.06825 (2023)"},{"key":"25_CR14","doi-asserted-by":"crossref","unstructured":"Kiela, D., et al.: Dynabench: rethinking benchmarking in NLP. In: Toutanova, K., et al. (eds.) Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 4110\u20134124. Association for Computational Linguistics, Online (2021)","DOI":"10.18653\/v1\/2021.naacl-main.324"},{"issue":"5","key":"25_CR15","doi-asserted-by":"publisher","first-page":"604","DOI":"10.1145\/324133.324140","volume":"46","author":"JM Kleinberg","year":"1999","unstructured":"Kleinberg, J.M.: Authoritative sources in a hyperlinked environment. J. ACM 46(5), 604\u2013632 (1999)","journal-title":"J. ACM"},{"key":"25_CR16","doi-asserted-by":"crossref","unstructured":"Lai, G., Xie, Q., Liu, H., Yang, Y., Hovy, E.: RACE: large-scale reading comprehension dataset from examinations. In: Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing, pp. 785\u2013794. Association for Computational Linguistics, Copenhagen, Denmark (2017)","DOI":"10.18653\/v1\/D17-1082"},{"key":"25_CR17","doi-asserted-by":"crossref","unstructured":"Li, Z., et al.: FlexKBQA: a flexible LLM-powered framework for few-shot knowledge base question answering. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, no. 17, pp. 18608\u201318616 (2024)","DOI":"10.1609\/aaai.v38i17.29823"},{"key":"25_CR18","doi-asserted-by":"crossref","unstructured":"Lunardi, R., Soprano, M., Coppola, P., Della Mea, V., Mizzaro, S., Roitero, K.: PILs of knowledge: a synthetic benchmark for evaluating question answering systems in healthcare. In: Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2025, pp. 3648\u20133658. Association for Computing Machinery, New York, NY, USA (2025)","DOI":"10.1145\/3726302.3730283"},{"key":"25_CR19","doi-asserted-by":"crossref","unstructured":"Mihaylov, T., Clark, P., Khot, T., Sabharwal, A.: Can a suit of armor conduct electricity? A new dataset for open book question answering. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 2381\u20132391. Association for Computational Linguistics, Brussels, Belgium (2018)","DOI":"10.18653\/v1\/D18-1260"},{"key":"25_CR20","doi-asserted-by":"crossref","unstructured":"Mitchell, M., et al.: Model cards for model reporting. In: Proceedings of the Conference on Fairness, Accountability, and Transparency, FAT* 2019, pp. 220\u2013229. Association for Computing Machinery, New York, NY, USA (2019)","DOI":"10.1145\/3287560.3287596"},{"issue":"11","key":"25_CR21","doi-asserted-by":"publisher","first-page":"989","DOI":"10.1002\/asi.10296","volume":"54","author":"S Mizzaro","year":"2003","unstructured":"Mizzaro, S.: Quality control in scholarly publishing: a new proposal. J. Am. Soc. Inform. Sci. Technol. 54(11), 989\u20131005 (2003)","journal-title":"J. Am. Soc. Inform. Sci. Technol."},{"key":"25_CR22","doi-asserted-by":"crossref","unstructured":"Mizzaro, S., Robertson, S.E.: HITS hits TREC: exploring IR evaluation results with network analysis. In: Proceedings of the 30th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2007, pp. 479\u2013486. Association for Computing Machinery, Amsterdam, The Netherlands (2007)","DOI":"10.1145\/1277741.1277824"},{"key":"25_CR23","doi-asserted-by":"crossref","unstructured":"Moffat, A., Zobel, J.: Rank-biased precision for measurement of retrieval effectiveness. ACM Trans. Inf. Syst. 27(1) (2008)","DOI":"10.1145\/1416950.1416952"},{"issue":"12","key":"25_CR24","doi-asserted-by":"publisher","first-page":"1067","DOI":"10.1002\/asi.1164","volume":"52","author":"C Peters","year":"2001","unstructured":"Peters, C., Braschler, M.: European research letter: cross-language system evaluation: the CLEF campaigns. J. Am. Soc. Inform. Sci. Technol. 52(12), 1067\u20131072 (2001)","journal-title":"J. Am. Soc. Inform. Sci. Technol."},{"key":"25_CR25","doi-asserted-by":"crossref","unstructured":"Podolak, J., Peri\u0107, L., Jani\u0107ijevi\u0107, M., Petcu, R.: Beyond reproducibility: advancing zero-shot LLM reranking efficiency with setwise insertion. In: Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2025, pp. 3205\u20133213. Association for Computing Machinery, New York, NY, USA (2025)","DOI":"10.1145\/3726302.3730323"},{"key":"25_CR26","doi-asserted-by":"crossref","unstructured":"Qin, Z., et al.: Large language models are effective text rankers with pairwise ranking prompting. In: Duh, K., Gomez, H., Bethard, S. (eds.) Findings of the Association for Computational Linguistics: NAACL 2024, pp. 1504\u20131518. Association for Computational Linguistics, Mexico City, Mexico (2024)","DOI":"10.18653\/v1\/2024.findings-naacl.97"},{"key":"25_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"605","DOI":"10.1007\/978-3-319-56608-5_55","volume-title":"Advances in Information Retrieval","author":"K Roitero","year":"2017","unstructured":"Roitero, K., Maddalena, E., Mizzaro, S.: Do easy topics predict effectiveness better than difficult topics? In: Jose, J.M., et al. (eds.) ECIR 2017. LNCS, vol. 10193, pp. 605\u2013611. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-56608-5_55"},{"key":"25_CR28","unstructured":"Roitero, K., Mizzaro, S., Soprano, M.: Bias and fairness in effectiveness evaluation by means of network analysis and mixture models. In: Proceedings of the 10th Italian Information Retrieval Workshop (IIR 2019), pp. 6\u20137. CEUR-WS, Padova, Italy (2019), extended Abstract, CEUR-WS Vol-2441"},{"key":"25_CR29","doi-asserted-by":"crossref","unstructured":"Roitero, K., Wright, D., Soprano, M., Augenstein, I., Mizzaro, S.: Efficiency and effectiveness of LLM-based summarization of evidence in crowdsourced fact-checking. In: Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2025, pp. 457\u2013467. Association for Computing Machinery, New York, NY, USA (2025)","DOI":"10.1145\/3726302.3729960"},{"issue":"5","key":"25_CR30","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1007\/s10791-008-9059-7","volume":"11","author":"T Sakai","year":"2008","unstructured":"Sakai, T., Kando, N.: On information retrieval metrics designed for evaluation with incomplete relevance assessments. Inf. Retrieval 11(5), 447\u2013470 (2008)","journal-title":"Inf. Retrieval"},{"key":"25_CR31","doi-asserted-by":"crossref","unstructured":"Sakai, T., Oard, D.W., Kando, N. (eds.): Evaluating Information Retrieval and Access Tasks: NTCIR\u2019s Legacy of Research Impact, The Information Retrieval Series, vol.\u00a043. Springer, Singapore (2021)","DOI":"10.1007\/978-981-15-5554-1"},{"issue":"4","key":"25_CR32","doi-asserted-by":"publisher","first-page":"247","DOI":"10.1561\/1500000009","volume":"4","author":"M Sanderson","year":"2010","unstructured":"Sanderson, M.: Test collection based evaluation of information retrieval systems. Found. Trends Inf. Retr. 4(4), 247\u2013375 (2010)","journal-title":"Found. Trends Inf. Retr."},{"key":"25_CR33","doi-asserted-by":"crossref","unstructured":"Sanderson, M., Zobel, J.: Information retrieval system evaluation: effort, sensitivity, and reliability. In: Proceedings of the 28th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2005), pp. 162\u2013169. ACM, Salvador, Brazil (2005)","DOI":"10.1145\/1076034.1076064"},{"key":"25_CR34","unstructured":"Soprano, M., Roitero, K., Mizzaro, S.: HITS hits readersourcing: validating peer review alternatives using network analysis. In: Proceedings of the 4th International Workshop on Bibliometric-enhanced Information Retrieval and Natural Language Processing for Digital Libraries (BIRNDL 2019) co-located with SIGIR 2019, pp. 70\u201382. CEUR-WS, Paris, France (2019), cEUR-WS Vol-2414"},{"key":"25_CR35","doi-asserted-by":"crossref","unstructured":"Sun, W., et al.: Is ChatGPT good at search? Investigating large language models as re-ranking agents. In: Bouamor, H., Pino, J., Bali, K. (eds.) Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 14918\u201314937. Association for Computational Linguistics, Singapore (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.923"},{"key":"25_CR36","doi-asserted-by":"crossref","unstructured":"Voorhees, E.M.: The evolution of cranfield. In: Ferro, N., Peters, C. (eds.) Information Retrieval Evaluation in a Changing World - Lessons Learned from 20 Years of CLEF, The Information Retrieval Series, vol.\u00a041, pp. 45\u201369. Springer (2019)","DOI":"10.1007\/978-3-030-22948-1_2"},{"key":"25_CR37","unstructured":"Voorhees, E.M.: Cranfield is dead; long live cranfield. In: Kando, N., Clarke, C.L.A., Kato, M.P., Liu, Y. (eds.) Proceedings of the 16th NTCIR Conference on Evaluation of Information Access Technologies, NTCIR 2022, Tokyo, Japan, 14\u201317 June 2022. National Institute of Informatics (NII) (2022)"},{"key":"25_CR38","doi-asserted-by":"crossref","unstructured":"Voorhees, E.M., Buckley, C.: The effect of topic set size on retrieval experiment error. In: Proceedings of the 25th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2002), pp. 316\u2013323. ACM, Tampere, Finland (2002)","DOI":"10.1145\/564376.564432"},{"key":"25_CR39","volume-title":"TREC: Experiment and Evaluation in Information Retrieval","year":"2005","unstructured":"Voorhees, E.M., Harman, D.K. (eds.): TREC: Experiment and Evaluation in Information Retrieval. The MIT Press, Cambridge (2005)"},{"issue":"1","key":"25_CR40","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1080\/00031305.1982.10482778","volume":"36","author":"CH Wagner","year":"1982","unstructured":"Wagner, C.H.: Simpson\u2019s paradox in real life. Am. Stat. 36(1), 46\u201348 (1982)","journal-title":"Am. Stat."},{"key":"25_CR41","doi-asserted-by":"crossref","unstructured":"Wang, L., Yang, N., Wei, F.: Query2doc: query expansion with large language models. In: Bouamor, H., Pino, J., Bali, K. (eds.) Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 9414\u20139423. Association for Computational Linguistics, Singapore (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.585"},{"key":"25_CR42","doi-asserted-by":"crossref","unstructured":"Welbl, J., Liu, N.F., Gardner, M.: Crowdsourcing multiple choice science questions. In: Proceedings of the 3rd Workshop on Noisy User-generated Text (W-NUT 2017), pp. 94\u2013106. Association for Computational Linguistics, Copenhagen, Denmark (2017)","DOI":"10.18653\/v1\/W17-4413"},{"key":"25_CR43","unstructured":"Yang, A., et al.: Qwen2 Technical Report. arXiv preprint arXiv:2407.10671 (2024)"},{"key":"25_CR44","doi-asserted-by":"publisher","first-page":"5985","DOI":"10.18653\/v1\/2023.findings-emnlp.398","volume-title":"Findings of the Association for Computational Linguistics: EMNLP 2023","author":"F Ye","year":"2023","unstructured":"Ye, F., Fang, M., Li, S., Yilmaz, E.: Enhancing conversational search: large language model-aided informative query rewriting. In: Bouamor, H., Pino, J., Bali, K. (eds.) Findings of the Association for Computational Linguistics: EMNLP 2023, pp. 5985\u20136006. Association for Computational Linguistics, Singapore (2023)"},{"key":"25_CR45","doi-asserted-by":"crossref","unstructured":"Yilmaz, E., Aslam, J.A.: A simple and efficient sampling method for estimating AP and NDCG. In: Proceedings of the 31st Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2008). ACM, Singapore (2008)","DOI":"10.1145\/1390334.1390437"},{"key":"25_CR46","doi-asserted-by":"crossref","unstructured":"Zellers, R., Holtzman, A., Bisk, Y., Farhadi, A., Choi, Y.: HellaSwag: can a machine really finish your sentence? In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics (ACL 2019), pp. 4791\u20134800. Association for Computational Linguistics, Florence, Italy (2019)","DOI":"10.18653\/v1\/P19-1472"},{"key":"25_CR47","doi-asserted-by":"crossref","unstructured":"Zhang, H., Yu, P.S., Zhang, J.: A systematic survey of text summarization: from statistical methods to large language models. ACM Comput. Surv. 57(11) (2025)","DOI":"10.1145\/3731445"},{"issue":"1","key":"25_CR48","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3748304","volume":"44","author":"Y Zhu","year":"2025","unstructured":"Zhu, Y., et al.: Large language models for information retrieval: a survey. ACM Trans. Inf. Syst. 44(1), 1\u201354 (2025)","journal-title":"ACM Trans. Inf. Syst."}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21300-6_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T13:07:41Z","timestamp":1774357661000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21300-6_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032212993","9783032213006"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21300-6_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"25 March 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Delft","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 March 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"48","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2026.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}