{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,5]],"date-time":"2025-04-05T04:11:11Z","timestamp":1743826271406,"version":"3.40.3"},"publisher-location":"Cham","reference-count":19,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031887192","type":"print"},{"value":"9783031887208","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-88720-8_56","type":"book-chapter","created":{"date-parts":[[2025,4,4]],"date-time":"2025-04-04T12:06:54Z","timestamp":1743768414000},"page":"366-372","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["ELOQUENT CLEF Shared Tasks for\u00a0Evaluation of\u00a0Generative Language Model Quality, 2025 Edition"],"prefix":"10.1007","author":[{"given":"Jussi","family":"Karlgren","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ekaterina","family":"Artemova","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ond\u0159ej","family":"Bojar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vladislav","family":"Mikhailov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Magnus","family":"Sahlgren","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Erik","family":"Velldal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lilja","family":"\u00d8vrelid","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,4,3]]},"reference":[{"key":"56_CR1","unstructured":"Bavaresco, A., et al.: LLMS instead of human judges? A large scale empirical study across 20 NLP evaluation tasks (2024). https:\/\/arxiv.org\/abs\/2406.18403"},{"key":"56_CR2","unstructured":"Bevendorff, J., et al.: Overview of the Voight-Kampff Generative AI authorship verification task at PAN and ELOQUENT 2024. In: Faggioli, G., Ferro, N., Vlachos, M., Galu\u0161\u010d\u00e1kov\u00e1, P., de\u00a0Herrera, A.G.S. (eds.) Working Notes of CLEF 2024 - Conference and Labs of the Evaluation Forum. CEUR-WS.org (2024)"},{"key":"56_CR3","doi-asserted-by":"crossref","unstructured":"Doddapaneni, S., Khan, M.S.U.R., Verma, S., Khapra, M.M.: Finding blind spots in evaluator LLMS with interpretable checklists (2024). https:\/\/arxiv.org\/abs\/2406.13439","DOI":"10.18653\/v1\/2024.emnlp-main.911"},{"key":"56_CR4","unstructured":"D\u00fcrlich, L., Gogoulou, E., Guillou, L., Nivre, J., Zahra, S.: Overview of the CLEF-2024 eloquent lab: task 2 on HalluciGen. In: Faggioli, G., Ferro, N., Vlachos, M., Galu\u0161\u010d\u00e1kov\u00e1, P., de\u00a0Herrera, A.G.S. (eds.) Working Notes of CLEF 2024 - Conference and Labs of the Evaluation Forum. CEUR-WS.org (2024)"},{"key":"56_CR5","doi-asserted-by":"publisher","unstructured":"Ettinger, A., Rao, S., Daum\u00e9\u00a0III, H., Bender, E.M.: Towards linguistically generalizable nlp systems: a workshop and shared task. In: Bender, E., Daum\u00e9\u00a0III, H., Ettinger, A., Rao, S. (eds.) Proceedings of the First Workshop on Building Linguistically Generalizable NLP Systems, pp. 1\u201310. Association for Computational Linguistics, Copenhagen, Denmark (Sep 2017). https:\/\/doi.org\/10.18653\/v1\/W17-5401, https:\/\/aclanthology.org\/W17-5401","DOI":"10.18653\/v1\/W17-5401"},{"key":"56_CR6","unstructured":"Huang, H., et al.: On the limitations of fine-tuned judge models for LLM evaluation (2024). https:\/\/arxiv.org\/abs\/2403.02839"},{"key":"56_CR7","doi-asserted-by":"crossref","unstructured":"Karlgren, J., et al.: Overview of ELOQUENT 2024\u2013shared tasks for evaluating generative language model quality. In: Goeuriot, L., et al. (eds.) Experimental IR Meets Multilinguality, Multimodality, and Interaction \u2013 Proceedings of the 15th International Conference of the CLEF Association (2024)","DOI":"10.1007\/978-3-031-71908-0_3"},{"key":"56_CR8","unstructured":"Karlgren, J., Talman, A.: ELOQUENT 2024 \u2013 Topical Quiz Task. In: Faggioli, G., Ferro, N., Vlachos, M., Galu\u0161\u010d\u00e1kov\u00e1, P., de\u00a0Herrera, A.G.S. (eds.) Working Notes of CLEF 2024 - Conference and Labs of the Evaluation Forum. CEUR-WS.org (2024)"},{"key":"56_CR9","unstructured":"Lambert, N., et\u00a0al.: RewardBench: Evaluating Reward Models for Language Modeling. arXiv preprint arXiv:2403.13787 (2024)"},{"key":"56_CR10","doi-asserted-by":"crossref","unstructured":"Manakul, P., Liusie, A., Gales, M.J.F.: SelfcheckGPT: Zero-resource black-box hallucination detection for generative large language models. arXiv:2303.08896 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.557"},{"key":"56_CR11","unstructured":"M\u00fcndler, N., He, J., Jenko, S., Vechev, M.: Self-contradictory hallucinations of large language models: Evaluation, detection and mitigation. arXiv:2305.15852 (2023)"},{"key":"56_CR12","unstructured":"Neralla, V., Bijl\u00a0de Vroe, S.: Evaluating Poro-34B-Chat and Mistral-7B-Instruct-v0.1: LLM system description for ELOQUENT at CLEF 2024. In: Faggioli, G., Ferro, N., Vlachos, M., Galu\u0161\u010d\u00e1kov\u00e1, P., de\u00a0Herrera, A.G.S. (eds.) Working Notes of CLEF 2024 - Conference and Labs of the Evaluation Forum. CEUR-WS.org (2024)"},{"key":"56_CR13","unstructured":"Sahlgren, M., Karlgren, J., D\u00fcrlich, L., Gogoulou, E., Talman, A., Zahra, S.: Eloquent 2024 \u2013 robustness task. In: Faggioli, G., Ferro, N., Vlachos, M., Galu\u0161\u010d\u00e1kov\u00e1, P., de\u00a0Herrera, A.G.S. (eds.) Working Notes of CLEF 2024 - Conference and Labs of the Evaluation Forum. CEUR-WS.org (2024)"},{"key":"56_CR14","unstructured":"Saunders, W., Yeh, C., Wu, J., Bills, S., Ouyang, L., Ward, J., Leike, J.: Self-critiquing models for assisting human evaluators. arXiv:2206.05802 (2022)"},{"key":"56_CR15","unstructured":"Simonsen, A.: Experimental report on robustness task - ELOQUENT Lab @ CLEF 2024. In: Faggioli, G., Ferro, N., Vlachos, M., Galu\u0161\u010d\u00e1kov\u00e1, P., de\u00a0Herrera, A.G.S. (eds.) Working Notes of CLEF 2024 - Conference and Labs of the Evaluation Forum. CEUR-WS.org (2024)"},{"key":"56_CR16","unstructured":"Tan, S., et al.: JudgeBench: A Benchmark for Evaluating LLM-based Judges. arXiv preprint arXiv:2410.12784 (2024)"},{"key":"56_CR17","unstructured":"Verga, P., et al.: Replacing judges with juries: Evaluating LLM generations with a panel of diverse models (2024). https:\/\/arxiv.org\/abs\/2404.18796"},{"key":"56_CR18","unstructured":"Wang, P., et al.: Large language models are not fair evaluators. In: Ku, L.W., Martins, A., Srikumar, V. (eds.) Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 9440\u20139450. Association for Computational Linguistics, Bangkok, Thailand, August 2024. https:\/\/aclanthology.org\/2024.acl-long.511"},{"key":"56_CR19","first-page":"46595","volume":"36","author":"L Zheng","year":"2023","unstructured":"Zheng, L., et al.: Judging LLM-as-a-judge with MT-bench and Chatbot Arena. Adv. Neural. Inf. Process. Syst. 36, 46595\u201346623 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-88720-8_56","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,4]],"date-time":"2025-04-04T12:07:03Z","timestamp":1743768423000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-88720-8_56"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031887192","9783031887208"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-88720-8_56","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"3 April 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lucca","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 April 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 April 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"47","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2025.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}