{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T23:18:50Z","timestamp":1784675930220,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,8]],"date-time":"2024-07-08T00:00:00Z","timestamp":1720396800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Dutch Research Council (NWO)","award":["KIVI.2019.009"],"award-info":[{"award-number":["KIVI.2019.009"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,8]]},"DOI":"10.1145\/3640794.3665889","type":"proceedings-article","created":{"date-parts":[[2024,7,7]],"date-time":"2024-07-07T06:24:56Z","timestamp":1720333496000},"page":"1-5","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Our Dialogue System Sucks - but Luckily we are at the Top of the Leaderboard!: A Discussion on Current Practices in NLP Evaluation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7284-0442","authenticated-orcid":false,"given":"Anouck","family":"Braggaar","sequence":"first","affiliation":[{"name":"Department of Communication and Cognition, Tilburg School of Humanities and Digital Sciences, Tilburg University, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6593-1661","authenticated-orcid":false,"given":"Linwei","family":"He","sequence":"additional","affiliation":[{"name":"Department of Communication and Cognition, Tilburg School of Humanities and Digital Sciences, Tilburg University, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7299-7992","authenticated-orcid":false,"given":"Jan","family":"De Wit","sequence":"additional","affiliation":[{"name":"Department of Communication and Cognition, Tilburg School of Humanities and Digital Sciences, Tilburg University, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,7,8]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"International higher education42","author":"Altbach Philip","year":"2006","unstructured":"Philip Altbach. 2006. The dilemmas of ranking. International higher education42 (2006)."},{"key":"e_1_3_2_1_2_1","volume-title":"https:\/\/www.anthropic.com\/news\/introducing-claude First published on","author":"Claude Introducing","year":"2023","unstructured":"Anthropic. 2023. Introducing Claude. (2023). https:\/\/www.anthropic.com\/news\/introducing-claude First published on March 14 2023. Last checked on 27 March 2024."},{"key":"e_1_3_2_1_3_1","volume-title":"Leak","author":"Balloccu Simone","year":"2024","unstructured":"Simone Balloccu, Patr\u00edcia Schmidtov\u00e1, Mateusz Lango, and Ondrej Dusek. 2024. Leak, Cheat, Repeat: Data Contamination and Evaluation Malpractices in Closed-Source LLMs. In Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers), Yvette Graham and Matthew Purver (Eds.). Association for Computational Linguistics, St. Julian\u2019s, Malta, 67\u201393. https:\/\/aclanthology.org\/2024.eacl-long.5"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","unstructured":"R.A. Blythe and W. Croft. 2010. Can a Science\u2014Humanities Collaboration Be Successful?Adaptive Behavior 18 1 (2010) 12\u201320. https:\/\/doi.org\/10.1177\/1059712309350969 arXiv:https:\/\/doi.org\/10.1177\/1059712309350969","DOI":"10.1177\/1059712309350969"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.nlp4posimpact-1.4"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.689"},{"key":"e_1_3_2_1_7_1","volume-title":"The case for participatory evaluation. Educational evaluation and policy analysis 14, 4","author":"Cousins J\u00a0Bradley","year":"1992","unstructured":"J\u00a0Bradley Cousins and Lorna\u00a0M Earl. 1992. The case for participatory evaluation. Educational evaluation and policy analysis 14, 4 (1992), 397\u2013418."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1177\/0265407520959463"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411763.3451778"},{"key":"e_1_3_2_1_10_1","volume-title":"Low-Cost Evaluations of Designed Conversations. In International Workshop on Chatbot Research and Design. Springer, 77\u201393","author":"de Wit Jan","year":"2023","unstructured":"Jan de Wit. 2023. Leveraging Large Language Models as Simulated Users for Initial, Low-Cost Evaluations of Designed Conversations. In International Workshop on Chatbot Research and Design. Springer, 77\u201393."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1111\/cogs.13230"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.393"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00607-021-01016-7"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11747-022-00841-2"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-68213-6_2"},{"key":"e_1_3_2_1_16_1","unstructured":"Pengfei Li Jianyi Yang Mohammad\u00a0A. Islam and Shaolei Ren. 2023. Making AI Less \"Thirsty\": Uncovering and Addressing the Secret Water Footprint of AI Models. arxiv:2304.03271\u00a0[cs.LG]"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3506695"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1230"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1142\/S0219877024500287"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.26034\/cm.jostrans.2024.4706"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.evalprogplan.2013.04.006"},{"key":"e_1_3_2_1_22_1","volume-title":"https:\/\/openai.com\/blog\/chatgpt First published on","author":"Introducing AI.","year":"2022","unstructured":"OpenAI. 2022. Introducing ChatGPT. (2022). https:\/\/openai.com\/blog\/chatgpt First published on November 30 2022. Last checked on 26 February 2024."},{"key":"e_1_3_2_1_23_1","unstructured":"Chanjun Park Hyeonseok Moon Seolhwa Lee Jaehyung Seo Sugyeong Eo and Heuiseok Lim. 2023. Self-Improving-Leaderboard(SIL): A Call for Real-World Centric Natural Language Processing Leaderboards. arxiv:2303.10888\u00a0[cs.CL]"},{"key":"e_1_3_2_1_24_1","unstructured":"Antonio Peque\u00f1o\u00a0IV. 2024. Google\u2019s Gemini Controversy Explained: AI Model Criticized By Musk and Others Over Alleged Bias. (2024). https:\/\/www.forbes.com\/sites\/antoniopequenoiv\/2024\/02\/26\/googles-gemini-controversy-explained-ai-model-criticized-by-musk-and-others-over-alleged-bias\/?sh=1b61569d4b99 First published on February 26 2024. Last checked on 27 March 2024."},{"key":"e_1_3_2_1_25_1","volume-title":"Conceptualising engagement with digital behaviour change interventions: a systematic review using principles from critical interpretive synthesis. Translational behavioral medicine 7, 2","author":"Perski Olga","year":"2017","unstructured":"Olga Perski, Ann Blandford, Robert West, and Susan Michie. 2017. Conceptualising engagement with digital behaviour change interventions: a systematic review using principles from critical interpretive synthesis. Translational behavioral medicine 7, 2 (2017), 254\u2013267."},{"key":"e_1_3_2_1_26_1","volume-title":"Introducing Gemini: our largest and most capable AI model. (2023). https:\/\/blog.google\/technology\/ai\/google-gemini-ai\/#sundar-note First published on","author":"Pichai Sundar","year":"2023","unstructured":"Sundar Pichai and Demis Hassabis. 2023. Introducing Gemini: our largest and most capable AI model. (2023). https:\/\/blog.google\/technology\/ai\/google-gemini-ai\/#sundar-note First published on December 6 2023. Last checked on 27 March 2024."},{"key":"e_1_3_2_1_27_1","unstructured":"Evgeniia Razumovskaia Ivan Vuli\u0107 and Anna Korhonen. 2024. Analyzing and Adapting Large Language Models for Few-Shot Multilingual NLU: Are We There Yet?arxiv:2403.01929\u00a0[cs.CL]"},{"key":"e_1_3_2_1_28_1","unstructured":"Ehud Reiter. 2022. I dont like leaderboards. https:\/\/ehudreiter.com\/2022\/11\/28\/leaderboards\/"},{"key":"e_1_3_2_1_29_1","volume-title":"Interdisciplinary research: Trend or transition. Items and issues 5, 1-2","author":"Rhoten Diana","year":"2004","unstructured":"Diana Rhoten. 2004. Interdisciplinary research: Trend or transition. Items and issues 5, 1-2 (2004), 6\u201311."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1021\/acs.est.3c01106"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.346"},{"key":"e_1_3_2_1_32_1","unstructured":"Anna Rogers. 2019. How the Transformers broke NLP leaderboards. https:\/\/hackingsemantics.xyz\/2019\/leaderboards\/"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-short.85"},{"key":"e_1_3_2_1_34_1","unstructured":"David Schlangen. 2023. Dialogue Games for Benchmarking Language Understanding: Motivation Taxonomy Strategy. arxiv:2304.07007\u00a0[cs.CL]"},{"key":"e_1_3_2_1_35_1","unstructured":"David Schlangen. 2023. What A Situated Language-Using Agent Must be Able to Do: A Top-Down Analysis. arxiv:2302.08590\u00a0[cs.CL]"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/175208.175217"},{"key":"e_1_3_2_1_37_1","volume-title":"Why \u2018containment rate","author":"Simms Kane","year":"2021","unstructured":"Kane Simms. 2021. Why \u2018containment rate\u2019 is NOT the best way to measure your chatbot or voicebot. (2021). https:\/\/vux.world\/why-containment-rate-is-not-the-best-way-to-measure-your-chatbot-or-voicebot\/ First published on February 25, 2021. Last checked on 1 April 2024.."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0119)","author":"Sivaprasad Prabhu\u00a0Teja","year":"2020","unstructured":"Prabhu\u00a0Teja Sivaprasad, Florian Mai, Thijs Vogels, Martin Jaggi, and Fran\u00e7ois Fleuret. 2020. Optimizer Benchmarking Needs to Account for Hyperparameter Tuning. In Proceedings of the 37th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0119), Hal\u00a0Daum\u00e9 III and Aarti Singh (Eds.). PMLR, 9036\u20139045. https:\/\/proceedings.mlr.press\/v119\/sivaprasad20a.html"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3571884.3597144"},{"key":"e_1_3_2_1_40_1","volume-title":"The methodology of participatory design. Technical communication 52, 2","author":"Spinuzzi Clay","year":"2005","unstructured":"Clay Spinuzzi. 2005. The methodology of participatory design. Technical communication 52, 2 (2005), 163\u2013174."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.368"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2020.101151"},{"key":"e_1_3_2_1_43_1","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric.\u00a0P Xing Hao Zhang Joseph\u00a0E. Gonzalez and Ion Stoica. 2023. Judging LLM-as-a-judge with MT-Bench and Chatbot Arena. arxiv:2306.05685\u00a0[cs.CL]"}],"event":{"name":"CUI '24: ACM Conversational User Interfaces 2024","location":"Luxembourg Luxembourg","acronym":"CUI '24","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["ACM Conversational User Interfaces 2024"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640794.3665889","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3640794.3665889","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T18:04:13Z","timestamp":1755885853000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640794.3665889"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,8]]},"references-count":43,"alternative-id":["10.1145\/3640794.3665889","10.1145\/3640794"],"URL":"https:\/\/doi.org\/10.1145\/3640794.3665889","relation":{},"subject":[],"published":{"date-parts":[[2024,7,8]]},"assertion":[{"value":"2024-07-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}