{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:15:25Z","timestamp":1784178925169,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","funder":[{"name":"Research Council of Norway","award":["309834"],"award-info":[{"award-number":["309834"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,7]]},"DOI":"10.1145\/3767695.3769478","type":"proceedings-article","created":{"date-parts":[[2025,12,3]],"date-time":"2025-12-03T17:14:58Z","timestamp":1764782098000},"page":"261-271","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Limitations of Current Evaluation Practices for Conversational Recommender Systems and the Potential of User Simulation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-0565-3210","authenticated-orcid":false,"given":"Nolwenn","family":"Bernard","sequence":"first","affiliation":[{"name":"TH K\u00f6ln, Cologne, Germany and University of Stavanger, Stavanger, Norway"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2762-721X","authenticated-orcid":false,"given":"Krisztian","family":"Balog","sequence":"additional","affiliation":[{"name":"University of Stavanger, Stavanger, Norway"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,12,6]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3616855.3635856"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539597.3573029"},{"key":"e_1_3_2_1_3_1","volume-title":"Computational Interaction","author":"Azzopardi Leif","unstructured":"Leif Azzopardi and Guido Zuccon. 2018. Economic Models of Interaction. In Computational Interaction. Oxford University Press."},{"key":"e_1_3_2_1_4_1","volume-title":"Building Economic Models of Human Computer Interaction. In Extended Abstracts of the 2019 CHI Conference on Human Factors in Computing Systems (CHI EA '19). 1-4.","author":"Azzopardi Leif","year":"2019","unstructured":"Leif Azzopardi and Guido Zuccon. 2019. Building Economic Models of Human Computer Interaction. In Extended Abstracts of the 2019 CHI Conference on Human Factors in Computing Systems (CHI EA '19). 1-4."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1561\/1500000098"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/2043932.2043996"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","first-page":"610","DOI":"10.1145\/3442188.3445922","volume-title":"Proceedings of the 2021 ACM Conference on Fairness, Accountability, and Transparency (FAccT '21)","author":"Bender Emily M.","year":"2021","unstructured":"Emily M. Bender, Timnit Gebru, Angelina McMillan-Major, and Shmargaret Shmitchell. 2021. On the Dangers of Stochastic Parrots: Can Language Models Be Too Big?. In Proceedings of the 2021 ACM Conference on Fairness, Accountability, and Transparency (FAccT '21). 610-623."},{"key":"e_1_3_2_1_8_1","volume-title":"The 10th ACM SIGIR \/ The 14th International Conference on the Theory of Information Retrieval (ICTIR '24)","author":"Bernard Nolwenn","year":"2024","unstructured":"Nolwenn Bernard and Krisztian Balog. 2024. Towards a Formal Characterization of User Simulation Objectives in Conversational Information Access. In The 10th ACM SIGIR \/ The 14th International Conference on the Theory of Information Retrieval (ICTIR '24)."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the Eighteenth ACM International Conference on Web Search and Data Mining (WSDM '25').","author":"Bernard Nolwenn","year":"2025","unstructured":"Nolwenn Bernard, Hideaki Joko, Faegheh Hasibi, and Krisztian Balog. 2025. CRS Arena: Crowdsourced Benchmarking of Conversational Recommender Systems. In Proceedings of the Eighteenth ACM International Conference on Web Search and Data Mining (WSDM '25')."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340631.3394856"},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP '19)","author":"Chen Qibin","year":"2019","unstructured":"Qibin Chen, Junyang Lin, Yichang Zhang, Ming Ding, Yukuo Cen, Hongxia Yang, and Jie Tang. 2019. Towards Knowledge-Based Recommender Dialog System. In Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP '19). 1803-1813."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","first-page":"15607","DOI":"10.18653\/v1\/2023.acl-long.870","volume-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (ACL '23)","author":"Chiang Cheng-Han","year":"2023","unstructured":"Cheng-Han Chiang and Hung-yi Lee. 2023. Can Large Language Models Be an Alternative to Human Evaluations?. In Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (ACL '23). 15607-15631."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/1864708.1864721"},{"key":"e_1_3_2_1_14_1","first-page":"1285","volume-title":"Vote Goat: Conversational Movie Recommendation. In The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval (SIGIR '18)","author":"Dalton Jeffrey","year":"2018","unstructured":"Jeffrey Dalton, Victor Ajayi, and Richard Main. 2018. Vote Goat: Conversational Movie Recommendation. In The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval (SIGIR '18). 1285-1288."},{"key":"e_1_3_2_1_15_1","unstructured":"Luke Friedman Sameer Ahuja David Allen Zhenning Tan Hakim Sidahmed Changbo Long Jun Xie Gabriel Schubiner Ajay Patel Harsh Lara Brian Chu Zexi Chen and Manoj Tiwari. 2023. Leveraging Large Language Models in Conversational Recommender Systems. arXiv:2305.07961 [cs.IR]"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2021.06.002"},{"key":"e_1_3_2_1_17_1","first-page":"3405","volume-title":"Proceedings of the 29th ACM International Conference on Information & Knowledge Management (CIKM '20)","author":"Habib Javeria","year":"2020","unstructured":"Javeria Habib, Shuo Zhang, and Krisztian Balog. 2020. IAI MovieBot: A Conversational Movie Recommender System. In Proceedings of the 29th ACM International Conference on Information & Knowledge Management (CIKM '20). 3405-3408."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","first-page":"13806","DOI":"10.18653\/v1\/2024.acl-long.745","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (ACL '24)","author":"Hashemi Helia","year":"2024","unstructured":"Helia Hashemi, Jason Eisner, Corby Rosset, Benjamin Van Durme, and Chris Kedzie. 2024. LLM-Rubric: A Multidimensional, Calibrated Approach to Automated Evaluation of Natural Language Texts. In Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (ACL '24). 13806-13834."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614949"},{"key":"e_1_3_2_1_20_1","unstructured":"Chen Huang Peixin Qin Yang Deng Wenqiang Lei Jiancheng Lv and Tat-Seng Chua. 2024. Concept - An Evaluation Protocol on Conversation Recommender Systems with System-centric and User-centric Factors. arXiv:2404.03304 [cs.CL]"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-022-10229-x"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453154"},{"key":"e_1_3_2_1_23_1","volume-title":"Martin","author":"Jurafsky Dan","year":"2023","unstructured":"Dan Jurafsky and James H. Martin. 2023. Chatbots and Dialogue Systems. In Speech and Language Processing (3rd edition ed.). Chapter 15."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3336191.3371769"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.5555\/3327546.3327641"},{"key":"e_1_3_2_1_27_1","first-page":"270","volume-title":"Proceedings of the 23rd Annual Meeting of the Special Interest Group on Discourse and Dialogue (SIGDIAL '22)","author":"Geishauser Christian","year":"2022","unstructured":"Hsien-chin Lin, Christian Geishauser, Shutong Feng, Nurul Lubis, Carel van Niekerk, Michael Heck, and Milica Gasic. 2022. GenTUS: Simulating User Behaviour and Language in Task-oriented Dialogues with Generative Transformers. In Proceedings of the 23rd Annual Meeting of the Special Interest Group on Discourse and Dialogue (SIGDIAL '22). 270-282."},{"key":"e_1_3_2_1_28_1","first-page":"445","volume-title":"Proceedings of the 22nd Annual Meeting of the Special Interest Group on Discourse and Dialogue (SIGDIAL '21)","author":"Lubis Nurul","year":"2021","unstructured":"Hsien-chin Lin, Nurul Lubis, Songbo Hu, Carel van Niekerk, Christian Geishauser, Michael Heck, Shutong Feng, and Milica Gasic. 2021. Domain-independent User Simulation with Transformers for Task-oriented Dialogue Systems. In Proceedings of the 22nd Annual Meeting of the Special Interest Group on Discourse and Dialogue (SIGDIAL '21). 445-456."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3460231.3475942"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.is.2022.102083"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627043.3659574"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1081"},{"key":"e_1_3_2_1_33_1","unstructured":"Arvind Neelakantan Tao Xu Raul Puri Alec Radford Jesse Michael Han Jerry Tworek Qiming Yuan Nikolas Tezak Jong Wook Kim Chris Hallacy Johannes Heidecke Pranav Shyam Boris Power Tyna Eloundou Nekoul Girish Sastry Gretchen Krueger David Schnurr Felipe Petroski Such Kenny Hsu Madeleine Thompson Tabarak Khan Toki Sherbakov Joanne Jang Peter Welinder and Lilian Weng. 2022. Text and Code Embeddings by Contrastive Pre-Training. arXiv:2201.10005 [cs.CL]"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.117539"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","first-page":"149","DOI":"10.3115\/1614108.1614146","volume-title":"Proc. of NAACL '07","author":"Schatzmann Jost","year":"2007","unstructured":"Jost Schatzmann, Blaise Thomson, Karl Weilhammer, Hui Ye, and Steve Young. 2007. Agenda-Based User Simulation for Bootstrapping a POMDP Dialogue System. In Proc. of NAACL '07. 149-152."},{"key":"e_1_3_2_1_36_1","first-page":"19","volume-title":"Proceedings of the 1st Workshop on Simulating Conversational Intelligence in Chat (SCI-CHAT '24)","author":"Sekulic Ivan","year":"2024","unstructured":"Ivan Sekulic, Silvia Terragni, Victor Guimar aes, Nghia Khau, Bruna Guedes, Modestas Filipavicius, Andre Ferreira Manso, and Roland Mathis. 2024. Reliable LLM-based User Simulator for Task-Oriented Dialogue Systems. In Proceedings of the 1st Workshop on Simulating Conversational Intelligence in Chat (SCI-CHAT '24). 19-35."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/2507157.2507160"},{"key":"e_1_3_2_1_38_1","volume-title":"Konstan","author":"Sun Ruixuan","year":"2024","unstructured":"Ruixuan Sun, Xinyi Li, Avinash Akella, and Joseph A. Konstan. 2024. Large Language Models as Conversational Movie Recommenders: A User Study. arXiv:2404.19093 [cs.IR]"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463241"},{"key":"e_1_3_2_1_40_1","first-page":"235","volume-title":"Conversational Recommender System. In The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval (SIGIR '18)","author":"Sun Yueming","year":"2018","unstructured":"Yueming Sun and Yi Zhang. 2018. Conversational Recommender System. In The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval (SIGIR '18). 235-244."},{"key":"e_1_3_2_1_41_1","unstructured":"Silvia Terragni Modestas Filipavicius Nghia Khau Bruna Guedes Andr\u00e9 Manso and Roland Mathis. 2023. In-Context Learning User Simulators for Task-Oriented Dialog Systems. arXiv:2306.00774 [cs.CL]"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622467.1622479"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240323.3240347"},{"key":"e_1_3_2_1_44_1","unstructured":"Maria Vlachou and Craig Macdonald. 2024. What Else Would I Like? A User Simulator using Alternatives for Improved Evaluation of Fashion Conversational Recommendation Systems. arXiv:2401.05783 [cs.IR]"},{"key":"e_1_3_2_1_45_1","volume-title":"BARCOR: Towards A Unified Framework for Conversational Recommendation Systems. arXiv:2203.14257 [cs.CL]","author":"Wang Ting-Chun","year":"2022","unstructured":"Ting-Chun Wang, Shang-Yu Su, and Yun-Nung Chen. 2022a. BARCOR: Towards A Unified Framework for Conversational Recommendation Systems. arXiv:2203.14257 [cs.CL]"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.621"},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining (KDD '22)","author":"Wang Xiaolei","year":"2022","unstructured":"Xiaolei Wang, Kun Zhou, Ji-Rong Wen, and Wayne Xin Zhao. 2022b. Towards Unified Conversational Recommender Systems via Knowledge-Enhanced Prompt Learning. In Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining (KDD '22). 1929-1937."},{"key":"e_1_3_2_1_48_1","first-page":"1","article-title":"A Survey of Joint Intent Detection and Slot Filling Models in Natural Language","volume":"55","author":"Weld Henry","year":"2022","unstructured":"Henry Weld, Xiaoqi Huang, Siqu Long, Josiah Poon, and Soyeon Caren Han. 2022. A Survey of Joint Intent Detection and Slot Filling Models in Natural Language Understanding. Comput. Surveys, Vol. 55, 8 (2022), 1-38.","journal-title":"Understanding. Comput. Surveys"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437963.3441791"},{"key":"e_1_3_2_1_50_1","first-page":"1490","volume-title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers) (NAACL '24)","author":"He Zhankui","year":"2024","unstructured":"Se-eun Yoon, Zhankui He, Jessica Echterhoff, and Julian McAuley. 2024. Evaluating Large Language Models as Generative User Simulators for Conversational Recommendation. In Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers) (NAACL '24). 1490-1504."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2023.3322403"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403202"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","first-page":"270","DOI":"10.18653\/v1\/2020.acl-demos.30","volume-title":"Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics: System Demonstrations (ACL '20)","author":"Zhang Yizhe","year":"2020","unstructured":"Yizhe Zhang, Siqi Sun, Michel Galley, Yen-Chun Chen, Chris Brockett, Xiang Gao, Jianfeng Gao, Jingjing Liu, and Bill Dolan. 2020. DIALOGPT : Large-Scale Generative Pre-training for Conversational Response Generation. In Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics: System Demonstrations (ACL '20). 270-278."},{"key":"e_1_3_2_1_54_1","volume-title":"Yaliang Li, and Ji-Rong Wen.","author":"Zhou Kun","year":"2021","unstructured":"Kun Zhou, Xiaolei Wang, Yuanhang Zhou, Chenzhan Shang, Yuan Cheng, Wayne Xin Zhao, Yaliang Li, and Ji-Rong Wen. 2021. CRSLab: An Open-Source Toolkit for Building Conversational Recommender System. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing: System Demonstrations (ACL-IJCNLP '21). 185-193."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403143"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"crossref","first-page":"1488","DOI":"10.1145\/3488560.3498514","volume-title":"Proceedings of the Fifteenth ACM International Conference on Web Search and Data Mining (WSDM '22)","author":"Zhou Yuanhang","year":"2022","unstructured":"Yuanhang Zhou, Kun Zhou, Wayne Xin Zhao, Cheng Wang, Peng Jiang, and He Hu. 2022. C\u00b2-CRS: Coarse-to-Fine Contrastive Learning for Conversational Recommender System. In Proceedings of the Fifteenth ACM International Conference on Web Search and Data Mining (WSDM '22). 1488-1496."}],"event":{"name":"SIGIR-AP 2025:Annual International ACM SIGIR Conference on Research and Development in Information Retrieval in the Asia Pacific Region","location":"Xi'an China","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 2025 Annual International ACM SIGIR Conference on Research and Development in Information Retrieval in the Asia Pacific Region"],"original-title":[],"deposited":{"date-parts":[[2025,12,3]],"date-time":"2025-12-03T17:19:15Z","timestamp":1764782355000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3767695.3769478"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,6]]},"references-count":56,"alternative-id":["10.1145\/3767695.3769478","10.1145\/3767695"],"URL":"https:\/\/doi.org\/10.1145\/3767695.3769478","relation":{},"subject":[],"published":{"date-parts":[[2025,12,6]]},"assertion":[{"value":"2025-12-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}