{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T16:01:09Z","timestamp":1780329669004,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":21,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,7]],"date-time":"2026-06-07T00:00:00Z","timestamp":1780790400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,8]]},"DOI":"10.1145\/3774935.3806172","type":"proceedings-article","created":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T14:44:13Z","timestamp":1780325053000},"page":"351-355","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Can Third-Party Annotators Reliably Evaluate Conversational Recommender Systems?"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-0918-8833","authenticated-orcid":false,"given":"Michael","family":"M\u00fcller","sequence":"first","affiliation":[{"name":"Department of Computer Science, University of Innsbruck, Innsbruck, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3934-6941","authenticated-orcid":false,"given":"Amir Reza","family":"Mohammadi","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Innsbruck, Innsbruck, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7337-524X","authenticated-orcid":false,"given":"Andreas","family":"Peintner","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Innsbruck, Innsbruck, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9111-5402","authenticated-orcid":false,"given":"Beatriz","family":"Barroso Gstrein","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Innsbruck, Innsbruck, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0978-7201","authenticated-orcid":false,"given":"G\u00fcnther","family":"Specht","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Innsbruck, Innsbruck, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3195-8273","authenticated-orcid":false,"given":"Eva","family":"Zangerle","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Innsbruck, Innsbruck, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,7]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"Ron Artstein and Massimo Poesio. 2008. Inter-coder agreement for computational linguistics. Computational linguistics 34 4 (2008) 555\u2013596.","DOI":"10.1162\/coli.07-034-R2"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","unstructured":"Christine Bauer Li Chen Nicola Ferro and Norbert Fuhr. 2025. Conversational Agents: A Framework for Evaluation (CAFE) (Dagstuhl Perspectives Workshop 24352). Dagstuhl Reports 14 8 (2025) 53\u201358. 10.4230\/DagRep.14.8.53","DOI":"10.4230\/DagRep.14.8.53"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"crossref","unstructured":"Nolwenn Bernard and Krisztian Balog. 2025. Limitations of Current Evaluation Practices for Conversational Recommender Systems and the Potential of User Simulation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.05624 (2025).","DOI":"10.1145\/3767695.3769478"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","unstructured":"Nuo Chen Quanyu Dai Xiaoyu Dong Piaohong Wang Qinglin Jia Zhaocheng Du Zhenhua Dong and Xiao-Ming Wu. 2025. Evaluating Conversational Recommender Systems via Large Language Models: A User-Centric Framework. arxiv:https:\/\/arXiv.org\/abs\/2501.09493\u00a0[cs.IR] 10.48550\/arxiv.2501.09493","DOI":"10.48550\/arxiv.2501.09493"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","unstructured":"Lukas Gienapp Tim Hagen Maik Fr\u00f6be Matthias Hagen Benno Stein Martin Potthast and Harrisen Scells. 2025. The Viability of Crowdsourcing for RAG Evaluation. (2025). arxiv:https:\/\/arXiv.org\/abs\/2504.15689\u00a0[cs.IR] 10.48550\/arXiv.2504.15689arXiv preprint.","DOI":"10.48550\/arXiv.2504.15689"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","unstructured":"Chen Huang Peixin Qin Yang Deng Wenqiang Lei Jiancheng Lv and Tat-Seng Chua. 2024. Concept \u2013 An Evaluation Protocol on Conversational Recommender Systems with System-centric and User-centric Factors. arxiv:https:\/\/arXiv.org\/abs\/2404.03304\u00a0[cs.CL] 10.48550\/arxiv.2404.03304","DOI":"10.48550\/arxiv.2404.03304"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","unstructured":"Dietmar Jannach. 2023. Evaluating conversational recommender systems. Artificial Intelligence Review 56 3 (mar 2023) 2365\u20132400. 10.1007\/s10462-022-10229-x","DOI":"10.1007\/s10462-022-10229-x"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","unstructured":"Dietmar Jannach Ahtsham Manzoor Wanling Cai and Li Chen. 2021. A Survey on Conversational Recommender Systems. ACM Comput. Surv. 54 5 (may 2021) 105:1\u2013105:36. 10.1145\/3453154","DOI":"10.1145\/3453154"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","unstructured":"Yucheng Jin Li Chen Wanling Cai and Xianglin Zhao. 2024. CRS-Que: A User-centric Evaluation Framework for Conversational Recommender Systems. ACM Trans. Recomm. Syst. 2 1 Article 2 (March 2024) 34\u00a0pages. 10.1145\/3631534","DOI":"10.1145\/3631534"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","unstructured":"Terry\u00a0K. Koo and Mae\u00a0Y. Li. 2016. A Guideline of Selecting and Reporting Intraclass Correlation Coefficients for Reliability Research. Journal of Chiropractic Medicine 15 2 (2016) 155\u2013163. 10.1016\/j.jcm.2016.02.012","DOI":"10.1016\/j.jcm.2016.02.012"},{"key":"e_1_3_3_1_12_2","unstructured":"Klaus Krippendorff. 2011. Computing Krippendorff\u2019s alpha-reliability. (2011)."},{"key":"e_1_3_3_1_13_2","series-title":"(NIPS\u201918)","first-page":"9748","volume-title":"Advances in Neural Information Processing Systems 31 (NIPS 2018)","author":"Li Raymond","year":"2018","unstructured":"Raymond Li, Samira Kahou, Hannes Schulz, Vincent Michalski, Laurent Charlin, and Chris Pal. 2018. Towards Deep Conversational Recommendations. In Advances in Neural Information Processing Systems 31 (NIPS 2018) (Montr\u00e9al, Canada) (NIPS\u201918). Curran Associates Inc., Red Hook, NY, USA, 9748\u20139758."},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/3627043.3659574"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"crossref","unstructured":"Kenneth\u00a0O McGraw and Seok\u00a0P Wong. 1996. Forming inferences about some intraclass correlation coefficients. Psychological methods 1 1 (1996) 30.","DOI":"10.1037\/1082-989X.1.1.30"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3705328.3748759"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/2043932.2043962"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Patrick\u00a0E Shrout and Joseph\u00a0L Fleiss. 1979. Intraclass correlations: uses in assessing rater reliability. Psychological bulletin 86 2 (1979) 420.","DOI":"10.1037\/\/0033-2909.86.2.420"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"crossref","unstructured":"SD Walter M Eliasziw and A Donner. 1998. Sample size and optimal designs for reliability studies. Statistics in medicine 17 1 (1998) 101\u2013110.","DOI":"10.1002\/(SICI)1097-0258(19980115)17:1<101::AID-SIM727>3.0.CO;2-E"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","unstructured":"Lingzhi Wang Shafiq Joty Wei Gao Xingshan Zeng and Kam-Fai Wong. 2024. Improving Conversational Recommender System Via Contextual and Time-Aware Modeling With Less Domain-Specific Knowledge. IEEE Transactions on Knowledge and Data Engineering 36 11 (2024) 6447\u20136461. 10.1109\/TKDE.2024.3397321","DOI":"10.1109\/TKDE.2024.3397321"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","unstructured":"Sojeong Yun and Youn-kyung Lim. 2025. User Experience with LLM-powered Conversational Recommendation Systems: A Case of Music Recommendation. arxiv:https:\/\/arXiv.org\/abs\/2502.15229\u00a0[cs] 10.1145\/3706598.3713347","DOI":"10.1145\/3706598.3713347"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3640457.3688133"}],"event":{"name":"UMAP '26: 34th ACM Conference on User Modeling, Adaptation and Personalization","location":"Gothenburg , Sweden","acronym":"UMAP '26","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 34th ACM Conference on User Modeling, Adaptation and Personalization"],"original-title":[],"deposited":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T15:08:49Z","timestamp":1780326529000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774935.3806172"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,7]]},"references-count":21,"alternative-id":["10.1145\/3774935.3806172","10.1145\/3774935"],"URL":"https:\/\/doi.org\/10.1145\/3774935.3806172","relation":{},"subject":[],"published":{"date-parts":[[2026,6,7]]},"assertion":[{"value":"2026-06-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}