{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T10:22:50Z","timestamp":1783074170687,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":22,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T00:00:00Z","timestamp":1783123200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,5]]},"DOI":"10.1145\/3786579.3804927","type":"proceedings-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T15:20:12Z","timestamp":1783005612000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Evaluating the Evaluator: LLM Evaluators for Patient-Facing AI Agents"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8179-0101","authenticated-orcid":false,"given":"Angela","family":"Mastrianni","sequence":"first","affiliation":[{"name":"Department of Population Health, NYU Grossman School of Medicine, New York, New York, USA and Department of Health Informatics, NYU Langone Health, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8586-450X","authenticated-orcid":false,"given":"Katerina","family":"Andreadis","sequence":"additional","affiliation":[{"name":"Department of Population Health, NYU Grossman School of Medicine, New York, New York, USA and Department of Health Informatics, NYU Langone Health, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4856-0141","authenticated-orcid":false,"given":"Ji","family":"Chen","sequence":"additional","affiliation":[{"name":"Medical Center Information Technology, NYU Langone Health, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4642-6798","authenticated-orcid":false,"given":"Danissa V","family":"Rodriguez","sequence":"additional","affiliation":[{"name":"Department of Population Health, NYU Grossman School of Medicine, New York, New York, USA and Department of Health Informatics, NYU Langone Health, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2099-0852","authenticated-orcid":false,"given":"Devin","family":"Mann","sequence":"additional","affiliation":[{"name":"Department of Population Health, NYU Grossman School of Medicine, New York, New York, USA and Department of Health Informatics, NYU Langone Health, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,4]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-48229-6_2"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.3233\/SHTI240565"},{"key":"e_1_3_3_1_4_2","unstructured":"Rahul\u00a0K Arora Jason Wei Rebecca\u00a0Soskin Hicks Preston Bowman Joaquin Qui\u00f1onero-Candela Foivos Tsimpourlas Michael Sharman Meghan Shah Andrea Vallone Alex Beutel et\u00a0al. 2025. Healthbench: Evaluating large language models towards improved human health. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.08775 (2025)."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Karthik\u00a0S Bhat Mohit Jain and Neha Kumar. 2021. Infrastructuring telehealth in (in) formal patient-doctor contexts. Proceedings of the ACM on Human-Computer Interaction 5 CSCW2 (2021) 1\u201328.","DOI":"10.1145\/3476064"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.63317\/4b94kue5ke3h"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Emma Croxford Yanjun Gao Elliot First Nicholas Pellegrino Miranda Schnier John Caskey Madeline Oguss Graham Wills Guanhua Chen Dmitriy Dligach et\u00a0al. 2025. Evaluating clinical AI summaries with large language models as judges. npj Digital Medicine 8 1 (2025) 640.","DOI":"10.1038\/s41746-025-02005-2"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.958"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.bionlp-1.19"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"crossref","unstructured":"Ariana Genovese Lars Hegstrom Srinivasagam Prabha Cesar\u00a0A Gomez-Cabello Syed\u00a0Ali Haider Bernardo Collaco Nadia\u00a0G Wood and Antonio\u00a0Jorge Forte. 2026. Artificial Authority: The Promise and Perils of LLM Judges in Healthcare. Bioengineering 13 1 (2026) 108.","DOI":"10.3390\/bioengineering13010108"},{"key":"e_1_3_3_1_11_2","unstructured":"Jiawei Gu Xuhui Jiang Zhichao Shi Hexiang Tan Xuehao Zhai Chengjin Xu Wei Li Yinghan Shen Shengjie Ma Honghao Liu et\u00a0al. 2024. A survey on llm-as-a-judge. The Innovation (2024)."},{"key":"e_1_3_3_1_12_2","unstructured":"Bill\u00a0Yuchen Lin Yuntian Deng Khyathi Chandu Faeze Brahman Abhilasha Ravichander Valentina Pyatkin Nouha Dziri Ronan\u00a0Le Bras and Yejin Choi. 2024. Wildbench: Benchmarking llms with challenging tasks from real users in the wild. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.04770 (2024)."},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"crossref","unstructured":"Angela Mastrianni Paige Kmetz-Cutrone Kathryn Chang Jonathan\u00a0Y Stein and Aleksandra Sarcevic. 2025. Beyond Decision Making: Considering Collaboration and Agency in the Design of AI-Based Decision-Support Systems for Fast-Response Medical Teams. Proceedings of the ACM on human-computer interaction 9 7 (2025) 1\u201328.","DOI":"10.1145\/3757457"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/3733816.3760760"},{"key":"e_1_3_3_1_15_2","unstructured":"Lorelli\u00a0S Nowell Jill\u00a0M Norris Deborah\u00a0E White and Nancy\u00a0J Moules. 2017. Thematic analysis: Striving to meet the trustworthiness criteria. International journal of qualitative methods 16 1 (2017) 1609406917733847."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.hucllm-1.2"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICHI61247.2024.00052"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Nina Singh Katharine Lawrence Katerina Andreadis Chelsea Twan and Devin\u00a0M Mann. 2024. Developing and scaling remote patient monitoring capacity in ambulatory practice. NEJM Catalyst Innovations in Care Delivery 5 6 (2024) CAT\u201323.","DOI":"10.1056\/CAT.23.0417"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3708359.3712091"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/2675133.2675277"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Ziyu Wang Hao Li Di Huang Hye-Sung Kim Chae-Won Shin and Amir\u00a0M Rahmani. 2025. Healthq: Unveiling questioning capabilities of llm chains in healthcare conversations. Smart Health (2025) 100570.","DOI":"10.1016\/j.smhl.2025.100570"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Liu Yang Iter Dan Yichong Xu Wang Shuohang Ruochen Xu and Zhu Chenguang. 2023. Gpteval: Nlg evaluation using gpt-4 with better human alignment. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.16634 (2023).","DOI":"10.18653\/v1\/2023.emnlp-main.153"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric Xing et\u00a0al. 2023. Judging llm-as-a-judge with mt-bench and chatbot arena. Advances in neural information processing systems 36 (2023) 46595\u201346623.","DOI":"10.52202\/075280-2020"}],"event":{"name":"IH '26: Interactive Health Conference","location":"Porto , Portugal","acronym":"IH '26","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2026 ACM Interactive Health Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3786579.3804927","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T10:06:00Z","timestamp":1783073160000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3786579.3804927"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,4]]},"references-count":22,"alternative-id":["10.1145\/3786579.3804927","10.1145\/3786579"],"URL":"https:\/\/doi.org\/10.1145\/3786579.3804927","relation":{},"subject":[],"published":{"date-parts":[[2026,7,4]]},"assertion":[{"value":"2026-07-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}