{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T14:14:27Z","timestamp":1783433667160,"version":"3.54.6"},"publisher-location":"Cham","reference-count":11,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032307095","type":"print"},{"value":"9783032307101","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T00:00:00Z","timestamp":1783468800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T00:00:00Z","timestamp":1783468800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-30710-1_44","type":"book-chapter","created":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T13:20:54Z","timestamp":1783430454000},"page":"368-377","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Can LLMs Accurately Score Medical Diagnoses and\u00a0Clinical Reasoning?"],"prefix":"10.1007","author":[{"given":"Amy","family":"Rouillard","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sitwala","family":"Mundia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linda","family":"Camara","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziyaad","family":"Dangor","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael Cameron","family":"Gramanie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ismail","family":"Kalla","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shabir A.","family":"Madhi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kajal","family":"Morar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Marlvin T.","family":"Ncube","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haroon","family":"Saloojee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bruce A.","family":"Bassett","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,8]]},"reference":[{"key":"44_CR1","unstructured":"Arora, R.K., et al.: Healthbench: evaluating large language models towards improved human health. arXiv preprint arXiv:2505.08775 (2025)"},{"key":"44_CR2","unstructured":"Bassett, B.A., et\u00a0al.: Evaluating multimodal LLMs for inpatient diagnosis: real-world performance, safety, and cost across ten frontier models. arXiv preprint arXiv:2604.16980 (2026)"},{"key":"44_CR3","unstructured":"Bedi, S., et al.: Holistic evaluation of large language models for medical tasks with MedHELM. Nat. Med.\u00a01\u20139 (2026)"},{"key":"44_CR4","doi-asserted-by":"crossref","unstructured":"Chiang, C.H., Lee, H.Y.: Can large language models be an alternative to human evaluations? In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 15607\u201315631 (2023)","DOI":"10.18653\/v1\/2023.acl-long.870"},{"key":"44_CR5","doi-asserted-by":"publisher","unstructured":"Croxford, E., et al.: Automating evaluation of AI text generation in healthcare with a large language model (LLM)-as-a-judge. medRxiv (2025). https:\/\/doi.org\/10.1101\/2025.04.22.25326219","DOI":"10.1101\/2025.04.22.25326219"},{"key":"44_CR6","doi-asserted-by":"crossref","unstructured":"Croxford, E., et al.: Evaluating clinical AI summaries with large language models as judges. NPJ Digit. Med. 8(1), 640 (2025)","DOI":"10.1038\/s41746-025-02005-2"},{"key":"44_CR7","doi-asserted-by":"publisher","unstructured":"Genovese, A., et al.: Artificial authority: the promise and perils of LLM judges in healthcare. Bioengineering 13(1) (2026). https:\/\/doi.org\/10.3390\/bioengineering13010108","DOI":"10.3390\/bioengineering13010108"},{"key":"44_CR8","doi-asserted-by":"crossref","unstructured":"Li, D., et al.: From Generation to Judgment: Opportunities and Challenges of LLM-as-a-Judge (2025)","DOI":"10.18653\/v1\/2025.emnlp-main.138"},{"key":"44_CR9","unstructured":"Reese, M.L., Zeneli, M., Ng, M., Haimes, J., Damien, A., Stade, E.: Using LLM-as-a-Judge\/Jury to Advance Scalable, Clinically-Validated Safety Evaluations of Model Responses to Users Demonstrating Psychosis. arXiv preprint arXiv:2604.02359 (2026)"},{"key":"44_CR10","doi-asserted-by":"publisher","unstructured":"Williams, G., Rutunda, S., Nzabakira, F., Mateen, B.A.: Human Evaluators vs. LLM-as-a-Judge: Toward Scalable, Real-Time Evaluation of GenAI in Global Health. medRxiv (2025). https:\/\/doi.org\/10.1101\/2025.10.27.25338910","DOI":"10.1101\/2025.10.27.25338910"},{"key":"44_CR11","doi-asserted-by":"crossref","unstructured":"Zheng, L., et al.: Judging LLM-as-a-judge with MT-bench and chatbot arena. In: Advances in Neural Information Processing Systems (NeurIPS), Datasets and Benchmarks Track (2023)","DOI":"10.52202\/075280-2020"}],"container-title":["Lecture Notes in Computer Science","Artificial Intelligence in Medicine"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-30710-1_44","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T13:21:10Z","timestamp":1783430470000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-30710-1_44"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,8]]},"ISBN":["9783032307095","9783032307101"],"references-count":11,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-30710-1_44","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,8]]},"assertion":[{"value":"8 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"The authors used ChatGPT for the formatting of LaTeX tables. After using this technology, the authors reviewed the results and take full responsibility for the contents of the manuscript.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration of AI Assistance"}},{"value":"AIME","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Artificial Intelligence in Medicine","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Ottawa, ON","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"aime2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/aime26.aimedicine.info\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}