{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T11:44:13Z","timestamp":1786103053790,"version":"3.56.0"},"reference-count":19,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,3,7]],"date-time":"2025-03-07T00:00:00Z","timestamp":1741305600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,3,7]],"date-time":"2025-03-07T00:00:00Z","timestamp":1741305600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["BMC Med Inform Decis Mak"],"DOI":"10.1186\/s12911-025-02954-4","type":"journal-article","created":{"date-parts":[[2025,3,7]],"date-time":"2025-03-07T15:08:21Z","timestamp":1741360101000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":308,"title":["A systematic review of large language model (LLM) evaluations in clinical medicine"],"prefix":"10.1186","volume":"25","author":[{"given":"Sina","family":"Shool","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sara","family":"Adimi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Reza","family":"Saboori Amleshi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ehsan","family":"Bitaraf","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Reza","family":"Golpira","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mahmood","family":"Tara","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,3,7]]},"reference":[{"key":"2954_CR1","unstructured":"Zhou H, Liu F, Gu B, Zou X, Huang J, Wu J et al. A survey of large language models in medicine: progress, application, and challenge. ArXiv Preprint. 2023;arXiv:231105112."},{"issue":"1","key":"2954_CR2","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1007\/s10916-023-01925-4","volume":"47","author":"M Cascella","year":"2023","unstructured":"Cascella M, Montomoli J, Bellini V, Bignami E. Evaluating the feasibility of ChatGPT in healthcare: an analysis of multiple clinical and research scenarios. J Med Syst. 2023;47(1):33.","journal-title":"J Med Syst"},{"key":"2954_CR3","doi-asserted-by":"crossref","unstructured":"Tustumi F, Andreollo NA, Aguilar-Nascimento, JEd. Future of the language models in healthcare: the role of chatGPT. ABCD arquivos brasileiros de cirurgia digestiva (s\u00e3o paulo). 2023;36:e1727.","DOI":"10.1590\/0102-672020230002e1727"},{"key":"2954_CR4","doi-asserted-by":"publisher","first-page":"e49324","DOI":"10.2196\/49324","volume":"25","author":"TI Wilhelm","year":"2023","unstructured":"Wilhelm TI, Roos J, Kaczmarczyk R. Large language models for therapy recommendations across 3 clinical specialties: comparative study. J Med Internet Res. 2023;25:e49324.","journal-title":"J Med Internet Res"},{"key":"2954_CR5","doi-asserted-by":"crossref","unstructured":"Lahat A, Klang E. Can advanced technologies help address the global increase in demand for specialized medical care and improve telehealth services? J Telemed Telecare. 2024;30(9).","DOI":"10.1177\/1357633X231155520"},{"key":"2954_CR6","unstructured":"Chen X, Xiang J, Lu S, Liu Y, He M, Shi D. Evaluating large language models in medical applications: a survey. ArXiv Preprint. 2024;arXiv:240507468."},{"issue":"5","key":"2954_CR7","first-page":"e39305","volume":"15","author":"M Karabacak","year":"2023","unstructured":"Karabacak M, Margetis K. Embracing large language models for medical applications: opportunities and challenges. Cureus. 2023;15(5):e39305.","journal-title":"Cureus"},{"key":"2954_CR8","doi-asserted-by":"crossref","unstructured":"Nazi ZA, Peng W. Large language models in healthcare and medical domain: A review. ArXiv Preprint. 2023;arXiv:240106775.","DOI":"10.3390\/informatics11030057"},{"key":"2954_CR9","doi-asserted-by":"publisher","first-page":"1380148","DOI":"10.3389\/fmed.2024.1380148","volume":"11","author":"A R\u00edos-Hoyo","year":"2024","unstructured":"R\u00edos-Hoyo A, Shan NL, Li A, Pearson AT, Pusztai L, Howard FM. Evaluation of large language models as a diagnostic aid for complex medical cases. Front Med. 2024;11:1380148.","journal-title":"Front Med"},{"key":"2954_CR10","unstructured":"Zhou H, Liu F, Gu B, Zou X, Huang J, Wu J et al. A survey of large language models in medicine: principles, applications, and challenges. arXiv preprint. 2023;arXiv:231105112."},{"key":"2954_CR11","doi-asserted-by":"crossref","unstructured":"Busch F, Hoffmann L, Rueger C, van Dijk EHC, Kader R, Ortiz-Prado E et al. Systematic review of large language models for patient care: current applications and challenges. medRxiv. 2024:2024.03.04.24303733.","DOI":"10.1101\/2024.03.04.24303733"},{"issue":"1","key":"2954_CR12","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1186\/s12911-024-02459-6","volume":"24","author":"Y-J Park","year":"2024","unstructured":"Park Y-J, Pillai A, Deng J, Guo E, Gupta M, Paget M, et al. Assessing the research landscape and clinical utility of large language models: a scoping review. BMC Med Inf Decis Mak. 2024;24(1):72.","journal-title":"BMC Med Inf Decis Mak"},{"issue":"10","key":"2954_CR13","doi-asserted-by":"publisher","first-page":"e2335924","DOI":"10.1001\/jamanetworkopen.2023.35924","volume":"6","author":"RH Perlis","year":"2023","unstructured":"Perlis RH, Fihn SD. Evaluating the application of large language models in clinical research contexts. JAMA Netw Open. 2023;6(10):e2335924\u2013e.","journal-title":"JAMA Netw Open"},{"key":"2954_CR14","doi-asserted-by":"publisher","first-page":"e56110","DOI":"10.2196\/56110","volume":"26","author":"JM Hoppe","year":"2024","unstructured":"Hoppe JM, Auer MK, Str\u00fcven A, Massberg S, Stremmel C. ChatGPT with GPT-4 outperforms emergency department physicians in diagnostic accuracy: retrospective analysis. J Med Internet Res. 2024;26:e56110.","journal-title":"J Med Internet Res"},{"issue":"1","key":"2954_CR15","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1007\/s44163-024-00135-2","volume":"4","author":"BP Mackey","year":"2024","unstructured":"Mackey BP, Garabet R, Maule L, Tadesse A, Cross J, Weingarten M. Evaluating ChatGPT-4 in medical education: an assessment of subject exam performance reveals limitations in clinical curriculum support for students. Discover Artif Intell. 2024;4(1):38.","journal-title":"Discover Artif Intell"},{"issue":"1","key":"2954_CR16","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1186\/s44247-023-00058-5","volume":"2","author":"D Ueda","year":"2024","unstructured":"Ueda D, Walston SL, Matsumoto T, Deguchi R, Tatekawa H, Miki Y. Evaluating GPT-4-based ChatGPT\u2019s clinical potential on the NEJM quiz. BMC Digit Health. 2024;2(1):4.","journal-title":"BMC Digit Health"},{"key":"2954_CR17","doi-asserted-by":"crossref","unstructured":"Alessandri-Bonetti MGR, Naegeli M, Liu HY, Egro FM. Assessing the soft tissue infection expertise of ChatGPT and Bard compared to IDSA recommendations. Ann Biomed Eng. 2023.","DOI":"10.1007\/s10439-023-03372-1"},{"key":"2954_CR18","unstructured":"Liu F, Zhou H, Hua Y, Rohanian O, Clifton L, Clifton DA. Large language models in healthcare: A comprehensive benchmark. MedRxiv. 2024:2024.04.24.24306315."},{"issue":"8","key":"2954_CR19","doi-asserted-by":"publisher","first-page":"1928","DOI":"10.1007\/s10439-024-03454-8","volume":"52","author":"S Garc\u00eda-M\u00e9ndez","year":"2024","unstructured":"Garc\u00eda-M\u00e9ndez S, de Arriba-P\u00e9rez F. Large language models and healthcare alliance: potential and challenges of two representative use cases. Ann Biomed Eng. 2024;52(8):1928\u201331.","journal-title":"Ann Biomed Eng"}],"container-title":["BMC Medical Informatics and Decision Making"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s12911-025-02954-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s12911-025-02954-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s12911-025-02954-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,7]],"date-time":"2025-03-07T15:08:24Z","timestamp":1741360104000},"score":1,"resource":{"primary":{"URL":"https:\/\/bmcmedinformdecismak.biomedcentral.com\/articles\/10.1186\/s12911-025-02954-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,7]]},"references-count":19,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["2954"],"URL":"https:\/\/doi.org\/10.1186\/s12911-025-02954-4","relation":{},"ISSN":["1472-6947"],"issn-type":[{"value":"1472-6947","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,7]]},"assertion":[{"value":"30 September 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 February 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 March 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The authors declare no competing interests.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"117"}}