{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,19]],"date-time":"2026-08-19T18:06:43Z","timestamp":1787162803655,"version":"build-2736575974"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,11,21]],"date-time":"2025-11-21T00:00:00Z","timestamp":1763683200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,21]],"date-time":"2025-11-21T00:00:00Z","timestamp":1763683200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Health Inf Sci Syst"],"DOI":"10.1007\/s13755-025-00397-9","type":"journal-article","created":{"date-parts":[[2025,11,21]],"date-time":"2025-11-21T13:12:58Z","timestamp":1763730778000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Large language models in healthcare: a systematic evaluation on medical Q\/A datasets"],"prefix":"10.1007","volume":"14","author":[{"given":"Khuzaima","family":"Tofeeq","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Asma","family":"Naseer","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5314-6113","authenticated-orcid":false,"given":"Aamir","family":"Wali","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,11,21]]},"reference":[{"key":"397_CR1","unstructured":"Brown TB, et al. Language models are few-shot learners. 2020;33:1877\u2013901 arXiv:2005.14165."},{"issue":"20","key":"397_CR2","doi-asserted-by":"publisher","first-page":"2776","DOI":"10.3390\/healthcare11202776","volume":"11","author":"P Yu","year":"2023","unstructured":"Yu P, Xu H, Hu X, Deng C. Leveraging generative ai and large language models: a comprehensive roadmap for healthcare integration. Healthcare. 2023;11(20):2776. https:\/\/doi.org\/10.3390\/healthcare11202776 (https:\/\/www.mdpi.com\/2227-9032\/11\/20\/2776).","journal-title":"Healthcare"},{"issue":"3","key":"397_CR3","doi-asserted-by":"publisher","first-page":"57","DOI":"10.3390\/informatics11030057","volume":"11","author":"ZA Nazi","year":"2024","unstructured":"Nazi ZA, Peng W. Large language models in healthcare and medical domain: a review. Informatics. 2024;11(3):57. https:\/\/doi.org\/10.3390\/informatics11030057 (https:\/\/www.mdpi.com\/2227-9709\/11\/3\/57).","journal-title":"Informatics"},{"issue":"1","key":"397_CR4","first-page":"261","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani A, et al. Attention is all you need. CoRR. 2017;30(1):261\u201372 arXiv:1706.03762.","journal-title":"CoRR"},{"key":"397_CR5","unstructured":"Wang Y, Zhao Y, Petzold L. Are large language models ready for healthcare? A comparative study on clinical language understanding; 2023. arXiv:2304.05368 [cs.CL]."},{"issue":"4","key":"397_CR6","doi-asserted-by":"publisher","first-page":"1234","DOI":"10.1093\/bioinformatics\/btz682","volume":"36","author":"J Lee","year":"2019","unstructured":"Lee J, et al. Biobert: a pre-trained biomedical language representation model for biomedical text mining. Bioinformatics. 2019;36(4):1234\u201340. https:\/\/doi.org\/10.1093\/bioinformatics\/btz682.","journal-title":"Bioinformatics"},{"key":"397_CR7","unstructured":"Huang K, Altosaar J, Ranganath R. Clinicalbert: modeling clinical notes and predicting hospital readmission; 2020. arXiv:1904.05342 [cs.CL]."},{"key":"397_CR8","unstructured":"Liu Z, et al. Radiology-llama2: best-in-class large language model for radiology; 2023. arXiv:2309.06419 [cs.CL]."},{"key":"397_CR9","unstructured":"Singhal K, et al. Towards expert-level medical question answering with large language models. 2023;31(3):943\u201350 arXiv:2305.09617 [cs.CL]."},{"key":"397_CR10","doi-asserted-by":"crossref","unstructured":"Labrak Y, et\u00a0al. Biomistral: a collection of open-source pretrained large language models for medical domains; 2024. arXiv:2402.10373 [cs.CL].","DOI":"10.18653\/v1\/2024.findings-acl.348"},{"key":"397_CR11","doi-asserted-by":"publisher","first-page":"1097","DOI":"10.3390\/biomedinformatics4020062","volume":"4","author":"K Nassiri","year":"2024","unstructured":"Nassiri K, Akhloufi M. Recent advances in large language models for healthcare. BioMedInformatics. 2024;4:1097\u2013143. https:\/\/doi.org\/10.3390\/biomedinformatics4020062.","journal-title":"BioMedInformatics"},{"key":"397_CR12","doi-asserted-by":"publisher","unstructured":"Dobhal U, Garcia C. Application of large language models in healthcare: a concise review; 2024. https:\/\/doi.org\/10.13140\/RG.2.2.32471.69284.","DOI":"10.13140\/RG.2.2.32471.69284"},{"issue":"2","key":"397_CR13","doi-asserted-by":"publisher","first-page":"605","DOI":"10.12669\/pjms.39.2.7653","volume":"39","author":"R KHAN","year":"2023","unstructured":"KHAN R, Jawaid M, Khan A, Sajjad M. Chatgpt\u2013reshaping medical education and clinical management. Pakistan J Med Sci. 2023;39(2):605. https:\/\/doi.org\/10.12669\/pjms.39.2.7653.","journal-title":"Pakistan J Med Sci"},{"issue":"2","key":"397_CR14","doi-asserted-by":"publisher","first-page":"16","DOI":"10.36922\/aih.2558","volume":"1","author":"SM Ummara Mumtaz","year":"2024","unstructured":"Ummara Mumtaz SM, Ahmed A. Llms-healthcare: current applications and challenges of large language models in various medical specialties. AIH. 2024;1(2):16. https:\/\/doi.org\/10.36922\/aih.2558.","journal-title":"AIH"},{"issue":"1","key":"397_CR15","doi-asserted-by":"publisher","first-page":"bbad493","DOI":"10.1093\/bib\/bbad493","volume":"25","author":"S Tian","year":"2024","unstructured":"Tian S, et al. Opportunities and challenges for chatgpt and large language models in biomedicine and health. Brief Bioinform. 2024;25(1):bbad493. https:\/\/doi.org\/10.1093\/bib\/bbad493.","journal-title":"Brief Bioinform"},{"issue":"1","key":"397_CR16","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1038\/s44184-024-00056-z","volume":"3","author":"EC Stade","year":"2024","unstructured":"Stade EC, et al. Large language models could change the future of behavioral healthcare: a proposal for responsible development and evaluation. NPJ Mental Health Res. 2024;3(1):12. https:\/\/doi.org\/10.1038\/s44184-024-00056-z.","journal-title":"NPJ Mental Health Res"},{"key":"397_CR17","doi-asserted-by":"crossref","unstructured":"Labrak Y, et\u00a0al. Biomistral: a collection of open-source pretrained large language models for medical domains; 2024. arXiv:2402.10373.","DOI":"10.18653\/v1\/2024.findings-acl.348"},{"key":"397_CR18","unstructured":"Gpt-4 technical report; 2024. arXiv:2303.08774 [cs.CL]."},{"key":"397_CR19","unstructured":"Chen Z, et\u00a0al. Meditron-70b: scaling medical pretraining for large language models; 2023. arXiv:2311.16079 [cs.CL]."},{"key":"397_CR20","unstructured":"Han T, et\u00a0al. Medalpaca\u2013an open-source collection of medical conversational AI models and training data; 2023. arXiv:2304.08247 [cs.CL]."},{"key":"397_CR21","unstructured":"Zhang X, et\u00a0al. Alpacare:instruction-tuned large language models for medical application; 2024. arXiv:2310.14558 [cs.CL]."},{"key":"397_CR22","doi-asserted-by":"crossref","unstructured":"Wang S, Zhao Z, Ouyang X, Wang Q, Shen D. Chatcad: interactive computer-aided diagnosis on medical image using large language models; 2023. arXiv:2302.07257 [cs.CV].","DOI":"10.1038\/s44172-024-00271-8"},{"issue":"6","key":"397_CR23","doi-asserted-by":"publisher","DOI":"10.1093\/bib\/bbac409","volume":"23","author":"R Luo","year":"2022","unstructured":"Luo R, et al. Biogpt: generative pre-trained transformer for biomedical text generation and mining. Brief Bioinform. 2022;23(6):bbac409. https:\/\/doi.org\/10.1093\/bib\/bbac409.","journal-title":"Brief Bioinform"},{"key":"397_CR24","unstructured":"Yang X, et\u00a0al. Gatortron: a large clinical language model to unlock patient information from unstructured electronic health records; 2022. arXiv:2203.03540."},{"key":"397_CR25","unstructured":"Bolton E, et\u00a0al. Biomedlm: a 2.7b parameter language model trained on biomedical text; 2024. arXivhttps:\/\/arxiv.org\/abs\/2403.18421arXiv:2403.18421 [cs.CL]."},{"key":"397_CR26","doi-asserted-by":"crossref","unstructured":"Yuan H. et\u00a0al. Biobart: pretraining and evaluation of a biomedical generative language model; 2022. arXiv:2204.03905 [cs.CL].","DOI":"10.18653\/v1\/2022.bionlp-1.9"},{"key":"397_CR27","doi-asserted-by":"publisher","unstructured":"Lu Q, Dou D, Nguyen T. Clinicalt5: a generative language model for clinical text; 2022:5436\u20135443. https:\/\/doi.org\/10.18653\/v1\/2022.findings-emnlp.398 .","DOI":"10.18653\/v1\/2022.findings-emnlp.398"},{"key":"397_CR28","unstructured":"Sherstinsky A. Fundamentals of recurrent neural network (RNN) and long short-term memory (LSTM) network. CoRR; 2018. arXiv:1808.03314 ."},{"key":"397_CR29","doi-asserted-by":"crossref","unstructured":"Shahid R, Wali A, Bashir M. Next word prediction for Urdu language using deep learning models. Comput. Speech Lang 2024;87:101635","DOI":"10.1016\/j.csl.2024.101635"},{"key":"397_CR30","doi-asserted-by":"crossref","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers) 2019 Jun (pp. 4171\u20134186)","DOI":"10.18653\/v1\/N19-1423"},{"key":"397_CR31","unstructured":"Liu Y, Ott M, Goyal N, Du J, Joshi M, Chen D, Levy O, Lewis M, Zettlemoyer L, Stoyanov V. Roberta: A robustly optimized bert pretraining approach. 2019. arXiv preprint arXiv:1907.11692"},{"key":"397_CR32","doi-asserted-by":"crossref","unstructured":"Jin Q, Dhingra B, Liu Z, Cohen WW, Lu X. Pubmedqa: a dataset for biomedical research question answering; 2019. arXiv:1909.06146 [cs.CL].","DOI":"10.18653\/v1\/D19-1259"},{"key":"397_CR33","unstructured":"Pal A, Umapathi LK, Sankarasubbu M. Medmcqa: a large-scale multi-subject multi-choice dataset for medical domain question answering; 2022. arXiv:https:\/\/arxiv.org\/abs\/2203.14371."},{"issue":"2","key":"397_CR34","doi-asserted-by":"publisher","first-page":"210","DOI":"10.7326\/m23-2772","volume":"177","author":"JA Omiye","year":"2024","unstructured":"Omiye JA, Gui H, Rezaei SJ, Zou J, Daneshjou R. Large language models in medicine: the potentials and pitfalls: a narrative review. Ann Intern Med. 2024;177(2):210\u201320. https:\/\/doi.org\/10.7326\/m23-2772.","journal-title":"Ann Intern Med"},{"key":"397_CR35","doi-asserted-by":"crossref","unstructured":"Novelli C, Casolari F, Hacker P, Spedicato G, Floridi L. Generative AI in EU law: liability, privacy, intellectual property, and cybersecurity;2024. arXiv:2401.07348.","DOI":"10.2139\/ssrn.4821952"}],"container-title":["Health Information Science and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13755-025-00397-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13755-025-00397-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13755-025-00397-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,21]],"date-time":"2025-11-21T13:13:09Z","timestamp":1763730789000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13755-025-00397-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,21]]},"references-count":35,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,12]]}},"alternative-id":["397"],"URL":"https:\/\/doi.org\/10.1007\/s13755-025-00397-9","relation":{},"ISSN":["2047-2501"],"issn-type":[{"value":"2047-2501","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,21]]},"assertion":[{"value":"3 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 November 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This statement is to certify that the author list is correct. The Authors also confirm that this research has not been published previously and that it is not under consideration for publication elsewhere. On behalf of all Co-Authors, the Corresponding Author shall bear full responsibility for the submission. There is no Conflict of interest. This research did not involve any human participants and\/or animals.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"The authors declare that they have no relevant financial or non-financial interests to disclose. There is no personal relationship that could influence the work reported in this paper. No funding was received for conducting this study. The authors have no Conflict of interest to declare that are relevant to the content of this article.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"2"}}