{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T01:54:21Z","timestamp":1784685261133,"version":"3.55.0"},"reference-count":60,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,5,2]],"date-time":"2025-05-02T00:00:00Z","timestamp":1746144000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,5,2]],"date-time":"2025-05-02T00:00:00Z","timestamp":1746144000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"name":"Ministry of Health & Welfare, Republic of Korea","award":["HR20C0021(3)"],"award-info":[{"award-number":["HR20C0021(3)"]}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2024-2020-0-0181"],"award-info":[{"award-number":["IITP-2024-2020-0-0181"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2023R1A2C3004176"],"award-info":[{"award-number":["NRF-2023R1A2C3004176"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["npj Digit. Med."],"DOI":"10.1038\/s41746-025-01653-8","type":"journal-article","created":{"date-parts":[[2025,5,2]],"date-time":"2025-05-02T16:46:00Z","timestamp":1746204360000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":53,"title":["Small language models learn enhanced reasoning skills from medical textbooks"],"prefix":"10.1038","volume":"8","author":[{"given":"Hyunjae","family":"Kim","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hyeon","family":"Hwang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiwoo","family":"Lee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sihyeon","family":"Park","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dain","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Taewhoo","family":"Lee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chanwoong","family":"Yoon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiwoong","family":"Sohn","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jungwoo","family":"Park","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Olga","family":"Reykhart","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Thomas","family":"Fetherston","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Donghee","family":"Choi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Soo Heon","family":"Kwak","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingyu","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jaewoo","family":"Kang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,2]]},"reference":[{"key":"1653_CR1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1186\/s12960-016-0176-x","volume":"15","author":"JX Liu","year":"2017","unstructured":"Liu, J. X., Goryakin, Y., Maeda, A., Bruckner, T. & Scheffler, R. Global health workforce labor market projections for 2030. Hum. Resour. Health 15, 1\u201312 (2017).","journal-title":"Hum. Resour. Health"},{"key":"1653_CR2","doi-asserted-by":"publisher","first-page":"274","DOI":"10.1017\/S174413311700055X","volume":"14","author":"RM Scheffler","year":"2019","unstructured":"Scheffler, R. M. & Arnold, D. R. Projecting shortages and surpluses of doctors and nurses in the OECD: what looms ahead. Health Econ. Policy Law 14, 274\u2013290 (2019).","journal-title":"Health Econ. Policy Law"},{"key":"1653_CR3","doi-asserted-by":"publisher","first-page":"1247","DOI":"10.1002\/nop2.1434","volume":"10","author":"AT Tamata","year":"2023","unstructured":"Tamata, A. T. & Mohammadnezhad, M. A systematic review study on the factors affecting shortage of nursing workforce in the hospitals. Nurs. Open 10, 1247\u20131257 (2023).","journal-title":"Nurs. Open"},{"key":"1653_CR4","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1038\/nature21056","volume":"542","author":"A Esteva","year":"2017","unstructured":"Esteva, A. et al. Dermatologist-level classification of skin cancer with deep neural networks. Nature 542, 115\u2013118 (2017).","journal-title":"Nature"},{"key":"1653_CR5","doi-asserted-by":"publisher","DOI":"10.1038\/s41746-020-0258-y","volume":"3","author":"B Norgeot","year":"2020","unstructured":"Norgeot, B. et al. Protected health information filter (philter): accurately and securely de-identifying free-text clinical notes. NPJ Digital Med. 3, 57 (2020).","journal-title":"NPJ Digital Med."},{"key":"1653_CR6","doi-asserted-by":"publisher","first-page":"1930","DOI":"10.1038\/s41591-023-02448-8","volume":"29","author":"AJ Thirunavukarasu","year":"2023","unstructured":"Thirunavukarasu, A. J. et al. Large language models in medicine. Nat. Med. 29, 1930\u20131940 (2023).","journal-title":"Nat. Med."},{"key":"1653_CR7","doi-asserted-by":"publisher","first-page":"bbad493","DOI":"10.1093\/bib\/bbad493","volume":"25","author":"S Tian","year":"2024","unstructured":"Tian, S. et al. Opportunities and challenges for chatgpt and large language models in biomedicine and health. Brief. Bioinforma. 25, bbad493 (2024).","journal-title":"Brief. Bioinforma."},{"key":"1653_CR8","doi-asserted-by":"publisher","first-page":"e0000198","DOI":"10.1371\/journal.pdig.0000198","volume":"2","author":"TH Kung","year":"2023","unstructured":"Kung, T. H. et al. Performance of ChatGPT on USMLE: potential for ai-assisted medical education using large language models. PLoS Digital health 2, e0000198 (2023).","journal-title":"PLoS Digital health"},{"key":"1653_CR9","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1038\/s41586-023-06291-2","volume":"620","author":"K Singhal","year":"2023","unstructured":"Singhal, K. et al. Large language models encode clinical knowledge. Nature 620, 172\u2013180 (2023).","journal-title":"Nature"},{"key":"1653_CR10","doi-asserted-by":"publisher","first-page":"943","DOI":"10.1038\/s41591-024-03423-7","volume":"31","author":"K Singhal","year":"2025","unstructured":"Singhal, K. et al. Toward expert-level medical question answering with large language models. Nat. Med. 31, 943\u2013950 (2025).","journal-title":"Nat. Med."},{"key":"1653_CR11","doi-asserted-by":"publisher","unstructured":"Nori, H., King, N., McKinney, S. M., Carignan, D. & Horvitz, E. Capabilities of gpt-4 on medical challenge problems. arXiv https:\/\/doi.org\/10.48550\/arXiv.2303.13375 (2023).","DOI":"10.48550\/arXiv.2303.13375"},{"key":"1653_CR12","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-023-43436-9","volume":"13","author":"D Brin","year":"2023","unstructured":"Brin, D. et al. Comparing chatgpt and gpt-4 performance in usmle soft skill assessments. Sci. Rep. 13, 16492 (2023).","journal-title":"Sci. Rep."},{"key":"1653_CR13","doi-asserted-by":"publisher","unstructured":"Xie, Y. et al. A preliminary study of o1 in medicine: are we closer to an AI doctor? arXiv https:\/\/doi.org\/10.48550\/arXiv.2409.15277 (2024).","DOI":"10.48550\/arXiv.2409.15277"},{"key":"1653_CR14","doi-asserted-by":"publisher","first-page":"AIoa2300068","DOI":"10.1056\/AIoa2300068","volume":"1","author":"C Zakka","year":"2024","unstructured":"Zakka, C. et al. Almanac-retrieval-augmented language models for clinical medicine. NEJM AI 1, AIoa2300068 (2024).","journal-title":"NEJM AI"},{"key":"1653_CR15","doi-asserted-by":"publisher","unstructured":"Tu, T. et al. Towards conversational diagnostic ai. arXiv https:\/\/doi.org\/10.48550\/arXiv.2401.05654 (2024).","DOI":"10.48550\/arXiv.2401.05654"},{"key":"1653_CR16","doi-asserted-by":"crossref","unstructured":"Eriksen, A. V., M\u00f6ller, S. & Ryg, J. Use of gpt-4 to diagnose complex clinical cases. (2023).","DOI":"10.1056\/AIp2300031"},{"key":"1653_CR17","unstructured":"OpenAI. Introducing chatgpt. https:\/\/openai.com\/blog\/chatgpt (2022)."},{"key":"1653_CR18","doi-asserted-by":"publisher","unstructured":"Achiam, J. et al. Gpt-4 technical report. arXiv https:\/\/doi.org\/10.48550\/arXiv.2303.08774 (2023).","DOI":"10.48550\/arXiv.2303.08774"},{"key":"1653_CR19","doi-asserted-by":"crossref","unstructured":"Li, X. & Zhang, T. An exploration on artificial intelligence application: from security, privacy and ethic perspective. In 2017 IEEE 2nd International Conference on Cloud Computing and Big Data Analysis (ICCCBDA), 416\u2013420 (IEEE, 2017).","DOI":"10.1109\/ICCCBDA.2017.7951949"},{"key":"1653_CR20","doi-asserted-by":"crossref","unstructured":"Bartoletti, I. AI in healthcare: ethical and privacy challenges. In Artificial Intelligence in Medicine: 17th Conference on Artificial Intelligence in Medicine, AIME 2019, Poznan, Poland, June 26\u201329, 2019, Proceedings 17, 7\u201310 (Springer, 2019).","DOI":"10.1007\/978-3-030-21642-9_2"},{"key":"1653_CR21","doi-asserted-by":"publisher","DOI":"10.1038\/s41746-023-00873-0","volume":"6","author":"B Mesk\u00f3","year":"2023","unstructured":"Mesk\u00f3, B. & Topol, E. J. The imperative for regulatory oversight of large language models (or generative AI) in healthcare. NPJ Digital Med. 6, 120 (2023).","journal-title":"NPJ Digital Med."},{"key":"1653_CR22","unstructured":"AI@Meta. Llama 3 model card. (2024)."},{"key":"1653_CR23","unstructured":"xAI. Open release of grok-1. https:\/\/x.ai\/blog\/grok-os (2024)."},{"key":"1653_CR24","doi-asserted-by":"publisher","unstructured":"Guo, D. et al. Deepseek-r1: incentivizing reasoning capability in llms via reinforcement learning. arXiv https:\/\/doi.org\/10.48550\/arXiv.2501.12948 (2025).","DOI":"10.48550\/arXiv.2501.12948"},{"key":"1653_CR25","doi-asserted-by":"publisher","unstructured":"Touvron, H. et al. Llama: open and efficient foundation language models. arXiv https:\/\/doi.org\/10.48550\/arXiv.2302.13971 (2023).","DOI":"10.48550\/arXiv.2302.13971"},{"key":"1653_CR26","doi-asserted-by":"publisher","unstructured":"Jiang, A. Q. et al. Mistral 7b. arXiv https:\/\/doi.org\/10.48550\/arXiv.2310.06825 (2023).","DOI":"10.48550\/arXiv.2310.06825"},{"key":"1653_CR27","unstructured":"Google. Gemma: introducing new state-of-the-art open models. https:\/\/blog.google\/technology\/developers\/gemma-open-models\/ (2024)."},{"key":"1653_CR28","doi-asserted-by":"crossref","unstructured":"Wu, C. et al. PMC-LLaMA: toward building open-source language models for medicine. J. Am. Med. Inf. Assoc. 31, 1833\u20131843 (2024).","DOI":"10.1093\/jamia\/ocae045"},{"key":"1653_CR29","doi-asserted-by":"publisher","unstructured":"Chen, Z. et al. Meditron-70b: scaling medical pretraining for large language models. arXiv https:\/\/doi.org\/10.48550\/arXiv.2311.16079 (2023).","DOI":"10.48550\/arXiv.2311.16079"},{"key":"1653_CR30","doi-asserted-by":"crossref","unstructured":"Labrak, Y. et al. Biomistral: a collection of open-source pretrained large language models for medical domains. In Findings of the Association for Computational Linguistics ACL 2024, 5848\u20135864 (Association for Computational Linguistics, 2024).","DOI":"10.18653\/v1\/2024.findings-acl.348"},{"key":"1653_CR31","doi-asserted-by":"crossref","unstructured":"Xie, Q. et al. Medical foundation large language models for comprehensive text analysis and beyond. npj Digit. Med. 8, 141 (2025).","DOI":"10.1038\/s41746-025-01533-1"},{"key":"1653_CR32","first-page":"24824","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J. et al. Chain-of-thought prompting elicits reasoning in large language models. Adv. Neural Inf. Process. Syst. 35, 24824\u201324837 (2022).","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1653_CR33","unstructured":"Wei, J. et al. Emergent abilities of large language models. Transactions on Machine Learning Research https:\/\/openreview.net\/forum?id=yzkSU5zdwD (2022)."},{"key":"1653_CR34","unstructured":"Tay, Y. et al. Ul2: Unifying language learning paradigms. In The Eleventh International Conference on Learning Representations (OpenReview.net, 2022)."},{"key":"1653_CR35","first-page":"1","volume":"25","author":"HW Chung","year":"2024","unstructured":"Chung, H. W. et al. Scaling instruction-finetuned language models. J. Mach. Learn. Res. 25, 1\u201353 (2024).","journal-title":"J. Mach. Learn. Res."},{"key":"1653_CR36","doi-asserted-by":"publisher","first-page":"6421","DOI":"10.3390\/app11146421","volume":"11","author":"D Jin","year":"2021","unstructured":"Jin, D. et al. What disease does this patient have? A large-scale open domain question answering dataset from medical exams. Appl. Sci. 11, 6421 (2021).","journal-title":"Appl. Sci."},{"key":"1653_CR37","doi-asserted-by":"crossref","unstructured":"Manes, I. et al. K-qa: a real-world medical q&a benchmark. In Proceedings of the 23rd Workshop on Biomedical Natural Language Processing, 277\u2013294 (Association for Computational Linguistics, 2024).","DOI":"10.18653\/v1\/2024.bionlp-1.22"},{"key":"1653_CR38","first-page":"e40895","volume":"15","author":"Y Li","year":"2023","unstructured":"Li, Y. et al. Chatdoctor: a medical chat model fine-tuned on a large language model meta-ai (llama) using medical domain knowledge. Cureus 15, e40895 (2023).","journal-title":"Cureus"},{"key":"1653_CR39","first-page":"9459","volume":"33","author":"P Lewis","year":"2020","unstructured":"Lewis, P. et al. Retrieval-augmented generation for knowledge-intensive nlp tasks. Adv. Neural Inf. Process. Syst. 33, 9459\u20139474 (2020).","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1653_CR40","first-page":"27730","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang, L. et al. Training language models to follow instructions with human feedback. Adv. Neural Inf. Process. Syst. 35, 27730\u201327744 (2022).","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1653_CR41","unstructured":"Pal, A., Umapathi, L. K. & Sankarasubbu, M. Medmcqa: a large-scale multi-subject multi-choice dataset for medical domain question answering. In Conference on Health, Inference, and Learning, 248\u2013260 (PMLR, 2022)."},{"key":"1653_CR42","unstructured":"Abacha, A. B., Agichtein, E., Pinter, Y. & Demner-Fushman, D. Overview of the medical question answering task at trec 2017 liveqa. In TREC, 1\u201312 (National Institute of Standards and Technology (NIST), 2017)."},{"key":"1653_CR43","doi-asserted-by":"crossref","unstructured":"Abacha, A. B. et al. Bridging the gap between consumers\u2019 medication questions and trusted answers. In MedInfo, 25\u201329 (IOS Press, 2019).","DOI":"10.3233\/SHTI190176"},{"key":"1653_CR44","doi-asserted-by":"publisher","unstructured":"Zhang, X. et al. Alpacare: Instruction-tuned large language models for medical application. arXiv https:\/\/doi.org\/10.48550\/arXiv.2310.14558 (2023).","DOI":"10.48550\/arXiv.2310.14558"},{"key":"1653_CR45","doi-asserted-by":"crossref","unstructured":"Wang, Y. et al. Self-Instruct: Aligning Language Models with Self-Generated Instructions. In Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, Vol 1: Long Papers (2023).","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"1653_CR46","unstructured":"Taori, R. et al. Stanford alpaca: an instruction-following llama model. https:\/\/github.com\/tatsu-lab\/stanford_alpaca (2023)."},{"key":"1653_CR47","first-page":"16344","volume":"35","author":"T Dao","year":"2022","unstructured":"Dao, T., Fu, D., Ermon, S., Rudra, A. & R\u00e9, C. Flashattention: fast and memory-efficient exact attention with io-awareness. Adv. Neural Inf. Process. Syst. 35, 16344\u201316359 (2022).","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1653_CR48","unstructured":"OpenAI. Gpt-4o system card. https:\/\/openai.com\/index\/gpt-4o-system-card\/ (2024)."},{"key":"1653_CR49","unstructured":"OpenAI. Openai o1-mini. https:\/\/openai.com\/index\/openai-o1-mini-advancing-cost-efficient-reasoning\/ (2024)."},{"key":"1653_CR50","unstructured":"OpenAI. Openai o3-mini. https:\/\/openai.com\/index\/openai-o3-mini\/ (2025)."},{"key":"1653_CR51","unstructured":"OpenAI. Introducing OpenAI o1. https:\/\/openai.com\/o1\/ (2024)."},{"key":"1653_CR52","doi-asserted-by":"publisher","unstructured":"Saab, K. et al. Capabilities of gemini models in medicine. arXiv https:\/\/doi.org\/10.48550\/arXiv.2404.18416 (2024).","DOI":"10.48550\/arXiv.2404.18416"},{"key":"1653_CR53","doi-asserted-by":"publisher","unstructured":"Toma, A. et al. Clinical camel: an open-source expert-level medical language model with dialogue-based knowledge encoding. arXiv https:\/\/doi.org\/10.48550\/arXiv.2305.12031 (2023).","DOI":"10.48550\/arXiv.2305.12031"},{"key":"1653_CR54","unstructured":"Chen, H., Fang, Z., Singla, Y. & Dredze, M. Benchmarking large language models on answering and explaining challenging medical questions. In Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies, Vol 1: Long Papers, 3563\u20133599 (2025)."},{"key":"1653_CR55","unstructured":"Hendrycks, D. et al. Measuring massive multitask language understanding. In International Conference on Learning Representations (OpenReview.net, 2020)."},{"key":"1653_CR56","doi-asserted-by":"publisher","first-page":"190","DOI":"10.1038\/s41746-024-01185-7","volume":"7","author":"Q Jin","year":"2024","unstructured":"Jin, Q. et al. Hidden flaws behind expert-level accuracy of multimodal GPT-4 vision in medicine. npj Digital Med. 7, 190 (2024).","journal-title":"npj Digital Med."},{"key":"1653_CR57","doi-asserted-by":"crossref","unstructured":"Kwon, W. et al. Efficient memory management for large language model serving with pagedattention. In Proceedings of the ACM SIGOPS 29th Symposium on Operating Systems Principles (Association for Computing Machinery, 2023).","DOI":"10.1145\/3600006.3613165"},{"key":"1653_CR58","doi-asserted-by":"publisher","unstructured":"Nori, H. et al. Can generalist foundation models outcompete special-purpose tuning? Case study in medicine. arXiv https:\/\/doi.org\/10.48550\/arXiv.2311.16452 (2023).","DOI":"10.48550\/arXiv.2311.16452"},{"key":"1653_CR59","doi-asserted-by":"crossref","unstructured":"Ko, M., Lee, J., Kim, H., Kim, G. & Kang, J. Look at the first sentence: position bias in question answering. In Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), 1109\u20131121 (Association for Computational Linguistics, 2020).","DOI":"10.18653\/v1\/2020.emnlp-main.84"},{"key":"1653_CR60","unstructured":"Abacha, A. B., Yim, W.-W., Fan, Y. & Lin, T. An empirical study of clinical note generation from doctor-patient encounters. In Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics, 2283\u20132294 (Association for Computational Linguistics, 2023)."}],"updated-by":[{"DOI":"10.1038\/s41746-025-01745-5","type":"correction","label":"Correction","source":"publisher","updated":{"date-parts":[[2025,6,6]],"date-time":"2025-06-06T00:00:00Z","timestamp":1749168000000}}],"container-title":["npj Digital Medicine"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.nature.com\/articles\/s41746-025-01653-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s41746-025-01653-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s41746-025-01653-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,7]],"date-time":"2025-06-07T18:03:02Z","timestamp":1749319382000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.nature.com\/articles\/s41746-025-01653-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,2]]},"references-count":60,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["1653"],"URL":"https:\/\/doi.org\/10.1038\/s41746-025-01653-8","relation":{},"ISSN":["2398-6352"],"issn-type":[{"value":"2398-6352","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,2]]},"assertion":[{"value":"26 January 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 April 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 June 2025","order":4,"name":"change_date","label":"Change Date","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Correction","order":5,"name":"change_type","label":"Change Type","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"A Correction to this paper has been published:","order":6,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"https:\/\/doi.org\/10.1038\/s41746-025-01745-5","URL":"https:\/\/doi.org\/10.1038\/s41746-025-01745-5","order":7,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"240"}}