{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T14:40:46Z","timestamp":1778769646566,"version":"3.51.4"},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T00:00:00Z","timestamp":1775433600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T00:00:00Z","timestamp":1775433600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Health Inf Sci Syst"],"DOI":"10.1007\/s13755-026-00449-8","type":"journal-article","created":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T15:58:07Z","timestamp":1775491087000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Enhancing LLM-based medical decision-making by test-time knowledge acquisition"],"prefix":"10.1007","volume":"14","author":[{"given":"Shipeng","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liuxin","family":"Bao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shikun","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Wan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,6]]},"reference":[{"key":"449_CR1","doi-asserted-by":"publisher","first-page":"1072","DOI":"10.1016\/j.jacr.2024.01.007","volume":"21","author":"HA Zaki","year":"2024","unstructured":"Zaki HA, et al. The application of large language models for radiologic decision making. J Am Coll Radiol. 2024;21:1072\u20138.","journal-title":"J Am Coll Radiol"},{"key":"449_CR2","unstructured":"Chen K, et al. MDTeamGPT: a self-evolving LLM-based multi-agent framework for multi-disciplinary team medical consultation. 2025. arXiv:2503.13856"},{"key":"449_CR3","doi-asserted-by":"crossref","unstructured":"Gekhman Z, et\u00a0al. Does fine-tuning LLMs on new knowledge encourage hallucinations? 2024. arXiv:2405.05904","DOI":"10.18653\/v1\/2024.emnlp-main.444"},{"key":"449_CR4","doi-asserted-by":"crossref","unstructured":"Kostikova A, et\u00a0al. LLLMs: a data-driven survey of evolving research on limitations of large language models. 2025. arXiv:2505.19240","DOI":"10.1145\/3801096"},{"key":"449_CR5","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O. Proximal policy optimization algorithms. 2017. arXiv:1707.06347"},{"key":"449_CR6","unstructured":"Shao Z, et\u00a0al. DeepSeekMath: pushing the limits of mathematical reasoning in open language models. 2024. arXiv:2402.03300"},{"key":"449_CR7","doi-asserted-by":"crossref","unstructured":"Li H, Ding L, Fang M, Tao D. Revisiting catastrophic forgetting in large language model tuning. 2024. arXiv:2406.04836","DOI":"10.18653\/v1\/2024.findings-emnlp.249"},{"key":"449_CR8","unstructured":"Cai Y, et\u00a0al. Training-free group relative policy optimization. 2025. arXiv:2510.08191"},{"key":"449_CR9","doi-asserted-by":"crossref","unstructured":"Suzgun M, Yuksekgonul M, Bianchi F, Jurafsky D, Zou J. Dynamic cheatsheet: test-time learning with adaptive memory. 2025. arXiv:2504.07952","DOI":"10.18653\/v1\/2026.eacl-long.333"},{"key":"449_CR10","unstructured":"Zhang Q, et\u00a0al. Agentic context engineering: evolving contexts for self-improving language models. 2025. arXiv:2510.04618"},{"key":"449_CR11","unstructured":"Zuo Y, et\u00a0al. TTRL: test-time reinforcement learning. 2025. arXiv:2504.16084"},{"key":"449_CR12","doi-asserted-by":"crossref","unstructured":"Zhang H, Li J, Wang Y, Song Y. Integrating automated knowledge extraction with large language models for explainable medical decision-making. In: 2023 IEEE international conference on bioinformatics and biomedicine (BIBM). IEEE; 2023. p. 1710\u201317.","DOI":"10.1109\/BIBM58861.2023.10385557"},{"key":"449_CR13","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijmedinf.2024.105501","volume":"188","author":"E Sblendorio","year":"2024","unstructured":"Sblendorio E, et al. Integrating human expertise & automated methods for a dynamic and multi-parametric evaluation of large language models\u2019 feasibility in clinical decision-making. Int J Med Inform. 2024;188:105501.","journal-title":"Int J Med Informatics"},{"key":"449_CR14","doi-asserted-by":"publisher","DOI":"10.1200\/PO-24-00478","volume":"8","author":"J Lammert","year":"2024","unstructured":"Lammert J, et al. Expert-guided large language models for clinical decision support in precision oncology. JCO Precis Oncol. 2024;8:e2400478.","journal-title":"JCO Precis Oncol"},{"key":"449_CR15","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1007\/s00432-024-05673-x","volume":"150","author":"A Lawson McLean","year":"2024","unstructured":"Lawson McLean A, Wu Y, Lawson McLean AC, Hristidis V. Large language models as decision aids in neuro-oncology: a review of shared decision-making applications. J Cancer Res Clin Oncol. 2024;150:139.","journal-title":"J Cancer Res Clin Oncol"},{"key":"449_CR16","doi-asserted-by":"crossref","unstructured":"Yang D, et\u00a0al. MedAide: information fusion and anatomy of medical intents via llm-based agent collaboration. Inf. Fusion. 2025;103743.","DOI":"10.1016\/j.inffus.2025.103743"},{"key":"449_CR17","unstructured":"Feng Y, Wang J, Zhou L, Lei Z, Li Y. DoctorAgent-RL: a multi-agent collaborative reinforcement learning system for multi-turn clinical dialogue. 2025. arXiv:2505.19630"},{"key":"449_CR18","unstructured":"Ghezloo F, et\u00a0al. Pathfinder: a multi-modal multi-agent system for medical diagnostic decision-making applied to histopathology. 2025. arXiv:2502.08916"},{"key":"449_CR19","doi-asserted-by":"crossref","unstructured":"Zhang W, et\u00a0al. Multi-agent reasoning for cardiovascular imaging phenotype analysis. Berlin: Springer; 2025. p. 429\u201339.","DOI":"10.1007\/978-3-032-04927-8_41"},{"key":"449_CR20","doi-asserted-by":"crossref","unstructured":"Yue L, Xing S, Chen J, Fu T. Clinicalagent: clinical trial multi-agent system with large language model-based reasoning. In: Proceedings of the 15th ACM international conference on bioinformatics, computational biology and health informatics. 2024. p. 1\u201310.","DOI":"10.1145\/3698587.3701359"},{"key":"449_CR21","unstructured":"Chen Z, et\u00a0al. Harnessing multiple large language models: a survey on llm ensemble. 2025. arXiv:2502.18036"},{"key":"449_CR22","doi-asserted-by":"crossref","unstructured":"Wang Y, et\u00a0al. MMLU-Pro: a more robust and challenging multi-task language understanding benchmark. Adv Neural Inf Process Syst. 2024;37:95266\u201390.","DOI":"10.52202\/079017-3018"},{"issue":"5","key":"449_CR23","doi-asserted-by":"publisher","first-page":"AIdbp2300192","DOI":"10.1056\/AIdbp2300192","volume":"1","author":"U Katz","year":"2024","unstructured":"Katz U, et al. GPT versus resident physicians-a benchmark based on official board scores. NEJM AI. 2024;1(5):AIdbp2300192.","journal-title":"NEJM AI"},{"key":"449_CR24","doi-asserted-by":"publisher","first-page":"6421","DOI":"10.3390\/app11146421","volume":"11","author":"D Jin","year":"2021","unstructured":"Jin D, et al. What disease does this patient have? A large-scale open domain question answering dataset from medical exams. Appl Sci. 2021;11:6421.","journal-title":"Appl Sci"},{"key":"449_CR25","unstructured":"Abdin M, et\u00a0al. Phi-4 technical report. 2024. arXiv:2412.08905"},{"key":"449_CR26","unstructured":"Yang A, et\u00a0al. Qwen2.5 technical report. 2024. arXiv:2412.15115"},{"key":"449_CR27","unstructured":"Team O. Open thoughts. 2025."},{"key":"449_CR28","unstructured":"Guo D, et\u00a0al. DeepSeek-R1: incentivizing reasoning capability in llms via reinforcement learning. 2025. arXiv:2501.12948"},{"key":"449_CR29","unstructured":"Jiang AQ, et\u00a0al. Mixtral of experts. 2024. arXiv:2401.04088"},{"key":"449_CR30","unstructured":"Xu C, et\u00a0al. WizardLM: empowering large language models to follow complex instructions. 2023. arXiv:2304.12244"},{"key":"449_CR31","unstructured":"Liu A, et\u00a0al. DeepSeek-V3 technical report. 2024. arXiv:2412.19437"},{"key":"449_CR32","unstructured":"Achiam J, et\u00a0al. GPT-4 technical report. 2023. arXiv:2303.08774"},{"key":"449_CR33","unstructured":"Powers DM. Evaluation: from precision, recall and F-measure to ROC, informedness, markedness and correlation. 2020. arXiv:2010.16061"},{"key":"449_CR34","unstructured":"Saah A, Hoover D. Sensitivity and specificity revisited significance of the terms in analytic and diagnostic language. Annales de Dermatologie et de Venereologie. 1998;125:291\u20134."},{"issue":"2","key":"449_CR35","doi-asserted-by":"publisher","first-page":"442","DOI":"10.1016\/0005-2795(75)90109-9","volume":"405","author":"BW Matthews","year":"1975","unstructured":"Matthews BW. Comparison of the predicted and observed secondary structure of T4 phage lysozyme. Biochim Biophys Acta. 1975;405(2):442\u201351.","journal-title":"Biochim Biophys Acta - Proteins Proteomics"},{"key":"449_CR36","doi-asserted-by":"publisher","first-page":"276","DOI":"10.11613\/BM.2012.031","volume":"22","author":"ML McHugh","year":"2012","unstructured":"McHugh ML. Interrater reliability: the kappa statistic. Biochem Medica. 2012;22:276\u201382.","journal-title":"Biochemia medica"},{"key":"449_CR37","unstructured":"Wang X, et\u00a0al. Self-consistency improves chain of thought reasoning in language models. 2022. arXiv:2203.11171"},{"issue":"1","key":"449_CR38","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1186\/s41235-021-00279-0","volume":"6","author":"S Meyen","year":"2021","unstructured":"Meyen S, Sigg DM, Luxburg UV, Franz VH. Group decisions based on confidence weighted majority voting. Cognit Res. 2021;6(1):18.","journal-title":"Cogn Res Princ Implic"},{"key":"449_CR39","doi-asserted-by":"crossref","unstructured":"Wang W, Wang Y, Huang H. Ranked voting based self-consistency of large language models. 2025. arXiv:2505.10772","DOI":"10.18653\/v1\/2025.findings-acl.744"}],"container-title":["Health Information Science and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13755-026-00449-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13755-026-00449-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13755-026-00449-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T15:58:18Z","timestamp":1775491098000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13755-026-00449-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,6]]},"references-count":39,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,12]]}},"alternative-id":["449"],"URL":"https:\/\/doi.org\/10.1007\/s13755-026-00449-8","relation":{},"ISSN":["2047-2501"],"issn-type":[{"value":"2047-2501","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,6]]},"assertion":[{"value":"8 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no financial or non-financial interests that could directly or indirectly influence, or be perceived as influencing, the work submitted for publication. Each author has thoroughly reviewed the content of the manuscript and confirms that no affiliations, financial involvements, or personal relationships pose any conflict of interest in the design, conduct, interpretation, or reporting of the research findings.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"51"}}