{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T01:03:30Z","timestamp":1784682210147,"version":"3.55.0"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T00:00:00Z","timestamp":1784678400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T00:00:00Z","timestamp":1784678400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Med Syst"],"DOI":"10.1007\/s10916-026-02440-y","type":"journal-article","created":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T00:38:07Z","timestamp":1784680687000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Multi-Axial Analysis of Clinical Reasoning in Large Language Models: Inter-Verifier Disagreement and Its Implications for Automated Evaluation"],"prefix":"10.1007","volume":"50","author":[{"given":"Hyunjung","family":"Byun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dahyoun","family":"Lee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Munyoung","family":"Jung","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Beakcheol","family":"Jang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,22]]},"reference":[{"key":"2440_CR1","doi-asserted-by":"publisher","unstructured":"Singhal, K., Azizi, S., Tu, T., Mahdavi, S. S., Wei, J., Chung, H. W., et\u00a0al., Large language models encode clinical knowledge. Nature 620(7972):172\u2013180, 2023. https:\/\/doi.org\/10.1038\/s41586-023-06291-2.","DOI":"10.1038\/s41586-023-06291-2"},{"key":"2440_CR2","doi-asserted-by":"publisher","unstructured":"McCoy, L. G., Swamy, R., Sagar, N., Wang, M., Bacchi, S., Fong, J. M. N., et\u00a0al., Assessment of large language models in clinical reasoning. NEJM AI. 2(10), 2025. https:\/\/doi.org\/10.1056\/AIdbp2500120","DOI":"10.1056\/AIdbp2500120"},{"key":"2440_CR3","doi-asserted-by":"publisher","unstructured":"Thirunavukarasu, A. J., Ting, D. S. J., Elangovan, K., Gutierrez, L., Tan, T. F., and Ting, D. S. W., Large language models in medicine. Nat. Med. 2023;29(8):1930\u20131940. https:\/\/doi.org\/10.1038\/s41591-023-02448-8.","DOI":"10.1038\/s41591-023-02448-8"},{"key":"2440_CR4","doi-asserted-by":"publisher","unstructured":"Kung, T. H., Cheatham, M., Medenilla, A., Sillos, C., De\u00a0Leon, L., Elepa\u00f1o, C., et\u00a0al., Performance of ChatGPT on USMLE: Potential for AI-assisted medical education using large language models. PLOS Digit. Health. 2(2):e0000198, 2023. https:\/\/doi.org\/10.1371\/journal.pdig.0000198.","DOI":"10.1371\/journal.pdig.0000198"},{"key":"2440_CR5","unstructured":"OpenAI, GPT-4 Technical Report. Available from: arxiv:2303.08774"},{"key":"2440_CR6","doi-asserted-by":"publisher","unstructured":"Bedi, S., Liu, Y., Orr-Ewing, L., Dash, D., Koyejo, S., Callahan, A., et\u00a0al., Testing and evaluation of health care applications of large language models: A systematic review. JAMA. 333(4):319\u2013328, 2025. https:\/\/doi.org\/10.1001\/jama.2024.21700.","DOI":"10.1001\/jama.2024.21700"},{"key":"2440_CR7","doi-asserted-by":"publisher","unstructured":"Vasey, B., Nagendran, M., Campbell, B., Clifton, D. A., Collins, G. S., Denaxas, S., et\u00a0al., Reporting guideline for the early-stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI. Nat. Med. 28(5):924\u2013933, 2022. https:\/\/doi.org\/10.1038\/s41591-022-01772-9.","DOI":"10.1038\/s41591-022-01772-9"},{"key":"2440_CR8","doi-asserted-by":"publisher","unstructured":"Liu, X., Cruz\u00a0Rivera, S., Moher, D., Calvert, M. J., and Denniston, A. K., Reporting guidelines for clinical trial reports for interventions involving artificial intelligence: the CONSORT-AI extension. Lancet Digit. Health. 2(10):e537\u2013e548, 2020. https:\/\/doi.org\/10.1016\/S2589-7500(20)30218-1.","DOI":"10.1016\/S2589-7500(20)30218-1"},{"key":"2440_CR9","doi-asserted-by":"publisher","unstructured":"Cruz\u00a0Rivera, S., Liu, X., Chan, A. W., Denniston, A. K, and Calvert, M. J., Guidelines for clinical trial protocols for interventions involving artificial intelligence: the SPIRIT-AI extension. BMJ. 370:m3210, 2020. https:\/\/doi.org\/10.1136\/bmj.m3210.","DOI":"10.1136\/bmj.m3210"},{"key":"2440_CR10","doi-asserted-by":"publisher","unstructured":"Choudhury, A., and Chaudhry, Z., Large language models and user trust: Consequence of self-referential learning loop and the deskilling of health care professionals. J. Med. Intern. Res. 26:e56764, 2024. https:\/\/doi.org\/10.2196\/56764.","DOI":"10.2196\/56764"},{"key":"2440_CR11","doi-asserted-by":"publisher","unstructured":"Ayers, J. W., Poliak, A., Dredze, M., Leas, E. C., Zhu, Z., Kelley, J. B., et\u00a0al., Comparing physician and artificial intelligence chatbot responses to patient questions posted to a public social media forum. JAMA Intern. Med. 183(6):589\u2013596, 2023. https:\/\/doi.org\/10.1001\/jamainternmed.2023.1838.","DOI":"10.1001\/jamainternmed.2023.1838"},{"key":"2440_CR12","doi-asserted-by":"publisher","unstructured":"Asgari, E., Monta\u00f1a-Brown, N., Dubois, M., Khalil, S., Balloch, J., Au\u00a0Yeung, J., et\u00a0al., A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation. NPJ Digit. Med. 8(1):274, 2025. https:\/\/doi.org\/10.1038\/s41746-025-01670-7.","DOI":"10.1038\/s41746-025-01670-7"},{"key":"2440_CR13","doi-asserted-by":"publisher","unstructured":"Huang, L., Yu, W., Ma, W., Zhong, W., Feng, Z., Wang, H., et\u00a0al., A survey on hallucination in large language models: Principles, taxonomy, challenges, and open questions. ACM Trans. Inf. Syst., 2025. https:\/\/doi.org\/10.1145\/3703155.","DOI":"10.1145\/3703155"},{"key":"2440_CR14","doi-asserted-by":"publisher","unstructured":"Cabral, S., Kanjee, Z., Wilson, P., Crowe, B., Abdulnour, R. E., and Rodman, A., Clinical reasoning of a generative artificial intelligence model compared with physicians. JAMA Intern. Med. 184(5):581\u2013583, 2024. https:\/\/doi.org\/10.1001\/jamainternmed.2024.0295.","DOI":"10.1001\/jamainternmed.2024.0295"},{"key":"2440_CR15","doi-asserted-by":"crossref","unstructured":"Kwon, T., Ong, K. T. i., Kang, D., Moon, S., Lee, J. R., Hwang, D., et\u00a0al., Large language models are clinical reasoners: reasoning-aware diagnosis framework with prompt-generated rationales. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol.\u00a038. p. 18417\u201318425, 2024.","DOI":"10.1609\/aaai.v38i16.29802"},{"key":"2440_CR16","unstructured":"Wang, X., Wei, J., Schuurmans, D., Le, Q. V., Chi, E. H., Narang, S., et\u00a0al., Self-consistency improves chain of thought reasoning in language models. In: International Conference on Learning Representations (ICLR 2023), 2023. Available from: https:\/\/openreview.net\/forum?id=1PL1NIMMrw."},{"key":"2440_CR17","doi-asserted-by":"publisher","unstructured":"Elazar, Y., Kassner, N., Ravfogel, S., Ravichander, A., Hovy, E., Sch\u00fctze, H., et\u00a0al., Measuring and improving consistency in pretrained language models. Trans. Assoc. Computat. Linguist. 9: 1012\u20131031, 2021. https:\/\/doi.org\/10.1162\/tacl_a_00410","DOI":"10.1162\/tacl_a_00410"},{"key":"2440_CR18","doi-asserted-by":"publisher","unstructured":"Sai, A. B., Mohankumar, A. K., and Khapra, M. M., A survey of evaluation metrics used for NLG systems. ACM Comput. Surv. 55(2):26:1\u201326:39, 2022. https:\/\/doi.org\/10.1145\/3485766.","DOI":"10.1145\/3485766"},{"key":"2440_CR19","doi-asserted-by":"publisher","unstructured":"Song, J. W., Park, J., Kim, J. H., and You, S. C., Large language model assistant for emergency department discharge documentation. JAMA Netw. Open 8(10):e2538427, 2025. https:\/\/doi.org\/10.1001\/jamanetworkopen.2025.38427.","DOI":"10.1001\/jamanetworkopen.2025.38427"},{"key":"2440_CR20","doi-asserted-by":"publisher","unstructured":"Williams, C. Y. K., Subramanian, C. R., Ali, S. S., Apolinario, M., Askin, E., Barish, P., et\u00a0al., Physician- and large language model-generated hospital discharge summaries. JAMA Intern. Med. 185(7):818\u2013825, 2025. https:\/\/doi.org\/10.1001\/jamainternmed.2025.0821.","DOI":"10.1001\/jamainternmed.2025.0821"},{"key":"2440_CR21","doi-asserted-by":"publisher","unstructured":"Seo, J., Choi, D., Kim, T., Cha, W. C., Kim, M., Yoo, H., et\u00a0al., Evaluation framework of large language models in medical documentation: Development and usability study. J. Med. Intern. Res. 26:e58329, 2024. https:\/\/doi.org\/10.2196\/58329.","DOI":"10.2196\/58329"},{"key":"2440_CR22","doi-asserted-by":"publisher","unstructured":"Huang, T., Safranek, C., Socrates, V., Chartash, D., Wright, D., Dilip, M., et\u00a0al., Patient-representing population\u2019s perceptions of gpt-generated versus standard emergency department discharge instructions: Randomized blind survey assessment. J. Med. Intern. Res. 26:e60336, 2024. https:\/\/doi.org\/10.2196\/60336.","DOI":"10.2196\/60336"},{"key":"2440_CR23","doi-asserted-by":"publisher","unstructured":"Feldman, J., Hochman, K. A., Guzman, B. V., Goodman, A., Weisstuch, J., and Testa, P., Scaling note quality assessment across an academic medical center with AI and GPT-4. NEJM Catal. Innov. Care Deliv. 5(5), 2024. https:\/\/doi.org\/10.1056\/CAT.23.0283.","DOI":"10.1056\/CAT.23.0283"},{"key":"2440_CR24","doi-asserted-by":"publisher","unstructured":"Jin, Q., Chen, F., Zhou, Y., Xu, Z., Cheung, J. M., Chen, R., et\u00a0al. Hidden flaws behind expert-level accuracy of multimodal GPT-4 Vision in medicine. NPJ Digit. Med. 7(1):190, 2024. https:\/\/doi.org\/10.1038\/s41746-024-01185-7.","DOI":"10.1038\/s41746-024-01185-7"},{"key":"2440_CR25","doi-asserted-by":"publisher","unstructured":"Johri, S., Jeong, J., Tran, B. A., Schlessinger, D. I., Wongvibulsin, S., Barnes, L. A., et\u00a0al., An evaluation framework for clinical use of large language models in patient interaction tasks. Nat. Med. 31(1):77\u201386, 2025. https:\/\/doi.org\/10.1038\/s41591-024-03328-5.","DOI":"10.1038\/s41591-024-03328-5"},{"key":"2440_CR26","doi-asserted-by":"publisher","unstructured":"Uzuner, \u00d6., South, B. R., Shen, S., and DuVall, S. L., 2010 i2b2\/VA challenge on concepts, assertions, and relations in clinical text. J. Am. Med. Inf. Assoc. 18(5):552\u2013556, 2011. https:\/\/doi.org\/10.1136\/amiajnl-2011-000203.","DOI":"10.1136\/amiajnl-2011-000203"},{"key":"2440_CR27","doi-asserted-by":"crossref","unstructured":"Liu, F., Shareghi, E., Meng, Z., Basaldella, M., and Collier, N., Self-alignment pretraining for biomedical entity representations. In: Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL-HLT). Online: Association for Computational Linguistics. p. 4228\u20134238, 2021. Available from: https:\/\/aclanthology.org\/2021.naacl-main.334\/.","DOI":"10.18653\/v1\/2021.naacl-main.334"},{"key":"2440_CR28","unstructured":"Kuhn, L., Gal, Y., and Farquhar, S., Semantic uncertainty: Linguistic invariances for uncertainty estimation in natural language generation. In: International Conference on Learning Representations (ICLR 2023), 2023. Available from: https:\/\/openreview.net\/forum?id=VD-AYtP0dve."},{"key":"2440_CR29","unstructured":"Wang, B., Chang, J., Qian, Y., Chen, G., Chen, J., Jiang, Z., et\u00a0al., DiReCT: Diagnostic reasoning for clinical notes via large language models. In: Advances in Neural Information Processing Systems (NeurIPS 2024), 2024. Available from: https:\/\/openreview.net\/forum?id=F7rAX6yiS2."},{"key":"2440_CR30","doi-asserted-by":"publisher","unstructured":"Croxford, E., Gao, Y., First, E., Pellegrino, N., Schnier, M., Caskey, J., et\u00a0al., Automating evaluation of AI text generation in healthcare with a Large Language Model (LLM)-as-a-Judge. medRxiv. 2025. Preprint. https:\/\/doi.org\/10.1101\/2025.04.22.25326219.","DOI":"10.1101\/2025.04.22.25326219"},{"key":"2440_CR31","unstructured":"Johnson, A. E. W., Bulgarelli, L., Pollard, T., Horng, S., Celi, L. A., and Mark, R., MIMIC-IV (version 2.2). PhysioNet. Available from: https:\/\/physionet.org\/content\/mimiciv\/2.2\/."},{"key":"2440_CR32","unstructured":"Johnson, A. E. W., Pollard, T., Horng, S., Celi, L. A., and Mark, R., MIMIC-IV-Note: Deidentified free-text clinical notes (version 2.2). PhysioNet. Available from: https:\/\/physionet.org\/content\/mimic-iv-note\/2.2\/."},{"key":"2440_CR33","doi-asserted-by":"publisher","unstructured":"Johnson, A. E. W., Bulgarelli, L., Shen, L., Gayles, A., Shammout, A., Horng, S., et\u00a0al., MIMIC-IV, a freely accessible electronic health record dataset. Scientif. Data. 10(1):1, 2023. https:\/\/doi.org\/10.1038\/s41597-022-01899-x.","DOI":"10.1038\/s41597-022-01899-x"},{"key":"2440_CR34","unstructured":"Jung, J., Brahman, F., and Choi, Y., Trust or escalate: LLM Judges with provable guarantees for human agreement. In: International Conference on Learning Representations (ICLR 2025), 2025. Available from: https:\/\/openreview.net\/forum?id=UHPnqSTBPO"},{"key":"2440_CR35","doi-asserted-by":"crossref","unstructured":"Chen, J., Cai, Z., Ji, K., Wang, X., Liu, W., Wang, R., et al., Towards medical complex reasoning with LLMs through medical verifiable problems. In: Findings of the Association for Computational Linguistics: ACL 2025. Association for Computational Linguistics, Vienna, Austria (2025), pp. 14552\u201314573. Available from: https:\/\/aclanthology.org\/2025.findings-acl.751\/","DOI":"10.18653\/v1\/2025.findings-acl.751"},{"key":"2440_CR36","unstructured":"Grattafiori, A., Dubey, A., Jauhri, A., Pandey, A., Kadian, A., Al-Dahle, A., et\u00a0al., The llama 3 herd of models. Available from: arxiv:2407.21783."},{"key":"2440_CR37","doi-asserted-by":"publisher","unstructured":"Bodenreider, O., The Unified Medical Language System (UMLS): integrating biomedical terminology. Nucl. Acids Res. 32(Database issue):D267\u2013D270, 2004. https:\/\/doi.org\/10.1093\/nar\/gkh061.","DOI":"10.1093\/nar\/gkh061"},{"key":"2440_CR38","doi-asserted-by":"crossref","unstructured":"Alsentzer, E., Murphy, J. R., Boag, W., Weng, W. H., Jin, D., Naumann, T., et\u00a0al., Publicly available clinical BERT embeddings. In: Proceedings of the 2nd Clinical Natural Language Processing Workshop. p. 72\u201378, 2019. Available from: https:\/\/aclanthology.org\/W19-1909\/.","DOI":"10.18653\/v1\/W19-1909"},{"key":"2440_CR39","doi-asserted-by":"crossref","unstructured":"Reimers, N., and Gurevych, I., Sentence-BERT: Sentence embeddings using siamese BERT-networks. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP). p. 3982\u20133992, 2019. Available from: https:\/\/aclanthology.org\/D19-1410\/","DOI":"10.18653\/v1\/D19-1410"},{"key":"2440_CR40","doi-asserted-by":"publisher","unstructured":"Gu, Y., Tinn, R., Cheng, H., Lucas, M., Usuyama, N., Liu, X, et\u00a0al., Domain-specific language model pretraining for biomedical natural language processing. ACM Trans. Comput. Healthc. 3(1):1\u201323, 2021. https:\/\/doi.org\/10.1145\/3458754.","DOI":"10.1145\/3458754"},{"key":"2440_CR41","doi-asserted-by":"crossref","unstructured":"Turpin, M., Michael, J., Perez, E., & Bowman, S. R., Language models don\u2019t always say what they think: Unfaithful explanations in chain-of-thought prompting. Adv. Neural Inf. Process. Syst. (NeurIPS) 36:74952\u201374965, 2023.","DOI":"10.52202\/075280-3275"},{"key":"2440_CR42","unstructured":"Shankar, R. D., Tu, S. W., and Musen, M. A., Medical arguments in an automated health care system. In: Argumentation for Consumers of Healthcare: Papers from the 2006 AAAI Spring Symposium. Technical Report SS-06-01. Menlo Park, CA: AAAI Press. p. 96\u2013104, 2006."}],"container-title":["Journal of Medical Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10916-026-02440-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10916-026-02440-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10916-026-02440-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T00:38:09Z","timestamp":1784680689000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10916-026-02440-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,22]]},"references-count":42,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,12]]}},"alternative-id":["2440"],"URL":"https:\/\/doi.org\/10.1007\/s10916-026-02440-y","relation":{},"ISSN":["1573-689X"],"issn-type":[{"value":"1573-689X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,22]]},"assertion":[{"value":"20 April 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 July 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","label":"Competing interests","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This study used de-identified data from the publicly available MIMIC-IV electronic health record dataset, which has been approved for use under a data use agreement with PhysioNet. No additional ethics approval was required for secondary analysis of de-identified data.","order":2,"name":"Ethics","label":"Ethics approval and consent to participate","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":3,"name":"Ethics","label":"Consent for publication","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":4,"name":"Ethics","label":"Clinical trial number","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"117"}}