{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:06:16Z","timestamp":1783209976193,"version":"3.54.6"},"reference-count":91,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s00371-026-04437-7","type":"journal-article","created":{"date-parts":[[2026,3,26]],"date-time":"2026-03-26T09:06:11Z","timestamp":1774515971000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Vision\u2013language foundation model driven agentic AI systems for healthcare"],"prefix":"10.1007","volume":"42","author":[{"given":"Lifeng","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinming","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiming","family":"Qin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haoxuan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nan","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,3,26]]},"reference":[{"issue":"4","key":"4437_CR1","doi-asserted-by":"publisher","first-page":"718","DOI":"10.1016\/j.dld.2024.01.191","volume":"56","author":"Y Zhang","year":"2024","unstructured":"Zhang, Y., et al.: Unexpectedly low accuracy of GPT-4 in identifying common liver diseases from CT scan images. Dig. Liver Dis. 56(4), 718\u2013720 (2024)","journal-title":"Dig. Liver Dis."},{"key":"4437_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2025.107819","author":"J Zhang","year":"2025","unstructured":"Zhang, J., et al.: Gradient amplification for gradient matching based dataset distillation. Neural Netw. (2025). https:\/\/doi.org\/10.1016\/j.neunet.2025.107819","journal-title":"Neural Netw."},{"key":"4437_CR3","doi-asserted-by":"crossref","unstructured":"Terleira Fern\u00e1ndez, Paula, et al. Assessing the object-detection skills of modern vision language models. International Conference on Disruptive Technologies, Tech Ethics and Artificial Intelligence. Cham: Springer Nature Switzerland, 2025.","DOI":"10.1007\/978-3-031-99474-6_26"},{"issue":"2","key":"4437_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3783994","volume":"22","author":"Z Shao","year":"2026","unstructured":"Shao, Z., et al.: Spatio-temporal disentanglement and constrained self-attention for multi-modal deception detection. ACM Trans. Multimedia Comput. Commun. Appl. 22(2), 1\u201320 (2026)","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl."},{"issue":"8","key":"4437_CR5","doi-asserted-by":"publisher","DOI":"10.7150\/thno.100786","volume":"15","author":"J Wang","year":"2025","unstructured":"Wang, J., et al.: Artificial intelligence-enhanced retinal imaging as a biomarker for systemic diseases. Theranostics 15(8), 3223 (2025)","journal-title":"Theranostics"},{"issue":"8","key":"4437_CR6","first-page":"864","volume":"66","author":"TY Wong","year":"2025","unstructured":"Wong, T.Y., et al.: EyeFM: a multi-modal vision-language copilot foundation model for eyecare. Investigative Ophthalmol. Visual Sci. 66(8), 864\u2013864 (2025)","journal-title":"Investigative Ophthalmol. Visual Sci."},{"issue":"6","key":"4437_CR7","doi-asserted-by":"publisher","first-page":"3871","DOI":"10.1007\/s00371-024-03391-6","volume":"40","author":"SG Ali","year":"2024","unstructured":"Ali, S.G., et al.: AI-enhanced digital technologies for myopia management: advancements, challenges, and future prospects. Vis. Comput. 40(6), 3871\u20133887 (2024)","journal-title":"Vis. Comput."},{"key":"4437_CR8","unstructured":"Vaswani, Ashish, et al. Attention is all you need. Adv. Neural Inf. Process. Syst. 30 (2017)."},{"key":"4437_CR9","doi-asserted-by":"crossref","unstructured":"Le, Phong, and Willem Zuidema. Quantifying the vanishing gradient and long distance dependency problem in recursive neural networks and recursive LSTMs. arXiv preprint arXiv:1603.00423 (2016).","DOI":"10.18653\/v1\/W16-1610"},{"key":"4437_CR10","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-025-03822-y","author":"J Wang","year":"2025","unstructured":"Wang, J., et al.: Temporal goal-aware transformer assisted visual reinforcement learning for virtual table tennis agent. Vis. Comput. (2025). https:\/\/doi.org\/10.1007\/s00371-025-03822-y","journal-title":"Vis. Comput."},{"key":"4437_CR11","doi-asserted-by":"publisher","DOI":"10.1136\/bjo-2024-326254","author":"H Chen","year":"2025","unstructured":"Chen, H., et al.: Can large language models fully automate or partially assist paper selection in systematic reviews? Br. J. Ophthalmol. (2025). https:\/\/doi.org\/10.1136\/bjo-2024-326254","journal-title":"Br. J. Ophthalmol."},{"issue":"2","key":"4437_CR12","doi-asserted-by":"publisher","DOI":"10.1002\/mef2.70019","volume":"4","author":"H Cheng","year":"2025","unstructured":"Cheng, H., et al.: The use of large language models and their association with enhanced impact in biomedical research and beyond. MedComm-Future Med. 4(2), e70019 (2025)","journal-title":"MedComm-Future Med."},{"key":"4437_CR13","unstructured":"Lu, Jiasen, et al. Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Adv. Neural Inf. Process. Syst. 32 (2019)."},{"key":"4437_CR14","doi-asserted-by":"crossref","unstructured":"Tan, Hao, and Mohit Bansal. Lxmert: learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490 (2019).","DOI":"10.18653\/v1\/D19-1514"},{"key":"4437_CR15","unstructured":"Devlin, Jacob, et al. Bert: Pre-training of deep bidirectional transformers for language understanding. Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, volume 1 (long and short papers). 2019."},{"key":"4437_CR16","doi-asserted-by":"crossref","unstructured":"Lin, Tsung-Yi, et al. Microsoft coco: common objects in context. European Conference on Computer Vision. Springer International Publishing, Cham, 2014.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"4437_CR17","doi-asserted-by":"crossref","unstructured":"Sharma, Piyush, et al. Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 2018.","DOI":"10.18653\/v1\/P18-1238"},{"issue":"6","key":"4437_CR18","doi-asserted-by":"publisher","first-page":"7478","DOI":"10.1109\/TNNLS.2022.3227717","volume":"35","author":"Y Liu","year":"2023","unstructured":"Liu, Y., et al.: A survey of visual transformers. IEEE Trans. Neural Networks and Learn. Syst. 35(6), 7478\u20137498 (2023)","journal-title":"IEEE Trans. Neural Networks and Learn. Syst."},{"key":"4437_CR19","unstructured":"Dosovitskiy, Alexey. An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"4437_CR20","doi-asserted-by":"crossref","unstructured":"Deng, Jia, et al. Imagenet: a large-scale hierarchical image database. 2009 IEEE Conference on Computer Vision and Pattern Recognition. Ieee, 2009.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"4437_CR21","unstructured":"Abnar, Samira, et al. Exploring the limits of large scale pre-training. arXiv preprint arXiv:2110.02095 (2021)."},{"key":"4437_CR22","doi-asserted-by":"crossref","unstructured":"Liu, Ze, et al. Swin transformer: hierarchical vision transformer using shifted windows. Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2021.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"4437_CR23","unstructured":"Radford, Alec, et al. Learning transferable visual models from natural language supervision. International Conference on Machine Learning. PmLR, 2021."},{"issue":"4","key":"4437_CR24","doi-asserted-by":"publisher","first-page":"2775","DOI":"10.1007\/s00371-023-02985-w","volume":"40","author":"S Masood","year":"2024","unstructured":"Masood, S., et al.: Deep choroid layer segmentation using hybrid features extraction from OCT images. Vis. Comput. 40(4), 2775\u20132792 (2024)","journal-title":"Vis. Comput."},{"key":"4437_CR25","unstructured":"Su, Weijie, et al. Vl-bert: Pre-training of generic visual-linguistic representations. arXiv preprint arXiv:1908.08530 (2019)."},{"issue":"1","key":"4437_CR26","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna, R., et al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. Int. J. Comput. Vis. 123(1), 32\u201373 (2017)","journal-title":"Int. J. Comput. Vis."},{"key":"4437_CR27","doi-asserted-by":"crossref","unstructured":"Goyal, Yash, et al. Making the v in vqa matter: elevating the role of image understanding in visual question answering. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2017.","DOI":"10.1109\/CVPR.2017.670"},{"key":"4437_CR28","doi-asserted-by":"crossref","unstructured":"Hudson, Drew A., and Christopher D. Manning. Gqa: a new dataset for real-world visual reasoning and compositional question answering. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2019.","DOI":"10.1109\/CVPR.2019.00686"},{"key":"4437_CR29","doi-asserted-by":"crossref","unstructured":"Vinyals, Oriol, et al. Show and tell: a neural image caption generator. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 2015.","DOI":"10.1109\/CVPR.2015.7298935"},{"issue":"12","key":"4437_CR30","doi-asserted-by":"publisher","first-page":"4404","DOI":"10.1109\/TMI.2024.3421644","volume":"43","author":"J Xiao","year":"2024","unstructured":"Xiao, J., et al.: Multi-label chest x-ray image classification with single positive labels. IEEE Trans. Med. Imaging 43(12), 4404\u20134418 (2024)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"4437_CR31","doi-asserted-by":"crossref","unstructured":"Li, Tingyao, and Bin Sheng. MSCE-LT: multi-label supervised contrastive enhancement for long-tailed retinal diseases recognition. 2024 IEEE International Conference on Bioinformatics and Biomedicine (BIBM). IEEE, 2024.","DOI":"10.1109\/BIBM62325.2024.10821772"},{"key":"4437_CR32","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2025.3590938","author":"M Xiong","year":"2025","unstructured":"Xiong, M., et al.: Adaptive clustering and weighted regularization contrastive learning framework for unsupervised person re-identification. IEEE Trans. Multimedia (2025). https:\/\/doi.org\/10.1109\/TMM.2025.3590938","journal-title":"IEEE Trans. Multimedia"},{"key":"4437_CR33","doi-asserted-by":"publisher","first-page":"113661","DOI":"10.1016\/j.engappai.2025.113661","volume":"166","author":"G Qiu","year":"2026","unstructured":"Qiu, G., Chen, Z., Dai, L., Li, P., Sheng, B.: Slimmable neural architecture design based on cross architecture and token distillation. Eng. Appl. Artif. Intell. 166, 113661 (2026)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"4437_CR34","first-page":"9694","volume":"34","author":"J Li","year":"2021","unstructured":"Li, J., et al.: Align before fuse: vision and language representation learning with momentum distillation. Adv. Neural Inf. Process. Syst. 34, 9694\u20139705 (2021)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"4437_CR35","doi-asserted-by":"crossref","unstructured":"Singh, Amanpreet, et al. Flava: a foundational language and vision alignment model. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2022.","DOI":"10.1109\/CVPR52688.2022.01519"},{"issue":"9","key":"4437_CR36","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., et al.: Learning to prompt for vision-language models. Int. J. Comput. Vis. 130(9), 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vis."},{"key":"4437_CR37","doi-asserted-by":"publisher","DOI":"10.1109\/TASE.2025.3604283","author":"J Zhu","year":"2025","unstructured":"Zhu, J., et al.: Semi-supervised privacy-preserving EEG-based motor imagery classification via self and adversarial training. IEEE Trans. Autom. Sci. Eng. (2025). https:\/\/doi.org\/10.1109\/TASE.2025.3604283","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"key":"4437_CR38","doi-asserted-by":"crossref","unstructured":"Hu, Yutao, et al. Omnimedvqa: a new large-scale comprehensive evaluation benchmark for medical lvlm. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2024.","DOI":"10.1109\/CVPR52733.2024.02093"},{"key":"4437_CR39","unstructured":"Li, Junnan, et al. Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. International Conference on Machine Learning. PMLR, 2023."},{"key":"4437_CR40","unstructured":"Zhu, Deyao, et al. Minigpt-4: enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592 (2023)."},{"key":"4437_CR41","doi-asserted-by":"publisher","first-page":"23716","DOI":"10.52202\/068431-1723","volume":"35","author":"JB Alayrac","year":"2022","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. Adv. Neural Inf. Process. Syst. 35, 23716\u201323736 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"4437_CR42","unstructured":"Team Gemini, et al. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"4437_CR43","doi-asserted-by":"crossref","unstructured":"Wang, Zifeng, et al. Medclip: Contrastive learning from unpaired medical images and text. Proceedings of the Conference on Empirical Methods in Natural Language Processing. Conference on Empirical Methods in Natural Language Processing. Vol. 2022. 2022.","DOI":"10.18653\/v1\/2022.emnlp-main.256"},{"key":"4437_CR44","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2025.3543001","author":"J Zhu","year":"2025","unstructured":"Zhu, J., et al.: SGG-Nets: generic rotation-invariant plugin networks for point cloud analysis. IEEE Trans. Multimedia (2025). https:\/\/doi.org\/10.1109\/TMM.2025.3543001","journal-title":"IEEE Trans. Multimedia"},{"issue":"1","key":"4437_CR45","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-025-62385-7","volume":"16","author":"C Wu","year":"2025","unstructured":"Wu, C., et al.: Towards generalist foundation model for radiology by leveraging web-scale 2d&3d medical data. Nat. Commun. 16(1), 7866 (2025)","journal-title":"Nat. Commun."},{"key":"4437_CR46","unstructured":"Bai, Fan, et al. M3d: Advancing 3d medical image analysis with multi-modal large language models. arXiv preprint arXiv:2404.00578 (2024)."},{"key":"4437_CR47","unstructured":"Lai, Haoran, et al. E3D-GPT: enhanced 3D visual foundation for medical vision-language model. arXiv preprint arXiv:2410.14200 (2024)."},{"key":"4437_CR48","doi-asserted-by":"publisher","first-page":"3404","DOI":"10.1038\/s41591-025-03900-7","volume":"31","author":"Y Wu","year":"2025","unstructured":"Wu, Y., Qian, B., Li, T., et al.: An eyecare foundation model for clinical assistance: a randomized controlled trial. Nat. Med. 31, 3404\u20133413 (2025)","journal-title":"Nat. Med."},{"issue":"2","key":"4437_CR49","doi-asserted-by":"publisher","DOI":"10.1002\/acm2.14268","volume":"25","author":"C Liu","year":"2024","unstructured":"Liu, C., et al.: Improvements to a GLCM\u2010based machine\u2010learning approach for quantifying posterior capsule opacification. J. Appl. Clin. Med. Phys. 25(2), e14268 (2024)","journal-title":"J. Appl. Clin. Med. Phys."},{"key":"4437_CR50","doi-asserted-by":"publisher","DOI":"10.1016\/j.preteyeres.2025.101353","author":"B Phipps","year":"2025","unstructured":"Phipps, B., et al.: AI image generation technology in ophthalmology: use, misuse and future applications. Prog. Retin. Eye Res. (2025). https:\/\/doi.org\/10.1016\/j.preteyeres.2025.101353","journal-title":"Prog. Retin. Eye Res."},{"key":"4437_CR51","doi-asserted-by":"crossref","unstructured":"He, Xuehai, et al. Towards visual question answering on pathology images. Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 2: Short Papers). 2021.","DOI":"10.18653\/v1\/2021.acl-short.90"},{"issue":"1","key":"4437_CR52","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/sdata.2018.251","volume":"5","author":"JJ Lau","year":"2018","unstructured":"Lau, J.J., et al.: A dataset of clinically generated visual questions and answers about radiology images. Sci. Data. 5(1), 1\u201310 (2018)","journal-title":"Sci. Data."},{"key":"4437_CR53","first-page":"28541","volume":"36","author":"C Li","year":"2023","unstructured":"Li, C., et al.: Llava-med: training a large language-and-vision assistant for biomedicine in one day. Adv. Neural Inf. Process. Syst. 36, 28541\u201328564 (2023)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"4437_CR54","unstructured":"Murari Vepa, Arvind, et al. A multimodal LLM approach for visual question answering on multiparametric 3D brain MRI. arXiv e-prints (2025): arXiv-2509."},{"issue":"2","key":"4437_CR55","doi-asserted-by":"publisher","first-page":"1097","DOI":"10.1007\/s00371-024-03386-3","volume":"41","author":"Y Zhou","year":"2025","unstructured":"Zhou, Y., et al.: GAMNet: a gated attention mechanism network for grading myopic traction maculopathy in OCT images. Vis. Comput. 41(2), 1097\u20131108 (2025)","journal-title":"Vis. Comput."},{"key":"4437_CR56","doi-asserted-by":"publisher","DOI":"10.1007\/s00330-024-11339-6","author":"S Lee","year":"2025","unstructured":"Lee, S., et al.: CXR-LLAVA: a multimodal large language model for interpreting chest X-ray images. Eur. Radiol. (2025). https:\/\/doi.org\/10.1007\/s00330-024-11339-6","journal-title":"Eur. Radiol."},{"key":"4437_CR57","doi-asserted-by":"crossref","unstructured":"Xin, Yu, et al. Med3dvlm: an efficient vision-language model for 3d medical image analysis. arXiv preprint arXiv:2503.20047 (2025).","DOI":"10.1109\/JBHI.2025.3604595"},{"key":"4437_CR58","unstructured":"Ates, Gorkem Can, et al. Dcformer: Efficient 3d vision-language modeling with decomposed convolutions. arXiv preprint arXiv:2502.05091 (2025)."},{"key":"4437_CR59","doi-asserted-by":"crossref","unstructured":"Kapadnis, Manav, et al. SERPENT-VLM: self-refining radiology report generation using vision language models. Proceedings of the 6th Clinical Natural Language Processing Workshop. 2024.","DOI":"10.18653\/v1\/2024.clinicalnlp-1.24"},{"issue":"8051","key":"4437_CR60","doi-asserted-by":"publisher","first-page":"769","DOI":"10.1038\/s41586-024-08378-w","volume":"638","author":"J Xiang","year":"2025","unstructured":"Xiang, J., et al.: A vision\u2013language foundation model for precision oncology. Nature 638(8051), 769\u2013778 (2025)","journal-title":"Nature"},{"issue":"3","key":"4437_CR61","doi-asserted-by":"publisher","first-page":"863","DOI":"10.1038\/s41591-024-02856-4","volume":"30","author":"MY Lu","year":"2024","unstructured":"Lu, M.Y., et al.: A visual-language foundation model for computational pathology. Nat. Med. 30(3), 863\u2013874 (2024)","journal-title":"Nat. Med."},{"key":"4437_CR62","doi-asserted-by":"crossref","unstructured":"Ma, Liangdi, et al. A vision\u2013language pretrained transformer for versatile clinical respiratory disease applications. Nature Biomed. Eng. (2025): 1\u201319.","DOI":"10.1038\/s41551-025-01544-z"},{"issue":"11","key":"4437_CR63","doi-asserted-by":"publisher","first-page":"3129","DOI":"10.1038\/s41591-024-03185-2","volume":"30","author":"K Zhang","year":"2024","unstructured":"Zhang, K., et al.: A generalist vision\u2013language foundation model for diverse biomedical tasks. Nat. Med. 30(11), 3129\u20133141 (2024)","journal-title":"Nat. Med."},{"key":"4437_CR64","doi-asserted-by":"crossref","unstructured":"Gai, Xiaotang, et al. Medthink: a rationale-guided framework for explaining medical visual question answering. Findings of the Association for Computational Linguistics: NAACL 2025. 2025.","DOI":"10.18653\/v1\/2025.findings-naacl.415"},{"key":"4437_CR65","unstructured":"Lin, Tianwei, et al. Healthgpt: a medical large vision-language model for unifying comprehension and generation via heterogeneous knowledge adaptation. arXiv preprint arXiv:2502.09838 (2025)."},{"key":"4437_CR66","unstructured":"Jiang, Songtao, et al. Hulu-med: a transparent generalist model towards holistic medical vision-language understanding. arXiv preprint arXiv:2510.08668 (2025)."},{"key":"4437_CR67","unstructured":"Xu, Jiao, et al. PulseMind: A multi-modal medical model for real-world clinical diagnosis. arXiv preprint arXiv:2601.07344 (2026)."},{"key":"4437_CR68","unstructured":"Nguyen, Thien, et al. QwenVLConnector: a fast, unified medical VLM chatbot for fine-grained clinical perception and text generation. MICCAI 2025 FLARE Challenge."},{"key":"4437_CR69","unstructured":"Xing, Yang, et al. MedVL-SAM2: a unified 3D medical vision-language model for multimodal reasoning and prompt-driven segmentation. arXiv preprint arXiv:2601.09879 (2026)."},{"issue":"2","key":"4437_CR70","doi-asserted-by":"publisher","first-page":"1061","DOI":"10.1007\/s00371-024-03384-5","volume":"41","author":"B Qian","year":"2025","unstructured":"Qian, B., et al.: HRDC challenge: a public benchmark for hypertension and hypertensive retinopathy classification from fundus images. Vis. Comput. 41(2), 1061\u20131077 (2025)","journal-title":"Vis. Comput."},{"key":"4437_CR71","doi-asserted-by":"crossref","unstructured":"Baghbanzadeh, Negin, et al. Advancing medical representation learning through high-quality data. International Conference on Medical Image Computing and Computer-Assisted Intervention. Cham: Springer Nature Switzerland, 2025.","DOI":"10.1007\/978-3-032-05169-1_3"},{"issue":"6","key":"4437_CR72","doi-asserted-by":"publisher","DOI":"10.1161\/JAHA.123.033584","volume":"13","author":"P Li","year":"2024","unstructured":"Li, P., et al.: Potential multidisciplinary use of large language models for addressing queries in cardio\u2010oncology. J. Am. Heart Assoc. 13(6), e033584 (2024)","journal-title":"J. Am. Heart Assoc."},{"key":"4437_CR73","unstructured":"Yan, Qiao, et al. Medhalltune: an instruction-tuning benchmark for mitigating medical hallucination in vision-language models. arXiv preprint arXiv:2502.20780 (2025)."},{"issue":"4","key":"4437_CR74","doi-asserted-by":"publisher","first-page":"1383","DOI":"10.1007\/s00146-021-01247-4","volume":"37","author":"S Han","year":"2022","unstructured":"Han, S., et al.: Aligning artificial intelligence with human values: reflections from a phenomenological perspective. AI Soc. 37(4), 1383\u20131395 (2022)","journal-title":"AI Soc."},{"issue":"8","key":"4437_CR75","doi-asserted-by":"publisher","first-page":"569","DOI":"10.1016\/S2213-8587(24)00154-2","volume":"12","author":"B Sheng","year":"2024","unstructured":"Sheng, B., et al.: Artificial intelligence for diabetes care: current and future prospects. Lancet Diabetes Endocrinol. 12(8), 569\u2013595 (2024)","journal-title":"Lancet Diabetes Endocrinol."},{"issue":"6464","key":"4437_CR76","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1126\/science.aax2342","volume":"366","author":"Z Obermeyer","year":"2019","unstructured":"Obermeyer, Z., et al.: Dissecting racial bias in an algorithm used to manage the health of populations. Science 366(6464), 447\u2013453 (2019)","journal-title":"Science"},{"issue":"11","key":"4437_CR77","doi-asserted-by":"publisher","first-page":"981","DOI":"10.1056\/NEJMp1714229","volume":"378","author":"DS Char","year":"2018","unstructured":"Char, D.S., Shah, N.H., Magnus, D.: Implementing machine learning in health care\u2014addressing ethical challenges. N. Engl. J. Med. 378(11), 981 (2018)","journal-title":"N. Engl. J. Med."},{"issue":"1","key":"4437_CR78","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1038\/s41746-020-0288-5","volume":"3","author":"D Cirillo","year":"2020","unstructured":"Cirillo, D., Catuara-Solarz, S., Morey, C., Guney, E., Subirats, L., Mellino, S., Gigante, A., Valencia, A., Rementeria, M.J., Chadha, A.S., Mavridis, N.: Sex and gender differences and biases in artificial intelligence for biomedicine and healthcare. NPJ Digital Medicine 3(1), 81 (2020)","journal-title":"NPJ Digital Medicine"},{"issue":"6","key":"4437_CR79","doi-asserted-by":"publisher","first-page":"e279","DOI":"10.1016\/j.jhep.2023.11.017","volume":"80","author":"Y Zhang","year":"2024","unstructured":"Zhang, Y., et al.: Preliminary fatty liver disease grading using general-purpose online large language models: ChatGPT-4 or Bard? J. Hepatol. 80(6), e279\u2013e281 (2024)","journal-title":"J. Hepatol."},{"key":"4437_CR80","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1038\/s44360-025-00024-7","volume":"1","author":"JCL Ong","year":"2026","unstructured":"Ong, J.C.L., Ning, Y., Yang, R., et al.: Large language models in global health. Nature Health 1, 35\u201347 (2026)","journal-title":"Nature Health"},{"issue":"5","key":"4437_CR81","doi-asserted-by":"publisher","first-page":"583","DOI":"10.1016\/j.scib.2024.01.004","volume":"69","author":"B Sheng","year":"2024","unstructured":"Sheng, B., et al.: Large language models for diabetes care: potentials and prospects. Sci. Bull. 69(5), 583\u2013588 (2024)","journal-title":"Sci. Bull."},{"issue":"12","key":"4437_CR82","doi-asserted-by":"publisher","first-page":"1384","DOI":"10.1038\/s41551-022-00872-8","volume":"6","author":"G Erion","year":"2022","unstructured":"Erion, G., et al.: A cost-aware framework for the development of AI models for healthcare applications. Nature Biomed. Eng. 6(12), 1384\u20131398 (2022)","journal-title":"Nature Biomed. Eng."},{"key":"4437_CR83","doi-asserted-by":"crossref","unstructured":"H. Li, D. Guo, W. Fan, M. Xu, and Y. Song, Multi-step jailbreaking privacy attacks on chatgpt, arXiv preprint arXiv:2304.05197, 2023.","DOI":"10.18653\/v1\/2023.findings-emnlp.272"},{"key":"4437_CR84","unstructured":"J. Duan, F. Kong, S. Wang, X. Shi, and K. Xu, Are diffusion models vulnerable to membership inference attacks? arXiv preprint arXiv:2302.01316, 2023."},{"key":"4437_CR85","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1038\/s44360-025-00016-7","volume":"1","author":"YM Hwang","year":"2026","unstructured":"Hwang, Y.M., Ng, M.Y., Pillai, M., et al.: The landscape of AI implementation in US hospitals. Nat. Health 1, 99\u2013112 (2026)","journal-title":"Nat. Health"},{"key":"4437_CR86","unstructured":"Zhou, Yukun, et al. Revealing the impact of pre-training data on medical foundation models. (2025)."},{"key":"4437_CR87","volume":"63","author":"H Chen","year":"2025","unstructured":"Chen, H., et al.: Large language models and global health equity: a roadmap for equitable adoption in LMICs. Lancet Reg. Health West. Pac. 63, 101707 (2025)","journal-title":"Lancet Reg. Health West. Pac."},{"key":"4437_CR88","doi-asserted-by":"publisher","DOI":"10.1038\/s41591-025-03787-4","author":"JCL Ong","year":"2025","unstructured":"Ong, J.C.L., et al.: International partnership for governing generative artificial intelligence models in medicine. Nat. Med. (2025). https:\/\/doi.org\/10.1038\/s41591-025-03787-4","journal-title":"Nat. Med."},{"key":"4437_CR89","doi-asserted-by":"crossref","unstructured":"Gehman, Samuel, et al. Realtoxicityprompts: evaluating neural toxic degeneration in language models. arXiv preprint arXiv:2009.11462 (2020).","DOI":"10.18653\/v1\/2020.findings-emnlp.301"},{"issue":"10","key":"4437_CR90","doi-asserted-by":"publisher","first-page":"2396","DOI":"10.1038\/s41591-023-02412-6","volume":"29","author":"S Gilbert","year":"2023","unstructured":"Gilbert, S., Harvey, H., Melvin, T., et al.: Large language model AI chatbots require approval as medical devices. Nature Medicine 29(10), 2396\u20132398 (2023)","journal-title":"Nature Medicine"},{"issue":"13","key":"4437_CR91","doi-asserted-by":"publisher","first-page":"1233","DOI":"10.1056\/NEJMsr2214184","volume":"388","author":"P Lee","year":"2023","unstructured":"Lee, P., Bubeck, S., Petro, J.: Benefits, limits, and risks of gpt-4 as an ai chatbot for medicine. N. Engl. J. Med. 388(13), 1233\u20131239 (2023)","journal-title":"N. Engl. J. Med."}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04437-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04437-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04437-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,9]],"date-time":"2026-04-09T13:39:59Z","timestamp":1775741999000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04437-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":91,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["4437"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04437-7","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3]]},"assertion":[{"value":"21 February 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"224"}}