{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T07:21:51Z","timestamp":1784359311155,"version":"3.55.0"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T00:00:00Z","timestamp":1766016000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T00:00:00Z","timestamp":1766016000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Biomed Semant"],"DOI":"10.1186\/s13326-025-00341-6","type":"journal-article","created":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T10:46:09Z","timestamp":1766054769000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["SimSUM \u2013 simulated benchmark with structured and unstructured medical records"],"prefix":"10.1186","volume":"16","author":[{"given":"Paloma","family":"Rabaey","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stefan","family":"Heytens","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Thomas","family":"Demeester","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,12,18]]},"reference":[{"key":"341_CR1","doi-asserted-by":"crossref","unstructured":"Ford E, Carroll JA, Smith HE, Scott D, Cassell JA. Extracting information from the text of electronic medical records to improve case detection: a systematic review. J Am Med Inf Assoc. 2016;23(5):1007\u201315.","DOI":"10.1093\/jamia\/ocv180"},{"issue":"1","key":"341_CR2","doi-asserted-by":"publisher","first-page":"7155","DOI":"10.1038\/s41598-020-62922-y","volume":"10","author":"Y Li","year":"2020","unstructured":"Li Y, Rao S, Solares JRA, Hassaine A, Ramakrishnan R, Canoy D, et al. BEHRT: transformer for electronic health records. Sci Rep. 2020;10(1):7155.","journal-title":"Sci Rep"},{"key":"341_CR3","doi-asserted-by":"publisher","first-page":"494","DOI":"10.1016\/j.eswa.2018.09.034","volume":"116","author":"G Mujtaba","year":"2019","unstructured":"Mujtaba G, Shuib L, Idris N, Hoo WL, Raj RG, Khowaja K, et al. Clinical text classification research trends: systematic literature review and open issues. Expert Syst Appl. 2019;116:494\u2013520.","journal-title":"Expert Syst Appl"},{"issue":"5","key":"341_CR4","doi-asserted-by":"publisher","first-page":"584","DOI":"10.1016\/j.cmi.2019.09.009","volume":"26","author":"N Peiffer-Smadja","year":"2020","unstructured":"Peiffer-Smadja N, Rawson TM, Ahmad R, Buchard A, et al. Machine learning for clinical decision support in infectious diseases: a narrative review of current applications. Clin Microbiol Infect. 2020;26(5):584\u201395.","journal-title":"Clin Microbiol Infect"},{"issue":"1","key":"341_CR5","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1038\/s41746-021-00455-y","volume":"4","author":"L Rasmy","year":"2021","unstructured":"Rasmy L, Xiang Y, Xie Z, Tao C, Zhi D. Med-BERT: pretrained contextualized embeddings on large-scale structured electronic health records for disease prediction. NPJ Digit Med. 2021;4(1):86.","journal-title":"NPJ Digit Med"},{"key":"341_CR6","unstructured":"Xu K, Lam M, Pang J, Gao X, Band C, Mathur P, et al. Multimodal machine learning for automated icd coding. Machine learning for healthcare conference PMLR. 2019. p. 197\u2013215."},{"key":"341_CR7","unstructured":"Huang K, Altosaar J, Ranganath R. Clinicalbert: modeling clinical notes and predicting hospital readmission. arXiv preprint arXiv:190405342. 2019."},{"key":"341_CR8","doi-asserted-by":"publisher","first-page":"5848","DOI":"10.18653\/v1\/2024.findings-acl.348","volume-title":"Findings of the association for computational linguistics: acl 2024","author":"Y Labrak","year":"2024","unstructured":"Labrak Y, Bazoge A, Morin E, Gourraud PA, Rouvier M, Dufour R. BioMistral: a collection of open-source pretrained large language models for medical domains. In: Findings of the association for computational linguistics: acl 2024. Bangkok, Thailand: Association for Computational Linguistics; 2024. p. 5848\u201364."},{"key":"341_CR9","unstructured":"Lehman E, Johnson A. Clinical-t5: large language models built using mimic clinical text. PhysioNet. 2023."},{"issue":"1","key":"341_CR10","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1109\/JBHI.2022.3217810","volume":"27","author":"S Liu","year":"2022","unstructured":"Liu S, Wang X, Hou Y, Li G, Wang H, Xu H, et al. Multimodal data matters: language model pre-training over structured and unstructured electronic health records. IEEE J Biomed and Health Inf. 2022;27(1):504\u201314.","journal-title":"IEEE J Biomed And Health Inf"},{"key":"341_CR11","doi-asserted-by":"crossref","unstructured":"Singhal K, Azizi S, Tu T, Mahdavi SS, Wei J, Chung HW, et al. Large language models encode clinical knowledge. Nature. 2023;620(7972):172\u2013180.","DOI":"10.1038\/s41586-023-06291-2"},{"key":"341_CR12","doi-asserted-by":"crossref","unstructured":"Zhang D, Yin C, Zeng J, Yuan X, Zhang P. Combining structured and unstructured data for predictive models: a deep learning approach. BMC Med Inf Decis Mak. 2020;20(1):280.","DOI":"10.1186\/s12911-020-01297-6"},{"key":"341_CR13","doi-asserted-by":"crossref","unstructured":"Quinn TP, Jacobs S, Senadeera M, Le V, Coghlan S. The three ghosts of medical ai: can the black-box present deliver? Artif Intel in Med. 2022;124:102158.","DOI":"10.1016\/j.artmed.2021.102158"},{"key":"341_CR14","doi-asserted-by":"crossref","unstructured":"Tian S, Jin Q, Yeganova L, Lai PT, Zhu Q, Chen X, et al. Opportunities and challenges for ChatGPT and large language models in biomedicine and health. Briefings In Bioinf. 2024;25(1):bbad493.","DOI":"10.1093\/bib\/bbad493"},{"key":"341_CR15","doi-asserted-by":"crossref","unstructured":"Zhao H, Chen H, Yang F, Liu N, Deng H, Cai H, et al. Explainability for large language models: a survey. ACM Trans On Intell Syst and Technol. 2024;15(2):1\u201338.","DOI":"10.1145\/3639372"},{"key":"341_CR16","doi-asserted-by":"crossref","unstructured":"Lundberg SM, Erion G, Chen H, DeGrave A, Prutkin JM, Nair B, et al. From local explanations to global understanding with explainable ai for trees. Nat Mach Intel. 2020;2(1):56\u201367.","DOI":"10.1038\/s42256-019-0138-9"},{"key":"341_CR17","doi-asserted-by":"crossref","unstructured":"Rudin C. Stop explaining black box machine learning models for high stakes decisions and use interpretable models instead. Nat Mach Intel. 2019;1(5):206\u201315.","DOI":"10.1038\/s42256-019-0048-x"},{"key":"341_CR18","doi-asserted-by":"crossref","unstructured":"Sanchez P, Voisey JP, Xia T, Watson HI, O\u2019Neil AQ, Tsaftaris SA. Causal machine learning for healthcare and precision medicine. R Soc Open Sci. 2022;9(8):220638.","DOI":"10.1098\/rsos.220638"},{"key":"341_CR19","doi-asserted-by":"crossref","unstructured":"W, Wang L, Rastegar-Mojarad M, Moon S, Shen F, Afzal N, et al. Clinical information extraction applications: a literature review. J Biomed Inf. 2018;77:34\u201349.","DOI":"10.1016\/j.jbi.2017.11.011"},{"key":"341_CR20","doi-asserted-by":"crossref","unstructured":"Hahn U, Oleynik M. Medical information extraction in the age of deep learning. Yearb of Med Inf. 2020;29(1):208\u201320.","DOI":"10.1055\/s-0040-1702001"},{"key":"341_CR21","doi-asserted-by":"crossref","unstructured":"Wang B, Xie Q, Pei J, Chen Z, Tiwari P, Li Z, et al. Pre-trained language models in biomedical domain: a systematic survey. ACM Comput Surv. 2023;56(3):1\u201352.","DOI":"10.1145\/3611651"},{"key":"341_CR22","doi-asserted-by":"crossref","unstructured":"Xu D, Chen W, Peng W, Zhang C, Xu T, Zhao X, et al. Large language models for generative information extraction: a survey. Front of Comput Sci. 2024;18(6):186357.","DOI":"10.1007\/s11704-024-40555-y"},{"key":"341_CR23","doi-asserted-by":"crossref","unstructured":"Sirocchi C, Bogliolo A, Montagna S. Medical-informed machine learning: integrating prior knowledge into medical decision systems. BMC Med Inf and Decis Mak. 2024;24(Suppl 4):186.","DOI":"10.1186\/s12911-024-02582-4"},{"key":"341_CR24","doi-asserted-by":"crossref","unstructured":"Wu X, Duan J, Pan Y, Li M. Medical knowledge graph: data sources, construction, reasoning, and applications. Big data mining and analytics. 2023;6(2):201\u201317.","DOI":"10.26599\/BDMA.2022.9020021"},{"key":"341_CR25","doi-asserted-by":"crossref","unstructured":"Kyrimi E, McLachlan S, Dube K, Neves MR, Fahmi A, Fenton N. A comprehensive scoping review of Bayesian networks in healthcare: past, present and future. Artif Intel in Med. 2021;117:102108.","DOI":"10.1016\/j.artmed.2021.102108"},{"key":"341_CR26","unstructured":"Koller D, Friedman N. Probabilistic graphical models: principles and techniques. Adaptive computation and machine learning. MIT Press; 2009."},{"key":"341_CR27","doi-asserted-by":"crossref","unstructured":"Johnson AE, Pollard TJ, Shen L, LwH L, Feng M, Ghassemi M, et al. MIMIC-III, a freely accessible critical care database. Sci Data. 2016;3(1):1\u20139.","DOI":"10.1038\/sdata.2016.35"},{"key":"341_CR28","doi-asserted-by":"crossref","unstructured":"Johnson AE, Bulgarelli L, Shen L, Gayles A, Shammout A, Horng S, et al. MIMIC-IV, a freely accessible electronic health record dataset. Sci Data. 2023;10(1):1.","DOI":"10.1038\/s41597-023-01945-2"},{"key":"341_CR29","unstructured":"Kwon Y, Kim J, Lee G, Bae S, Kyung D, Cha W, et al. Ehrcon: dataset for checking consistency between unstructured notes and structured tables in electronic health Records. Adv Neural Inf Process Syst. 2024;37."},{"key":"341_CR30","doi-asserted-by":"crossref","unstructured":"Lin AY, Arabandi S, Beale T, Duncan WD, Hicks A, Hogan WR, et al. Improving the quality and utility of electronic health record data through ontologies. Standards. 2023;3(3):316\u201340.","DOI":"10.3390\/standards3030023"},{"key":"341_CR31","doi-asserted-by":"crossref","unstructured":"D\u2019Oosterlinck K, Remy F, Deleu J, Demeester T, Develder C, Zaporojets K, et al. BioDEX: large-scale biomedical adverse drug event extraction for real-world pharmacovigilance. Findings of The Assoc For Comput Linguistics: EMNLP. 2023; 2023;13425\u201354.","DOI":"10.18653\/v1\/2023.findings-emnlp.896"},{"key":"341_CR32","doi-asserted-by":"crossref","unstructured":"Kefeli J, Tatonetti N. TCGA-Reports: a machine-readable pathology report resource for benchmarking text-based ai models. Patterns. 2024;5(3).","DOI":"10.1016\/j.patter.2024.100933"},{"key":"341_CR33","unstructured":"Rabaey P, Arno H, Heytens S, Demeester T. SynSUM -- Synthetic benchmark with structured and unstructured medical records. GenAI4Health, Workshop on Large Language Models and Generative AI for Health at AAAI 2025; 2025."},{"key":"341_CR34","doi-asserted-by":"crossref","unstructured":"Oni\u015bko A, Druzdzel MJ, Wasyluk H. Learning Bayesian network parameters from small data sets: application of noisy-OR gates. Int J Approximate Reasoning. 2001;27(2):165\u201382.","DOI":"10.1016\/S0888-613X(01)00039-1"},{"key":"341_CR35","doi-asserted-by":"crossref","unstructured":"Shwartz-Ziv R, Armon A. Tabular data: deep learning is not all you need. Inf Fusion. 2022;81:84\u201390.","DOI":"10.1016\/j.inffus.2021.11.011"},{"key":"341_CR36","doi-asserted-by":"crossref","unstructured":"Remy F, Demuynck K, Demeester T. BioLORD-2023: semantic textual representations fusing large language models and clinical knowledge graph insights. J Am Med Inf Assoc. 2024;02;p. ocae029.","DOI":"10.1093\/jamia\/ocae029"},{"key":"341_CR37","doi-asserted-by":"crossref","unstructured":"Rabaey P, Deleu J, Heytens S, Demeester T. Clinical reasoning over tabular Data and text with bayesian networks. International Conference on Artificial Intelligence in Medicine Springer. 2024. p. 229\u201350.","DOI":"10.1007\/978-3-031-66538-7_24"},{"key":"341_CR38","unstructured":"Arno H, Rabaey P, Demeester T. From text to treatment effects: a meta-learning approach to handling text-based confounding. Causal Representation Learning Workshop at NeurIPS. 2024;2024."},{"key":"341_CR39","unstructured":"Ma Y, Frauen D, Schweisthal J, Feuerriegel S. LLM-Driven treatment effect estimation under inference time text confounding. arXiv preprint arXiv:250702843. 2025."},{"key":"341_CR40","doi-asserted-by":"crossref","unstructured":"Hernandez M, Epelde G, Alberdi A, Cilla R, Rankin D. Synthetic data generation for tabular health records: a systematic review. Neurocomputing. 2022;493:28\u201345.","DOI":"10.1016\/j.neucom.2022.04.053"},{"key":"341_CR41","doi-asserted-by":"crossref","unstructured":"Guan J, Li R, Yu S, Zhang X. A method for generating synthetic electronic medical record text. IEEE\/ACM Trans On Comput Biol and Bioinf. 2019;18(1):173\u201382.","DOI":"10.1109\/TCBB.2019.2948985"},{"key":"341_CR42","doi-asserted-by":"crossref","unstructured":"Ibrahim M, Al Khalil Y, Amirrajab S, Sun C, Breeuwer M, Pluim J, et al. Generative ai for synthetic data across multiple medical modalities: a systematic review of recent developments and challenges. Comput In Biol and Med. 2025;189:109834.","DOI":"10.1016\/j.compbiomed.2025.109834"},{"issue":"1","key":"341_CR43","doi-asserted-by":"publisher","first-page":"63","DOI":"10.1038\/s41746-018-0070-0","volume":"1","author":"SH Lee","year":"2018","unstructured":"Lee SH. Natural language generation for electronic health records. NPJ Digit Med. 2018;1(1):63.","journal-title":"NPJ Digit Med"},{"key":"341_CR44","doi-asserted-by":"crossref","unstructured":"Wang Z, Sun J. Prompt EHR: conditional electronic healthcare records generation with prompt learning. Proceedings of the Conference on Empirical Methods in Natural Language Processing. Conference on Empirical Methods in Natural Language Processing. 2022;p. 2873, vol. 2022.","DOI":"10.18653\/v1\/2022.emnlp-main.185"},{"key":"341_CR45","unstructured":"OpenAI: Models GPT\u20134o. 2024. Online. https:\/\/platform.openai.com\/docs\/models\/gpt-4o;. Accessed 12 Aug 2024."},{"key":"341_CR46","volume-title":"Content analysis: an introduction to its methodology","author":"K Krippendorff","year":"2018","unstructured":"Krippendorff K. Content analysis: an introduction to its methodology. Sage publications; 2018."},{"key":"341_CR47","doi-asserted-by":"crossref","unstructured":"Ankan A. Abinash Panda. pgmpy: probabilistic graphical models using Python. Proceedings of the 14th Python in Science Conference. 2015. p. 6\u201311.","DOI":"10.25080\/Majora-7b98e3ed-001"}],"container-title":["Journal of Biomedical Semantics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13326-025-00341-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s13326-025-00341-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13326-025-00341-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T10:46:12Z","timestamp":1766054772000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1186\/s13326-025-00341-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,18]]},"references-count":47,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["341"],"URL":"https:\/\/doi.org\/10.1186\/s13326-025-00341-6","relation":{},"ISSN":["2041-1480"],"issn-type":[{"value":"2041-1480","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,18]]},"assertion":[{"value":"28 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 October 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The authors declare no competing interests.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"20"}}