{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T17:48:13Z","timestamp":1782496093050,"version":"3.54.5"},"reference-count":65,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100018806","name":"Department of Science and Technology of Hubei Province","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100018806","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002855","name":"Ministry of Science and Technology of the People&apos;s Republic of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002855","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.knosys.2026.116524","type":"journal-article","created":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T16:25:22Z","timestamp":1782318322000},"page":"116524","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MKDS: Multi-source knowledge-driven data synthesis framework for effective domain adaptation of large language models"],"prefix":"10.1016","volume":"349","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0118-5217","authenticated-orcid":false,"given":"Qihuang","family":"Zhong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinzhao","family":"Gong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ke","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fei","family":"Liao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3907-8820","authenticated-orcid":false,"given":"Juhua","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Du","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116524_b1","series-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b2","series-title":"Deepseek-v3 technical report","author":"Liu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b3","series-title":"A survey of large language models","author":"Zhao","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b4","series-title":"Findings of the Association for Computational Linguistics: ACL 2024","article-title":"Biomistral: A collection of open-source pretrained large language models for medical domains","author":"Labrak","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b5","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2023","first-page":"10859","article-title":"HuatuoGPT, towards taming language model to be a doctor","author":"Zhang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b6","series-title":"MedAlpaca\u2013an open-source collection of medical conversational AI models and training data","author":"Han","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b7","series-title":"Alpacare: Instruction-tuned large language models for medical application","author":"Zhang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b8","doi-asserted-by":"crossref","first-page":"55006","DOI":"10.52202\/075280-2400","article-title":"Lima: Less is more for alignment","volume":"36","author":"Zhou","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116524_b9","series-title":"Wizardmath: Empowering mathematical reasoning for large language models via reinforced evol-instruct","author":"Luo","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b10","series-title":"Knowledge-tuning large language models with structured medical knowledge bases for reliable response generation in Chinese","author":"Wang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b11","series-title":"The 61st Annual Meeting of the Association for Computational Linguistics","article-title":"Self-instruct: Aligning language models with self-generated instructions","author":"Wang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b12","series-title":"Best practices and lessons learned on synthetic data for language models","author":"Liu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b13","series-title":"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing","article-title":"Rethinking the role of demonstrations: What makes in-context learning work?","author":"Min","year":"2022"},{"issue":"4","key":"10.1016\/j.knosys.2026.116524_b14","doi-asserted-by":"crossref","first-page":"1116","DOI":"10.26599\/BDMA.2024.9020044","article-title":"Medbench: A comprehensive, standardized, and reliable benchmarking system for evaluating chinese medical large language models","volume":"7","author":"Liu","year":"2024","journal-title":"Big Data Min. Anal."},{"key":"10.1016\/j.knosys.2026.116524_b15","series-title":"The llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b16","series-title":"Qwen2. 5 technical report","author":"Yang","year":"2024"},{"issue":"8","key":"10.1016\/j.knosys.2026.116524_b17","doi-asserted-by":"crossref","first-page":"1930","DOI":"10.1038\/s41591-023-02448-8","article-title":"Large language models in medicine","volume":"29","author":"Thirunavukarasu","year":"2023","journal-title":"Nature Med."},{"issue":"1","key":"10.1016\/j.knosys.2026.116524_b18","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1186\/s12911-025-02954-4","article-title":"A systematic review of large language model (LLM) evaluations in clinical medicine","volume":"25","author":"Shool","year":"2025","journal-title":"BMC Med. Inform. Decis. Mak."},{"issue":"2","key":"10.1016\/j.knosys.2026.116524_b19","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2022.103221","article-title":"NRAND: An efficient and robust dismantling approach for infectious disease network","volume":"60","author":"Akhtar","year":"2023","journal-title":"Inf. Process. Manage."},{"key":"10.1016\/j.knosys.2026.116524_b20","series-title":"Findings of the Association for Computational Linguistics: ACL 2025","first-page":"24085","article-title":"KaFT: Knowledge-aware fine-tuning for boosting LLMs\u2019 domain-specific question-answering performance","author":"Zhong","year":"2025"},{"key":"10.1016\/j.knosys.2026.116524_b21","series-title":"Resolving knowledge conflicts in domain-specific data selection: A case study on medical instruction-tuning","author":"Zhong","year":"2025"},{"key":"10.1016\/j.knosys.2026.116524_b22","series-title":"Aloe: A family of fine-tuned open healthcare LLMs","author":"Gururajan","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b23","series-title":"Med42-v2: A suite of clinical llms","author":"Christophe","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b24","article-title":"Taiyi: A bilingual fine-tuned large language model for diverse biomedical tasks","author":"Luo","year":"2023","journal-title":"J. Am. Med. Inform. Assoc. : JAMIA"},{"issue":"6","key":"10.1016\/j.knosys.2026.116524_b25","doi-asserted-by":"crossref","first-page":"bbac409","DOI":"10.1093\/bib\/bbac409","article-title":"BioGPT: generative pre-trained transformer for biomedical text generation and mining","volume":"23","author":"Luo","year":"2022","journal-title":"Brief. Bioinform."},{"key":"10.1016\/j.knosys.2026.116524_b26","first-page":"ocae045","article-title":"PMC-LLaMA: toward building open-source language models for medicine","author":"Wu","year":"2024","journal-title":"J. Am. Med. Inform. Assoc."},{"issue":"7972","key":"10.1016\/j.knosys.2026.116524_b27","doi-asserted-by":"crossref","first-page":"172","DOI":"10.1038\/s41586-023-06291-2","article-title":"Large language models encode clinical knowledge","volume":"620","author":"Singhal","year":"2023","journal-title":"Nature"},{"key":"10.1016\/j.knosys.2026.116524_b28","first-page":"1","article-title":"Toward expert-level medical question answering with large language models","author":"Singhal","year":"2025","journal-title":"Nature Med."},{"key":"10.1016\/j.knosys.2026.116524_b29","series-title":"A survey of large language models in medicine: Progress, application, and challenge","author":"Zhou","year":"2023"},{"issue":"6","key":"10.1016\/j.knosys.2026.116524_b30","article-title":"Chatdoctor: A medical chat model fine-tuned on a large language model meta-ai (llama) using medical domain knowledge","volume":"15","author":"Li","year":"2023","journal-title":"Cureus"},{"key":"10.1016\/j.knosys.2026.116524_b31","series-title":"Large language model alignment: A survey","author":"Shen","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b32","series-title":"Coig-cqia: Quality is all you need for chinese instruction fine-tuning","author":"Bai","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b33","series-title":"The 61st Annual Meeting of the Association for Computational Linguistics","article-title":"Unnatural instructions: Tuning language models with (almost) no human labor","author":"Honovich","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b34","series-title":"Stanford alpaca: An instruction-following LLaMA model","author":"Taori","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b35","series-title":"Enhancing chat language models by scaling high-quality instructional conversations","author":"Ding","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b36","series-title":"Wizardlm: Empowering large language models to follow complex instructions","author":"Xu","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b37","series-title":"Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","first-page":"6491","article-title":"A survey on rag meeting llms: Towards retrieval-augmented large language models","author":"Fan","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b38","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/s41019-025-00335-5","article-title":"Retrieval-augmented generation for ai-generated content: A survey","author":"Zhao","year":"2026","journal-title":"Data Sci. Eng."},{"key":"10.1016\/j.knosys.2026.116524_b39","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.109660","article-title":"Multilingual entity alignment by abductive knowledge reasoning on multiple knowledge graphs","volume":"139","author":"Akhtar","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"issue":"10","key":"10.1016\/j.knosys.2026.116524_b40","doi-asserted-by":"crossref","first-page":"10098","DOI":"10.1109\/TKDE.2023.3250499","article-title":"Knowledge graph augmented network towards multiview representation learning for aspect-based sentiment analysis","volume":"35","author":"Zhong","year":"2023","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"10.1016\/j.knosys.2026.116524_b41","article-title":"Pythia-RAG: Retrieval-augmented generation over a unified multimodal knowledge graph for enhanced QA","author":"Ali","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116524_b42","doi-asserted-by":"crossref","unstructured":"A. Khan, Z. Ali, A.A. Irfanullah, P. Kefalas, Talk2Doc: A Patient Q&A system using Retrieval-Augmented Generation with Weighted Knowledge Graphs and LLMs, in: Proceedings of the 21st International Conference on Intelligent Computing, ICIC 2025, Ningbo, China, 2025, pp. 595\u2013610.","DOI":"10.65286\/icic.v21i4.47743"},{"key":"10.1016\/j.knosys.2026.116524_b43","series-title":"A survey on medical large language models: Technology, application, trustworthiness, and future directions","author":"Liu","year":"2024"},{"issue":"2","key":"10.1016\/j.knosys.2026.116524_b44","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3597307","article-title":"Biases in large language models: origins, inventory, and discussion","volume":"15","author":"Navigli","year":"2023","journal-title":"ACM J. Data Inf. Qual."},{"key":"10.1016\/j.knosys.2026.116524_b45","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Acl 2023): Long Papers, Vol 1","first-page":"14014","article-title":"Mitigating label biases for in-context learning","author":"Fei","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b46","series-title":"Findings of the Association for Computational Linguistics ACL 2024","first-page":"11065","article-title":"On LLMs-driven synthetic data generation, curation, and evaluation: A survey","author":"Long","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b47","series-title":"The Twelfth International Conference on Learning Representations","article-title":"AlpaGasus: Training a better alpaca with fewer data","author":"Chen","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b48","series-title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)","first-page":"6184","article-title":"Cmb: A comprehensive medical benchmark in chinese","author":"Wang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b49","article-title":"C-eval: A multi-level multi-discipline chinese evaluation suite for foundation models","volume":"36","author":"Huang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116524_b50","series-title":"Qilin-med: Multi-stage knowledge injection advanced medical large language model","author":"Ye","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b51","series-title":"Cmmlu: Measuring massive multitask language understanding in chinese","author":"Li","year":"2023"},{"issue":"1","key":"10.1016\/j.knosys.2026.116524_b52","doi-asserted-by":"crossref","first-page":"44","DOI":"10.1038\/s44401-025-00038-z","article-title":"Enabling doctor-centric medical AI with LLMs through workflow-aligned tasks and benchmarks","volume":"2","author":"Xie","year":"2025","journal-title":"npj Health Syst."},{"key":"10.1016\/j.knosys.2026.116524_b53","series-title":"Baichuan 2: Open large-scale language models","author":"Yang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b54","series-title":"A benchmark for long-form medical question answering","author":"Hosseini","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b55","series-title":"Huatuo-26m, a large-scale chinese medical qa dataset","author":"Li","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b56","series-title":"Disc-medllm: Bridging general large language models and real-world medical consultation","author":"Bao","year":"2023"},{"key":"10.1016\/j.knosys.2026.116524_b57","series-title":"Chinese-medical-dialogue-data","author":"Toyhom","year":"2019"},{"key":"10.1016\/j.knosys.2026.116524_b58","series-title":"Aqulia-Med LLM: Pioneering full-process open-source medical language models","author":"Zhao","year":"2024"},{"key":"10.1016\/j.knosys.2026.116524_b59","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 3: System Demonstrations)","article-title":"LlamaFactory: Unified efficient fine-tuning of 100+ language models","author":"Zheng","year":"2024"},{"issue":"2","key":"10.1016\/j.knosys.2026.116524_b60","first-page":"3","article-title":"Lora: Low-rank adaptation of large language models","volume":"1","author":"Hu","year":"2022","journal-title":"Iclr"},{"issue":"14","key":"10.1016\/j.knosys.2026.116524_b61","doi-asserted-by":"crossref","first-page":"6421","DOI":"10.3390\/app11146421","article-title":"What disease does this patient have? a large-scale open domain question answering dataset from medical exams","volume":"11","author":"Jin","year":"2021","journal-title":"Appl. Sci."},{"key":"10.1016\/j.knosys.2026.116524_b62","series-title":"Conference on Health, Inference, and Learning","first-page":"248","article-title":"Medmcqa: A large-scale multi-subject multi-choice dataset for medical domain question answering","author":"Pal","year":"2022"},{"key":"10.1016\/j.knosys.2026.116524_b63","series-title":"GLM-5: from vibe coding to agentic engineering","author":"Zeng","year":"2026"},{"issue":"Jan","key":"10.1016\/j.knosys.2026.116524_b64","first-page":"1","article-title":"Statistical comparisons of classifiers over multiple data sets","volume":"7","author":"Dem\u0161ar","year":"2006","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.knosys.2026.116524_b65","first-page":"1","article-title":"Wilcoxon signed-rank test","author":"Woolson","year":"2007","journal-title":"Wiley Encycl. Clin. Trials"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126012505?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126012505?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T16:55:07Z","timestamp":1782492907000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126012505"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":65,"alternative-id":["S0950705126012505"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116524","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MKDS: Multi-source knowledge-driven data synthesis framework for effective domain adaptation of large language models","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116524","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"116524"}}