{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T19:06:16Z","timestamp":1783969576240,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":33,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234974","type":"print"},{"value":"9789819234981","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T00:00:00Z","timestamp":1783987200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T00:00:00Z","timestamp":1783987200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3498-1_1","type":"book-chapter","created":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T18:44:44Z","timestamp":1783968284000},"page":"3-14","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["TCMBenchEval: A Benchmark with Authentic Clinical Cases for Evaluating Large Language Models in Traditional Chinese Medicine"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7170-4441","authenticated-orcid":false,"given":"Yu","family":"Tong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-1807-3594","authenticated-orcid":false,"given":"Weihao","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6778-9697","authenticated-orcid":false,"given":"Chuipu","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,14]]},"reference":[{"key":"1_CR1","unstructured":"Wei, S., et al.: BianCang: A Traditional Chinese Medicine Large Language Model. arXiv preprint https:\/\/arxiv.org\/abs\/2411.11027 (2024)."},{"key":"1_CR2","unstructured":"Chen, Y., et al.: BianQue: Balancing the Questioning and Suggestion Ability of Health LLMs with Multi-turn Health Conversations Polished by ChatGPT. arXiv preprint https:\/\/arxiv.org\/abs\/2310.15896 (2023)."},{"key":"1_CR3","unstructured":"Yan, X., Xue, D.: Sunsimiao: Chinese Medicine LLM. GitHub repository, https:\/\/github.com\/thomas-yanxin\/Sunsimiao. Accessed 11 May 2026"},{"issue":"14","key":"1_CR4","doi-asserted-by":"publisher","first-page":"6421","DOI":"10.3390\/app11146421","volume":"11","author":"D Jin","year":"2021","unstructured":"Jin, D., Pan, E., Oufattole, N., Weng, W.H., Fang, H., Szolovits, P.: What disease does this patient have? A large-scale open domain question answering dataset from medical exams. Appl. Sci. 11(14), 6421 (2021)","journal-title":"Appl. Sci."},{"key":"1_CR5","first-page":"248","volume-title":"Conference on Health, Inference, and Learning","author":"A Pal","year":"2022","unstructured":"Pal, A., Umapathi, L.K., Sankarasubbu, M.: MedMCQA: a large-scale multi-subject multi-choice dataset for medical domain question answering. In: Conference on Health, Inference, and Learning, pp. 248\u2013260. PMLR (2022)"},{"key":"1_CR6","unstructured":"Yue, W., et al.: TCMBench: A Comprehensive Benchmark for Evaluating Large Language Models in Traditional Chinese Medicine. arXiv preprint https:\/\/arxiv.org\/abs\/2406.01126 (2024)."},{"key":"1_CR7","unstructured":"Zuo, Y., et al.: MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding. arXiv preprint https:\/\/arxiv.org\/abs\/2501.18362 (2025)."},{"key":"1_CR8","unstructured":"Kong, S., et al.: MTCMB: A Multi-Task Benchmark Framework for Evaluating LLMs on Knowledge, Reasoning, and Safety in Traditional Chinese Medicine. arXiv preprint https:\/\/arxiv.org\/abs\/2506.01252 (2025)."},{"key":"1_CR9","unstructured":"Wang, X., et al.: CMB: A Comprehensive Medical Benchmark in Chinese. arXiv preprint https:\/\/arxiv.org\/abs\/2308.08833 (2024)."},{"key":"1_CR10","doi-asserted-by":"publisher","first-page":"8862","DOI":"10.18653\/v1\/2021.emnlp-main.698","volume-title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","author":"J Li","year":"2021","unstructured":"Li, J., Zhong, S., Chen, K.: MLEC-QA: a Chinese multi-choice biomedical question answering dataset. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, pp. 8862\u20138874. Association for Computational Linguistics, Online and Punta Cana, Dominican Republic (2021)"},{"key":"1_CR11","first-page":"17709","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 38","author":"Y Cai","year":"2024","unstructured":"Cai, Y., et al.: MedBench: a large-scale Chinese benchmark for evaluating medical large language models. In: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 38, pp. 17709\u201317717. AAAI Press (2024)"},{"key":"1_CR12","unstructured":"Zhang, H., et al.: Qibo: A Large Language Model for Traditional Chinese Medicine. arXiv preprint https:\/\/arxiv.org\/abs\/2403.16056 (2024)."},{"key":"1_CR13","unstructured":"Huang, T., et al.: TCM-3CEval: A Triaxial Benchmark for Assessing Responses from Large Language Models in Traditional Chinese Medicine. arXiv preprint https:\/\/arxiv.org\/abs\/2503.07041 (2025)."},{"key":"1_CR14","unstructured":"Huang, T., et al.: TCM-5CEval: Extended Deep Evaluation Benchmark for LLMs\u2019 Comprehensive Clinical Research Competence in Traditional Chinese Medicine. arXiv preprint https:\/\/arxiv.org\/abs\/2511.13169 (2025)."},{"issue":"1","key":"1_CR15","doi-asserted-by":"publisher","first-page":"437","DOI":"10.1038\/s41597-025-04772-9","volume":"12","author":"Z Wang","year":"2025","unstructured":"Wang, Z., et al.: TCMEval-SDT: a benchmark dataset for syndrome differentiation thought of traditional Chinese medicine. Sci. Data. 12(1), 437 (2025)","journal-title":"Sci. Data"},{"key":"1_CR16","unstructured":"Zhou, C., Wang, J., Qin, J., Wang, Y., Sun, L., Dai, W.: OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology. arXiv preprint https:\/\/arxiv.org\/abs\/2502.01243 (2025)."},{"key":"1_CR17","unstructured":"Xie, J., et al.: TCM-Ladder: A Benchmark for Multimodal Question Answering on Traditional Chinese Medicine. arXiv preprint https:\/\/arxiv.org\/abs\/2505.24063 (2025)."},{"key":"1_CR18","unstructured":"Liu, A., et al.: DeepSeek-V3 Technical Report. arXiv preprint https:\/\/arxiv.org\/abs\/2412.19437 (2024)."},{"key":"1_CR19","unstructured":"Yang, A., et al.: Qwen3 Technical Report. arXiv preprint https:\/\/arxiv.org\/abs\/2505.09388 (2025)."},{"key":"1_CR20","unstructured":"M2 Team, et al.: Baichuan-M2: Scaling Medical Capability with Large Verifier System. arXiv preprint https:\/\/arxiv.org\/abs\/2509.02208 (2025)."},{"key":"1_CR21","unstructured":"Agarwal, S., et al.: GPT-OSS-120B and GPT-OSS-20B Model Card. arXiv preprint https:\/\/arxiv.org\/abs\/2508.10925 (2025)."},{"key":"1_CR22","volume-title":"Working Notes of CLEF 2024","author":"V Neralla","year":"2024","unstructured":"Neralla, V., de Vroe, S.B.: Evaluating Poro-34B-chat and mistral-7B-instruct-v0.1: LLM system description for ELOQUENT at CLEF 2024. In: Working Notes of CLEF 2024 (2024)"},{"key":"1_CR23","unstructured":"Yang, A., Yang, B., Hui, B., Zheng, B., Yu, B., et al.: Qwen2 Technical Report. arXiv preprint https:\/\/arxiv.org\/abs\/2407.10671 (2024)."},{"key":"1_CR24","unstructured":"Chen, J., et al.: HuatuoGPT-o1: Towards Medical Complex Reasoning with LLMs. arXiv preprint https:\/\/arxiv.org\/abs\/2412.18925 (2024)."},{"key":"1_CR25","unstructured":"Wang, H., et al.: HuaTuo: Tuning LLaMA Model with Chinese Medical Knowledge. arXiv preprint https:\/\/arxiv.org\/abs\/2304.06975 (2023)."},{"key":"1_CR26","doi-asserted-by":"crossref","unstructured":"Tian, Y., Gan, R., Song, Y., Zhang, J., Zhang, Y.: ChiMed-GPT: A Chinese Medical Large Language Model with Full Training Regime and Better Alignment to Human Preferences. arXiv preprint https:\/\/arxiv.org\/abs\/2311.06025 (2024).","DOI":"10.18653\/v1\/2024.acl-long.386"},{"issue":"9","key":"1_CR27","doi-asserted-by":"publisher","first-page":"1865","DOI":"10.1093\/jamia\/ocae037","volume":"31","author":"L Luo","year":"2024","unstructured":"Luo, L., et al.: Taiyi: a bilingual fine-tuned large language model for diverse biomedical tasks. J. Am. Med. Inform. Assoc. 31(9), 1865\u20131874 (2024)","journal-title":"J. Am. Med. Inform. Assoc."},{"key":"1_CR28","unstructured":"Winning Health Technology: WiNGPT2-14B-Base: A Large Language Model for Medical Applications. Hugging Face Model Hub, https:\/\/huggingface.co\/winninghealth\/WiNGPT2-14B-Base. Accessed 11 May 2026"},{"key":"1_CR29","unstructured":"Developer, E.: DeepSeek-R1-Medical-COT: A Medical Chain-of-Thought Large Language Model. Hugging Face Model Hub. https:\/\/huggingface.co\/emredeveloper\/DeepSeek-R1-Medical-COT. Accessed 11 May 2026"},{"key":"1_CR30","first-page":"311","volume-title":"Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics","author":"K Papineni","year":"2002","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: BLEU: a method for automatic evaluation of machine translation. In: Isabelle, P., Charniak, E., Lin, D. (eds.) Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318. Association for Computational Linguistics, Philadelphia, Pennsylvania, USA (2002)"},{"key":"1_CR31","first-page":"74","volume-title":"Text Summarization Branches out","author":"C-Y Lin","year":"2004","unstructured":"Lin, C.-Y.: ROUGE: a package for automatic evaluation of summaries. In: Text Summarization Branches out, pp. 74\u201381. Association for Computational Linguistics, Barcelona, Spain (2004)"},{"key":"1_CR32","unstructured":"Zhang, T., Kishore, V., Wu, F., Weinberger, K.Q., Artzi, Y.: BERTScore: Evaluating Text Generation with BERT. arXiv preprint https:\/\/arxiv.org\/abs\/1904.09675 (2020)."},{"key":"1_CR33","unstructured":"Cheng, Z., et al.: TCM-Eval: An Expert-Level Dynamic and Extensible Benchmark for Traditional Chinese Medicine. arXiv preprint https:\/\/arxiv.org\/abs\/2511.07148 (2025)."}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3498-1_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T18:44:46Z","timestamp":1783968286000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3498-1_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,14]]},"ISBN":["9789819234974","9789819234981"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3498-1_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,14]]},"assertion":[{"value":"14 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}