{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T06:42:57Z","timestamp":1777012977709,"version":"3.51.4"},"reference-count":58,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T00:00:00Z","timestamp":1776988800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T00:00:00Z","timestamp":1776988800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"funder":[{"DOI":"10.13039\/501100003593","name":"Conselho Nacional de Desenvolvimento Cient\u00edfico e Tecnol\u00f3gico","doi-asserted-by":"publisher","award":["404771\/2024-6, 406354\/2023-5,312755\/2023-6, 313053\/2023-5"],"award-info":[{"award-number":["404771\/2024-6, 406354\/2023-5,312755\/2023-6, 313053\/2023-5"]}],"id":[{"id":"10.13039\/501100003593","id-type":"DOI","asserted-by":"publisher"}]},{"name":"MAI\/DAI","award":["68\/2022"],"award-info":[{"award-number":["68\/2022"]}]},{"name":"Maria Emilia Foundation","award":["01\/2023"],"award-info":[{"award-number":["01\/2023"]}]},{"name":"INCITE FAPESB","award":["PIE0002\/2022"],"award-info":[{"award-number":["PIE0002\/2022"]}]},{"DOI":"10.13039\/501100002322","name":"CAPES","doi-asserted-by":"crossref","id":[{"id":"10.13039\/501100002322","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100006181","name":"Funda\u00e7\u00e3o de Amparo \u00e0 Pesquisa do Estado da Bahia","doi-asserted-by":"publisher","award":["1589\/2021"],"award-info":[{"award-number":["1589\/2021"]}],"id":[{"id":"10.13039\/501100006181","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Universidade Federal Da Bahia"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Health Inf Sci Syst"],"abstract":"<jats:title>Abstract<\/jats:title>\n                  <jats:p>The effective use of Large Language Models (LLMs) for generating coherent and informative content in specialized domains has largely been driven by the development of robust evaluation strategies. Based on this assumption, we introduce HemoQAL, a domain-specific question-and-answer (Q&amp;A) dataset on hemophilia, derived from recent scientific publications and clinical guidelines. Our main contribution lies in a fine-grained evaluation of the quality of LLM-generated content. First, we carried out a human evaluation in which medical experts assessed the factual accuracy and educational value of the generated Q&amp;A pairs. Second, we conducted a semantic similarity analysis to quantitatively evaluate the alignment between each Q&amp;A pair and its original source material. These lightweight, scalable semantic metrics offer an efficient alternative to more resource-intensive human or LLM-based evaluation pipelines. Our findings show that integrating expert review with semantic similarity measures improves the reliability and trustworthiness of LLM-generated medical content, contributing to the development of dependable AI tools in health informatics.<\/jats:p>","DOI":"10.1007\/s13755-026-00458-7","type":"journal-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T06:01:22Z","timestamp":1777010482000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Fine-grained evaluation of a domain-specific Q&amp;A dataset to support trustworthy medical language models"],"prefix":"10.1007","volume":"14","author":[{"given":"Rafael da C.","family":"Fonseca","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ricardo A.","family":"Rios","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rodrigo","family":"Castaldoni","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Adrielle A.","family":"Carvalho","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tiago J. S.","family":"Lopes","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Caio L. B.","family":"Andrade","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Braian V. G.","family":"Bispo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"La\u00eds R.","family":"Mota","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6992-977X","authenticated-orcid":false,"given":"Tatiane N.","family":"Rios","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,24]]},"reference":[{"key":"458_CR1","unstructured":"Zhao WX, Zhou K, Li J, Tang T, Wang X, Hou Y, Min Y, Zhang B, Zhang J, Dong Z, et al. A survey of large language models. arXiv:2303.18223 [Preprint]. 2023. Available from: http:\/\/arxiv.org\/abs\/2303.18223"},{"key":"458_CR2","doi-asserted-by":"publisher","first-page":"1350306","DOI":"10.3389\/frai.2023.1350306","volume":"6","author":"A Zubiaga","year":"2024","unstructured":"Zubiaga A. Natural language processing in the era of large language models. Front Artif Intell. 2024;6:1350306.","journal-title":"Front Artif Intell"},{"key":"458_CR3","unstructured":"Naveed H, Khan AU, Qiu S, Saqib M, Anwar S, Usman M, Akhtar N, Barnes N, Mian A. A comprehensive overview of large language models. arXiv:2307.06435 [Preprint]. 2023. Available from: http:\/\/arxiv.org\/abs\/2307.06435"},{"issue":"2","key":"458_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3605943","volume":"56","author":"B Min","year":"2023","unstructured":"Min B, Ross H, Sulem E, Veyseh APB, Nguyen TH, Sainz O, et al. Recent advances in natural language processing via large pre-trained language models: a survey. ACM Comput Surv. 2023;56(2):1\u201340.","journal-title":"ACM Comput Surv"},{"issue":"8","key":"458_CR5","first-page":"1","volume":"57","author":"J Zheng","year":"2025","unstructured":"Zheng J, Qiu S, Shi C, Ma Q. Towards lifelong learning of large language models: a survey. ACM Comput Surv. 2025;57(8):1\u201335.","journal-title":"ACM Comput Surv"},{"key":"458_CR6","doi-asserted-by":"crossref","unstructured":"Gururangan S, Marasovi\u0107 A, Swayamdipta S, Lo K, Beltagy I, Downey D, Smith NA. Don\u2019t stop pretraining: Adapt language models to domains and tasks. arXiv:2004.10964 [Preprint]. 2020. http:\/\/arxiv.org\/abs\/2004.10964","DOI":"10.18653\/v1\/2020.acl-main.740"},{"issue":"4","key":"458_CR7","doi-asserted-by":"publisher","first-page":"1234","DOI":"10.1093\/bioinformatics\/btz682","volume":"36","author":"J Lee","year":"2020","unstructured":"Lee J, Yoon W, Kim S, Kim D, Kim S, So CH, et al. Biobert: a pre-trained biomedical language representation model for biomedical text mining. Bioinformatics. 2020;36(4):1234\u201340.","journal-title":"Bioinformatics"},{"key":"458_CR8","unstructured":"Wu S, Irsoy O, Lu S, Dabravolski V, Dredze M, Gehrmann S, Kambadur P, Rosenberg D, Mann G. Bloomberggpt: a large language model for finance. arXiv:2303.17564 [Preprint]. 2023. Available from http:\/\/arxiv.org\/abs\/2303.17564"},{"issue":"7972","key":"458_CR9","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1038\/s41586-023-06291-2","volume":"620","author":"K Singhal","year":"2023","unstructured":"Singhal K, Azizi S, Tu T, Mahdavi SS, Wei J, Chung HW, et al. Large language models encode clinical knowledge. Nature. 2023;620(7972):172\u201380.","journal-title":"Nature"},{"key":"458_CR10","unstructured":"Jiaxi Cui1, ZLBCYYHLBLYT, Munan Ning1, Yuan, L. Chatlaw: A multi-agent collaborative legal assistant with knowledge graph enhanced mixture-of-experts large language model. arXiv:2306.16092 [Preprint]. 2024. http:\/\/arxiv.org\/abs\/2306.16092"},{"key":"458_CR11","unstructured":"Dai H, Li Y, Liu Z, Zhao L, Wu Z, Song S, Shen Y, Zhu D, Li X, Li S et al. AD-autoGPT: an autonomous GPT for Alzheimer\u2019s disease infodemiology. arXiv:2306.10095 [Preprint]. 2023. Available from http:\/\/arxiv.org\/abs\/2306.10095"},{"issue":"4","key":"458_CR12","doi-asserted-by":"publisher","first-page":"3007","DOI":"10.1109\/JBHI.2024.3478809","volume":"29","author":"Y Li","year":"2024","unstructured":"Li Y, Zheng X, Li J, Dai Q, Wang CD, Chen M. LKAN: LLM-based knowledge-aware attention network for clinical staging of liver cancer. IEEE J Biomed Health Inform. 2024;29(4):3007\u201320.","journal-title":"IEEE J Biomed Health Inform"},{"key":"458_CR13","first-page":"464","volume-title":"International workshop on machine learning in medical imaging","author":"Z Liu","year":"2023","unstructured":"Liu Z, Zhong A, Li Y, Yang L, Ju C, Wu Z, et al. Tailoring large language models to radiology: a preliminary approach to llm adaptation for a highly specialized domain. In: International workshop on machine learning in medical imaging. Cham: Springer; 2023. p. 464\u201373."},{"key":"458_CR14","unstructured":"Zhao X, Lu J, Deng C, Zheng C, Wang J, Chowdhury T, Yun L, Cui H, Xuchao Z, Zhao T, et al. Domain specialization as the key to make large language models disruptive: A comprehensive survey. arXiv:2305.18703 [Preprint]. 2023. Available from http:\/\/arxiv.org\/abs\/2305.18703"},{"key":"458_CR15","unstructured":"Li Q, Cui L, Kong L, Bi W. Collaborative evaluation: exploring the synergy of large language models and humans for open-ended generation evaluation. 2023."},{"key":"458_CR16","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu W-J. Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th annual meeting of the association for computational linguistics. 2002. p. 311\u20138","DOI":"10.3115\/1073083.1073135"},{"key":"458_CR17","unstructured":"Lin C-Y. Rouge: A package for automatic evaluation of summaries. In: Text summarization branches out. 2004. p. 74\u201381."},{"key":"458_CR18","doi-asserted-by":"publisher","first-page":"685","DOI":"10.18653\/v1\/N18-1063","volume-title":"Proceedings of the 2018 conference of the North American Chapter of the Association for Computational Linguistics: human language technologies","author":"E Sulem","year":"2018","unstructured":"Sulem E, Abend O, Rappoport A. Semantic structural evaluation for text simplification. In: Walker M, Ji H, Stent A, editors. Proceedings of the 2018 conference of the North American Chapter of the Association for Computational Linguistics: human language technologies. New Orleans: Association for Computational Linguistics; 2018. p. 685\u201396. https:\/\/doi.org\/10.18653\/v1\/N18-1063."},{"key":"458_CR19","unstructured":"Fu J, Ng S-K, Jiang Z, Liu P. Gptscore: evaluate as you desire. arXiv:2302.04166 [Preprint]. 2023. Available from http:\/\/arxiv.org\/abs\/2302.04166"},{"key":"458_CR20","unstructured":"Shi W, Lin\u00a0M, Vosoughi, S. Judging the judges: A systematic investigation of position bias in pairwise comparative assessments by llms. 2024."},{"key":"458_CR21","doi-asserted-by":"crossref","unstructured":"Hada R, Gumma V, Wynter A, Diddee H, Ahmed M, Choudhury M, Bali K, Sitaram S. Are large language model-based evaluators the solution to scaling up multilingual evaluation? arXiv:2309.07462 [Preprint]. 2023. https:\/\/arxiv.org\/abs\/2309.07462","DOI":"10.18653\/v1\/2024.findings-eacl.71"},{"key":"458_CR22","unstructured":"Liu P, Mao J. Coas-core: Chain-of-aspects prompting for nlg evaluation. arXiv:2312.10355 [Preprint]. 2023. https:\/\/arxiv.org\/abs\/2312.10355"},{"key":"458_CR23","doi-asserted-by":"publisher","DOI":"10.1002\/9781118398258","volume-title":"Textbook of hemophilia","author":"CA Lee","year":"2014","unstructured":"Lee CA, Berntorp EE, Hoots WK. Textbook of hemophilia. 3rd ed. Chichester: Wiley; 2014.","edition":"3"},{"issue":"4","key":"458_CR24","doi-asserted-by":"publisher","first-page":"556","DOI":"10.1590\/1414-462x202028040484","volume":"28","author":"AA Ferreira","year":"2020","unstructured":"Ferreira AA, Brum IV, Souza JVDL, Leite ICG. Cost analysis of hemophilia treatment in a Brazilian public blood center. Cadernos Sa\u00fade Coletiva. 2020;28(4):556\u201366.","journal-title":"Cadernos Sa\u00fade Coletiva"},{"key":"458_CR25","doi-asserted-by":"publisher","DOI":"10.1201\/9781420059458","volume-title":"Text mining: classification, clustering, and applications","author":"AN Srivastava","year":"2009","unstructured":"Srivastava AN, Sahami M. Text mining: classification, clustering, and applications. 1st ed. Boca Raton: Chapman & Hall\/CRC; 2009.","edition":"1"},{"issue":"1","key":"458_CR26","doi-asserted-by":"publisher","first-page":"155","DOI":"10.1017\/S1351324916000334","volume":"23","author":"KW Church","year":"2017","unstructured":"Church KW. Word2vec. Nat Lang Eng. 2017;23(1):155\u201362.","journal-title":"Nat Lang Eng"},{"key":"458_CR27","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K. Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (long and Short Papers). 2019. p. 4171\u201386."},{"key":"458_CR28","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I. Attention is all you need. Advances in neural information processing systems. 2017. p. 30."},{"key":"458_CR29","unstructured":"Abnar S, Dehghani M, Neyshabur B, Sedghi H. Exploring the limits of large scale pre-training. arXiv:2110.02095 [Preprint]. 2021. http:\/\/arxiv.org\/abs\/2110.02095"},{"issue":"1","key":"458_CR30","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1162\/coli.2007.33.1.41","volume":"33","author":"D Moll\u00e1","year":"2007","unstructured":"Moll\u00e1 D, Vicedo JL. Question answering in restricted domains: an overview. Comput Linguist. 2007;33(1):41\u201361.","journal-title":"Comput Linguist"},{"issue":"2","key":"458_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3490238","volume":"55","author":"Q Jin","year":"2022","unstructured":"Jin Q, Yuan Z, Xiong G, Yu Q, Ying H, Tan C, et al. Biomedical question answering: a survey of approaches and challenges. ACM Comput Surv. 2022;55(2):1\u201336.","journal-title":"ACM Comput Surv"},{"issue":"12","key":"458_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3571730","volume":"55","author":"Z Ji","year":"2023","unstructured":"Ji Z, Lee N, Frieske R, Yu T, Su D, Xu Y, et al. Survey of hallucination in natural language generation. ACM Comput Surv. 2023;55(12):1\u201338.","journal-title":"ACM Comput Surv"},{"key":"458_CR33","unstructured":"Zhou H, Gu B, Zou X, Li Y, Chen SS, Zhou P, Liu J, Hua Y, Mao C, Wu X et al. A survey of large language models in medicine: Progress, application, and challenge. arXiv:2311.05112 [Preprint]. 2023. Available from: http:\/\/arxiv.org\/abs\/2311.05112"},{"key":"458_CR34","unstructured":"Omiye JA, Gui H, Rezaei SJ, Zou J, Daneshjou R. Large language models in medicine: the potentials and pitfalls. arXiv:2309.00087 [Preprint]. 2023. Available from: http:\/\/arxiv.org\/abs\/2309.00087"},{"key":"458_CR35","unstructured":"Singhal K, Tu T, Gottweis J, Sayres R, Wulczyn E, Hou L, Clark K, Pfohl S, Cole-Lewis H, Neal D, et al. Towards expert-level medical question answering with large language models. arXiv:2305.09617 [Preprint]. 2023. http:\/\/arxiv.org\/abs\/2305.09617"},{"key":"458_CR36","unstructured":"Nori H, Lee YT, Zhang S, Carignan D, Edgar R, Fusi N, King N, Larson J, Li Y, Liu W, et al. Can generalist foundation models outcompete special-purpose tuning? case study in medicine. arXiv:2311.16452 [Preprint]. 2023. http:\/\/arxiv.org\/abs\/2311.16452"},{"issue":"6","key":"458_CR37","first-page":"1","volume":"15","author":"Y Li","year":"2023","unstructured":"Li Y, Li Z, Zhang K, Dan R, Jiang S, Zhang Y. Chatdoctor: a medical chat model fine-tuned on a large language model meta-ai (llama) using medical domain knowledge. Cureus. 2023;15(6):1.","journal-title":"Cureus"},{"key":"458_CR38","unstructured":"Han T, Adams LC, Papaioannou J-M, Grundmann P, Oberhauser T, L\u00f6ser A, Truhn D, Bressem KK. Medalpaca\u2013an open-source collection of medical conversational ai models and training data. arXiv:2304.08247 [Preprint]. 2023. http:\/\/arxiv.org\/abs\/2304.08247"},{"key":"458_CR39","unstructured":"Wu C, Zhang X, Zhang Y, Wang Y, Xie W. Pmc-llama: further finetuning llama on medical papers. arXiv:2304.14454 [Preprint]. 2023. Available from http:\/\/arxiv.org\/abs\/2304.14454"},{"key":"458_CR40","unstructured":"Wang H, Liu C, Xi N, Qiang Z, Zhao S, Qin B, Liu T. Huatuo: Tuning llama model with chinese medical knowledge. arXiv:2304.06975 [Preprint]. 2023. Available from http:\/\/arxiv.org\/abs\/2304.06975"},{"key":"458_CR41","unstructured":"Toma A, Lawler PR, Ba J, Krishnan RG, Rubin BB, Wang B. Clinical camel: an open-source expert-level medical language model with dialogue-based knowledge encoding. arXiv:2305.12031 [Preprint]. 2023. Available from http:\/\/arxiv.org\/abs\/2305.12031"},{"issue":"1","key":"458_CR42","doi-asserted-by":"publisher","first-page":"098","DOI":"10.1093\/bioadv\/vbac098","volume":"3","author":"TJ Lopes","year":"2023","unstructured":"Lopes TJ, Rios RA, Rios TN, Alencar BM, Ferreira MV, Morishita E. Computational analyses reveal fundamental properties of the at structure related to thrombosis. Bioinform Adv. 2023;3(1):098.","journal-title":"Bioinform Adv"},{"issue":"3","key":"458_CR43","doi-asserted-by":"publisher","first-page":"2257","DOI":"10.1109\/JBHI.2024.3514659","volume":"29","author":"PC Sukhwal","year":"2024","unstructured":"Sukhwal PC, Rajan V, Kankanhalli A. A joint llm-kg system for disease q&a. IEEE J Biomed Health Inform. 2024;29(3):2257\u201370.","journal-title":"IEEE J Biomed Health Inform"},{"issue":"2","key":"458_CR44","doi-asserted-by":"publisher","first-page":"1487","DOI":"10.1002\/widm.1487","volume":"13","author":"LC Budler","year":"2023","unstructured":"Budler LC, Gosak L, Stiglic G. Review of artificial intelligence-based question-answering systems in healthcare. Wiley Interdiscip Rev Data Min Knowl Discov. 2023;13(2):1487.","journal-title":"Wiley Interdiscip Rev Data Min Knowl Discov"},{"issue":"12","key":"458_CR45","doi-asserted-by":"publisher","first-page":"6074","DOI":"10.1109\/JBHI.2023.3316750","volume":"27","author":"J Qiu","year":"2023","unstructured":"Qiu J, Li L, Sun J, Peng J, Shi P, Zhang R, et al. Large ai models in health informatics: applications, challenges, and the future. IEEE J Biomed Health Inform. 2023;27(12):6074\u201387.","journal-title":"IEEE J Biomed Health Inform"},{"key":"458_CR46","unstructured":"Li D, Jiang B, Huang L, Beigi A, Zhao C, Tan Z, Bhattacharjee A, Jiang Y, Chen C, Wu T, et al. From generation to judgment: opportunities and challenges of llm-as-a-judge. arXiv:2411.16594 [Preprint]. 2024. Available from: http:\/\/arxiv.org\/abs\/2411.16594"},{"key":"458_CR47","doi-asserted-by":"crossref","unstructured":"Wong SM, Leung H, Wong KY. Efficiency in language understanding and generation: an evaluation of four open-source large language models. 2024.","DOI":"10.21203\/rs.3.rs-4063228\/v1"},{"key":"458_CR48","doi-asserted-by":"crossref","unstructured":"Risch J, M\u00f6ller T, Gutsch J, Pietsch M. Semantic answer similarity for evaluating question answering models. arXiv:2108.06130 [Preprint]. 2021. Available from: http:\/\/arxiv.org\/abs\/2108.06130","DOI":"10.18653\/v1\/2021.mrqa-1.15"},{"key":"458_CR49","doi-asserted-by":"crossref","unstructured":"Kamalloo E, Dziri N, Clarke CL, Rafiei D. Evaluating open-domain question answering in the era of large language models. arXiv:2305.06984 [Preprint]. 2023. Available from: http:\/\/arxiv.org\/abs\/2305.06984","DOI":"10.18653\/v1\/2023.acl-long.307"},{"key":"458_CR50","doi-asserted-by":"crossref","unstructured":"Ezzini S, Abualhaija S, Arora C, Sabetzadeh M. Ai-based question answering assistance for analyzing natural-language requirements. In: 2023 IEEE\/ACM 45th international conference on software engineering (ICSE). IEEE; 2023. p. 1277\u201389.","DOI":"10.1109\/ICSE48619.2023.00113"},{"key":"458_CR51","first-page":"1","volume":"2024","author":"A Srivastava","year":"2024","unstructured":"Srivastava A, Memon A. Towards robust evaluation: a comprehensive taxonomy of datasets and metrics for open domain question answering in the era of large language models. IEEE Access. 2024;2024:1\u201310.","journal-title":"IEEE Access"},{"key":"458_CR52","doi-asserted-by":"crossref","unstructured":"Kamalloo E, Upadhyay S, Lin J. Towards robust qa evaluation via open llms. In: Proceedings of the 47th international ACM SIGIR conference on research and development in information retrieval. 2024. p. 2811\u201316.","DOI":"10.1145\/3626772.3657675"},{"key":"458_CR53","unstructured":"Finardi P, Avila L, Castaldoni R, Gengo P, Larcher C, Piau M, Costa P, Carid\u00e1 V. The chronicles of rag: the retriever, the chunk and the generator. arXiv:2401.07883 [Preprint]. 2024. Available from http:\/\/arxiv.org\/abs\/2401.07883"},{"key":"458_CR54","doi-asserted-by":"crossref","unstructured":"Rajpurkar P, Zhang J, Lopyrev K, Liang P. Squad: 100,000+ questions for machine comprehension of text. arXiv:1606.05250 [Preprint]. 2016. Available from: http:\/\/arxiv.org\/abs\/1606.05250","DOI":"10.18653\/v1\/D16-1264"},{"issue":"7","key":"458_CR55","doi-asserted-by":"publisher","first-page":"175","DOI":"10.1007\/s10462-024-10824-0","volume":"57","author":"X Huang","year":"2024","unstructured":"Huang X, Ruan W, Huang W, Jin G, Dong Y, Wu C, et al. A survey of safety and trustworthiness of large language models through the lens of verification and validation. Artif Intell Rev. 2024;57(7):175.","journal-title":"Artif Intell Rev"},{"issue":"4","key":"458_CR56","first-page":"35","volume":"24","author":"A Singhal","year":"2001","unstructured":"Singhal A, et al. Modern information retrieval: a brief overview. IEEE Data Eng Bull. 2001;24(4):35\u201343.","journal-title":"IEEE Data Eng Bull"},{"key":"458_CR57","doi-asserted-by":"crossref","unstructured":"Es S, James J, Anke LE, Schockaert S. Ragas: Automated evaluation of retrieval augmented generation. In: Proceedings of the 18th conference of the European chapter of the association for computational linguistics: system demonstrations. 2024. p. 150\u20138.","DOI":"10.18653\/v1\/2024.eacl-demo.16"},{"key":"458_CR58","doi-asserted-by":"crossref","unstructured":"Yang Y, Li Z, Dong Q, Xia H, Sui Z. Can large multimodal models uncover deep semantics behind images? arXiv:2402.11281 [Preprint]. 2024. Available from: http:\/\/arxiv.org\/abs\/2402.11281","DOI":"10.18653\/v1\/2024.findings-acl.113"}],"container-title":["Health Information Science and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13755-026-00458-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13755-026-00458-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13755-026-00458-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T06:01:38Z","timestamp":1777010498000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13755-026-00458-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,24]]},"references-count":58,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,12]]}},"alternative-id":["458"],"URL":"https:\/\/doi.org\/10.1007\/s13755-026-00458-7","relation":{},"ISSN":["2047-2501"],"issn-type":[{"value":"2047-2501","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,24]]},"assertion":[{"value":"15 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"There are no conflict of interest to declare.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"All experts who participated in the human evaluation phase of the LLM-generated results provided informed consent prior to their involvement in the study.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed consent"}}],"article-number":"60"}}