{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T05:36:52Z","timestamp":1743140212054,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":28,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819608645"},{"type":"electronic","value":"9789819608652"}],"license":[{"start":{"date-parts":[[2024,12,6]],"date-time":"2024-12-06T00:00:00Z","timestamp":1733443200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,6]],"date-time":"2024-12-06T00:00:00Z","timestamp":1733443200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-0865-2_22","type":"book-chapter","created":{"date-parts":[[2024,12,5]],"date-time":"2024-12-05T06:29:28Z","timestamp":1733380168000},"page":"269-279","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Evaluating Large Language Models for Healthcare: Insights from MCQ Evaluation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-0589-3088","authenticated-orcid":false,"given":"Shuangshuang","family":"Lin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9247-8682","authenticated-orcid":false,"given":"Hamzah Bin","family":"Osop","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3237-1213","authenticated-orcid":false,"given":"Miao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8732-6252","authenticated-orcid":false,"given":"Xinxian","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,6]]},"reference":[{"issue":"4","key":"22_CR1","doi-asserted-by":"publisher","first-page":"100324","DOI":"10.1016\/j.xops.2023.100324","volume":"3","author":"F Antaki","year":"2023","unstructured":"Antaki, F., Touma, S., Milad, D., El-Khoury, J., Duval, R.: Evaluating the performance of ChatGPT in ophthalmology: an analysis of its successes and shortcomings. Ophthalmol. Sci. 3(4), 100324 (2023). https:\/\/doi.org\/10.1016\/j.xops.2023.100324","journal-title":"Ophthalmol. Sci."},{"key":"22_CR2","doi-asserted-by":"crossref","unstructured":"Bang, Y., et al.: A multitask, multilingual, multimodal evaluation of ChatGPT on reasoning, hallucination, and interactivity. In: Proceedings of the 13th International Joint Conference on Natural Language Processing and the 3rd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 675\u2013718, November 2023","DOI":"10.18653\/v1\/2023.ijcnlp-main.45"},{"key":"22_CR3","doi-asserted-by":"crossref","unstructured":"Bast, H., Buchhold, B., Haussmann, E.: Semantic search on text and knowledge bases. Found. Trends\u00ae Information Retr. 10(2\u20133), 119\u2013271 (2016)","DOI":"10.1561\/1500000032"},{"key":"22_CR4","first-page":"1877","volume":"33","author":"B Mann","year":"2020","unstructured":"Mann, B., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"22_CR5","doi-asserted-by":"crossref","unstructured":"He, P., Huang, J., Li, M.: Text keyword extraction based on GPT. In: 2024 27th International Conference on Computer Supported Cooperative Work in Design (CSCWD), pp. 1394\u20131398. IEEE, May 2024","DOI":"10.1109\/CSCWD61410.2024.10580849"},{"key":"22_CR6","unstructured":"Feng, Z., et al.: Trends in Integration of Knowledge and Large Language Models: A Survey and Taxonomy of Methods, Benchmarks, and Applications (2023). arXiv preprint arXiv:2311.05876"},{"key":"22_CR7","unstructured":"Guu, K., Lee, K., Tung, Z., Pasupat, P., Chang, M.W.: REALM: retrieval-augmented language model pre-training. In: Proceedings of the 37th International Conference on Machine Learning, pp. 3929\u20133938, July 2020"},{"key":"22_CR8","doi-asserted-by":"crossref","unstructured":"Huang, L., et al.: A survey on hallucination in large language models: principles, taxonomy, challenges, and open questions (2023). arXiv preprint arXiv:2311.05232","DOI":"10.1145\/3703155"},{"key":"22_CR9","unstructured":"Jiang, A.Q., et al.: Mistral 7B (2023). arXiv preprint arXiv:2310.06825"},{"key":"22_CR10","doi-asserted-by":"publisher","first-page":"102274","DOI":"10.1016\/j.lindif.2023.102274","volume":"103","author":"E Kasneci","year":"2023","unstructured":"Kasneci, E., et al.: ChatGPT for good? On opportunities and challenges of large language models for education. Learn. Individ. Differ. 103, 102274 (2023). https:\/\/doi.org\/10.1016\/j.lindif.2023.102274","journal-title":"Learn. Individ. Differ."},{"key":"22_CR11","unstructured":"Khattab, O., et al.: Demonstrate-search-predict: Composing retrieval and language models for knowledge-intensive NLP (2023). arXiv preprint arXiv:2212.14024"},{"key":"22_CR12","first-page":"22199","volume":"35","author":"T Kojima","year":"2022","unstructured":"Kojima, T., Gu, S.S., Reid, M., Matsuo, Y., Iwasawa, Y.: Large language models are zero-shot reasoners. Adv. Neural. Inf. Process. Syst. 35, 22199\u201322213 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"22_CR13","doi-asserted-by":"crossref","unstructured":"Labrak, Y., Bazoge, A., Morin, E., Gourraud, P.-A., Rouvier, M., Dufour, R.: BioMistral: a collection of open-source pretrained large language models for medical domains. In Findings of the Association for Computational Linguistics ACL 2024, pp. 5848\u20135864, Bangkok, Thailand and virtual meeting. Association for Computational Linguistics, August 2024","DOI":"10.18653\/v1\/2024.findings-acl.348"},{"key":"22_CR14","doi-asserted-by":"crossref","unstructured":"Li, J., Cheng, X., Zhao, W.X., Nie, J.Y., Wen, J.R.: Halueval: a large-scale hallucination evaluation benchmark for large language models. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 6449\u20136464, December 2023","DOI":"10.18653\/v1\/2023.emnlp-main.397"},{"key":"22_CR15","doi-asserted-by":"publisher","unstructured":"Li\u00e9vin, V., Hother, C.E., Motzfeldt, A.G., Winther, O.: Can large language models reason about medical questions? Patterns (New York, N.Y.) 5(3), 100943 (2024). https:\/\/doi.org\/10.1016\/j.patter.2024.100943","DOI":"10.1016\/j.patter.2024.100943"},{"key":"22_CR16","doi-asserted-by":"crossref","unstructured":"Lin, S., Hilton, J., Evans, O.: TruthfulQA: measuring how models mimic human falsehoods. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 3214\u20133252, Dublin, Ireland. Association for Computational Linguistics, May 2022","DOI":"10.18653\/v1\/2022.acl-long.229"},{"key":"22_CR17","doi-asserted-by":"crossref","unstructured":"Mallen, A., Asai, A., Zhong, V., Das, R., Khashabi, D., Hajishirzi, H.: When not to trust language models: investigating effectiveness of parametric and non-parametric memories. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 9802\u20139822. Association for Computational Linguistics, Toronto, Canada (2023)","DOI":"10.18653\/v1\/2023.acl-long.546"},{"issue":"2","key":"22_CR18","doi-asserted-by":"publisher","first-page":"210","DOI":"10.7326\/M23-2772","volume":"177","author":"JA Omiye","year":"2023","unstructured":"Omiye, J.A., Gui, H., Rezaei, S.J., Zou, J., Daneshjou, R.: Large language models in medicine: the potentials and pitfalls. Ann. Intern. Med. 177(2), 210\u2013220 (2023). https:\/\/doi.org\/10.7326\/M23-2772","journal-title":"Ann. Intern. Med."},{"key":"22_CR19","unstructured":"Pal, A., Umapathi, L.K., Sankarasubbu, M.: Medmcqa: a large-scale multi-subject multi-choice dataset for medical domain question answering. In: Conference on Health, Inference, and Learning, Proceedings of Machine Learning Research, pp. 248\u2013260, April 2022"},{"key":"22_CR20","doi-asserted-by":"publisher","unstructured":"Peng, C., et al.: A study of generative large language model for medical research and healthcare. NPJ Digit. Med. 6(1) (2023). https:\/\/doi.org\/10.1038\/s41746-023-00958-w","DOI":"10.1038\/s41746-023-00958-w"},{"key":"22_CR21","doi-asserted-by":"crossref","unstructured":"Shi, W., et al.: Replug: retrieval-augmented black-box language models. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 8371\u20138384, Mexico City, Mexico. Association for Computational Linguistics (2023)","DOI":"10.18653\/v1\/2024.naacl-long.463"},{"key":"22_CR22","unstructured":"Si, C., et al.: Prompting GPT-3 to be reliable. In: The Eleventh International Conference on Learning Representations, (ICLR 2023) (2022)"},{"key":"22_CR23","doi-asserted-by":"crossref","unstructured":"Siddiqi, S., Sharan, A.: Keyword and keyphrase extraction techniques: a literature review. Int. J. Comput. Appl. 109(2) (2015)","DOI":"10.5120\/19161-0607"},{"key":"22_CR24","doi-asserted-by":"crossref","unstructured":"Stupans, I.: Multiple choice questions: can they examine application of knowledge? Pharmacy Educ. 6(1) (2006)","DOI":"10.1080\/15602210600567916"},{"key":"22_CR25","unstructured":"Touvron, H., et al.: Llama 2: open foundation and fine-tuned chat models (2023). arXiv preprint arXiv:2307.09288"},{"key":"22_CR26","doi-asserted-by":"crossref","unstructured":"Pal, A., Umapathi, L.K., Sankarasubbu, M.: Med-HALT: medical domain hallucination test for large language models. In: Proceedings of the 27th Conference on Computational Natural Language Learning (CoNLL), pp. 314\u2013334, December 2023","DOI":"10.18653\/v1\/2023.conll-1.21"},{"key":"22_CR27","unstructured":"Wu, C., Zhang, X., Zhang, Y., Wang, Y., Xie, W.: PMC-llama: further finetuning llama on medical papers (2023). arXiv preprint arXiv:2304.14454"},{"key":"22_CR28","unstructured":"Zhu, K., et al.: PromptBench: Towards Evaluating the Robustness of Large Language Models on Adversarial Prompts (2023). arXiv preprint arXiv:2306.04528"}],"container-title":["Lecture Notes in Computer Science","Sustainability and Empowerment in the Context of Digital Libraries"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-0865-2_22","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,5]],"date-time":"2024-12-05T11:06:11Z","timestamp":1733396771000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-0865-2_22"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,6]]},"ISBN":["9789819608645","9789819608652"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-0865-2_22","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,12,6]]},"assertion":[{"value":"6 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interest"}},{"value":"ICADL","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Asian Digital Libraries","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Bandar Sunway","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Malaysia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icadl2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icadl.net\/icadl2024\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}