{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,6]],"date-time":"2026-01-06T05:49:03Z","timestamp":1767678543740,"version":"3.48.0"},"reference-count":57,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/aiccsa66935.2025.11315481","type":"proceedings-article","created":{"date-parts":[[2026,1,5]],"date-time":"2026-01-05T18:35:55Z","timestamp":1767638155000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["Information Reasoning and Question Answering in Healthcare: A PubMedQA Benchmark Study"],"prefix":"10.1109","author":[{"given":"Dina","family":"Sayed","sequence":"first","affiliation":[{"name":"University of Basel,Databases and Information Systems Research Group,Basel,Switzerland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Heiko","family":"Schuldt","sequence":"additional","affiliation":[{"name":"University of Basel,Databases and Information Systems Research Group,Basel,Switzerland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463251"},{"key":"ref2","first-page":"523","article-title":"Medseer: A medical controversial information retrieval system based on credible sources","volume-title":"Linking Theory and Practice of Digital Libraries - 26th International Conference on Theory and Practice of Digital Libraries, TPDL 2022","volume":"13541","author":"Sayed"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3458754"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.bionlp-1.16"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d19-1259"},{"key":"ref6","first-page":"8003","article-title":"Linkbert: Pretraining language models with document links","volume":"abs\/2203.15827","author":"Yasunaga","year":"2022","journal-title":"CoRR"},{"key":"ref7","first-page":"5848","article-title":"Biomistral: A collection of open-source pretrained large language models for medical domains","volume-title":"Findings of the Association for Computational Linguistics, ACL 2024. Bangkok, Thailand and virtual meeting","author":"Labrak"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1093\/jamia\/ocae045"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-short.47"},{"key":"ref10","article-title":"Meditron70b: Scaling medical pretraining for large language models","volume":"abs\/2311.16079","author":"Chen","year":"2023","journal-title":"CoRR"},{"key":"ref11","article-title":"Capabilities of gpt-4 on medical challenge problems","volume":"abs\/2303.13375","author":"Nori","year":"2023","journal-title":"CoRR"},{"article-title":"The claude 3 model family: Opus, sonnet, haiku","year":"2024","author":"Anthropic","key":"ref12"},{"key":"ref13","article-title":"Towards expert-level medical question answering with large language models","volume":"abs\/2305.09617","author":"Singhal","year":"2023","journal-title":"CoRR"},{"key":"ref14","article-title":"Can generalist foundation models outcompete special-purpose tuning? case study in medicine","volume":"abs\/2311.16452","author":"Nori","year":"2023","journal-title":"CoRR"},{"year":"2024","key":"ref15","article-title":"Phi-3 medium 4k instruct"},{"year":"2023","key":"ref16","article-title":"Pubmed"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.5603\/KP.a2015.0092"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/p14-5010"},{"key":"ref19","first-page":"29242936","article-title":"Boolq: Exploring the surprising difficulty of natural yes\/no questions","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL-HLT","volume":"1","author":"Clark"},{"key":"ref20","first-page":"8596","article-title":"This is not a dataset: A large negation benchmark to challenge large language models","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, EMNLP 2023.","author":"Garc\u00eda-Ferrero"},{"key":"ref21","first-page":"101","article-title":"Language models are not naysayers: an analysis of language models on negation benchmarks","volume-title":"Proceedings of the 12th Joint Conference on Lexical and Computational Semantics (*SEM 2023","author":"Truong"},{"key":"ref22","article-title":"Negation blindness in large language models: Unveiling the no syndrome in image generation","volume":"abs\/2409.00105","author":"Nadeem","year":"2024","journal-title":"CoRR"},{"key":"ref23","article-title":"Investigating and addressing hallucinations of 11 ms in tasks involving negation","volume":"abs\/2406.05494","author":"Varshney","year":"2024","journal-title":"CoRR"},{"key":"ref24","article-title":"Numerologic: Number encoding for enhanced 11 ms \u2018 numerical reasoning","volume":"abs\/2404.00459","author":"Schwartz","year":"2024","journal-title":"CoRR"},{"key":"ref25","first-page":"37","article-title":"Mathprompter: Mathematical reasoning using large language models","volume-title":"Proceedings of the The 61st Annual Meeting of the Association for Computational Linguistics: Industry Track, ACL 2023.","author":"Imani"},{"key":"ref26","article-title":"Evaluating mathematical reasoning beyond accuracy","volume":"abs\/2404.05692","author":"Xia","year":"2024","journal-title":"CoRR"},{"key":"ref27","first-page":"24824","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume":"35","author":"Wei","year":"2022","journal-title":"Advances in neural information processing systems"},{"article-title":"Self-consistency improves chain of thought reasoning in language models","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Wang","key":"ref28"},{"key":"ref29","article-title":"Large language model guided tree-of-thought","volume":"abs\/2305.08291","author":"Long","year":"2023","journal-title":"CoRR"},{"key":"ref30","article-title":"Tree of thoughts: Deliberate problem solving with large language models","volume":"36","author":"Yao","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref31","article-title":"Large language models as an indirect reasoner: Contrapositive and contradiction for automated reasoning","volume":"abs\/2402.03667","author":"Zhang","year":"2024","journal-title":"CoRR"},{"key":"ref32","first-page":"9215","article-title":"Are 11 ms capable of data-based statistical and causal reasoning? benchmarking advanced quantitative reasoning with data","volume-title":"Findings of the Association for Computational Linguistics, ACL 2024","author":"Liu"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1016\/S0090-3019(03)00016-8"},{"year":"2024","key":"ref34","article-title":"Llama"},{"volume-title":"Mistral","year":"2024","key":"ref35"},{"year":"2024","key":"ref36","article-title":"Phi"},{"key":"ref37","first-page":"41714186","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proceedings of naacL-HLT","volume":"1","author":"Devlin"},{"article-title":"Openllama: An open reproduction of llama","year":"2023","author":"Geng","key":"ref38"},{"key":"ref39","article-title":"Phi-3 technical report: A highly capable language model locally on your phone","volume-title":"CoRR","volume":"abs\/2404.14219","author":"Abdin","year":"2024"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1038\/s41746-023-00896-7"},{"year":"2024","key":"ref41","article-title":"Axios hq conclusion generator"},{"year":"2024","key":"ref42","article-title":"Hypotenuse conclusion generator"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1016\/j.annepidem.2024.06.008"},{"year":"2024","key":"ref44","article-title":"Yeschat conclusion generator"},{"article-title":"Lora: Low-rank adaptation of large language models","volume-title":"The Tenth International Conference on Learning Representations, ICLR 2022","author":"Hu","key":"ref45"},{"key":"ref46","article-title":"A rank stabilization scaling factor for fine-tuning with lora","volume":"abs\/2312.03732","author":"Kalajdzievski","year":"2023","journal-title":"CoRR"},{"key":"ref47","first-page":"74","article-title":"Rouge: A package for automatic evaluation of summaries","author":"Lin","year":"2004","journal-title":"Text summarization branches out."},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1186\/s12859-015-0564-6"},{"key":"ref50","first-page":"248","article-title":"Medmcqa: A largescale multi-subject multi-choice dataset for medical domain question answering","volume-title":"Proceedings of the Conference on Health, Inference, and Learning, ser. Proceedings of Machine Learning Research","volume":"174","author":"Pal"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.3390\/app11146421"},{"key":"ref52","article-title":"Measuring massive multitask language understanding","author":"Hendrycks","year":"2020","journal-title":"arXiv preprint arXiv:2009.03300"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1611835114"},{"key":"ref54","article-title":"An empirical study of catastrophic forgetting in large language models during continual fine-tuning","volume":"abs\/2308.08747","author":"Luo","year":"2023","journal-title":"CoRR"},{"key":"ref55","first-page":"1416","article-title":"Mitigating catastrophic forgetting in large language models with selfsynthesized rehearsal","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics","volume":"1","author":"Huang"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.398"},{"key":"ref57","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume":"36","author":"Rafailov","year":"2024","journal-title":"Advances in Neural Information Processing Systems"}],"event":{"name":"2025 IEEE\/ACS 22nd International Conference on Computer Systems and Applications (AICCSA)","start":{"date-parts":[[2025,10,19]]},"location":"Doha, Qatar","end":{"date-parts":[[2025,10,22]]}},"container-title":["2025 IEEE\/ACS 22nd International Conference on Computer Systems and Applications (AICCSA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11315137\/11315140\/11315481.pdf?arnumber=11315481","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,6]],"date-time":"2026-01-06T05:45:26Z","timestamp":1767678326000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11315481\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":57,"URL":"https:\/\/doi.org\/10.1109\/aiccsa66935.2025.11315481","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}