{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T19:11:51Z","timestamp":1784056311589,"version":"3.55.0"},"reference-count":81,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,1,2]],"date-time":"2025-01-02T00:00:00Z","timestamp":1735776000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,2]],"date-time":"2025-01-02T00:00:00Z","timestamp":1735776000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SN COMPUT. SCI."],"DOI":"10.1007\/s42979-024-03533-6","type":"journal-article","created":{"date-parts":[[2025,1,3]],"date-time":"2025-01-03T01:28:55Z","timestamp":1735867735000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Large Language Model Evaluation Criteria Framework in Healthcare: Fuzzy MCDM Approach"],"prefix":"10.1007","volume":"6","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1410-8031","authenticated-orcid":false,"given":"Hamzeh Mohammad","family":"Alabool","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,1,2]]},"reference":[{"key":"3533_CR1","volume-title":"Planning for agi and beyond","author":"S Altman","year":"2023","unstructured":"Altman S. Planning for agi and beyond. OpenAI Blog; 2023."},{"issue":"6","key":"3533_CR2","doi-asserted-by":"publisher","first-page":"887","DOI":"10.3390\/healthcare11060887","volume":"11","author":"M Sallam","year":"2023","unstructured":"Sallam M. ChatGPT utility in healthcare education, research, and practice: systematic review on the promising perspectives and valid concerns. Healthcare. 2023;11(6):887.","journal-title":"Healthcare"},{"issue":"2","key":"3533_CR3","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pdig.0000198","volume":"2","author":"TH Kung","year":"2023","unstructured":"Kung TH, Cheatham M, Medenilla A, Sillos C, De Leon L, Elepa\u00f1o C, Madriaga M, Aggabao R, Diaz-Candido G, Maningo J, Tseng V. Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models. PLoS Digit Health. 2023;2(2): e0000198.","journal-title":"PLoS Digit Health"},{"issue":"4","key":"3533_CR4","doi-asserted-by":"publisher","first-page":"e179","DOI":"10.1016\/S2589-7500(23)00048-1","volume":"5","author":"SR Ali","year":"2023","unstructured":"Ali SR, Dobbs TD, Hutchings HA, Whitaker IS. Using ChatGPT to write patient clinic letters. The Lancet Digital Health. 2023;5(4):e179\u201381.","journal-title":"The Lancet Digital Health"},{"issue":"7947","key":"3533_CR5","doi-asserted-by":"publisher","first-page":"224","DOI":"10.1038\/d41586-023-00288-7","volume":"614","author":"EA Van Dis","year":"2023","unstructured":"Van Dis EA, Bollen J, Zuidema W, van Rooij R, Bockting CL. ChatGPT: five priorities for research. Nature. 2023;614(7947):224\u20136.","journal-title":"Nature"},{"issue":"4","key":"3533_CR6","doi-asserted-by":"publisher","first-page":"405","DOI":"10.1016\/S1473-3099(23)00113-5","volume":"23","author":"A Howard","year":"2023","unstructured":"Howard A, Hope W, Gerada A. ChatGPT and antimicrobial advice: the end of the consulting infection doctor? Lancet Infect Dis. 2023;23(4):405\u20136.","journal-title":"Lancet Infect Dis"},{"key":"3533_CR7","doi-asserted-by":"publisher","DOI":"10.1016\/j.lindif.2023.102274","volume":"103","author":"E Kasneci","year":"2023","unstructured":"Kasneci E, Se\u00dfler K, K\u00fcchemann S, Bannert M, Dementieva D, Fischer F, Gasser U, Groh G, G\u00fcnnemann S, H\u00fcllermeier E, Krusche S. ChatGPT for good? On opportunities and challenges of large language models for education. Learn Individ Differ. 2023;103: 102274.","journal-title":"Learn Individ Differ"},{"issue":"5","key":"3533_CR8","doi-asserted-by":"publisher","first-page":"1233","DOI":"10.1039\/D3DD00113J","volume":"2","author":"KM Jablonka","year":"2023","unstructured":"Jablonka KM, Ai Q, Al-Feghali A, Badhwar S, Bocarsly JD, Bran AM, Bringuier S, Brinson LC, Choudhary K, Circi D, Cox S. 14 examples of how LLMs can transform materials science and chemistry: a reflection on a large language model hackathon. Digit Discov. 2023;2(5):1233\u201350.","journal-title":"Digit Discov"},{"issue":"1","key":"3533_CR9","doi-asserted-by":"publisher","first-page":"104","DOI":"10.37284\/eaje.6.1.1272","volume":"6","author":"M Aluga","year":"2023","unstructured":"Aluga M. Application of CHATGPT in civil engineering. East Afr J Eng. 2023;6(1):104\u201312.","journal-title":"East Afr J Eng"},{"key":"3533_CR10","doi-asserted-by":"crossref","unstructured":"Ogundare O, Madasu S, Wiggins N. Industrial engineering with large language models: a case study of ChatGPT's performance on Oil & Gas problems.\u00a02023. arXiv preprint arXiv:2304.14354.","DOI":"10.1109\/ICCMA59762.2023.10374622"},{"issue":"1","key":"3533_CR11","doi-asserted-by":"publisher","first-page":"237","DOI":"10.1162\/coli_a_00502","volume":"50","author":"C Ziems","year":"2024","unstructured":"Ziems C, Held W, Shaikh O, Chen J, Zhang Z, Yang D. Can large language models transform computational social science? Comput Linguist. 2024;50(1):237\u201391.","journal-title":"Comput Linguist."},{"key":"3533_CR12","unstructured":"Tamkin A, Brundage M, Clark J, Ganguli, D. Understanding the capabilities, limitations, and societal impact of large language models.\u00a02021. arXiv preprint arXiv:2102.02503."},{"key":"3533_CR13","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijinfomgt.2023.102642","volume":"71","author":"YK Dwivedi","year":"2023","unstructured":"Dwivedi YK, Kshetri N, Hughes L, Slade EL, Jeyaraj A, Kar AK, Baabdullah AM, Koohang A, Raghavan V, Ahuja M, Albanna H. \u201cSo what if ChatGPT wrote it?\u201d Multidisciplinary perspectives on opportunities, challenges and implications of generative conversational AI for research, practice and policy. Int J Inf Manag. 2023;71: 102642.","journal-title":"Int J Inf Manag"},{"key":"3533_CR14","doi-asserted-by":"publisher","DOI":"10.1016\/j.frl.2023.104643","volume":"58","author":"A Alonso-Robisco","year":"2023","unstructured":"Alonso-Robisco A, Carb\u00f3 JM. Analysis of CBDC narrative by central banks using large language models. Financ Res Lett. 2023;58: 104643.","journal-title":"Financ Res Lett"},{"issue":"2","key":"3533_CR15","doi-asserted-by":"publisher","first-page":"102","DOI":"10.69554\/CNMI7720","volume":"8","author":"JP Sleiman","year":"2023","unstructured":"Sleiman JP. Generative artificial intelligence and large language models for digital banking: First outlook and perspectives. J Digit Bank. 2023;8(2):102\u201317.","journal-title":"J Digit Bank"},{"issue":"6","key":"3533_CR16","doi-asserted-by":"publisher","first-page":"23","DOI":"10.18775\/ijied.1849-7551-7020.2015.86.2003","volume":"8","author":"S Sood","year":"2023","unstructured":"Sood S, Pattinson H. Marketing education renaissance through big data curriculum: developing marketing expertise using AI large language models. Int J Innov Econ Dev. 2023;8(6):23\u201340.","journal-title":"Int J Innov Econ Dev"},{"key":"3533_CR17","doi-asserted-by":"publisher","first-page":"3450","DOI":"10.1016\/j.procs.2023.10.340","volume":"225","author":"M Orzo\u0142","year":"2023","unstructured":"Orzo\u0142 M, Szopik-Depczy\u0144ska K. ChatGPT as an innovative tool for increasing sales in online stores. Proc Comput Sci. 2023;225:3450\u20139.","journal-title":"Proc Comput Sci"},{"issue":"4","key":"3533_CR18","doi-asserted-by":"publisher","DOI":"10.1016\/j.tbench.2023.100089","volume":"2","author":"A Haleem","year":"2022","unstructured":"Haleem A, Javaid M, Singh RP. An era of ChatGPT as a significant futuristic support tool: a study on features, abilities, and challenges. BenchCouncil Trans Benchmarks Stand Eval. 2022;2(4): 100089.","journal-title":"BenchCouncil Trans Benchmarks Stand Eval"},{"key":"3533_CR19","doi-asserted-by":"crossref","unstructured":"Suzgun M, Scales N, Sch\u00e4rli N, Gehrmann S, Tay Y, Chung HW, Chowdhery A, Le QV, Chi EH, Zhou D, Wei J. Challenging big-bench tasks and whether chain-of-thought can solve them.\u00a02022. arXiv preprint arXiv:2210.09261.","DOI":"10.18653\/v1\/2023.findings-acl.824"},{"key":"3533_CR20","unstructured":"Srivastava A, Rastogi A, Rao A, Shoeb AAM, Abid A, Fisch A, Brown AR, Santoro A, Gupta A, Garriga-Alonso A, Kluska, A. Beyond the imitation game: Quantifying and extrapolating the capabilities of language models.\u00a02022. arXiv preprint arXiv:2206.04615."},{"key":"3533_CR21","unstructured":"Liang, P, Bommasani, R, Lee, T, Tsipras, D, Soylu, D, Yasunaga, M, Zhang, Y, Narayanan, D, Wu, Y, Kumar, A. and Newman, B, 2022. Holistic evaluation of language models.\u00a0arXiv preprint arXiv:2211.09110."},{"key":"3533_CR22","doi-asserted-by":"crossref","unstructured":"Shah RS, Chawla K, Eidnani D, Shah A, Du W, Chava S, Raman N, Smiley C, Chen J, Yang D. When flue meets flang: Benchmarks and large pre-trained language model for financial domain.\u00a02022. arXiv preprint arXiv:2211.00083.","DOI":"10.18653\/v1\/2022.emnlp-main.148"},{"key":"3533_CR23","unstructured":"Huang Y, Bai Y, Zhu Z, Zhang J, Zhang J, Su T, Liu J, Lv C, Zhang Y, Lei J, Qi F. C-eval: A multi-level multi-discipline chinese evaluation suite for foundation models.\u00a02023. arXiv preprint arXiv:2305.08322."},{"key":"3533_CR24","doi-asserted-by":"crossref","unstructured":"Zhong W, Cui R, Guo Y, Liang Y, Lu S, Wang Y, Saied A, Chen W, Duan, N. Agieval: a human-centric benchmark for evaluating foundation models.\u00a02023. arXiv preprint arXiv:2304.06364.","DOI":"10.18653\/v1\/2024.findings-naacl.149"},{"key":"3533_CR25","unstructured":"Zheng L, Chiang WL, Sheng Y, Zhuang S, Wu Z, Zhuang Y, Lin Z, Li Z, Li D, Xing E, Zhang, H. Judging LLM-as-a-judge with MT-Bench and Chatbot Arena.\u00a02023. arXiv preprint arXiv:2306.05685."},{"key":"3533_CR26","unstructured":"Lee M, Srivastava M, Hardy A, Thickstun J, Durmus E, Paranjape A, Gerard-Ursin I, Li XL, Ladhak F, Rong F, Wang RE. Evaluating human-language model interaction.\u00a02022. arXiv preprint arXiv:2212.09746."},{"key":"3533_CR27","unstructured":"Ye S, Kim D, Kim S, Hwang H, Kim S, Jo Y, Thorne J, Kim J, Seo M. Flask: Fine-grained language model evaluation based on alignment skill sets.\u00a02023. arXiv preprint arXiv:2307.10928."},{"key":"3533_CR28","doi-asserted-by":"crossref","unstructured":"Bang Y, Cahyawijaya S, Lee N, Dai W, Su D, Wilie B, Lovenia H, Ji Z, Yu T, Chung W, Do QV. A multitask, multilingual, multimodal evaluation of chatgpt on reasoning, hallucination, and interactivity.\u00a02023. arXiv preprint arXiv:2302.04023.","DOI":"10.18653\/v1\/2023.ijcnlp-main.45"},{"key":"3533_CR29","unstructured":"Chen M, Tworek J, Jun H, Yuan Q, Pinto HPDO, Kaplan J, Edwards H, Burda Y, Joseph N, Brockman G, Ray A. Evaluating large language models trained on code.\u00a02021. arXiv preprint arXiv:2107.03374."},{"key":"3533_CR30","doi-asserted-by":"crossref","unstructured":"Li, M, Song, F, Yu, B, Yu, H, Li, Z, Huang, F. and Li, Y, 2023. Api-bank: A benchmark for tool-augmented llms.\u00a0arXiv preprint arXiv:2304.08244.","DOI":"10.18653\/v1\/2023.emnlp-main.187"},{"key":"3533_CR31","unstructured":"Singhal K, Azizi S, Tu T, Mahdavi SS, Wei J, Chung HW, Scales N, Tanwani A, Cole-Lewis H, Pfohl S, Payne P. Large language models encode clinical knowledge.\u00a02022. arXiv preprint arXiv:2212.13138."},{"key":"3533_CR32","unstructured":"Hendrycks D, Burns C, Basart S, Zou A, Mazeika M, Song D, Steinhardt. Measuring massive multitask language understanding.\u00a02020. arXiv preprint arXiv:2009.03300."},{"key":"3533_CR33","unstructured":"Zhao WX, Zhou K, Li J, Tang T, Wang X, Hou Y, Min Y, Zhang B, Zhang J, Dong Z, Du Y. A survey of large language models.\u00a02023. arXiv preprint arXiv:2303.18223."},{"key":"3533_CR34","unstructured":"Zeng H. Measuring massive multitask chinese understanding. 2023. arXiv preprint arXiv:2304.12986."},{"key":"3533_CR35","unstructured":"Liu C, Jin R, Ren Y, Yu L, Dong T, Peng X, Zhang S, Peng J, Zhang P, Lyu Q, Su X. M3KE: A massive multi-level multi-subject knowledge evaluation benchmark for Chinese large language models. 2023. arXiv preprint arXiv:2305.10263."},{"key":"3533_CR36","unstructured":"Zhouhong G, Xiaoxuan Z, Haoning Y, Lin Z, Jianchen W, Sihang J, Zhuozhi X, Zihan L, Qianyu H, Rui X, Wenhao, H. Xiezhi: An ever-updating benchmark for holistic domain knowledge evaluation.\u00a02023. arXiv preprint arXiv:2304.11679,\u00a02."},{"key":"3533_CR37","doi-asserted-by":"crossref","unstructured":"Fu, J, Ng, S.K, Jiang, Z. and Liu, P, 2023. Gptscore: Evaluate as you desire.\u00a0arXiv preprint arXiv:2302.04166","DOI":"10.18653\/v1\/2024.naacl-long.365"},{"key":"3533_CR38","volume-title":"Alpacaeval: An automatic evaluator of instruction-following models","author":"X Li","year":"2023","unstructured":"Li X, Zhang T, Dubois Y, Taori R, Gulrajani I, Guestrin C, Liang P, Hashimoto TB. Alpacaeval: An automatic evaluator of instruction-following models. GitHub Repository; 2023."},{"key":"3533_CR39","unstructured":"Tang Q, Deng Z, Lin H, Han X, Liang Q, Sun L. Toolalpaca: Generalized tool learning for language models with 3000 simulated cases. CoRR, abs\/2306.05301, 2023. 10.48550.\u00a0arXiv preprint arXiv.2306.05301."},{"key":"3533_CR40","unstructured":"Xu Q, Hong F, Li B, Hu C, Chen Z, Zhang J. On the Tool manipulation capability of open-source large language models.\u00a02023. arXiv preprint arXiv:2305.16504."},{"key":"3533_CR41","unstructured":"Qin Y, Liang S, Ye Y, Zhu K, Yan L, Lu Y, Lin Y, Cong X, Tang X, Qian B, Zhao, S. Toolllm: facilitating large language models to master 16000+ real-world apis.\u00a02023. arXiv preprint arXiv:2307.16789."},{"key":"3533_CR42","unstructured":"Liu Z, Yao W, Zhang J, Xue L, Heinecke S, Murthy R, Feng Y, Chen Z, Niebles JC, Arpit D, Xu R. Bolaa: Benchmarking and orchestrating llm-augmented autonomous agents.\u00a02023. arXiv preprint arXiv:2308.05960."},{"key":"3533_CR43","unstructured":"Guha N, Ho DE, Nyarko J, R\u00e9, C. Legalbench: Prototyping a collaborative benchmark for legal reasoning.\u00a02022. arXiv preprint arXiv:2209.06120."},{"key":"3533_CR44","unstructured":"Jain N, Saifullah K, Wen Y, Kirchenbauer J, Shu M, Saha A, Goldblum M, Geiping J, Goldstein T. Bring Your Own Data! Self-Supervised Evaluation for Large Language Models.\u00a02023. arXiv preprint arXiv:2306.13651."},{"key":"3533_CR45","unstructured":"Bai Y, Ying J, Cao Y, Lv X, He Y, Wang X, Yu J, Zeng K, Xiao Y, Lyu H, Zhang, J. Benchmarking foundation models with language-model-as-an-Examiner.\u00a02023. arXiv preprint arXiv:2306.04181."},{"key":"3533_CR46","unstructured":"Chan CM, Chen W, Su Y, Yu J, Xue W, Zhang S, F, J, Liu, Z. Chateval: Towards better llm-based evaluators through multi-agent debate.\u00a02023. arXiv preprint arXiv:2308.07201."},{"issue":"2","key":"3533_CR47","doi-asserted-by":"publisher","first-page":"120","DOI":"10.1576\/toag.7.2.120.27071","volume":"7","author":"S Thangaratinam","year":"2005","unstructured":"Thangaratinam S, Redman CW. The delphi technique. Obstet Gynaecol. 2005;7(2):120\u20135.","journal-title":"Obstet Gynaecol"},{"key":"3533_CR48","unstructured":"IEEE Standards Association and Others, IEEE STD 1061\u20131998, IEEE standard for a software quality metrics methodology, 1998."},{"key":"3533_CR49","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1016\/S1479-3709(07)11003-7","volume-title":"Empirical methods for bioethics: a primer","author":"J Forman","year":"2007","unstructured":"Forman J, Damschroder L. Qualitative content analysis. In: Jacoby L, Siminoff LA, editors.\u00a0Empirical methods for bioethics: a primer. Leeds: Emerald Group Publishing Limited; 2007. pp. 39\u201362."},{"issue":"1","key":"3533_CR50","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1016\/0165-0114(78)90029-5","volume":"1","author":"LA Zadeh","year":"1978","unstructured":"Zadeh LA. Fuzzy sets as a basis for a theory of possibility. Fuzzy Sets Syst. 1978;1(1):3\u201328.","journal-title":"Fuzzy Sets Syst"},{"key":"3533_CR51","volume-title":"The analytic hierarchy process","author":"TL Saaty","year":"1980","unstructured":"Saaty TL. The analytic hierarchy process. New York: McGraw-Hill; 1980."},{"issue":"22","key":"3533_CR52","doi-asserted-by":"publisher","first-page":"6653","DOI":"10.1080\/00207543.2017.1334976","volume":"55","author":"A Emrouznejad","year":"2017","unstructured":"Emrouznejad A, Marra M. The state of the art development of AHP (1979\u20132017): A literature review with a social network analysis. Int J Prod Res. 2017;55(22):6653\u201375.","journal-title":"Int J Prod Res"},{"key":"3533_CR53","doi-asserted-by":"crossref","unstructured":"Zhu K, Wang J, Zhou J, Wang Z, Chen H, Wang Y, Yang L, Ye W, Gong NZ, Zhang Y, Xie X. PromptBench: towards evaluating the robustness of large language models on adversarial prompts.\u00a02023. arXiv preprint arXiv:2306.04528.","DOI":"10.1145\/3689217.3690621"},{"key":"3533_CR54","doi-asserted-by":"crossref","unstructured":"Huovinen, L. Assessing usability of large language models in education. 2024.","DOI":"10.31237\/osf.io\/p54gk"},{"key":"3533_CR55","unstructured":"Yang, Z, Sun, Z, Yue, T.Z, Devanbu, P. and Lo, D, 2024. Robustness, security, privacy, explainability, efficiency, and usability of large language models for code.\u00a0arXiv preprint arXiv:2403.07506."},{"key":"3533_CR56","unstructured":"Wang J, Hu X, Hou W, Chen H, Zheng R, Wang Y, Yang L, Huang H, Ye W, Geng X, Jiao B. On the robustness of chatgpt: an adversarial and out-of-distribution perspective.\u00a02023. arXiv preprint arXiv:2302.12095."},{"key":"3533_CR57","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.101861","volume":"99","author":"J Koco\u0144","year":"2023","unstructured":"Koco\u0144 J, Cichecki I, Kaszyca O, Kochanek M, Szyd\u0142o D, Baran J, Bielaniewicz J, Gruza M, Janz A, Kanclerz K, Koco\u0144 A. ChatGPT: jack of all trades, master of none. Inform Fusion. 2023;99: 101861.","journal-title":"Inform Fusion"},{"key":"3533_CR58","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1016\/j.iotcps.2023.04.003","volume":"3","author":"PP Ray","year":"2023","unstructured":"Ray PP. ChatGPT: a comprehensive review on background, applications, key challenges, bias, ethics, limitations and future scope. Internet of Things Cyber-Phys Syst. 2023;3:121\u201354. https:\/\/doi.org\/10.1016\/j.iotcps.2023.04.003.","journal-title":"Internet of Things Cyber-Phys Syst"},{"key":"3533_CR59","doi-asserted-by":"crossref","unstructured":"Ross A, Chen N, Hang EZ, Glassman EL, Doshi-Velez F. Evaluating the interpretability of generative models by interactive reconstruction. In:\u00a0Proceedings of the 2021 CHI Conference on Human Factors in Computing Systems, 2021;\u00a0pp. 1\u201315.","DOI":"10.1145\/3411764.3445296"},{"key":"3533_CR60","unstructured":"Liu Y, Yao Y, Ton JF, Zhang X, Guo R, Cheng H, Klochkov Y, Taufiq MF, Li H. Trustworthy llms: a survey and guideline for evaluating large language models' alignment.\u00a02023. arXiv preprint arXiv:2308.05374."},{"key":"3533_CR61","doi-asserted-by":"crossref","unstructured":"Ganguli D, Hernandez D, Lovitt L, Askell A, Bai Y, Chen A, Conerly T, Dassarma N, Drain D, Elhage N, El Showk, S. Predictability and surprise in large generative models. In:\u00a0Proceedings of the 2022 ACM Conference on Fairness, Accountability, and Transparency, 2022, pp. 1747\u20131764.","DOI":"10.1145\/3531146.3533229"},{"key":"3533_CR62","doi-asserted-by":"crossref","unstructured":"Johnson, D, Goodman, R, Patrinely, J, Stone, C, Zimmerman, E, Donald, R, Chang, S, Berkowitz, S, Finn, A, Jahangir, E. and Scoville, E, 2023. Assessing the accuracy and reliability of AI-generated medical responses: an evaluation of the Chat-GPT model.\u00a0Research square.","DOI":"10.21203\/rs.3.rs-2566942\/v1"},{"key":"3533_CR63","doi-asserted-by":"crossref","unstructured":"Wang B, Wei C, Liu Z, Lin G, Chen NF. Resilience of Large Language Models for Noisy Instructions.\u00a02024. arXiv preprint arXiv:2404.09754.","DOI":"10.18653\/v1\/2024.findings-emnlp.697"},{"issue":"5","key":"3533_CR64","doi-asserted-by":"publisher","first-page":"842","DOI":"10.3390\/electronics13050842","volume":"13","author":"B Hannon","year":"2024","unstructured":"Hannon B, Kumar Y, Gayle D, Li JJ, Morreale P. Robust testing of AI language model resiliency with novel adversarial prompts. Electronics. 2024;13(5):842.","journal-title":"Electronics"},{"key":"3533_CR65","doi-asserted-by":"crossref","unstructured":"Zhang Z, Shen, G, Tao G, Cheng, S, Zhang, X. On large language models\u2019 resilience to coercive interrogation. In:\u00a02024 IEEE Symposium on Security and Privacy (SP), 2024, pp. 252\u2013252). IEEE Computer Society.","DOI":"10.1109\/SP54263.2024.00208"},{"key":"3533_CR66","doi-asserted-by":"crossref","unstructured":"Agarwal UK, Chan A, Pattabiraman K. Resilience assessment of large language models under transient hardware faults. In:\u00a02023 IEEE 34th International Symposium on Software Reliability Engineering (ISSRE), 2023; p. 659\u2013670. IEEE.","DOI":"10.1109\/ISSRE59848.2023.00052"},{"key":"3533_CR67","doi-asserted-by":"crossref","unstructured":"Wagner N, Ultes S. On the controllability of large language models for dialogue interaction. In:\u00a0Proceedings of the 25th Annual Meeting of the Special Interest Group on Discourse and Dialogue, 2024; p. 216\u2013221.","DOI":"10.18653\/v1\/2024.sigdial-1.19"},{"key":"3533_CR68","unstructured":"Liang, X, Wang, H, Wang, Y, Song, S, Yang, J, Niu, S, Hu, J, Liu, D, Yao, S, Xiong, F, and Li. Controllable text generation for large language models: a survey.\u00a02024. arXiv preprint arXiv:2408.12599."},{"key":"3533_CR69","doi-asserted-by":"crossref","unstructured":"Chen Z, Deng Y, Du W. Trusta: Reasoning about Assurance Cases with Formal Methods and Large Language Models.\u00a02023. arXiv preprint arXiv:2309.12941.","DOI":"10.2139\/ssrn.4956833"},{"key":"3533_CR70","unstructured":"Widyasari R, Lo D, Liao L. Beyond ChatGPT: Enhancing Software Quality Assurance Tasks with Diverse LLMs and Validation Techniques.\u00a02024. arXiv preprint arXiv:2409.01001."},{"key":"3533_CR71","unstructured":"Owen D. How predictable is language model benchmark performance? 2024. arXiv preprint arXiv:2401.04757."},{"key":"3533_CR72","unstructured":"Ruan Y, Maddison CJ, Hashimoto T. Observational Scaling laws and the predictability of language model performance. 2024. arXiv preprint arXiv:2405.10938."},{"key":"3533_CR73","doi-asserted-by":"crossref","unstructured":"Wu T, Terry M, Cai CJ. Ai chains: Transparent and controllable human-ai interaction by chaining large language model prompts. In:\u00a0Proceedings of the 2022 CHI Conference on human factors in computing systems, 2022; pp. 1\u201322.","DOI":"10.1145\/3491102.3517582"},{"issue":"3","key":"3533_CR74","doi-asserted-by":"publisher","first-page":"721","DOI":"10.3350\/cmh.2023.0089","volume":"29","author":"YH Yeo","year":"2023","unstructured":"Yeo YH, Samaan JS, Ng WH, Ting PS, Trivedi H, Vipani A, Ayoub W, Yang JD, Liran O, Spiegel B, Kuo A. Assessing the performance of ChatGPT in answering questions regarding cirrhosis and hepatocellular carcinoma. Clin Mol Hepatol. 2023;29(3):721.","journal-title":"Clin Mol Hepatol"},{"issue":"3","key":"3533_CR75","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3641289","volume":"15","author":"Y Chang","year":"2024","unstructured":"Chang Y, Wang X, Wang J, Wu Y, Yang L, Zhu K, Chen H, Yi X, Wang C, Wang Y, Ye W. A survey on evaluation of large language models. ACM Trans Syst Technol. 2024;15(3):1\u201345.","journal-title":"ACM Trans Syst Technol"},{"issue":"1","key":"3533_CR76","doi-asserted-by":"publisher","first-page":"277","DOI":"10.1017\/S1537592722001116","volume":"21","author":"C Von Soest","year":"2023","unstructured":"Von Soest C. Why do we speak to experts? Reviving the strength of the expert interview method. Perspect Polit. 2023;21(1):277\u201387.","journal-title":"Perspect Polit"},{"key":"3533_CR77","doi-asserted-by":"publisher","DOI":"10.1016\/j.heliyon.2024.e35996","author":"S Bera","year":"2024","unstructured":"Bera S, et al. m-Polar interval-valued fuzzy hypergraphs and its application in decision-making problems. Heliyon. 2024. https:\/\/doi.org\/10.1016\/j.heliyon.2024.e35996.","journal-title":"Heliyon."},{"key":"3533_CR78","doi-asserted-by":"publisher","unstructured":"Kale CP, et al. Deep learning applications to classify cross-topic natural language texts based on their argumentation. In: IET Conference Proceedings, Volume 2023, Issue 39, Pages 40 - 492023, 4th International Conference on Distributed Sensing and Intelligent Systems, ICDSIS 2023, Dubai 21 December 2023 through 23 December 2023.Code 202184. https:\/\/doi.org\/10.1049\/icp.2024.0478.","DOI":"10.1049\/icp.2024.0478"},{"key":"3533_CR79","doi-asserted-by":"publisher","first-page":"10412","DOI":"10.1038\/s41598-024-58741-0","volume":"14","author":"S Satpathy","year":"2024","unstructured":"Satpathy S, Khalaf OI, Shukla DK, et al. Consumer electronics based smart technologies for enhanced terahertz healthcare having an integration of split learning with medical imaging. Sci Rep. 2024;14:10412. https:\/\/doi.org\/10.1038\/s41598-024-58741-0.","journal-title":"Sci Rep"},{"issue":"17","key":"3533_CR80","doi-asserted-by":"publisher","DOI":"10.1016\/j.heliyon.2024.e36773","volume":"10","author":"KN Rao","year":"2024","unstructured":"Rao KN, et al. An efficient brain tumor detection and classification using pre-trained convolutional neural network models. Heliyon. 2024;10(17): e36773. https:\/\/doi.org\/10.1016\/j.heliyon.2024.e36773.","journal-title":"Heliyon."},{"issue":"7","key":"3533_CR81","doi-asserted-by":"publisher","first-page":"19645","DOI":"10.3934\/math.2024958","volume":"9","author":"N Madhusundar","year":"2024","unstructured":"Madhusundar N, Rajendran S, Khalaf OI, Hamam H. Deep-learning-based intelligent neonatal seizure identification using spatial and spectral GNN optimized with the Aquila algorithm. AIMS Math. 2024;9(7):19645\u201369. https:\/\/doi.org\/10.3934\/math.2024958.","journal-title":"AIMS Math"}],"container-title":["SN Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-024-03533-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42979-024-03533-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-024-03533-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,3]],"date-time":"2025-01-03T02:04:37Z","timestamp":1735869877000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42979-024-03533-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,2]]},"references-count":81,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,1]]}},"alternative-id":["3533"],"URL":"https:\/\/doi.org\/10.1007\/s42979-024-03533-6","relation":{},"ISSN":["2661-8907"],"issn-type":[{"value":"2661-8907","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,2]]},"assertion":[{"value":"1 September 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 November 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 January 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The author declares no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}},{"value":"Not Applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Research Involving Human and \/or Animals"}},{"value":"The research was conducted in strict accordance with ethical guidelines for human subjects<b>.<\/b>","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed Consent"}}],"article-number":"57"}}