{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,19]],"date-time":"2025-11-19T18:26:57Z","timestamp":1763576817351,"version":"3.45.0"},"publisher-location":"Gdansk, Poland; Belgrade, Serbia","reference-count":26,"publisher":"University of Gdansk, Department of Business Informatics & University of Belgrade, Faculty of Organizational Sciences","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.62036\/isd.2025.50","type":"proceedings-article","created":{"date-parts":[[2025,11,19]],"date-time":"2025-11-19T00:45:28Z","timestamp":1763513128000},"source":"Crossref","is-referenced-by-count":0,"title":["Expert Versus Metric-Based Evaluation: Testing the Reliability of Evaluation Metrics in Large Language Models Assessment"],"prefix":"10.62036","author":[{"given":"Bartlomiej","family":"Balsamski","sequence":"first","affiliation":[{"name":"Krakow University of Economics, Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jakub","family":"Kanclerz","sequence":"additional","affiliation":[{"name":"Krakow University of Economics, Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dariusz","family":"Put","sequence":"additional","affiliation":[{"name":"Krakow University of Economics, Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Janusz","family":"Stal","sequence":"additional","affiliation":[{"name":"Krakow University of Economics, Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"48375","published-online":{"date-parts":[[2025,11,17]]},"reference":[{"key":"ref0","doi-asserted-by":"publisher","unstructured":"1. Chen, G. H., Chen, S., Liu, Z., Jiang, F., Wang, B.: Humans or LLMs as the Judge? A Study on Judgement Bias. Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pp. 8301-8327 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.474"},{"key":"ref1","doi-asserted-by":"publisher","unstructured":"2. Chen, Y., Wang, R., Jiang, H., Shi, S., Xu, R.: Exploring the Use of Large Language Models for Reference-Free Text Quality Evaluation: An Empirical Study. Findings of the Association for Computational Linguistics: IJCNLP-AACL, pp. 361-374 (2023)","DOI":"10.18653\/v1\/2023.findings-ijcnlp.32"},{"key":"ref2","doi-asserted-by":"publisher","unstructured":"3. Chiang, C.H., Lee, H.: A Closer Look into Using Large Language Models for Automatic Evaluation. Findings of the Association for Computational Linguistics: EMNLP 2023, pp. 8928-8942 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.599"},{"key":"ref3","doi-asserted-by":"publisher","unstructured":"4. Dong, Y.R., Hu, T.M., Collier, N.: Can LLM be a Personalized Judge? Findings of the Association for Computational Linguistics: EMNLP 2024, pp. 10126-10141 (2024)","DOI":"10.18653\/v1\/2024.findings-emnlp.592"},{"key":"ref4","doi-asserted-by":"publisher","unstructured":"5. Fangkai, Y., Pu, Z., Zezhong, W., Lu, W., Jue, Z., Mohit, G., Qingwei, L., Saravan, R., Dongmei, Z.: Empower Large Language Model to Perform Better on Industrial Domain-Specific Question Answering. https:\/\/arxiv.org\/abs\/2305.11541 (2023)","DOI":"10.18653\/v1\/2023.emnlp-industry.29"},{"key":"ref5","unstructured":"6. Gu, J., Jiang, X., Shi, Z., Tan, H., Zhai, X., Xu, C., Li, W., Shen, Y., Ma, S., Liu, H., Wang, S., Zhang, K., Wang, Y., Gao, W., Ni, L., Guo, J.,: A Survey on LLM-as-a-Judge. https:\/\/arxiv.org\/abs\/2411.15594 (2024)"},{"key":"ref6","unstructured":"7. Iriste, A.I., Katane, I.: The Use of Expert Evaluation Method in Social Science Research. Baltic Journal of European Studies, 9(1), pp. 78-97 (2019)"},{"issue":"3","key":"ref7","doi-asserted-by":"publisher","first-page":"100583","DOI":"10.1016\/j.taml.2025.100583","article-title":"DeepSeek vs. ChatGPT vs. Claude: A Comparative Study for Scientific Computing and Scientific Machine Learning Tasks","volume":"15","author":"Jiang","year":"2025","unstructured":"8. Jiang, Q., Gao, Z., Karniadakis, G.E.: DeepSeek vs. ChatGPT vs. Claude: A Comparative Study for Scientific Computing and Scientific Machine Learning Tasks. Theoretical and Applied Mechanics Letters, 15(3), 100583 (2025)","journal-title":"Theoretical and Applied Mechanics Letters"},{"key":"ref8","doi-asserted-by":"publisher","unstructured":"9. Laskar, M.T.R., et al.: A Systematic Survey and Critical Review on Evaluating Large Language Models: Challenges, Limitations, and Recommendations. Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pp. 1378513816 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.764"},{"key":"ref9","unstructured":"10. Li, H., Dong, Q., Chen, J., Su, H., Zhou, Y., Ai, Q., Ye, Z., Liu, Y.: LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods. https:\/\/arxiv.org\/abs\/2412.05579 (2024)"},{"key":"ref10","unstructured":"11. Lin, C.Y.: ROUGE: A Package for Automatic Evaluation of Summaries. Text Summarization Branches Out, Association for Computational Linguistics, pp. 74-81 (2004)"},{"key":"ref11","doi-asserted-by":"publisher","unstructured":"12. Nasirov, R.: The Role of Claude 3.5 Sonet and ChatGPT-4 in Posterior Cervical Fusion Patient Guidance. World Neurosurg, 197:123889 (2025)","DOI":"10.1016\/j.wneu.2025.123889"},{"key":"ref12","unstructured":"13. Ociepa, K., Flis, \u0141., Wr\u00f3bel, K., Gwo\u017adziej, A., Kinas, R.: Bielik 7B v0.1: A Polish Language Model - Development, Insights, and Evaluation. https:\/\/arxiv.org\/abs\/2410.18565 (2024)"},{"key":"ref13","doi-asserted-by":"publisher","unstructured":"14. Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: BLEU: a Method for Automatic Evaluation of Machine Translation. Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics (ACL), pp. 311-318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"ref14","unstructured":"15. Pit, P., Linden, T., Mendoza, A.: Generative Artificial Intelligence in Higher Education: One Year Later. Americas Conference on Information Systems (AMCIS), 11 https:\/\/aisel.aisnet.org\/amcis2024\/is_education\/is_education\/11 (2024)"},{"key":"ref15","unstructured":"16. Poliakova, Y., Novosad, Z.: Application of the Expert Evaluation Method in the Analysis of Trends and Priorities of the Educational Process. Zeszyty Naukowe Wy\u017cszej Szko\u0142y Bankowej w Poznaniu, 95(4), pp. 73-81 (2021)"},{"key":"ref16","doi-asserted-by":"publisher","unstructured":"17. Roumeliotis, K.I., Tselikas, N.D., Nasiopoulos, D.K.: LLMs in e-commerce: A Comparative Analysis of GPT and LLaMA Models in Product Review Evaluation. Natural Language Processing Journal, 6(1):100056 (2024)","DOI":"10.1016\/j.nlp.2024.100056"},{"key":"ref17","unstructured":"18. Schroeder, K., Wood-Doughty, Z.: Can You Trust LLM Judgments? Reliability of LLMas-a-Judge. https:\/\/arxiv.org\/abs\/2412.12509v2 (2024)"},{"key":"ref18","doi-asserted-by":"publisher","unstructured":"19. Sharma, B., Ghawaly, J., McCleary, K., Webb, A.M., Baggili, I.: ForensicLLM: A Local Large Language Model for Digital Forensics. Forensic Science International: Digital Investigation, 52:301872 (2025)","DOI":"10.1016\/j.fsidi.2025.301872"},{"key":"ref19","doi-asserted-by":"publisher","unstructured":"20. Singh, J., Joesph, M.H., Jabbar, K.A.: Rule-based Chatbot for Student Enquiries. Journal of Physics: Conference Series, 1228(1):012060. IOP Publishing (2019)","DOI":"10.1088\/1742-6596\/1228\/1\/012060"},{"key":"ref20","doi-asserted-by":"publisher","unstructured":"21. Slimani, T.: Description and Evaluation of Semantic Similarity Measures Approaches, International Journal of Computer Applications, 80(10), pp. 25-33 (2013)","DOI":"10.5120\/13897-1851"},{"key":"ref21","doi-asserted-by":"publisher","unstructured":"22. Tan, Y., Min, D., Li, Y., Li, W., Xue, N., Chen, Y., Qi, G.: Can ChatGPT Replace Traditional KBQA Models? An In-depth Analysis of the Question Answering Performance of the GPT LLM Family. The Semantic Web - ISWC 2023, pp. 348-367 (2023)","DOI":"10.1007\/978-3-031-47240-4_19"},{"key":"ref22","unstructured":"23. van Schaik, T.A.: A List of Metrics for Evaluating LLM-generated Content. https:\/\/learn.microsoft.com\/en-us\/ai\/playbook\/technology-guidance\/generativeai\/working-with-llms\/evaluation\/list-of-eval-metrics (2024)"},{"key":"ref23","doi-asserted-by":"publisher","unstructured":"24. von Soest, C.: Why Do We Speak to Experts? Reviving the Strength of the Expert Interview Method. Perspectives on Politics 21(1), pp. 1-11 (2022)","DOI":"10.1017\/S1537592722001116"},{"key":"ref24","unstructured":"25. Zhang, T., Kishore, V., Wu, F., Weinberger, K.Q., Artzi, T.: BERTScore: Evaluating Text Generation with BERT. International Conference on Learning Representations (ICLR) (2020)"},{"key":"ref25","unstructured":"26. Zheng, L., et al.: Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena. 37th Conference on Neural Information Processing Systems (NeurIPS) (2023)"}],"event":{"name":"33rd International Conference on Information Systems Development","start":{"date-parts":[[2025,9,3]]},"location":"Belgrade, Serbia","end":{"date-parts":[[2025,9,5]]},"acronym":"ISD 2025"},"container-title":["International Conference on Information Systems Development","Proceedings of the 33rd International Conference on Information Systems Development"],"original-title":[],"link":[{"URL":"https:\/\/aisel.aisnet.org\/cgi\/viewcontent.cgi?article=1729&amp;context=isd2014&amp;unstamped=1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,19]],"date-time":"2025-11-19T18:15:53Z","timestamp":1763576153000},"score":1,"resource":{"primary":{"URL":"https:\/\/aisel.aisnet.org\/isd2014\/proceedings2025\/datascience\/20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,17]]},"references-count":26,"URL":"https:\/\/doi.org\/10.62036\/isd.2025.50","relation":{},"ISSN":["2938-5202"],"issn-type":[{"type":"print","value":"2938-5202"}],"subject":[],"published":{"date-parts":[[2025,11,17]]}}}