{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T16:15:30Z","timestamp":1783527330660,"version":"3.55.0"},"reference-count":61,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T00:00:00Z","timestamp":1753315200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100000883","name":"University of Bristol","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100000883","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computers and Education: Artificial Intelligence"],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1016\/j.caeai.2025.100449","type":"journal-article","created":{"date-parts":[[2025,7,30]],"date-time":"2025-07-30T16:01:06Z","timestamp":1753891266000},"page":"100449","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":4,"special_numbering":"C","title":["How well can LLMs grade essays in Arabic?"],"prefix":"10.1016","volume":"9","author":[{"given":"Rayed","family":"Ghazawi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Edwin","family":"Simpson","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.caeai.2025.100449_br0010","author":"Achiam"},{"key":"10.1016\/j.caeai.2025.100449_br0020","series-title":"Proceedings of the 37th IEEE\/ACM international conference on automated software engineering","first-page":"1","article-title":"Few-shot training LLMs for project-specific code-summarization","author":"Ahmed","year":"2022"},{"key":"10.1016\/j.caeai.2025.100449_br0030","doi-asserted-by":"crossref","first-page":"45","DOI":"10.20943\/01201605.4550","article-title":"An automated system for essay scoring of online exams in Arabic based on stemming techniques and Levenshtein edit operations","volume":"13","author":"Al-Shalabi","year":"2016","journal-title":"International Journal of Computer Science Issues"},{"key":"10.1016\/j.caeai.2025.100449_br0040","article-title":"AEGD: Arabic essay grading dataset for machine learning","author":"Al-Shargabi","year":"2021","journal-title":"Journal of Theoretical and Applied Information Technology"},{"key":"10.1016\/j.caeai.2025.100449_br0050","doi-asserted-by":"crossref","first-page":"103","DOI":"10.3233\/AIC-130586","article-title":"A hybrid automatic scoring system for Arabic essays","volume":"27","author":"Alghamdi","year":"2014","journal-title":"AI Communications"},{"key":"10.1016\/j.caeai.2025.100449_br0060","series-title":"2019 IEEE international symposium on signal processing and information technology (ISSPIT)","article-title":"Automatic evaluation for Arabic essays: A rule-based system","author":"Alqahtani","year":"2019"},{"key":"10.1016\/j.caeai.2025.100449_br0070","series-title":"Proceedings of the 17th international conference on natural language processing (ICON)","first-page":"181","article-title":"Automated Arabic essay evaluation","author":"Alqahtani","year":"2020"},{"key":"10.1016\/j.caeai.2025.100449_br0080","series-title":"Proceedings of the thirteenth language resources and evaluation conference","first-page":"6340","article-title":"Masader: Metadata sourcing for Arabic text and speech data resources","author":"Alyafeai","year":"2022"},{"key":"10.1016\/j.caeai.2025.100449_br0090","doi-asserted-by":"crossref","first-page":"764","DOI":"10.3390\/electronics13040764","article-title":"Prediction of Arabic legal rulings using large language models","volume":"13","author":"Ammar","year":"2024","journal-title":"Electronics"},{"key":"10.1016\/j.caeai.2025.100449_br0100","author":"Bari"},{"key":"10.1016\/j.caeai.2025.100449_br0110","series-title":"Proceedings of the 19th workshop on innovative use of NLP for building educational applications (BEA 2024)","first-page":"309","article-title":"LLMs in short answer scoring: Limitations and promise of zero-shot and few-shot approaches","author":"Chamieh","year":"2024"},{"key":"10.1016\/j.caeai.2025.100449_br0120","series-title":"Automated essay scoring by maximizing human-machine agreement","first-page":"1741","author":"Chen","year":"2013"},{"key":"10.1016\/j.caeai.2025.100449_br0130","series-title":"2018 international conference on Asian language processing (IALP)","first-page":"378","article-title":"Relevance-based automated essay scoring via hierarchical recurrent model","author":"Chen","year":"2018"},{"key":"10.1016\/j.caeai.2025.100449_br0140","doi-asserted-by":"crossref","first-page":"213","DOI":"10.1037\/h0026256","article-title":"Weighted kappa: Nominal scale agreement provision for scaled disagreement or partial credit","volume":"70","author":"Cohen","year":"1968","journal-title":"Psychological Bulletin"},{"key":"10.1016\/j.caeai.2025.100449_br0150","series-title":"Proceedings of the second Arabic natural language processing conference","first-page":"729","article-title":"Arabic train at NADI 2024 shared task: LLMs' ability to translate Arabic dialects into Modern Standard Arabic","author":"Demidova","year":"2024"},{"key":"10.1016\/j.caeai.2025.100449_br0160","series-title":"16th international conference on educational data mining, EDM 2023","first-page":"103","article-title":"Evaluating quadratic weighted kappa as the standard performance metric for automated essay scoring","author":"Doewes","year":"2023"},{"key":"10.1016\/j.caeai.2025.100449_br0170","series-title":"Proceedings of the 21st conference on computational natural language learning (CoNLL 2017)","first-page":"153","article-title":"Attention-based recurrent convolutional neural network for automatic essay scoring","author":"Dong","year":"2017"},{"key":"10.1016\/j.caeai.2025.100449_br0180","author":"Edwards"},{"key":"10.1016\/j.caeai.2025.100449_br0190","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/1644879.1644881","article-title":"Arabic natural language processing: Challenges and solutions","volume":"8","author":"Farghaly","year":"2009","journal-title":"ACM Transactions on Asian Language Information Processing (TALIP)"},{"key":"10.1016\/j.caeai.2025.100449_br0200","author":"Franci"},{"key":"10.1016\/j.caeai.2025.100449_br0210","doi-asserted-by":"crossref","first-page":"1165","DOI":"10.1007\/s10639-020-10300-6","article-title":"Automated students Arabic essay scoring using trained neural network by E-Jaya optimization to support personalized system of instruction","volume":"26","author":"Gaheen","year":"2021","journal-title":"Education and Information Technologies"},{"key":"10.1016\/j.caeai.2025.100449_br0220","author":"Gema"},{"key":"10.1016\/j.caeai.2025.100449_br0230","author":"Geng"},{"key":"10.1016\/j.caeai.2025.100449_br0240","author":"Ghazawi"},{"key":"10.1016\/j.caeai.2025.100449_br0250","doi-asserted-by":"crossref","unstructured":"Ghosn, Y., El Sardouk, O., Jabbour, Y., Jrad, M., Hussein Kamareddine, M., Abbas, N., Saade, C., & Abi Ghanem, A. (2023). ChatGPT 4 versus ChatGPT 3.5 on the final FRCR part a sample questions. Assessing performance and accuracy of explanations. medRxiv, 2023\u201309.","DOI":"10.1101\/2023.09.06.23295144"},{"key":"10.1016\/j.caeai.2025.100449_br0260","series-title":"Proceedings of the eleventh ACM conference on learning@ scale","first-page":"300","article-title":"Can large language models make the grade? An empirical study evaluating LLMs' ability to mark short answer questions in k-12 education","author":"Henkel","year":"2024"},{"key":"10.1016\/j.caeai.2025.100449_br0270","first-page":"1","article-title":"Can llms grade open response reading comprehension questions? An empirical study using the roars dataset","author":"Henkel","year":"2024","journal-title":"International Journal of Artificial Intelligence in Education"},{"key":"10.1016\/j.caeai.2025.100449_br0280","first-page":"3","article-title":"Lora: Low-rank adaptation of large language models","volume":"1","author":"Hu","year":"2022","journal-title":"International Conference on Learning Representations"},{"key":"10.1016\/j.caeai.2025.100449_br0290","series-title":"Human language technologies (volume 1: Long papers)","first-page":"8139","article-title":"AceGPT, localizing large language models in Arabic","author":"Huang","year":"2024"},{"key":"10.1016\/j.caeai.2025.100449_br0300","doi-asserted-by":"crossref","DOI":"10.1016\/j.lindif.2023.102274","article-title":"ChatGPT for good? On opportunities and challenges of large language models for education","volume":"103","author":"Kasneci","year":"2023","journal-title":"Learning and Individual Differences"},{"key":"10.1016\/j.caeai.2025.100449_br0310","author":"Katuka"},{"key":"10.1016\/j.caeai.2025.100449_br0320","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2023.101861","article-title":"ChatGPT: Jack of all trades, master of none","volume":"99","author":"Koco\u0144","year":"2023","journal-title":"Information Fusion"},{"key":"10.1016\/j.caeai.2025.100449_br0330","author":"Koto"},{"key":"10.1016\/j.caeai.2025.100449_br0340","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1007\/s40547-024-00143-4","article-title":"Sentiment analysis in the age of generative AI","volume":"11","author":"Krugmann","year":"2024","journal-title":"Customer Needs and Solutions"},{"key":"10.1016\/j.caeai.2025.100449_br0350","series-title":"Proceedings of ArabicNLP 2023","first-page":"101","article-title":"Beyond English: Evaluating LLMs for Arabic grammatical error correction","author":"Kwon","year":"2023"},{"key":"10.1016\/j.caeai.2025.100449_br0360","article-title":"Fine-tuning ChatGPT for automatic scoring","volume":"6","author":"Latif","year":"2024","journal-title":"Computers and Education: Artificial Intelligence"},{"key":"10.1016\/j.caeai.2025.100449_br0370","doi-asserted-by":"crossref","DOI":"10.1111\/exsy.13068","article-title":"Enhanced hybrid neural network for automated essay scoring","volume":"39","author":"Li","year":"2022","journal-title":"Expert Systems"},{"key":"10.1016\/j.caeai.2025.100449_br0380","author":"Li"},{"key":"10.1016\/j.caeai.2025.100449_br0390","doi-asserted-by":"crossref","DOI":"10.1016\/j.metrad.2023.100017","article-title":"Summary of ChatGPT-related research and perspective towards the future of large language models","volume":"1","author":"Liu","year":"2023","journal-title":"Meta-Radiology"},{"key":"10.1016\/j.caeai.2025.100449_br0400","series-title":"International conference on intelligent systems design and applications","first-page":"1164","article-title":"Arabic automatic essay scoring systems: An overview study","author":"Machhout","year":"2021"},{"key":"10.1016\/j.caeai.2025.100449_br0410","series-title":"Proceedings of the 2024 joint international conference on computational linguistics, language resources and evaluation (LREC-COLING 2024)","first-page":"2777","article-title":"Can large language models automatically score proficiency of written essays?","author":"Mansour","year":"2024"},{"key":"10.1016\/j.caeai.2025.100449_br0420","series-title":"Proceedings of the eleventh international conference on language resources and evaluation (LREC 2018)","first-page":"33","article-title":"ASAP++: Enriching the ASAP automated essay grading dataset with essay attribute scores","author":"Mathias","year":"2018"},{"key":"10.1016\/j.caeai.2025.100449_br0430","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"15991","article-title":"Crosslingual generalization through multitask finetuning","author":"Muennighoff","year":"2023"},{"key":"10.1016\/j.caeai.2025.100449_br0440","first-page":"238","article-title":"The imminence of... grading essays by computer","volume":"47","author":"Page","year":"1966","journal-title":"Phi Delta Kappan"},{"key":"10.1016\/j.caeai.2025.100449_br0450","doi-asserted-by":"crossref","first-page":"2074","DOI":"10.3390\/app14052074","article-title":"A review of current trends, techniques, and challenges in large language models (LLMs)","volume":"14","author":"Patil","year":"2024","journal-title":"Applied Sciences"},{"key":"10.1016\/j.caeai.2025.100449_br0460","article-title":"Language model tokenizers introduce unfairness between languages","volume":"36","author":"Petrov","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.caeai.2025.100449_br0470","series-title":"Proceedings of the 2015 conference on empirical methods in natural language processing","first-page":"431","article-title":"Flexible domain adaptation for automated essay scoring using correlated linear regression","author":"Phandi","year":"2015"},{"key":"10.1016\/j.caeai.2025.100449_br0480","doi-asserted-by":"crossref","first-page":"2495","DOI":"10.1007\/s10462-021-10068-2","article-title":"An automated essay scoring systems: A systematic literature review","volume":"55","author":"Ramesh","year":"2022","journal-title":"Artificial Intelligence Review"},{"key":"10.1016\/j.caeai.2025.100449_br0490","article-title":"The role of ChatGPT in higher education: Benefits, challenges, and future research directions","volume":"6","author":"Rasul","year":"2023","journal-title":"Journal of Applied Learning and Teaching"},{"key":"10.1016\/j.caeai.2025.100449_br0500","author":"Sengupta"},{"key":"10.1016\/j.caeai.2025.100449_br0510","series-title":"Computational linguistics, speech and image processing for Arabic language","first-page":"59","article-title":"Challenges in Arabic natural language processing","author":"Shaalan","year":"2019"},{"key":"10.1016\/j.caeai.2025.100449_br0520","series-title":"Proceedings of the 16th international conference on computational processing of Portuguese","first-page":"33","article-title":"A new benchmark for automatic essay scoring in Portuguese","author":"Silveira","year":"2024"},{"key":"10.1016\/j.caeai.2025.100449_br0530","series-title":"2023 20th ACS\/IEEE international conference on computer systems and applications (AICCSA)","first-page":"1","article-title":"Fine tuning of large language models for Arabic language","author":"Tamer","year":"2023"},{"key":"10.1016\/j.caeai.2025.100449_br0540","author":"Touvron"},{"key":"10.1016\/j.caeai.2025.100449_br0550","author":"Touvron"},{"key":"10.1016\/j.caeai.2025.100449_br0560","author":"\u00dcst\u00fcn"},{"key":"10.1016\/j.caeai.2025.100449_br0570","author":"Wang"},{"key":"10.1016\/j.caeai.2025.100449_br0580","author":"Xiao"},{"key":"10.1016\/j.caeai.2025.100449_br0600","series-title":"Findings of the association for computational linguistics: EMNLP 2020","first-page":"1560","article-title":"Enhancing automated essay scoring performance via fine-tuning pre-trained language models with combination of regression and ranking","author":"Yang","year":"2020"},{"key":"10.1016\/j.caeai.2025.100449_br0590","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"22466","article-title":"Unveiling the tapestry of automated essay scoring: A comprehensive investigation of accuracy, fairness, and generalizability","author":"Yang","year":"2024"},{"key":"10.1016\/j.caeai.2025.100449_br0610","series-title":"Proceedings of the 49th annual meeting of the association for computational linguistics: Human language technologies","first-page":"180","article-title":"A new dataset and method for automatically grading ESOL texts","author":"Yannakoudakis","year":"2011"}],"container-title":["Computers and Education: Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S2666920X2500089X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S2666920X2500089X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T07:12:21Z","timestamp":1777273941000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S2666920X2500089X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12]]},"references-count":61,"alternative-id":["S2666920X2500089X"],"URL":"https:\/\/doi.org\/10.1016\/j.caeai.2025.100449","relation":{},"ISSN":["2666-920X"],"issn-type":[{"value":"2666-920X","type":"print"}],"subject":[],"published":{"date-parts":[[2025,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"How well can LLMs grade essays in Arabic?","name":"articletitle","label":"Article Title"},{"value":"Computers and Education: Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.caeai.2025.100449","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2025 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"100449"}}