{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T15:58:08Z","timestamp":1784390288622,"version":"3.55.0"},"reference-count":39,"publisher":"Polish Information Processing Society","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.15439\/2025f4862","type":"proceedings-article","created":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T07:44:23Z","timestamp":1761119063000},"page":"451-460","source":"Crossref","is-referenced-by-count":1,"title":["AraXLM: Evaluating Arabic Diacritization Tools for Cross-Language Plagiarism Detection"],"prefix":"10.15439","volume":"43","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4193-6230","authenticated-orcid":true,"given":"Mona","family":"Alshehri","sequence":"first","affiliation":[{"name":"Dept. of Informatics, University of Sussex"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8872-7786","authenticated-orcid":true,"given":"Natalia","family":"Beloff","sequence":"additional","affiliation":[{"name":"Dept. of Informatics, University of Sussex"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8686-2274","authenticated-orcid":true,"given":"Martin","family":"White","sequence":"additional","affiliation":[{"name":"Dept. of Informatics, University of Sussex"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"6175","published-online":{"date-parts":[[2025,10,15]]},"reference":[{"key":"ref1","doi-asserted-by":"publisher","unstructured":"M. Elyaakoubi and A. Lazrek, \u201cJustify just or just justify,\u201d Journal of Electronic Publishing, vol. 13, no. 1, 2010. https:\/\/dx.doi.org\/10.3998\/3336451.0013.105.","DOI":"10.3998\/3336451.0013.105"},{"key":"ref2","unstructured":"R. Rjeily, Cultural Connectives: Bridging the Latin and Arabic Alphabets, vol. 1. Brooklyn, NY: Mark Batty Publisher, 2021."},{"key":"ref3","unstructured":"M. Hssini and A. Lazrek, \u201cDesign of Arabic Diacritical Marks,\u201d International Journal of Computer Science, vol. 8, no. 3, May 2011."},{"key":"ref4","unstructured":"M. Maamouri, A. Bies, and S. Kulick,  'Diacritization: A Challenge to Arabic Treebank Annotation and Parsing\u2019, the International Conference on the Challenge of Arabic for NLP\/MT , pp. 35-47, 2006."},{"key":"ref5","unstructured":"S. Alzahrani, \u201cArabic plagiarism detection using word correlation in N-Grams with K-overlapping approach,\u201d Taif, 2015. https:\/\/ceur-ws.org\/Vol-1587\/T5-2.pdf"},{"key":"ref6","doi-asserted-by":"publisher","unstructured":"E. M. B. Nagoudi et al., \u201c2L-APD: A two-level plagiarism detection system for Arabic documents,\u201d Cybernetics and Information Technologies, vol. 18, no. 1, pp. 124\u2013138, 2018. https:\/\/dx.doi.org\/10.2478\/cait-2018-0011.","DOI":"10.2478\/cait-2018-0011"},{"key":"ref7","unstructured":"B. Akanksha et al., \u201cA survey on plagiarism detection,\u201d International Journal of Computer Applications, vol. 10, no. 8, pp. 2359\u20132365, 2017.  http:\/\/www.ripublication.com"},{"key":"ref8","doi-asserted-by":"publisher","unstructured":"M. F. Akan et al., \u201cAn analysis of Arabic-English translation: Problems and prospects,\u201d Advances in Language and Literary Studies, vol. 10, no. 1, p. 58, Feb. 2019. https:\/\/dx.doi.org\/10.7575\/aiac.alls.v.10n.1p.58.","DOI":"10.7575\/aiac.alls.v.10n.1p.58"},{"key":"ref9","doi-asserted-by":"publisher","unstructured":"M. Alshehri, N. Beloff, and M. White, \u201cAraXLM: New XLM-RoBERTa based method for plagiarism detection in Arabic text,\u201d in Intelligent Computing, K. Arai, Ed., Cham: Springer, 2024, pp. 81\u201396. https:\/\/dx.doi.org\/10.1007\/978-3-031-62277-9_6","DOI":"10.1007\/978-3-031-62277-9_6"},{"key":"ref10","unstructured":"M. M. Elmallah et al., \u201cArabic diacritization using morphologically informed character-level model,\u201d in Proc. LREC-COLING 2024, Torino, Italy: ELRA and ICCL, May 2024, pp. 1446\u20131454. https:\/\/aclanthology.org\/2024.lrec-main.128\/"},{"key":"ref11","unstructured":"K. Shaalan and Khaled, \u201cRule-based approach in Arabic natural language processing,\u201d International Journal on Information and Communication Technologies, vol. 3, p. 11, May 2010."},{"key":"ref12","doi-asserted-by":"publisher","unstructured":"A. Chennoufi and A. Mazroui, \u201cMorphological, syntactic and diacritics rules for automatic diacritization of Arabic sentences,\u201d Journal of King Saud University - Computer and Information Sciences, vol. 29, no. 2, pp. 156\u2013163, 2017. https:\/\/dx.doi.org\/10.1016\/j.jksuci.2016.06.004.","DOI":"10.1016\/j.jksuci.2016.06.004"},{"key":"ref13","doi-asserted-by":"crossref","unstructured":"M. M\u00e9zard and J.-P. Nadal, \u201cLearning in feedforward layered networks: the tiling algorithm,\u201d Journal of Physics A, vol. 22, pp. 2191\u20132203, 1989. https:\/\/api.semanticscholar.org\/CorpusID:44826720","DOI":"10.1088\/0305-4470\/22\/12\/019"},{"key":"ref14","doi-asserted-by":"publisher","unstructured":"W. Almanaseer et al., \u201cA deep belief network classification approach for automatic diacritization of Arabic text,\u201d Applied Sciences, vol. 11, no. 11, 2021. https:\/\/dx.doi.org\/10.3390\/app11115228.","DOI":"10.3390\/app11115228"},{"key":"ref15","unstructured":"N. Srivastava et al., \u201cDropout: A simple way to prevent neural networks from overfitting,\u201d Journal of Machine Learning Research, vol. 15, no. 1, pp. 1929\u20131958, Jan. 2014."},{"key":"ref16","doi-asserted-by":"publisher","unstructured":"Y. Alalawi et al., \u201cA CNN-based Arabic diacritic symbol recognition system using domain adaptation,\u201d in Proc. 8th Int. Conf. Sustainable Information Engineering and Technology (SIET), New York, USA: ACM, 2023, pp. 23\u201332. https:\/\/dx.doi.org\/10.1145\/3626641.3627212.","DOI":"10.1145\/3626641.3627212"},{"key":"ref17","doi-asserted-by":"publisher","unstructured":"H. Hewamalage et al., \u201cRecurrent neural networks for time series forecasting: Current status and future directions,\u201d International Journal of Forecasting, vol. 37, no. 1, pp. 388\u2013427, 2021. https:\/\/dx.doi.org\/10.1016\/j.ijforecast.2020.06.008.","DOI":"10.1016\/j.ijforecast.2020.06.008"},{"key":"ref18","doi-asserted-by":"publisher","unstructured":"Y. Belinkov and J. Glass, \u201cArabic diacritization with recurrent neural networks,\u201d in Proc. EMNLP 2015, Lisbon, Portugal: ACL, Sep. 2015, pp. 2281\u20132285. https:\/\/dx.doi.org\/10.18653\/v1\/D15-1274.","DOI":"10.18653\/v1\/D15-1274"},{"key":"ref19","unstructured":"A. Vaswani et al., \u201cAttention is all you need,\u201d in Proc. NeurIPS 2017, 2017. http:\/\/arxiv.org\/abs\/1706.03762"},{"key":"ref20","doi-asserted-by":"publisher","unstructured":"A. Assad et al., \u201cTransformer-based automatic Arabic text diacritization,\u201d Sustainable Engineering and Innovation, vol. 6, no. 2, pp. 285\u2013296, Nov. 2024. https:\/\/dx.doi.org\/10.37868\/sei.v6i2.id305.","DOI":"10.37868\/sei.v6i2.id305"},{"key":"ref21","doi-asserted-by":"publisher","unstructured":"Y. Bengio et al., \u201cLearning long-term dependencies with gradient descent is difficult,\u201d IEEE Transactions on Neural Networks, vol. 5, no. 2, pp. 157\u2013166, Mar. 1994. https:\/\/dx.doi.org\/10.1109\/72.279181.","DOI":"10.1109\/72.279181"},{"key":"ref22","doi-asserted-by":"publisher","unstructured":"A. Gillioz, J. Casas, E. Mugellini and O. Abou Khaled, \u201cOverview of the Transformer-based Models for NLP Tasks,\u201d Proceedings of the Federated Conference on Computer Science and Information Systems, vol. 21, pp. 179\u2013183, 2020, https:\/\/dx.doi.org\/10.15439\/2020F20.","DOI":"10.15439\/2020F20"},{"key":"ref23","doi-asserted-by":"publisher","unstructured":"R. Al-Sabri and J. Gao, \u201cLAMAD: A linguistic attentional model for Arabic text diacritization,\u201d in Findings of the Association for Computational Linguistics: EMNLP 2021, Punta Cana, Dominican Republic: ACL, Nov. 2021, pp. 3757\u20133764. https:\/\/dx.doi.org\/10.18653\/v1\/2021.findings-emnlp.317.","DOI":"10.18653\/v1\/2021.findings-emnlp.317"},{"key":"ref24","doi-asserted-by":"publisher","unstructured":"M. Al-Badrashiny et al., \u201cA layered language model based hybrid approach to automatic full diacritization of Arabic,\u201d in Proc. 3rd Arabic NLP Workshop, Valencia, Spain: ACL, Apr. 2017, pp. 177\u2013184. https:\/\/dx.doi.org\/10.18653\/v1\/W17-1321.","DOI":"10.18653\/v1\/W17-1321"},{"key":"ref25","doi-asserted-by":"publisher","unstructured":"H. Alaqel and K. El Hindi, \u201cImproving diacritical Arabic speech recognition: Transformer-based models with transfer learning and hybrid data augmentation,\u201d Information, vol. 16, no. 3, 2025. https:\/\/dx.doi.org\/10.3390\/info16030161.","DOI":"10.3390\/info16030161"},{"key":"ref26","unstructured":"O. Obeid et al., \u201cCAMeL Tools: An open source Python toolkit for Arabic NLP,\u201d 2020. http:\/\/qatsdemo.cloudapp.net\/farasa\/"},{"key":"ref27","doi-asserted-by":"publisher","unstructured":"A. Abdelali et al., \u201cFarasa: A fast and furious segmenter for Arabic,\u201d in Proc. NAACL Demonstrations, San Diego, CA: ACL, Jun. 2016, pp. 11\u201316. https:\/\/dx.doi.org\/10.18653\/v1\/N16-3003.","DOI":"10.18653\/v1\/N16-3003"},{"key":"ref28","doi-asserted-by":"crossref","unstructured":"F. Alasmary et al., \u201cCATT: Character-based Arabic Tashkeel Transformer,\u201d arXiv preprint, vol. abs\/2407.03236, 2024. https:\/\/api.semanticscholar.org\/CorpusID:270924323","DOI":"10.18653\/v1\/2024.arabicnlp-1.21"},{"key":"ref29","doi-asserted-by":"crossref","unstructured":"B. Al-Rfooh et al., \u201cFine-Tashkeel: Fine-tuning byte-level models for accurate Arabic text diacritization,\u201d in Proc. IEEE JEEIT 2023, pp. 199\u2013204, 2023.  https:\/\/api.semanticscholar.org\/CorpusID:257767345","DOI":"10.1109\/JEEIT58638.2023.10185725"},{"key":"ref30","doi-asserted-by":"publisher","unstructured":"B. M. King, \u201cAnalysis of variance,\u201d in International Encyclopedia of Education, 3rd ed., Jan. 2009, pp. 32\u201336. https:\/\/dx.doi.org\/10.1016\/B978-0-08-044894-7.01306-3.","DOI":"10.1016\/B978-0-08-044894-7.01306-3"},{"key":"ref31","unstructured":"A. Lazrek, \u201cArabic mathematical notation,\u201d National Institute of Standards and Technology, USA, 2006.  https:\/\/www.w3.org\/TR\/2006\/NOTE-arabic-math-20060131\/"},{"key":"ref32","doi-asserted-by":"publisher","unstructured":"K. Darwish et al., \u201cArabic diacritization: Stats, rules, and hacks,\u201d in Proc. 3rd Arabic NLP Workshop, Valencia, Spain: ACL, Apr. 2017, pp. 9\u201317. https:\/\/dx.doi.org\/10.18653\/v1\/W17-1302.","DOI":"10.18653\/v1\/W17-1302"},{"key":"ref33","doi-asserted-by":"publisher","unstructured":"A. Fadel et al., \u201cNeural Arabic text diacritization: State of the art results and a novel approach for Arabic NLP downstream tasks,\u201d ACM Transactions on Asian and Low-Resource Language Information Processing, vol. 21, no. 1, Jan. 2022. https:\/\/dx.doi.org\/10.1145\/3470849.","DOI":"10.1145\/3470849"},{"key":"ref34","unstructured":"O. Obeid et al., \u201cCAMeL Tools: An open source Python toolkit for Arabic NLP,\u201d in Proc. LREC 2020, Marseille, France: ELRA, May 2020, pp. 7022\u20137032. https:\/\/aclanthology.org\/2020.lrec-1.868\/"},{"key":"ref35","doi-asserted-by":"publisher","unstructured":"T. Zerrouki, \u201cTowards an open platform for Arabic language processing,\u201d 2020. https:\/\/dx.doi.org\/10.13140\/RG.2.2.29882.82881.","DOI":"10.13140\/RG.2.2.29882.82881"},{"key":"ref36","doi-asserted-by":"publisher","unstructured":"A. Fadel et al., \u201cNeural Arabic text diacritization: State of the art results and a novel approach for machine translation,\u201d in Proc. 6th Workshop on Asian Translation, Hong Kong, China: ACL, Nov. 2019, pp. 215\u2013225. https:\/\/dx.doi.org\/10.18653\/v1\/D19-5229.","DOI":"10.18653\/v1\/D19-5229"},{"key":"ref37","doi-asserted-by":"crossref","unstructured":"D. Cer, M. Diab, E. Agirre, I. Lopez-Gazpio, and L. Specia, \u201cSemEval-2017 Task 1: Semantic Textual Similarity \u2013 Multilingual and Cross-lingual Focused Evaluation,\u201d in Proceedings of the 11th International Workshop on Semantic Evaluation (SemEval-2017), Vancouver, Canada, Aug. 2017, pp. 1\u201314.","DOI":"10.18653\/v1\/S17-2001"},{"key":"ref38","unstructured":"Y. Tian, Y. Song, H. Xia, Y. Li, and Q. Zhang, \u201cECNU at SemEval-2017 Task 1: Leverage Kernel-Based Traditional NLP Features and Distributed Word Representations for Semantic Textual Similarity Estimation,\u201d in Proc. 11th Int. Workshop on Semantic Evaluation (SemEval-2017), Vancouver, Canada, Aug. 2017, pp. 125\u2013131. https:\/\/aclanthology.org\/S17-2015"},{"key":"ref39","doi-asserted-by":"crossref","unstructured":"H. Wu, H. Huang, P. Jian, Y. Guo, and C. Su, \u201cBIT at SemEval-2017 Task 1: Using Semantic Information Space to Evaluate Semantic Textual Similarity,\u201d in Proc. 11th Int. Workshop on Semantic Evaluation (SemEval-2017), Vancouver, Canada, Aug. 2017, pp. 77\u201384.\n https:\/\/aclanthology.org\/S17-2007","DOI":"10.18653\/v1\/S17-2007"}],"event":{"name":"20th Conference on Computer Science and Intelligence Systems (FedCSIS)","theme":"Computer Science and Intelligence Systems","location":"Krak\u00f3w, Poland","acronym":"FedCSIS","number":"20","start":{"date-parts":[[2025,9,14]]},"end":{"date-parts":[[2025,9,17]]}},"container-title":["Annals of Computer Science and Information Systems","Proceedings of the 20th Conference on Computer Science and Intelligence Systems (FedCSIS)"],"original-title":[],"deposited":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T07:48:26Z","timestamp":1761119306000},"score":1,"resource":{"primary":{"URL":"https:\/\/annals-csis.org\/Volume_43\/drp\/4862.html"}},"subtitle":[],"proceedings-subject":"Computer Science and Information Systems","short-title":[],"issued":{"date-parts":[[2025,10,15]]},"references-count":39,"URL":"https:\/\/doi.org\/10.15439\/2025f4862","relation":{},"ISSN":["2300-5963"],"issn-type":[{"value":"2300-5963","type":"print"}],"subject":[],"published":{"date-parts":[[2025,10,15]]}}}