{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,17]],"date-time":"2025-11-17T12:24:05Z","timestamp":1763382245491,"version":"3.45.0"},"reference-count":75,"publisher":"Association for Computing Machinery (ACM)","issue":"11","funder":[{"name":"Deanship of Scientific Research at King Saud University"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["ACM Trans. Asian Low-Resour. Lang. Inf. Process."],"published-print":{"date-parts":[[2025,11,30]]},"abstract":"<jats:p>\n                    This article presents an approach to readability estimation that focuses on conceptual rather than linguistic complexity, using the extensive SaudiTextBooks textbooks. We introduce\n                    <jats:bold>DARES 2.0<\/jats:bold>\n                    , an enhanced concept-based readability training dataset designed to estimate the readability of Saudi educational texts. Building on DARES 1.0, DARES 2.0 extends the scope of conceptual complexity by replacing repetitive concepts and manually revising the input features with unique terms and their surrounding contexts from the SaudiTextBooks, spanning grades 1 to 12. The refined DARES 2.0 is employed to fine-tune pre-trained transformer models, including XLM-R Base, mBERT, AraELECTRA, AraBERTv2, and CAMeLBERTmix. The findings suggest that both the dataset and experimental setup require further development to ensure a larger, higher-quality dataset and to support more extensive fine-tuning experiments, in addition to exploring transfer learning from other languages and enhancing the diversity and richness of Arabic concepts. These developments pave the way for further advancements in concept-based readability estimation in educational contexts in future work.\n                  <\/jats:p>","DOI":"10.1145\/3770070","type":"journal-article","created":{"date-parts":[[2025,10,3]],"date-time":"2025-10-03T10:58:32Z","timestamp":1759489112000},"page":"1-21","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Complex Concept-Based Readability Estimation from Arabic Curriculum"],"prefix":"10.1145","volume":"24","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5155-8096","authenticated-orcid":false,"given":"Sultan","family":"Almujaiwel","sequence":"first","affiliation":[{"name":"Humanities and Social Sciences, King Saud University","place":["Riyadh, Saudi Arabia"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9222-1959","authenticated-orcid":false,"given":"Damith","family":"Premasiri","sequence":"additional","affiliation":[{"name":"Computing and Communications, Lancaster University","place":["Lancaster, United Kingdom of Great Britain and Northern Ireland"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3207-3821","authenticated-orcid":false,"given":"Tharindu","family":"Ranasinghe","sequence":"additional","affiliation":[{"name":"Computing and Communications, Lancaster University","place":["Lancaster, United Kingdom of Great Britain and Northern Ireland"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6136-3898","authenticated-orcid":false,"given":"Mo","family":"El-Haj","sequence":"additional","affiliation":[{"name":"Computing and Communications, Lancaster University","place":["Lancaster, United Kingdom of Great Britain and Northern Ireland"]},{"name":"Computer Science, VinUniversity","place":["Lancaster, United Kingdom of Great Britain and Northern Ireland"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2522-066X","authenticated-orcid":false,"given":"Ruslan","family":"Mitkov","sequence":"additional","affiliation":[{"name":"Computing and Communications, Lancaster University","place":["Lancaster, United Kingdom of Great Britain and Northern Ireland"]}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,17]]},"reference":[{"key":"e_1_3_3_2_2","doi-asserted-by":"crossref","first-page":"506","DOI":"10.1109\/ICDIM.2008.4746711","volume-title":"2008 Third International Conference on Digital Information Management","author":"Al-Ajlan Amani A.","year":"2008","unstructured":"Amani A. Al-Ajlan, Hend S. Al-Khalifa, and AbdulMalik S. Al-Salman. 2008. Towards the development of an automatic readability measurements for Arabic language. In 2008 Third International Conference on Digital Information Management. 506\u2013511. DOI:10.1109\/ICDIM.2008.4746711"},{"key":"e_1_3_3_3_2","first-page":"3053","volume-title":"Proceedings of the Twelfth Language Resources and Evaluation Conference","author":"Khalil M. Al","year":"2020","unstructured":"M. Al Khalil, N. Habash, and Z. Jiang. 2020. A large-scale leveled readability lexicon for standard Arabic. In Proceedings of the Twelfth Language Resources and Evaluation Conference. European Language Resources Association, Marseille, France, 3053\u20133062. Retrieved fromhttps:\/\/aclanthology.org\/2020.lrec-1.373"},{"key":"e_1_3_3_4_2","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2017.10.109"},{"key":"e_1_3_3_5_2","volume-title":"Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC 2018)","author":"Khalil M. Al","year":"2018","unstructured":"M. Al Khalil, H. Saddiki, N. Habash, and L. Alfalasi. 2018. A leveled reading corpus of modern standard Arabic. In Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC 2018). European Language Resources Association (ELRA), Miyazaki, Japan. Retrieved from https:\/\/aclanthology.org\/L18-1366"},{"issue":"4","key":"e_1_3_3_6_2","first-page":"370","article-title":"AARI: Automatic Arabic readability index","volume":"11","author":"Tamimi A.-K. Al","year":"2014","unstructured":"A.-K. Al Tamimi, M. Jaradat, N. AlJarrah, and S. Ghanem. 2014. AARI: Automatic Arabic readability index. International Arab Journal of Information Technology 11, 4 (2014), 370\u2013378.","journal-title":"International Arab Journal of Information Technology"},{"key":"e_1_3_3_7_2","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-022-09577-5"},{"key":"e_1_3_3_8_2","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2016.04.017"},{"key":"e_1_3_3_9_2","first-page":"9","volume-title":"Proceedings of the 4th Workshop on Open-Source Arabic Corpora and Processing Tools, with a Shared Task on Offensive Language Detection","author":"Antoun Wissam","year":"2020","unstructured":"Wissam Antoun, Fady Baly, and Hazem Hajj. 2020. AraBERT: Transformer-based model for Arabic language understanding. In Proceedings of the 4th Workshop on Open-Source Arabic Corpora and Processing Tools, with a Shared Task on Offensive Language Detection. European Language Resource Association, Marseille, France, 9\u201315. Retrieved fromhttps:\/\/aclanthology.org\/2020.osact-1.2"},{"key":"e_1_3_3_10_2","first-page":"191","volume-title":"Proceedings of the Sixth Arabic Natural Language Processing Workshop","author":"Antoun Wissam","year":"2021","unstructured":"Wissam Antoun, Fady Baly, and Hazem Hajj. 2021. AraELECTRA: Pre-training text discriminators for Arabic language understanding. In Proceedings of the Sixth Arabic Natural Language Processing Workshop. Nizar Habash, Houda Bouamor, Hazem Hajj, Walid Magdy, Wajdi Zaghouani, Fethi Bougares, Nadi Tomeh, Ibrahim Abu Farha, and Samia Touileb (Eds.). Association for Computational Linguistics, Kyiv, Ukraine (Virtual), 191\u2013195. Retrieved from https:\/\/aclanthology.org\/2021.wanlp-1.20"},{"issue":"4","key":"e_1_3_3_11_2","doi-asserted-by":"crossref","first-page":"518","DOI":"10.1037\/edu0000225","article-title":"Reading demands in secondary school: Does the linguistic complexity of textbooks increase with grade level and the academic orientation of the school track?","volume":"110","author":"Berendes Karin","year":"2018","unstructured":"Karin Berendes, Sowmya Vajjala, Detmar Meurers, Doreen Bryant, Wolfgang Wagner, Maria Chinkina, and Ulrich Trautwein. 2018. Reading demands in secondary school: Does the linguistic complexity of textbooks increase with grade level and the academic orientation of the school track? Journal of Educational Psychology 110, 4 (2018), 518.","journal-title":"Journal of Educational Psychology"},{"key":"e_1_3_3_12_2","doi-asserted-by":"publisher","DOI":"10.6025\/jdim\/2021\/19\/3\/75-82"},{"key":"e_1_3_3_13_2","doi-asserted-by":"publisher","DOI":"10.1007\/s10772-018-9528-3"},{"key":"e_1_3_3_14_2","volume-title":"A Frequency Dictionary of Arabic: Core Vocabulary for Learners","author":"Buckwalter T.","year":"2011","unstructured":"T. Buckwalter and D. Parkinson. 2011. A Frequency Dictionary of Arabic: Core Vocabulary for Learners. Routledge Frequency Dictionaries."},{"key":"e_1_3_3_15_2","first-page":"79","volume-title":"Proceedings of the 5th International Conference on Arabic Language Processing","author":"Cavalli-Sforza Violetta","year":"2014","unstructured":"Violetta Cavalli-Sforza, Mariam El Mezouar, and Hind Saddiki. 2014. Matching an Arabic text to a learners\u2019 curriculum. In Proceedings of the 5th International Conference on Arabic Language Processing. 79\u201388."},{"key":"e_1_3_3_16_2","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2018.10.459"},{"key":"e_1_3_3_17_2","first-page":"48","volume-title":"Proceedings of the 16th Workshop on Innovative Use of NLP for Building Educational Applications","author":"Chatzipanagiotidis Savvas","year":"2021","unstructured":"Savvas Chatzipanagiotidis, Maria Giagkou, and Detmar Meurers. 2021. Broad linguistic complexity analysis for greek readability classification. In Proceedings of the 16th Workshop on Innovative Use of NLP for Building Educational Applications. Association for Computational Linguistics, Online, 48\u201358. Retrieved from https:\/\/aclanthology.org\/2021.bea-1.5"},{"key":"e_1_3_3_18_2","first-page":"544","volume-title":"Proceedings of the 27th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR \u201904)","author":"Collins-Thompson Kevyn","year":"2004","unstructured":"Kevyn Collins-Thompson and Jamie Callan. 2004. Information retrieval for language tutoring: An overview of the REAP project. In Proceedings of the 27th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR \u201904). Association for Computing Machinery, New York, NY, USA, 544\u2013545. DOI:10.1145\/1008992.1009112"},{"key":"e_1_3_3_19_2","doi-asserted-by":"crossref","first-page":"8440","DOI":"10.18653\/v1\/2020.acl-main.747","volume-title":"Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics","author":"Conneau Alexis","year":"2020","unstructured":"Alexis Conneau, Kartikay Khandelwal, Naman Goyal, Vishrav Chaudhary, Guillaume Wenzek, Francisco Guzm\u00e1n, Edouard Grave, Myle Ott, Luke Zettlemoyer, and Veselin Stoyanov. 2020. Unsupervised cross-lingual representation learning at scale. In Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics. Dan Jurafsky, Joyce Chai, Natalie Schluter, and Joel Tetreault (Eds.). Association for Computational Linguistics, Online, 8440\u20138451. DOI:10.18653\/v1\/2020.acl-main.747"},{"key":"e_1_3_3_20_2","volume-title":"Proceedings of the Annual Meeting of the Cognitive Science Society","volume":"29","author":"Crossley Scott A.","year":"2007","unstructured":"Scott A. Crossley, David F. Dufty, Philip M. McCarthy, and Danielle S. McNamara. 2007. Toward a new readability: A mixed model approach. In Proceedings of the Annual Meeting of the Cognitive Science Society, Vol. 29."},{"key":"e_1_3_3_21_2","doi-asserted-by":"publisher","DOI":"10.1002\/j.1545-7249.2008.tb00142.x"},{"key":"e_1_3_3_22_2","doi-asserted-by":"publisher","DOI":"10.3758\/s13428-022-01802-x"},{"key":"e_1_3_3_23_2","doi-asserted-by":"publisher","DOI":"10.1080\/0163853X.2017.1296264"},{"issue":"1","key":"e_1_3_3_24_2","first-page":"19","article-title":"The concept of readability","volume":"26","author":"Dale Edgar","year":"1949","unstructured":"Edgar Dale and Jeanne S. Chall. 1949. The concept of readability. Elementary English 26, 1 (1949), 19\u201326.","journal-title":"Elementary English"},{"key":"e_1_3_3_25_2","doi-asserted-by":"publisher","DOI":"10.17250\/khisli.35..201809.006"},{"key":"e_1_3_3_26_2","unstructured":"B. A. K. Dawood. 1977. The Relationship between Readability and Selected Language Variables. [Doctoral Dissertation University of Jordan]. (1977)."},{"issue":"3","key":"e_1_3_3_27_2","doi-asserted-by":"crossref","first-page":"293","DOI":"10.1017\/S1351324912000344","article-title":"Using the crowd for readability prediction","volume":"20","author":"Clercq Orph\u00e9e De","year":"2014","unstructured":"Orph\u00e9e De Clercq, V\u00e9ronique Hoste, Bart Desmet, Philip Van Oosten, Martine De Cock, and Lieve Macken. 2014. Using the crowd for readability prediction. Natural Language Engineering 20, 3 (2014), 293\u2013325.","journal-title":"Natural Language Engineering"},{"key":"e_1_3_3_28_2","first-page":"4171","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers). Association for Computational Linguistics, Minneapolis, Minnesota, 4171\u20134186. DOI:10.18653\/v1\/N19-1423"},{"key":"e_1_3_3_29_2","first-page":"103","article-title":"DARES: Dataset for Arabic readability estimation of school materials","author":"El-Haj Mo","year":"2024","unstructured":"Mo El-Haj, Sultan Almujaiwel, Damith Premasiri, Tharindu Ranasinghe, and Ruslan Mitkov. 2024. DARES: Dataset for Arabic readability estimation of school materials. In Proceedings of the Workshop on DeTermIt! Evaluating Text Difficulty in a Multilingual Context @ LREC-COLING 2024. ELRA and ICCL, Torino, Italia, 103\u2013113. Retrieved from https:\/\/aclanthology.org\/2024.determit-1.10","journal-title":"Proceedings of the Workshop on DeTermIt! Evaluating Text Difficulty in a Multilingual Context @ LREC-COLING 2024"},{"key":"e_1_3_3_30_2","first-page":"250","article-title":"OSMAN \u2013 A novel Arabic readability metric","author":"El-Haj M.","year":"2016","unstructured":"M. El-Haj and P. Rayson. 2016. OSMAN \u2013 A novel Arabic readability metric. In Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC\u201916).European Language Resources Association (ELRA), Portoro\u017e, Slovenia, 250\u2013255. Retrieved from https:\/\/aclanthology.org\/L16-1038","journal-title":"Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC\u201916)"},{"key":"e_1_3_3_31_2","doi-asserted-by":"crossref","first-page":"78","DOI":"10.1007\/978-3-031-15925-1_6","volume-title":"International Conference on Computational and Corpus-Based Phraseology","author":"El-Madkouri Mohamed","year":"2022","unstructured":"Mohamed El-Madkouri and Beatriz Soto Aranda. 2022. Readability and communication in machine translation of Arabic phraseologisms into Spanish. In International Conference on Computational and Corpus-Based Phraseology. Springer, 78\u201389."},{"key":"e_1_3_3_32_2","first-page":"229","volume-title":"Proceedings of the 12th Conference of the European Chapter of the ACL (EACL 2009)","author":"Feng Lijun","year":"2009","unstructured":"Lijun Feng, No\u00e9mie Elhadad, and Matt Huenerfauth. 2009. Cognitively motivated features for readability assessment. In Proceedings of the 12th Conference of the European Chapter of the ACL (EACL 2009), Alex Lascarides, Claire Gardent, and Joakim Nivre (Eds.). Association for Computational Linguistics, Athens, Greece, 229\u2013237. Retrieved from https:\/\/aclanthology.org\/E09-1027"},{"key":"e_1_3_3_33_2","first-page":"276","article-title":"A comparison of features for automatic readability assessment","author":"Feng Lijun","year":"2010","unstructured":"Lijun Feng, Martin Jansche, Matt Huenerfauth, and No\u00e9mie Elhadad. 2010. A comparison of features for automatic readability assessment. In Coling 2010: Posters. Coling 2010 Organizing Committee, Beijing, China, 276\u2013284. Retrieved from https:\/\/aclanthology.org\/C10-2032","journal-title":"Coling 2010: Posters"},{"issue":"3","key":"e_1_3_3_34_2","doi-asserted-by":"crossref","first-page":"221","DOI":"10.1037\/h0057532","article-title":"A new readability yardstick","volume":"32","author":"Flesch R.","year":"1948","unstructured":"R. Flesch. 1948. A new readability yardstick. Journal of Applied Psychology 32, 3 (1948), 221\u2013233.","journal-title":"Journal of Applied Psychology"},{"key":"e_1_3_3_35_2","first-page":"9","volume-title":"Workshop on Free\/Open-Source Arabic Corpora and Corpora Processing Tools Workshop Programme","author":"Forsyth Jonathan","year":"2014","unstructured":"Jonathan Forsyth. 2014. Automatic readability prediction for modern standard Arabic. In Workshop on Free\/Open-Source Arabic Corpora and Corpora Processing Tools Workshop Programme. 9."},{"key":"e_1_3_3_36_2","first-page":"466","volume-title":"Proceedings of the 2012 Joint Conference on Empirical Methods in Natural Language Processing and Computational Natural Language Learning","author":"Fran\u00e7ois Thomas","year":"2012","unstructured":"Thomas Fran\u00e7ois and C\u00e9drick Fairon. 2012. An \u201cAI readability\u201d formula for French as a foreign language. In Proceedings of the 2012 Joint Conference on Empirical Methods in Natural Language Processing and Computational Natural Language Learning. Association for Computational Linguistics, Jeju Island, Korea, 466\u2013477. Retrieved from https:\/\/aclanthology.org\/D12-1043"},{"key":"e_1_3_3_37_2","doi-asserted-by":"crossref","unstructured":"T. Fran\u00e7ois. 2015. When readability meets computational linguistics: A new paradigm in readability. Revue Fran\u00e7aise de Linguistique Appliqu\u00e9e 20 2 (2015) 79\u201397.","DOI":"10.3917\/rfla.202.0079"},{"key":"e_1_3_3_38_2","volume-title":"The Technique of Clear Writing","author":"Gunning R.","year":"1952","unstructured":"R. Gunning. 1952. The Technique of Clear Writing. McGraw-Hill."},{"key":"e_1_3_3_39_2","doi-asserted-by":"crossref","unstructured":"Nizar Habash Hanada Taha-Thomure Khalid N. Elmadani Zeina Zeino and Abdallah Abushmaes. 2024. Guidelines for fine-grained sentence-level Arabic readability annotation. arXiv:2410.08674. Retrieved from https:\/\/arxiv.org\/abs\/\/2410.08674","DOI":"10.18653\/v1\/2025.law-1.30"},{"key":"e_1_3_3_40_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-demos.24"},{"key":"e_1_3_3_41_2","first-page":"460","volume-title":"Human Language Technologies 2007: The Conference of the North American Chapter of the Association for Computational Linguistics; Proceedings of the Main Conference","author":"Heilman Michael","year":"2007","unstructured":"Michael Heilman, Kevyn Collins-Thompson, Jamie Callan, and Maxine Eskenazi. 2007. Combining lexical and grammatical features to improve readability measures for first and second language texts. In Human Language Technologies 2007: The Conference of the North American Chapter of the Association for Computational Linguistics; Proceedings of the Main Conference. Association for Computational Linguistics, Rochester, New York, 460\u2013467. Retrieved from https:\/\/aclanthology.org\/N07-1058"},{"key":"e_1_3_3_42_2","first-page":"611","volume-title":"Proceedings of the International Conference on Recent Advances in Natural Language Processing (RANLP 2021)","author":"Imperial Joseph Marvin","year":"2021","unstructured":"Joseph Marvin Imperial. 2021. BERT embeddings for automatic readability assessment. In Proceedings of the International Conference on Recent Advances in Natural Language Processing (RANLP 2021), Ruslan Mitkov and Galia Angelova (Eds.). INCOMA Ltd., Held Online, 611\u2013618. Retrieved from https:\/\/aclanthology.org\/2021.ranlp-1.69"},{"key":"e_1_3_3_43_2","volume-title":"Proceedings of the Sixth Arabic Natural Language Processing Workshop","author":"Inoue Go","year":"2021","unstructured":"Go Inoue, Bashar Alhafni, Nurpeiis Baimukan, Houda Bouamor, and Nizar Habash. 2021. The interplay of variant, size, and task type in Arabic pre-trained language models. In Proceedings of the Sixth Arabic Natural Language Processing Workshop. Association for Computational Linguistics, Kyiv, Ukraine (Online)."},{"issue":"5","key":"e_1_3_3_44_2","doi-asserted-by":"crossref","first-page":"433","DOI":"10.1002\/asi.24123","article-title":"GRAW+: A two-view graph propagation method with word coupling for readability assessment","volume":"70","author":"Jiang Zhiwei","year":"2019","unstructured":"Zhiwei Jiang, Qing Gu, Yafeng Yin, Jianxiang Wang, and Daoxu Chen. 2019. GRAW+: A two-view graph propagation method with word coupling for readability assessment. Journal of the Association for Information Science and Technology 70, 5 (2019), 433\u2013447.","journal-title":"Journal of the Association for Information Science and Technology"},{"key":"e_1_3_3_45_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-demos.11"},{"key":"e_1_3_3_46_2","first-page":"105","article-title":"Automatic difficulty classification of Arabic sentences","author":"Khallaf Nouran","year":"2021","unstructured":"Nouran Khallaf and Serge Sharoff. 2021. Automatic difficulty classification of Arabic sentences. In Proceedings of the Sixth Arabic Natural Language Processing Workshop. Association for Computational Linguistics, Kyiv, Ukraine (Virtual), 105\u2013114. Retrieved from https:\/\/aclanthology.org\/2021.wanlp-1.11","journal-title":"Proceedings of the Sixth Arabic Natural Language Processing Workshop"},{"key":"e_1_3_3_47_2","doi-asserted-by":"crossref","unstructured":"J. P. Kincaid R. P. Fishburne Jr R. L. Rogers and B. S. Chissom. 1975. Derivation Of New Readability Formulas (Automated Readability Index Fog Count and Flesch Reading Ease Formula) For Navy Enlisted Personnel. Naval Technical Training Command Millington TN.","DOI":"10.21236\/ADA006655"},{"key":"e_1_3_3_48_2","doi-asserted-by":"crossref","first-page":"175","DOI":"10.18653\/v1\/2021.emnlp-demo.21","volume-title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing: System Demonstrations","author":"Lhoest Quentin","year":"2021","unstructured":"Quentin Lhoest, Albert Villanova del Moral, Yacine Jernite, Abhishek Thakur, Patrick von Platen, Suraj Patil, Julien Chaumond, Mariama Drame, Julien Plu, Lewis Tunstall, et\u00a0al. 2021. Datasets: A community library for natural language processing. In Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing: System Demonstrations. Heike Adel and Shuming Shi (Eds.). Association for Computational Linguistics, Online and Punta Cana, Dominican Republic, 175\u2013184. DOI:10.18653\/v1\/2021.emnlp-demo.21"},{"issue":"3","key":"e_1_3_3_49_2","first-page":"e170\u2013e173","article-title":"Readability of patient educational materials in english versus Arabic","volume":"3","author":"Malik Abdulaziz","year":"2019","unstructured":"Abdulaziz Malik, Mahmoud El-Haj, and Michael K. Paasche-Orlow. 2019. Readability of patient educational materials in english versus Arabic. Health Lit. Res. Pract. 3, 3 (July2019), e170\u2013e173.","journal-title":"Health Lit. Res. Pract."},{"issue":"8","key":"e_1_3_3_50_2","first-page":"639","article-title":"SMOG Grading - a new readability formula","volume":"12","author":"McLaughlin G. H.","year":"1969","unstructured":"G. H. McLaughlin. 1969. SMOG Grading - a new readability formula. Journal of Reading 12, 8 (1969), 639\u2013646.","journal-title":"Journal of Reading"},{"key":"e_1_3_3_51_2","doi-asserted-by":"crossref","first-page":"33","DOI":"10.1007\/978-3-030-45439-5_3","volume-title":"Advances in Information Retrieval: 42nd European Conference on IR Research, ECIR 2020, Lisbon, Portugal, April 14\u201317, 2020, Proceedings, Part I 42","author":"Meng Changping","year":"2020","unstructured":"Changping Meng, Muhao Chen, Jie Mao, and Jennifer Neville. 2020. Readnet: A hierarchical transformer framework for web article readability analysis. In Advances in Information Retrieval: 42nd European Conference on IR Research, ECIR 2020, Lisbon, Portugal, April 14\u201317, 2020, Proceedings, Part I 42. Springer, 33\u201349."},{"key":"e_1_3_3_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/3571510"},{"key":"e_1_3_3_53_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-73500-9_9"},{"key":"e_1_3_3_54_2","doi-asserted-by":"crossref","first-page":"404","DOI":"10.18653\/v1\/2023.bea-1.33","volume-title":"Proceedings of the 18th Workshop on Innovative Use of NLP for Building Educational Applications (BEA 2023)","author":"North Kai","year":"2023","unstructured":"Kai North, Alphaeus Dmonte, Tharindu Ranasinghe, Matthew Shardlow, and Marcos Zampieri. 2023. ALEXSIS+: Improving substitute generation and selection for lexical simplification with information retrieval. In Proceedings of the 18th Workshop on Innovative Use of NLP for Building Educational Applications (BEA 2023). Association for Computational Linguistics, Toronto, Canada, 404\u2013413. DOI:10.18653\/v1\/2023.bea-1.33"},{"key":"e_1_3_3_55_2","doi-asserted-by":"publisher","unstructured":"Kai North Tharindu Ranasinghe Matthew Shardlow and Marcos Zampieri. 2025. Deep learning approaches to lexical simplification: A survey. Journal of Intelligent Information Systems 63 1 (2025) 111\u2013134. DOI:10.1007\/s10844-024-00882-9","DOI":"10.1007\/s10844-024-00882-9"},{"key":"e_1_3_3_56_2","first-page":"6057","article-title":"ALEXSIS-PT: A new resource for Portuguese lexical simplification","author":"North Kai","year":"2022","unstructured":"Kai North, Marcos Zampieri, and Tharindu Ranasinghe. 2022. ALEXSIS-PT: A new resource for Portuguese lexical simplification. In Proceedings of the 29th International Conference on Computational Linguistics. International Committee on Computational Linguistics. Gyeongju, Republic of Korea, 6057\u20136062. Retrieved from https:\/\/aclanthology.org\/2022.coling-1.529","journal-title":"Proceedings of the 29th International Conference on Computational Linguistics"},{"issue":"1","key":"e_1_3_3_57_2","first-page":"168","article-title":"A corpus-based readability formula for estimate of Arabic texts reading difficulty","volume":"21","author":"Nuraihan M. D.","year":"2013","unstructured":"M. D. Nuraihan, H. Haslina, and A. A. Normaziah. 2013. A corpus-based readability formula for estimate of Arabic texts reading difficulty. World Applied Sciences Journal 21, 1 (2013), 168\u2013173.","journal-title":"World Applied Sciences Journal"},{"key":"e_1_3_3_58_2","first-page":"7022","volume-title":"Proceedings of the 12th Language Resources and Evaluation Conference","author":"Obeid Ossama","year":"2020","unstructured":"Ossama Obeid, Nasser Zalmout, Salam Khalifa, Dima Taji, Mai Oudah, Bashar Alhafni, Go Inoue, Fadhl Eryani, Alexander Erdmann, and Nizar Habash. 2020. CAMeL tools: An open source python toolkit for Arabic natural language processing. In Proceedings of the 12th Language Resources and Evaluation Conference. European Language Resources Association, Marseille, France, 7022\u20137032. Retrieved fromhttps:\/\/www.aclweb.org\/anthology\/2020.lrec-1.868"},{"key":"e_1_3_3_59_2","first-page":"88","volume-title":"Proceedinsg of the 5th Workshop on Open-Source Arabic Corpora and Processing Tools with Shared Tasks on Qur\u2019an QA and Fine-Grained Hate Speech Detection","author":"Premasiri Damith","year":"2022","unstructured":"Damith Premasiri, Tharindu Ranasinghe, Wajdi Zaghouani, and Ruslan Mitkov. 2022. DTW at Qur\u2019an QA 2022: Utilising transfer learning with transformers for question answering in a low-resource domain. In Proceedinsg of the 5th Workshop on Open-Source Arabic Corpora and Processing Tools with Shared Tasks on Qur\u2019an QA and Fine-Grained Hate Speech Detection. European Language Resources Association, Marseille, France, 88\u201395. Retrieved from https:\/\/aclanthology.org\/2022.osact-1.10\/"},{"key":"e_1_3_3_60_2","first-page":"1","volume-title":"2015 IEEE\/ACS 12th International Conference of Computer Systems and Applications (AICCSA)","author":"Saddiki Hind","year":"2015","unstructured":"Hind Saddiki, Karim Bouzoubaa, and Violetta Cavalli-Sforza. 2015. Text readability for Arabic as a foreign language. In 2015 IEEE\/ACS 12th International Conference of Computer Systems and Applications (AICCSA). IEEE, 1\u20138."},{"key":"e_1_3_3_61_2","volume-title":"Proceedings of the Sixth International Conference on Language Resources and Evaluation (LREC\u201908)","author":"Sato Satoshi","year":"2008","unstructured":"Satoshi Sato, Suguru Matsuyoshi, and Yohsuke Kondoh. 2008. Automatic assessment of Japanese text readability based on a textbook corpus. In Proceedings of the Sixth International Conference on Language Resources and Evaluation (LREC\u201908). European Language Resources Association (ELRA), Marrakech, Morocco. Retrieved from http:\/\/www.lrec-conf.org\/proceedings\/lrec2008\/pdf\/165_paper.pdf"},{"key":"e_1_3_3_62_2","article-title":"Automated Readability Index","author":"Senter R. J.","year":"1967","unstructured":"R. J. Senter and E. A. Smith. 1967. Automated Readability Index. Wright-Patterson Air Force Base: iii. AMRL-TR-6620 (1967).","journal-title":"Wright-Patterson Air Force Base: iii. AMRL-TR-6620"},{"key":"e_1_3_3_63_2","first-page":"38","volume-title":"Proceedings of the 3rd Workshop on Tools and Resources for People with REAding DIfficulties (READI) @ LREC-COLING 2024","author":"Shardlow Matthew","year":"2024","unstructured":"Matthew Shardlow, Fernando Alva-Manchego, Riza Batista-Navarro, Stefan Bott, Saul Calderon Ramirez, R\u00e9mi Cardon, Thomas Fran\u00e7ois, Akio Hayakawa, Andrea Horbach, Anna H\u00fclsing, et\u00a0al. 2024. An extensible massively multilingual lexical simplification pipeline dataset using the multiLS Framework. In Proceedings of the 3rd Workshop on Tools and Resources for People with REAding DIfficulties (READI) @ LREC-COLING 2024. ELRA and ICCL, Torino, Italia, 38\u201346. Retrieved from https:\/\/aclanthology.org\/2024.readi-1.4"},{"key":"e_1_3_3_64_2","first-page":"571","volume-title":"Proceedings of the 19th Workshop on Innovative Use of NLP for Building Educational Applications (BEA 2024)","author":"Shardlow Matthew","year":"2024","unstructured":"Matthew Shardlow, Fernando Alva-Manchego, Riza Batista-Navarro, Stefan Bott, Saul Calderon Ramirez, R\u00e9mi Cardon, Thomas Fran\u00e7ois, Akio Hayakawa, Andrea Horbach, Anna H\u00fclsing, et\u00a0al. 2024. The BEA 2024 shared task on the multilingual lexical simplification pipeline. In Proceedings of the 19th Workshop on Innovative Use of NLP for Building Educational Applications (BEA 2024). Association for Computational Linguistics, Mexico City, Mexico, 571\u2013589. Retrieved from https:\/\/aclanthology.org\/2024.bea-1.51"},{"key":"e_1_3_3_65_2","first-page":"4096","volume-title":"Proceedings of the 26th International Joint Conference on Artificial Intelligence","author":"\u0160tajner Sanja","year":"2017","unstructured":"Sanja \u0160tajner, Simone Paolo Ponzetto, and Heiner Stuckenschmidt. 2017. Automatic assessment of absolute sentence complexity. In Proceedings of the 26th International Joint Conference on Artificial Intelligence. 4096\u20134102."},{"key":"e_1_3_3_66_2","article-title":"Measuring reading comprehension with the lexile framework","author":"Stenner A. J.","year":"1996","unstructured":"A. J. Stenner. 1996. Measuring reading comprehension with the lexile framework. MetaMetrics, Inc. Paper Presented at the 4th North American Conference on Adolescent\/Adult Literacy. Washington, D.C. (1996).","journal-title":"MetaMetrics, Inc. Paper Presented at the 4th North American Conference on Adolescent\/Adult Literacy. Washington, D.C."},{"key":"e_1_3_3_67_2","doi-asserted-by":"publisher","DOI":"10.1162\/coli.09-036-r2-08-050"},{"key":"e_1_3_3_68_2","first-page":"5366","volume-title":"Proceedings of the Thirteenth Language Resources and Evaluation Conference","author":"Vajjala Sowmya","year":"2022","unstructured":"Sowmya Vajjala. 2022. Trends, limitations and open challenges in automatic readability assessment research. In Proceedings of the Thirteenth Language Resources and Evaluation Conference. European Language Resources Association, Marseille, France, 5366\u20135377. Retrieved from https:\/\/aclanthology.org\/2022.lrec-1.574"},{"key":"e_1_3_3_69_2","doi-asserted-by":"crossref","first-page":"297","DOI":"10.18653\/v1\/W18-0535","volume-title":"Proceedings of the Thirteenth Workshop on Innovative Use of NLP for Building Educational Applications","author":"Vajjala Sowmya","year":"2018","unstructured":"Sowmya Vajjala and Ivana Lu\u010di\u0107. 2018. OneStopEnglish corpus: A new corpus for automatic readability assessment and text simplification. In Proceedings of the Thirteenth Workshop on Innovative Use of NLP for Building Educational Applications. Association for Computational Linguistics, New Orleans, Louisiana, 297\u2013304. DOI:10.18653\/v1\/W18-0535"},{"key":"e_1_3_3_70_2","first-page":"163","article-title":"On improving the accuracy of readability classification using insights from second language acquisition","author":"Vajjala Sowmya","year":"2012","unstructured":"Sowmya Vajjala and Detmar Meurers. 2012. On improving the accuracy of readability classification using insights from second language acquisition. In Proceedings of the Seventh Workshop on Building Educational Applications Using NLP (NAACL HLT \u201912). Association for Computational Linguistics, USA, 163\u2013173.","journal-title":"Proceedings of the Seventh Workshop on Building Educational Applications Using NLP"},{"key":"e_1_3_3_71_2","unstructured":"John Vidler and Paul Rayson. 2023. UCREL - Hex: A shared hybrid multiprocessor system. Retrieved from https:\/\/github.com\/UCREL\/hex. (2023). Accessed: 2024."},{"key":"e_1_3_3_72_2","doi-asserted-by":"crossref","first-page":"141","DOI":"10.18653\/v1\/2022.bea-1.19","volume-title":"Proceedings of the 17th Workshop on Innovative Use of NLP for Building Educational Applications (BEA 2022)","author":"Weiss Zarah","year":"2022","unstructured":"Zarah Weiss and Detmar Meurers. 2022. Assessing sentence readability for German language learners with broad linguistic modeling or readability formulas: When do linguistic insights make a difference?. In Proceedings of the 17th Workshop on Innovative Use of NLP for Building Educational Applications (BEA 2022). 141\u2013153."},{"key":"e_1_3_3_73_2","doi-asserted-by":"crossref","first-page":"38","DOI":"10.18653\/v1\/2020.emnlp-demos.6","volume-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations","author":"Wolf Thomas","year":"2020","unstructured":"Thomas Wolf, Lysandre Debut, Victor Sanh, Julien Chaumond, Clement Delangue, Anthony Moi, Pierric Cistac, Tim Rault, Remi Louf, Morgan Funtowicz, et\u00a0al. 2020. Transformers: State-of-the-art natural language processing. In Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, Qun Liu and David Schlangen (Eds.). Association for Computational Linguistics, Online, 38\u201345. DOI:10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_3_74_2","first-page":"12","volume-title":"Proceedings of the 11th Workshop on Innovative Use of NLP for Building Educational Applications","author":"Xia Menglin","year":"2016","unstructured":"Menglin Xia, Ekaterina Kochmar, and Ted Briscoe. 2016. Text readability assessment for second language learners. In Proceedings of the 11th Workshop on Innovative Use of NLP for Building Educational Applications. 12\u201322."},{"key":"e_1_3_3_75_2","doi-asserted-by":"crossref","first-page":"49","DOI":"10.1145\/2700648.2809852","volume-title":"Proceedings of the 17th International ACM SIGACCESS Conference on Computers & Accessibility","author":"Yaneva Victoria","year":"2015","unstructured":"Victoria Yaneva, Irina Temnikova, and Ruslan Mitkov. 2015. Accessible texts for autism: An eye-tracking study. In Proceedings of the 17th International ACM SIGACCESS Conference on Computers & Accessibility. 49\u201357."},{"key":"e_1_3_3_76_2","doi-asserted-by":"publisher","DOI":"10.4304\/tpls.2.1.43-53"}],"container-title":["ACM Transactions on Asian and Low-Resource Language Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3770070","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,17]],"date-time":"2025-11-17T12:20:56Z","timestamp":1763382056000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3770070"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,17]]},"references-count":75,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,11,30]]}},"alternative-id":["10.1145\/3770070"],"URL":"https:\/\/doi.org\/10.1145\/3770070","relation":{},"ISSN":["2375-4699","2375-4702"],"issn-type":[{"type":"print","value":"2375-4699"},{"type":"electronic","value":"2375-4702"}],"subject":[],"published":{"date-parts":[[2025,11,17]]},"assertion":[{"value":"2024-10-22","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2025-09-22","order":2,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2025-11-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}