{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T07:40:23Z","timestamp":1766734823691,"version":"3.48.0"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T00:00:00Z","timestamp":1766707200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T00:00:00Z","timestamp":1766707200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"vice presidency for Science and Technology, Iran","award":["11\/61955"],"award-info":[{"award-number":["11\/61955"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SN COMPUT. SCI."],"DOI":"10.1007\/s42979-025-04612-y","type":"journal-article","created":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T07:35:30Z","timestamp":1766734530000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["HmBlogs: A Comprehensive Corpus and Benchmarking Study for Persian Word Embedding and Language Modeling"],"prefix":"10.1007","volume":"7","author":[{"given":"Hamzeh Motahari","family":"Khansari","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7027-7529","authenticated-orcid":false,"given":"Mehrnoush","family":"Shamsfard","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mostafa","family":"Masumi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Seyed Soroush","family":"Majd","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,12,26]]},"reference":[{"issue":"5","key":"4612_CR1","doi-asserted-by":"publisher","first-page":"382","DOI":"10.1016\/j.knosys.2009.05.002","volume":"22","author":"A AleAhmad","year":"2009","unstructured":"AleAhmad A, Amiri H, Darrudi E, Rahgozar M, Oroumchian F. Hamshahri: A standard Persian text collection. Knowl Based Syst. 2009;22(5):382\u20137. https:\/\/doi.org\/10.1016\/j.knosys.2009.05.002.","journal-title":"Knowl Based Syst"},{"key":"4612_CR2","unstructured":"Sabeti B, Firouzjaee HA, Choobbasti AJ, Najafabadi SM, Vaheb A. LREC. Mirastext: An automatically generated text corpus for persian. in Proceedings of the eleventh international conference on language resources and evaluation (2018). 2018."},{"key":"4612_CR3","doi-asserted-by":"publisher","first-page":"143","DOI":"10.1007\/s10579-010-9132-x","volume":"45","author":"M Bijankhan","year":"2011","unstructured":"Bijankhan M, Sheykhzadegan J, Bahrani M. Ghayoomi. Lessons from Building a Persian written corpus: Peykare. Lang Resour Evaluation. 2011;45:143\u201364. https:\/\/doi.org\/10.1007\/s10579-010-9132-x.","journal-title":"Lang Resour Evaluation"},{"key":"4612_CR4","unstructured":"Persian Wikipedia. Wikipedia. https:\/\/en.wikipedia.org\/w\/index.php?title=Persian_Wikipedia&oldid=1014015085. Accessed 20 March 2024, 31 March 2024."},{"key":"4612_CR5","unstructured":"List of Wikipedias. https:\/\/en.wikipedia.org\/wiki\/List_of_Wikipedias. Accessed 20 March 2024."},{"key":"4612_CR6","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1016\/j.chb.2015.11.038","volume":"57","author":"A AleAhmad","year":"2016","unstructured":"AleAhmad A, Zahedi M, Rahgozar M, Moshiri B. IrBlogs: A standard collection for studying Persian bloggers. Comput Hum Behav. 2016;57:195\u2013207. https:\/\/doi.org\/10.1016\/j.chb.2015.11.038.","journal-title":"Comput Hum Behav"},{"key":"4612_CR7","unstructured":"Khashabi D. persian-raw-text. https:\/\/github.com\/persiannlp\/persian-raw-text. Accessed March 31, 2024."},{"key":"4612_CR8","unstructured":"Common Crawl. https:\/\/commoncrawl.org\/. Accessed 1 April 2021."},{"key":"4612_CR9","doi-asserted-by":"publisher","unstructured":"Nguyen T et al. Culturax: A cleaned, enormous, and multilingual dataset for large Language models in 167 Languages. ArXiv Preprint ArXiv:2309.09400. 2023; https:\/\/doi.org\/10.48550\/arXiv.2309.09400","DOI":"10.48550\/arXiv.2309.09400"},{"key":"4612_CR10","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2201.06642","author":"J Abadji","year":"2022","unstructured":"Abadji J, O. Suarez P, Romary L, Sagot B. Towards a cleaner document-oriented multilingual crawled corpus. ArXiv Preprint arXiv:2201 06642. 2022. https:\/\/doi.org\/10.48550\/arXiv.2201.06642.","journal-title":"ArXiv Preprint arXiv:2201 06642"},{"key":"4612_CR11","doi-asserted-by":"publisher","unstructured":"Sabouri S, Rahmati E, Gooran S. H. Sameti. naab: A ready-to-use plug-and-play corpus for Farsi. arXiv preprint arXiv:2208.13486. 2022; https:\/\/doi.org\/10.48550\/arXiv.2208.13486","DOI":"10.48550\/arXiv.2208.13486"},{"key":"4612_CR12","doi-asserted-by":"publisher","unstructured":"Masumi M, Majd SS, Shamsfard M, Beigy H. FaBERT: Pre-training BERT on Persian Blogs. ArXiv Preprint ArXiv:2402.06617. 2024; https:\/\/doi.org\/10.48550\/arXiv.2402.06617","DOI":"10.48550\/arXiv.2402.06617"},{"key":"4612_CR13","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805","author":"J Devlin","year":"2018","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K. Bert: Pre-training of deep bidirectional Transformers for Language Understanding. ArXiv Preprint arXiv. 2018. https:\/\/doi.org\/10.48550\/arXiv.1810.04805. :1810.04805.","journal-title":"ArXiv Preprint arXiv"},{"key":"4612_CR14","doi-asserted-by":"publisher","first-page":"3831","DOI":"10.1007\/s11063-021-10528-4","volume":"53","author":"M Farahani","year":"2021","unstructured":"Farahani M, Gharachorloo M, Farahani M, Manthouri M, Parsbert. Transformer-based model for Persian Language Understanding. Neural Process Lett. 2021;53:3831\u201347. https:\/\/doi.org\/10.1007\/s11063-021-10528-4.","journal-title":"Neural Process Lett"},{"key":"4612_CR15","unstructured":"Shamsfard M et al. Semi automatic development of farsnet; the persian wordnet. in Proceedings of 5th global WordNet conference, Mumbai, India. 2010, vol. 29."},{"key":"4612_CR16","doi-asserted-by":"crossref","unstructured":"Shamsfard M, Jafari HS. M. Ilbeygi. STeP-1: A Set of Fundamental Tools for Persian Text Processing. in LREC. 2010.","DOI":"10.1109\/NLPKE.2009.5313844"},{"key":"4612_CR17","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1301.3781","author":"T Mikolov","year":"2013","unstructured":"Mikolov T, Chen K, Corrado G, Dean J. Efficient Estimation of word representations in vector space. ArXiv Preprint arXiv:1301 3781. 2013. https:\/\/doi.org\/10.48550\/arXiv.1301.3781.","journal-title":"ArXiv Preprint arXiv:1301 3781"},{"key":"4612_CR18","doi-asserted-by":"publisher","unstructured":"Zahedi MS, Bokaei MH, Shoeleh F, Yadollahi MM, Doostmohammadi E. M. Farhoodi. Persian word embedding evaluation benchmarks. in Electrical Engineering (ICEE), Iranian Conference on. 2018: IEEE. pp. 1583\u20131588. https:\/\/doi.org\/10.1109\/ICEE.2018.8472549","DOI":"10.1109\/ICEE.2018.8472549"},{"key":"4612_CR19","doi-asserted-by":"crossref","unstructured":"Mahmoudi SE, Shamsfard M. SAT Based Analogy Evaluation Framework For Persian Word Embeddings. in 2022 12th International Conference on Computer and Knowledge Engineering (ICCKE). 2022: IEEE. pp. 355\u2013360.","DOI":"10.1109\/ICCKE57176.2022.9960082"},{"key":"4612_CR20","doi-asserted-by":"publisher","unstructured":"Camacho-Collados J, Pilehvar MT, Collier N, Navigli R. Semeval-2017 task 2: Multilingual and cross-lingual semantic word similarity. in Proceedings of the 11th international workshop on semantic evaluation (SemEval-2017). 2017. pp. 15\u201326. https:\/\/doi.org\/10.18653\/v1\/S17-2002","DOI":"10.18653\/v1\/S17-2002"},{"key":"4612_CR21","doi-asserted-by":"publisher","unstructured":"Pennington J, Socher R. C. D. Manning. Glove: Global vectors for word representation. in Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP). 2014. pp. 1532\u20131543. https:\/\/doi.org\/10.3115\/v1\/D14-1162","DOI":"10.3115\/v1\/D14-1162"},{"key":"4612_CR22","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1162\/tacl_a_00051","volume":"5","author":"P Bojanowski","year":"2017","unstructured":"Bojanowski P, Grave E, Joulin A, Mikolov T. Enriching word vectors with subword information. Trans Association Comput Linguistics. 2017;5:135\u201346. https:\/\/doi.org\/10.1162\/tacl_a_00051.","journal-title":"Trans Association Comput Linguistics"},{"key":"4612_CR23","unstructured":"\u0158eh\u016f\u0159ek R. P. Sojka. Software framework for topic modelling with large corpora. 2010."},{"key":"4612_CR24","unstructured":"Bairathi R. Bairathirahul\/mikolov-analogies. https:\/\/github.com\/bairathirahul\/mikolov-analogies. Accessed."},{"key":"4612_CR25","unstructured":"Geiping J, Goldstein T, Cramming. Training a Language Model on a single GPU in one day. in International Conference on Machine Learning. 2023: PMLR. pp. 11117\u201311143."},{"key":"4612_CR26","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1907.11692","author":"Y Liu","year":"2019","unstructured":"Liu Y, et al. Roberta: A robustly optimized Bert pretraining approach. ArXiv Preprint arXiv:1907 11692. 2019. https:\/\/doi.org\/10.48550\/arXiv.1907.11692.","journal-title":"ArXiv Preprint arXiv:1907 11692"},{"key":"4612_CR27","doi-asserted-by":"publisher","unstructured":"Ushio A, Espinosa-Anke L, Schockaert S, Camacho-Collados J. BERT is to NLP what AlexNet is to CV: can pre-trained Language models identify analogies? ArXiv Preprint. 2021;arXiv:210504949. https:\/\/doi.org\/10.48550\/arXiv.2105.04949.","DOI":"10.48550\/arXiv.2105.04949"},{"key":"4612_CR28","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1911.02116","author":"A Conneau","year":"2019","unstructured":"Conneau A, et al. Unsupervised cross-lingual representation learning at scale. ArXiv Preprint arXiv:1911 02116. 2019. https:\/\/doi.org\/10.48550\/arXiv.1911.02116.","journal-title":"ArXiv Preprint arXiv:1911 02116"},{"key":"4612_CR29","doi-asserted-by":"publisher","unstructured":"Ghafouri A, Abbasi MA. H. Naderi. AriaBERT: A Pre-trained Persian BERT Model for Natural Language Understanding. 2023; https:\/\/doi.org\/10.21203\/rs.3.rs-3558473\/v1","DOI":"10.21203\/rs.3.rs-3558473\/v1"},{"key":"4612_CR30","doi-asserted-by":"publisher","unstructured":"Aghajani M, Badri A, Beigy H. ParsTwiNER: A corpus for named entity recognition at informal Persian. in Proceedings of the Seventh Workshop on Noisy User-generated Text (W-NUT 2021). 2021. pp. 131\u2013136. https:\/\/doi.org\/10.18653\/v1\/2021.wnut-1.16","DOI":"10.18653\/v1\/2021.wnut-1.16"},{"key":"4612_CR31","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1801.09936","author":"MS Shahshahani","year":"2018","unstructured":"Shahshahani MS, Mohseni M, Shakery A, Faili H. PEYMA: A tagged corpus for Persian named entities. ArXiv Preprint arXiv. 2018. https:\/\/doi.org\/10.48550\/arXiv.1801.09936. :1801.09936.","journal-title":"ArXiv Preprint arXiv"},{"key":"4612_CR32","doi-asserted-by":"publisher","unstructured":"Fetahu B, Chen Z, Kar S, Rokhlenko O, Malmasi S. MultiCoNER v2: a Large Multilingual dataset for Fine-grained and Noisy Named Entity Recognition. arXiv preprint arXiv:2310.13213. 2023; https:\/\/doi.org\/10.48550\/arXiv.2310.13213","DOI":"10.48550\/arXiv.2310.13213"},{"key":"4612_CR33","doi-asserted-by":"publisher","unstructured":"Amirkhani H, AzariJafari M, Faridan-Jahromi S, Kouhkan Z, Pourjafari Z, Amirak A. Farstail: A Persian natural Language inference dataset. Soft Comput. 2023;1\u201313. https:\/\/doi.org\/10.1007\/s00500-023-08959-3.","DOI":"10.1007\/s00500-023-08959-3"},{"key":"4612_CR34","doi-asserted-by":"publisher","unstructured":"Rahimi Z, ShamsFard. M. A Knowledge-Based Approach for Recognizing Textual Entailments with a Focus on Causality and Contradiction. Available at SSRN 4526759. 2024; https:\/\/doi.org\/10.2139\/ssrn.4526759","DOI":"10.2139\/ssrn.4526759"},{"key":"4612_CR35","doi-asserted-by":"publisher","first-page":"1147","DOI":"10.1162\/tacl_a_00419","volume":"9","author":"D Khashabi","year":"2021","unstructured":"Khashabi D, et al. Parsinlu: a suite of Language Understanding challenges for Persian. Trans Association Comput Linguistics. 2021;9:1147\u201362. https:\/\/doi.org\/10.1162\/tacl_a_00419.","journal-title":"Trans Association Comput Linguistics"},{"key":"4612_CR36","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2023.101486","author":"K Darvishi","year":"2023","unstructured":"Darvishi K, Shahbodaghkhan N, Abbasiantaeb Z, Momtazi S. Comput Speech Lang. 2023;80:101486. https:\/\/doi.org\/10.1016\/j.csl.2023.101486. PQuAD: A Persian question answering dataset."},{"key":"4612_CR37","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2312.04362","author":"HH Hemati","year":"2023","unstructured":"Hemati HH, Toghyani A, Souri A, Alavian SH, Sameti H, Beigy H. PCoQA: Persian conversational question answering dataset. ArXiv Preprint arXiv:2312 04362. 2023. https:\/\/doi.org\/10.48550\/arXiv.2312.04362.","journal-title":"ArXiv Preprint arXiv:2312 04362"},{"key":"4612_CR38","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2004.05328","author":"JPR Sharami","year":"2020","unstructured":"Sharami JPR, Sarabestani PA, Mirroshandel SA. Deepsentipers: novel deep learning models trained over proposed augmented Persian sentiment corpus. ArXiv Preprint arXiv:2004 05328. 2020. https:\/\/doi.org\/10.48550\/arXiv.2004.05328.","journal-title":"ArXiv Preprint arXiv:2004 05328"},{"key":"4612_CR39","unstructured":"Asli SAA, Sabeti B, Majdabadi Z, Golazizian P, Fahmi R, Momenzadeh O. Optimizing annotation effort using active learning strategies: A sentiment analysis case study in persian. in Proceedings of the Twelfth Language Resources and Evaluation Conference. 2020. pp. 2855\u20132861."},{"key":"4612_CR40","unstructured":"Golazizian P, Sabeti B, Asli SAA, Majdabadi Z, Momenzadeh O. R. Fahmi. Irony detection in Persian language: A transfer learning approach using emoji prediction. in Proceedings of the Twelfth Language Resources and Evaluation Conference. 2020. pp. 2839\u20132845."},{"key":"4612_CR41","doi-asserted-by":"publisher","unstructured":"Zhang C, Doan KD, Liao Q, Abdul-Mageed M. The Skipped Beat: A Study of Sociopragmatic Understanding in LLMs for 64 Languages. arXiv preprint arXiv:2310.14557. 2023; https:\/\/doi.org\/10.48550\/arXiv.2310.14557","DOI":"10.48550\/arXiv.2310.14557"},{"key":"4612_CR42","unstructured":"English Corpora. https:\/\/www.english-corpora.org\/. Accessed 1 April 2021."},{"key":"4612_CR43","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2406.00867","author":"P Falakaflaki","year":"2024","unstructured":"Falakaflaki P. Shamsfard. Formality style transfer in persian. arXiv preprint. 2024. https:\/\/doi.org\/10.48550\/arXiv.2406.00867. arXiv:2406.00867https:\/\/doi.org\/."},{"key":"4612_CR44","doi-asserted-by":"publisher","unstructured":"Arshia FZ, Sadidpour SS. Enhancing Persian Word Sense Disambiguation with Large Language Models Techniques and Applications. in 2024 14th International Conference on Computer and Knowledge Engineering (ICCKE). 2024: IEEE. pp. 162\u2013167. https:\/\/doi.org\/10.1109\/ICCKE65377.2024.10874479","DOI":"10.1109\/ICCKE65377.2024.10874479"},{"key":"4612_CR45","unstructured":"Mahdavi Mortazavi M, Shamsfard M. Emotion recognition in Persian text using FaBERT, presented at the International Conference on Artificial Intelligence, 2025."},{"key":"4612_CR46","unstructured":"Farhanian SE. Providing a Solution for Selecting a Polarity Detection Model in Persian Text Streams, Master\u2019s, The School of Computer Engineering, Iran University of Science and Technology (IUST), 2024."},{"key":"4612_CR47","doi-asserted-by":"crossref","unstructured":"VarastehNezhad A, Tavasoli R, Masumi M, Majd SS. M. Shamsfard. Evaluating LLMs in Persian News Summarization. in 2024 15th International Conference on Information and Knowledge Technology (IKT). 2024: IEEE. pp. 195\u2013201.","DOI":"10.1109\/IKT65497.2024.10892758"}],"container-title":["SN Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-025-04612-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42979-025-04612-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-025-04612-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T07:35:38Z","timestamp":1766734538000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42979-025-04612-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,26]]},"references-count":47,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,1]]}},"alternative-id":["4612"],"URL":"https:\/\/doi.org\/10.1007\/s42979-025-04612-y","relation":{"references":[{"id-type":"doi","id":"10.1016\/j.csl.2023.101486","asserted-by":"subject"},{"id-type":"doi","id":"10.48550\/arXiv.2406.00867","asserted-by":"subject"}]},"ISSN":["2661-8907"],"issn-type":[{"value":"2661-8907","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,26]]},"assertion":[{"value":"26 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"On behalf of all authors, the corresponding author states that there is no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}},{"value":"The submitted work is an original research work, neither published nor submitted to any other journal. The Research does not involve experiments on human participants and\/or animals.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}}],"article-number":"37"}}