{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T09:19:39Z","timestamp":1780391979310,"version":"3.54.1"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2025,4,26]],"date-time":"2025-04-26T00:00:00Z","timestamp":1745625600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,4,26]],"date-time":"2025-04-26T00:00:00Z","timestamp":1745625600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SN COMPUT. SCI."],"DOI":"10.1007\/s42979-025-03891-9","type":"journal-article","created":{"date-parts":[[2025,4,26]],"date-time":"2025-04-26T11:42:19Z","timestamp":1745667739000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Context-Aware Data Cleaning: Optimizing Bengali Text for Contextual Text Classification"],"prefix":"10.1007","volume":"6","author":[{"given":"Moshiur Rahman","family":"Faisal","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abdur Rahman","family":"Fahad","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shahriyar Zaman","family":"Ridoy","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jannat","family":"Sultana","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zinnat Fowzia","family":"Ria","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Md Hasibur","family":"Rahman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mohammed Arif","family":"Uddin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4514-6279","authenticated-orcid":false,"given":"Rashedur M.","family":"Rahman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,4,26]]},"reference":[{"issue":"3","key":"3891_CR1","doi-asserted-by":"publisher","first-page":"3713","DOI":"10.1007\/s11042-022-13428-4","volume":"82","author":"D Khurana","year":"2023","unstructured":"Khurana D, Koli A, Khatter K, Singh S. Natural language processing: state of the art, current trends and challenges. Multimed Tools Appl. 2023;82(3):3713\u201344.","journal-title":"Multimed Tools Appl"},{"key":"3891_CR2","doi-asserted-by":"publisher","first-page":"443","DOI":"10.1016\/j.neucom.2021.05.103","volume":"470","author":"I Lauriola","year":"2022","unstructured":"Lauriola I, Lavelli A, Aiolli F. An introduction to deep learning in natural language processing: models, techniques, and tools. Neurocomputing. 2022;470:443\u201356.","journal-title":"Neurocomputing"},{"key":"3891_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2021.101311","volume":"72","author":"D Cunliffe","year":"2022","unstructured":"Cunliffe D, Vlachidis A, Williams D, Tudhope D. Natural language processing for under-resourced languages: developing a Welsh natural language toolkit. Comput Speech Lang. 2022;72: 101311.","journal-title":"Comput Speech Lang"},{"key":"3891_CR4","doi-asserted-by":"crossref","unstructured":"Hasan T, Bhattacharjee A, Samin K, Hasan M, Basak M, Rahman MS, Shahriyar R. Not low-resource anymore: Aligner ensembling, batch filtering, and new datasets for Bengali-English machine translation. In Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), Association for Computational Linguistics; 2020. pp. 2612\u20132623.","DOI":"10.18653\/v1\/2020.emnlp-main.207"},{"key":"3891_CR5","doi-asserted-by":"crossref","unstructured":"Shruti AC, Rifat RH, Kamal M, Alam MGR (2023) A comparative study on bengali speech sentiment analysis based on audio data. In 2023 IEEE International Conference on Big Data and Smart Computing (BigComp). IEEE. pp. 219\u2013226.","DOI":"10.1109\/BigComp57234.2023.00043"},{"issue":"3","key":"3891_CR6","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1017\/S1351324922000213","volume":"29","author":"CP Chai","year":"2023","unstructured":"Chai CP. Comparison of text preprocessing methods. Nat Lang Eng. 2023;29(3):509\u201353.","journal-title":"Nat Lang Eng"},{"key":"3891_CR7","unstructured":"Wynne HS, Kotzor S, Zhou B, Lahiri A. The effect of phonological and morphological overlap on the processing of Bengali words. J South Asian Linguist. 2020;11(2)."},{"issue":"1","key":"3891_CR8","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1016\/j.ipm.2013.08.006","volume":"50","author":"AK Uysal","year":"2014","unstructured":"Uysal AK, Gunal S. The impact of preprocessing on text classification. Inf Process Manage. 2014;50(1):104\u201312.","journal-title":"Inf Process Manage"},{"issue":"6","key":"3891_CR9","first-page":"22","volume":"16","author":"AI Kadhim","year":"2018","unstructured":"Kadhim AI. An evaluation of preprocessing techniques for text classification. Int J Comput Sci Inf Secur (IJCSIS). 2018;16(6):22\u201332.","journal-title":"Int J Comput Sci Inf Secur (IJCSIS)"},{"issue":"4","key":"3891_CR10","first-page":"207","volume":"4","author":"J Kaur","year":"2018","unstructured":"Kaur J, Buttar PK. A systematic review on stopword removal algorithms. Int J Future Revol Comput Sci Commun Eng. 2018;4(4):207\u201310.","journal-title":"Int J Future Revol Comput Sci Commun Eng"},{"key":"3891_CR11","unstructured":"Wachirapong F, Nelimarkka M. Exploring the effect of preprocessing techniques on the topic modeling in social science data. 2023"},{"key":"3891_CR12","doi-asserted-by":"publisher","first-page":"164681","DOI":"10.1109\/ACCESS.2021.3134154","volume":"9","author":"MF Mridha","year":"2021","unstructured":"Mridha MF, Wadud MAH, Hamid MA, Monowar MM, Abdullah-Al-Wadud M, Alamri A. L-boost: Identifying offensive texts from social media post in bengali. Ieee Access. 2021;9:164681\u201399.","journal-title":"Ieee Access"},{"key":"3891_CR13","doi-asserted-by":"crossref","unstructured":"Karim MR, Chakravarthi BR, McCrae JP, Cochez M. Classification benchmarks for under-resourced bengali language based on multichannel convolutional-lstm network. In 2020 IEEE 7th international conference on Data Science and Advanced Analytics (DSAA). IEEE. 2020 pp. 390\u2013399.","DOI":"10.1109\/DSAA49011.2020.00053"},{"key":"3891_CR14","first-page":"50","volume-title":"Text detergent: the systematic combination of text pre-processing techniques for social media sentiment analysis. In: International Conference of Reliable Information and Communication Technology","author":"UH Hair Zaki","year":"2021","unstructured":"Hair Zaki UH, Ibrahim R, Abd Halim S, Kamsani II. Text detergent: the systematic combination of text pre-processing techniques for social media sentiment analysis. In: International Conference of Reliable Information and Communication Technology. Cham: Springer International Publishing; 2021. p. 50\u201361."},{"key":"3891_CR15","doi-asserted-by":"publisher","first-page":"549","DOI":"10.1016\/j.procs.2016.06.095","volume":"89","author":"T Singh","year":"2016","unstructured":"Singh T, Kumari M. Role of text preprocessing in twitter sentiment analysis. Proc Comput Sci. 2016;89:549\u201354.","journal-title":"Proc Comput Sci"},{"key":"3891_CR16","doi-asserted-by":"crossref","unstructured":"Srivastava A, Makhija P, Gupta A. Noisy text data: Achilles' heel of BERT. In Proceedings of the Sixth Workshop on Noisy User-generated Text (W-NUT 2020). 2020. pp. 16\u201321.","DOI":"10.18653\/v1\/2020.wnut-1.3"},{"key":"3891_CR17","doi-asserted-by":"crossref","unstructured":"Sun Y, Jiang H. Contextual text denoising with Masked Language Model. Proceedings of the 5th Workshop on Noisy User-Generated Text (W-NUT 2019). 2019","DOI":"10.18653\/v1\/D19-5537"},{"issue":"17","key":"3891_CR18","doi-asserted-by":"publisher","first-page":"8172","DOI":"10.3390\/app11178172","volume":"11","author":"J Khan","year":"2021","unstructured":"Khan J, Lee S. Enhancement of text analysis using context-aware normalization of social media informal text. Appl Sci. 2021;11(17):8172.","journal-title":"Appl Sci"},{"key":"3891_CR19","doi-asserted-by":"crossref","unstructured":"Siino M, Tinnirello I, La Cascia M. Is text preprocessing still worth the time? A comparative survey on the influence of popular preprocessing methods on Transformers and traditional classifiers. Information Systems. 2023. p 102342.","DOI":"10.1016\/j.is.2023.102342"},{"issue":"17","key":"3891_CR20","doi-asserted-by":"publisher","first-page":"8765","DOI":"10.3390\/app12178765","volume":"12","author":"MA Palomino","year":"2022","unstructured":"Palomino MA, Aider F. Evaluating the effectiveness of text pre-processing in sentiment analysis. Appl Sci. 2022;12(17):8765.","journal-title":"Appl Sci"},{"key":"3891_CR21","doi-asserted-by":"crossref","unstructured":"Bilal Mohammad. Bridging the language gap: evaluating text data pre-processing and classification techniques in Urdu sentiment analysis. 2024.","DOI":"10.62019\/abbdm.v4i02.183"},{"key":"3891_CR22","doi-asserted-by":"crossref","unstructured":"Nema NK, Shukla V, Pimpalkar A, Tandan SR. sentiment analysis of depression and anxiety social media tweets using TF-IDF weighting and supervised learning algorithm. In 2024 OPJU International Technology Conference (OTCON) on Smart Computing for Innovation and Advancement in Industry 4.0. IEEE. 2024. pp. 1\u20136.","DOI":"10.1109\/OTCON60325.2024.10687916"},{"key":"3891_CR23","doi-asserted-by":"crossref","unstructured":"Ghanem FA, Padma MC, Alkhatib R. Elevating the precision of summarization for short text in social media using preprocessing techniques. In 2023 IEEE International Conference on High Performance Computing & Communications, Data Science & Systems, Smart City & Dependability in Sensor, Cloud & Big Data Systems & Application (HPCC\/DSS\/SmartCity\/DependSys). IEEE. 2023. pp. 408\u2013416.","DOI":"10.1109\/HPCC-DSS-SmartCity-DependSys60770.2023.00063"},{"issue":"23","key":"3891_CR24","doi-asserted-by":"publisher","first-page":"8631","DOI":"10.3390\/app10238631","volume":"10","author":"V Maslej-Kre\u0161\u0148\u00e1kov\u00e1","year":"2020","unstructured":"Maslej-Kre\u0161\u0148\u00e1kov\u00e1 V, Sarnovsk\u00fd M, Butka P, Machov\u00e1 K. Comparison of deep learning models and various text pre-processing techniques for the toxic comments classification. Appl Sci. 2020;10(23):8631.","journal-title":"Appl Sci"},{"key":"3891_CR25","doi-asserted-by":"crossref","unstructured":"Ridoy SZ, Sultana J, Ria ZF, Uddin MA, Rahman MH, Rahman RM. An Efficient Text Cleaning Pipeline for Clinical Text for Transformer Encoder Models. In 2024 IEEE 12th International Conference on Intelligent Systems (IS) IEEE. 2024. pp. 1\u20139.","DOI":"10.1109\/IS61756.2024.10705199"},{"issue":"2","key":"3891_CR26","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1007\/s42979-022-01028-w","volume":"3","author":"MA Iqbal","year":"2022","unstructured":"Iqbal MA, Das A, Sharif O, Hoque MM, Sarker IH. Bemoc: a corpus for identifying emotion in bengali texts. SN Comput Sci. 2022;3(2):135.","journal-title":"SN Comput Sci"},{"key":"3891_CR27","doi-asserted-by":"crossref","unstructured":"Islam KI, Kar S, Islam MS, Amin MR. SentNoB: A dataset for analysing sentiment on noisy Bangla texts. In Findings of the Association for Computational Linguistics: EMNLP. 2021. pp. 3265\u20133271.","DOI":"10.18653\/v1\/2021.findings-emnlp.278"},{"key":"3891_CR28","doi-asserted-by":"crossref","unstructured":"Sourav MS, Mahmud MS, Zheng H, Aljaidi M, Talukder MS, Sulaiman RB, Nur AH, Al-Qerem A. Transformer-based text classification on unified bangla multi-class emotion corpus. In 2024 25th International Arab Conference on Information Technology (ACIT), IEEE; 2024. pp. 1\u20137.","DOI":"10.1109\/ACIT62805.2024.10877210"},{"key":"3891_CR29","doi-asserted-by":"crossref","unstructured":"Islam KI, Yuvraz T, Islam MS, Hassan E. EmoNoBa: A Dataset for Analyzing Fine-Grained Emotions on Noisy Bangla Texts. In Proceedings of the 2nd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics and the 12th International Joint Conference on Natural Language Processing. 2022. pp. 128\u2013134.","DOI":"10.18653\/v1\/2022.aacl-short.17"},{"key":"3891_CR30","volume-title":"Encyclopedia of machine learning","author":"C Sammut","year":"2011","unstructured":"Sammut C, Webb GI. Encyclopedia of machine learning. New York: Springer Science & Business Media; 2011."},{"key":"3891_CR31","doi-asserted-by":"crossref","unstructured":"Pennington J, Socher R, Manning CD. Glove: Global vectors for word representation. In Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP). 2014. pp. 1532\u20131543.","DOI":"10.3115\/v1\/D14-1162"},{"key":"3891_CR32","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1162\/tacl_a_00051","volume":"5","author":"P Bojanowski","year":"2017","unstructured":"Bojanowski P, Grave E, Joulin A, Mikolov T. Enriching word vectors with subword information. Trans Assoc Comput Linguist. 2017;5:135\u201346.","journal-title":"Trans Assoc Comput Linguist"},{"key":"3891_CR33","unstructured":"Kenton JDMWC, Toutanova LK. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of naacL-HLT vol. 1. 2019. p. 2"},{"issue":"4","key":"3891_CR34","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1109\/5254.708428","volume":"13","author":"MA Hearst","year":"1998","unstructured":"Hearst MA, Dumais ST, Osuna E, Platt J, Scholkopf B. Support vector machines. IEEE Intell Syst Appl. 1998;13(4):18\u201328.","journal-title":"IEEE Intell Syst Appl"},{"key":"3891_CR35","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1023\/A:1010933404324","volume":"45","author":"L Breiman","year":"2001","unstructured":"Breiman L. Random forests. Mach Learn. 2001;45:5\u201332.","journal-title":"Mach Learn."},{"key":"3891_CR36","doi-asserted-by":"crossref","unstructured":"Chen T, Guestrin C. Xgboost: a scalable tree boosting system. In Proceedings of the 22nd acm sigkdd international conference on knowledge discovery and data mining. 2016. pp. 785\u2013794.","DOI":"10.1145\/2939672.2939785"},{"key":"3891_CR37","unstructured":"McCallum A, Nigam K. A comparison of event models for naive bayes text classification. In AAAI-98 workshop on learning for text categorization. 752, (1): 41\u201348. 1998"},{"issue":"1","key":"3891_CR38","doi-asserted-by":"publisher","first-page":"1717","DOI":"10.4249\/scholarpedia.1717","volume":"2","author":"K Fukushima","year":"2007","unstructured":"Fukushima K. Neocognitron. Scholarpedia. 2007;2(1):1717.","journal-title":"Scholarpedia"},{"issue":"11","key":"3891_CR39","doi-asserted-by":"publisher","first-page":"2673","DOI":"10.1109\/78.650093","volume":"45","author":"M Schuster","year":"1997","unstructured":"Schuster M, Paliwal KK. Bidirectional recurrent neural networks. IEEE Trans Signal Process. 1997;45(11):2673\u201381.","journal-title":"IEEE Trans Signal Process"},{"key":"3891_CR40","doi-asserted-by":"crossref","unstructured":"Bhattacharjee A, Hasan T, Ahmad WU, Samin K, Islam MS, Iqbal A, Rahman MS, Shahriyar R. BanglaBERT: Language model pretraining and benchmarks for low-resource language understanding evaluation in Bangla. In Findings of the Association for Computational Linguistics: NAACL.  Association for Computational Linguistics. Seattle, United States; 2022. pp.  1318\u20131327.","DOI":"10.18653\/v1\/2022.findings-naacl.98"},{"key":"3891_CR41","unstructured":"Clark K, Luong MT, Le QV, Manning CD. Electra: Pre-training text encoders as discriminators rather than generators. In 8th International Conference on Learning Representations, ICLR 2020, Addis Ababa, Ethiopia, April 26-30, 2020. OpenReview.net."},{"key":"3891_CR42","doi-asserted-by":"crossref","unstructured":"Conneau A, Khandelwal K, Goyal N, Chaudhary V, Wenzek G, Guzm\u00e1n F, Grave E, Ott M, Zettlemoyer L, Stoyanov V. Unsupervised cross-lingual representation learning at scale. In Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, online. Association for Computational Linguistics; 2020. pp. 8440\u20138451.","DOI":"10.18653\/v1\/2020.acl-main.747"},{"issue":"3","key":"3891_CR43","doi-asserted-by":"publisher","first-page":"130","DOI":"10.1108\/eb046814","volume":"14","author":"MF Porter","year":"1980","unstructured":"Porter MF. An algorithm for suffix stripping. Program. 1980;14(3):130\u20137.","journal-title":"Program"},{"key":"3891_CR44","doi-asserted-by":"crossref","unstructured":"Chowdhury Rahman, Hasibur Rahman, Samiha Zakir, Mohammad Rafsan, and Mohammed Eunus Ali. BSpell: A CNN-Blended BERT Based Bangla Spell Checker. In Proceedings of the First Workshop on Bangla Language Processing (BLP-2023). Association for Computational Linguistics, Singapore. 2023. pp 7\u201317","DOI":"10.18653\/v1\/2023.banglalp-1.2"},{"key":"3891_CR45","doi-asserted-by":"crossref","unstructured":"Prechelt L. Early stopping-but when? In Neural Networks: Tricks of the trade. Springer, Berlin Heidelberg. 2002. pp. 55\u201369","DOI":"10.1007\/3-540-49430-8_3"},{"key":"3891_CR46","doi-asserted-by":"crossref","unstructured":"Smith LN. Cyclical learning rates for training neural networks. In 2017 IEEE winter conference on applications of computer vision (WACV). IEEE. 2017. pp. 464\u2013472","DOI":"10.1109\/WACV.2017.58"},{"key":"3891_CR47","doi-asserted-by":"publisher","unstructured":"Mahmud MR, Afrin M, Razzaque MA, Miller E, Iwashige J. \u201cA rule based bengali stemmer,\u201d 2014 International Conference on Advances in Computing, Communications and Informatics (ICACCI), New Delhi, 2014. pp. 2750\u20132756. https:\/\/doi.org\/10.1109\/ICACCI.2014.6968484","DOI":"10.1109\/ICACCI.2014.6968484"},{"key":"3891_CR48","unstructured":"Sarker S. Bnlp: Natural language processing toolkit for bengali language. arXiv preprint arXiv:2102.00405. 2021"}],"container-title":["SN Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-025-03891-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42979-025-03891-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-025-03891-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,27]],"date-time":"2025-04-27T00:01:25Z","timestamp":1745712085000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42979-025-03891-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,26]]},"references-count":48,"journal-issue":{"issue":"5","published-online":{"date-parts":[[2025,6]]}},"alternative-id":["3891"],"URL":"https:\/\/doi.org\/10.1007\/s42979-025-03891-9","relation":{},"ISSN":["2661-8907"],"issn-type":[{"value":"2661-8907","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,4,26]]},"assertion":[{"value":"26 March 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 March 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 April 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not Applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed Consent"}},{"value":"Not Applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Research Involving Human and \/or Animals"}}],"article-number":"422"}}