{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,15]],"date-time":"2025-10-15T00:30:57Z","timestamp":1760488257976,"version":"build-2065373602"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100020950","name":"National Science and Technology Council","doi-asserted-by":"publisher","award":["NSTC111\u20102222\u2010E\u2010110\u2010006\u2010MY3"],"award-info":[{"award-number":["NSTC111\u20102222\u2010E\u2010110\u2010006\u2010MY3"]}],"id":[{"id":"10.13039\/501100020950","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Science and Technology Council","award":["NSTC112\u20102628\u2010E\u2010110\u2010001\u2010MY3"],"award-info":[{"award-number":["NSTC112\u20102628\u2010E\u2010110\u2010001\u2010MY3"]}]},{"name":"National Science and Technology Council","award":["NSTC113\u20102634\u2010F\u2010110\u2010001\u2010MBK"],"award-info":[{"award-number":["NSTC113\u20102634\u2010F\u2010110\u2010001\u2010MBK"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,28]]},"DOI":"10.1145\/3732437.3732756","type":"proceedings-article","created":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T10:33:54Z","timestamp":1760438034000},"page":"103-108","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["An Effective Text Data Augmentation Method for Filtering Inappropriate Webpages"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5327-5304","authenticated-orcid":false,"given":"You-Cheng","family":"Chen","sequence":"first","affiliation":[{"name":"Department of Computer Science and Engineering, National Sun Yat-sen University, Kaohsiung, Taiwan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-4867-3835","authenticated-orcid":false,"given":"Cheng-Han","family":"Hsieh","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, National Sun Yat-sen University, Kaohsiung, Taiwan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8662-7944","authenticated-orcid":false,"given":"Shou-Chuan","family":"Lai","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, Ming Chuan University, Taipei, Taiwan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0140-160X","authenticated-orcid":false,"given":"Whai-En","family":"Chen","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, National Ilan University, Ilan, Taiwan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0128-4052","authenticated-orcid":false,"given":"Chun-Wei","family":"Tsai","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, National Sun Yat-sen University, Kaohsiung, Taiwan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,14]]},"reference":[{"key":"e_1_3_3_1_2_2","first-page":"1","volume-title":"Proceedings of the International Symposium on Innovations in Intelligent Systems and Applications","author":"Akbulut Akhan","year":"2012","unstructured":"Akhan Akbulut, Fatma Patlar, Coskun Bayrak, Engin Mendi, and Josh Hanna. 2012. Agent based pornography filtering system. In Proceedings of the International Symposium on Innovations in Intelligent Systems and Applications. 1\u20135."},{"key":"e_1_3_3_1_3_2","first-page":"7383","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","volume":"34","author":"Anaby-Tavor Ateret","year":"2020","unstructured":"Ateret Anaby-Tavor, Boaz Carmeli, Esther Goldbraich, Amir Kantor, George Kour, Segev Shlomov, Naama Tepper, and Naama Zwerdling. 2020. Do not have enough data? Deep learning to the rescue!. In Proceedings of the AAAI Conference on Artificial Intelligence , Vol.\u00a034. 7383\u20137390."},{"key":"e_1_3_3_1_4_2","unstructured":"David\u00a0M. Blei Andrew\u00a0Y. Ng and Michael\u00a0I. Jordan. 2003. Latent Dirichlet allocation. Journal of Machine Learning Research 3 (2003) 993\u20131022."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Leo Breiman. 2001. Random forests. Machine Learning 45 (2001) 5\u201332.","DOI":"10.1023\/A:1010933404324"},{"key":"e_1_3_3_1_6_2","first-page":"1877","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared\u00a0D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel Ziegler, Jeffrey Wu, Clemens Winter, Chris Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. 2020. Language models are few-shot learners. In Proceedings of the International Conference on Neural Information Processing Systems. 1877\u20131901."},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Ricardo Campos V\u00edtor Mangaravite Arian Pasquali Al\u00edpio Jorge C\u00e9lia Nunes and Adam Jatowt. 2020. YAKE! Keyword extraction from single documents using multiple local features. Information Sciences 509 (2020) 257\u2013289.","DOI":"10.1016\/j.ins.2019.09.013"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Francisco Charte Antonio\u00a0J. Rivera Mar\u00eda\u00a0J. del Jesus and Francisco Herrera. 2015. Addressing imbalance in multilabel classification: Measures and random resampling algorithms. Neurocomputing 163 (2015) 3\u201316.","DOI":"10.1016\/j.neucom.2014.08.091"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939785"},{"key":"e_1_3_3_1_10_2","first-page":"217","volume-title":"Proceedings of the International Conference on Artificial Intelligence in Education","author":"Cochran Keith","year":"2023","unstructured":"Keith Cochran, Clayton Cohn, Jean\u00a0Francois Rouet, and Peter Hastings. 2023. Improving automated evaluation of student text responses using GPT-3.5 for text data augmentation. In Proceedings of the International Conference on Artificial Intelligence in Education. 217\u2013228."},{"key":"e_1_3_3_1_11_2","volume-title":"Database download","author":"contributors Wikipedia","year":"2024","unstructured":"Wikipedia contributors. 2024. Database download. Retrieved Sep. 28, 2024 from https:\/\/dumps.wikimedia.org\/zhwiki\/"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Corinna Cortes and Vladimir Vapnik. 1995. Support-vector networks. Machine learning 20 (1995) 273\u2013297.","DOI":"10.1023\/A:1022627411411"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"crossref","unstructured":"T. Cover and P. Hart. 1967. Nearest neighbor pattern classification. Information Theory 13 1 (1967) 21\u201327.","DOI":"10.1109\/TIT.1967.1053964"},{"key":"e_1_3_3_1_14_2","unstructured":"Jacob Devlin Ming-Wei Chang Kenton Lee and Kristina Toutanova. 2018. BERT: Pre-training of deep bidirectional transformers for language understanding. arxiv:https:\/\/arXiv.org\/abs\/1810.04805"},{"key":"e_1_3_3_1_15_2","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey et\u00a0al. 2024. The Llama 3 herd of models. arxiv:https:\/\/arXiv.org\/abs\/2407.21783"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Susan\u00a0T. Dumais. 2004. Latent semantic analysis. Annual Review of Information Science and Technology 38 (2004) 189\u2013230.","DOI":"10.1002\/aris.1440380105"},{"key":"e_1_3_3_1_17_2","first-page":"968","volume-title":"Proceedings of the International Conference on Findings of the Association for Computational Linguistics","author":"Feng Steven\u00a0Y","year":"2021","unstructured":"Steven\u00a0Y Feng, Varun Gangal, Jason Wei, Sarath Chandar, Soroush Vosoughi, Teruko Mitamura, and Eduard Hovy. 2021. A survey of data augmentation approaches for NLP. In Proceedings of the International Conference on Findings of the Association for Computational Linguistics. 968\u2013988."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"C\u00e9sar Ferri Jos\u00e9 Hern\u00e1ndez-Orallo and R Modroiu. 2009. An experimental comparison of performance measures for classification. Pattern Recognition Letters 30 1 (2009) 27\u201338.","DOI":"10.1016\/j.patrec.2008.08.010"},{"key":"e_1_3_3_1_19_2","unstructured":"Hongyu Guo Yongyi Mao and Richong Zhang. 2019. Augmenting data with mixup for sentence classification: An empirical study. arxiv:https:\/\/arXiv.org\/abs\/1905.08941"},{"key":"e_1_3_3_1_20_2","first-page":"135","volume-title":"Proceedings of the International Conference on Collaboration and Internet Computing","author":"Hancock John","year":"2022","unstructured":"John Hancock, Justin\u00a0M. Johnson, and Taghi\u00a0M. Khoshgoftaar. 2022. A comparative approach to threshold optimization for classifying imbalanced data. In Proceedings of the International Conference on Collaboration and Internet Computing. 135\u2013142."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long short-term memory. Neural Computation 9 8 (1997) 1735\u20131780.","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_3_1_22_2","first-page":"156","volume-title":"Proceedings of the International Conference on Computer and Technology Applications","author":"Izzah Nur","year":"2018","unstructured":"Nur Izzah, Indra Budi, and Samuel Louvan. 2018. Classification of pornographic content on Twitter using support vector machine and Naive Bayes. In Proceedings of the International Conference on Computer and Technology Applications. 156\u2013160."},{"key":"e_1_3_3_1_23_2","unstructured":"Jaehun Jung Peter West Liwei Jiang Faeze Brahman Ximing Lu Jillian Fisher Taylor Sorensen and Yejin Choi. 2023. Impossible distillation: From low-quality model to high-quality dataset & model for summarization and paraphrasing. arxiv:https:\/\/arXiv.org\/abs\/2305.16635"},{"key":"e_1_3_3_1_24_2","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems","author":"Ke Guolin","year":"2017","unstructured":"Guolin Ke, Qi Meng, Thomas Finley, Taifeng Wang, Wei Chen, Weidong Ma, Qiwei Ye, and Tie-Yan Liu. 2017. LightGBM: A highly efficient gradient boosting decision tree. In Proceedings of the International Conference on Neural Information Processing Systems."},{"key":"e_1_3_3_1_25_2","first-page":"1188","volume-title":"Proceedings of the International Conference on Machine Learning","author":"Le Quoc","year":"2014","unstructured":"Quoc Le and Tomas Mikolov. 2014. Distributed representations of sentences and documents. In Proceedings of the International Conference on Machine Learning. 1188\u20131196."},{"key":"e_1_3_3_1_26_2","first-page":"564","volume-title":"Proceedings of the International Conference on Pacific Symposium Biocomputing","author":"Leslie Christina","year":"2001","unstructured":"Christina Leslie, Eleazar Eskin, and William\u00a0Stafford Noble. 2001. The spectrum kernel: A string kernel for SVM protein classification. In Proceedings of the International Conference on Pacific Symposium Biocomputing. 564\u2013575."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"crossref","unstructured":"Ke Li Yalei Wu Yu Nan Pengfei Li and Yang Li. 2017. Hierarchical multi-class classification in multimodal spacecraft data using DNN and weighted support vector machine. Neurocomputing 259 (2017) 55\u201365.","DOI":"10.1016\/j.neucom.2016.08.131"},{"key":"e_1_3_3_1_28_2","unstructured":"Yinhan Liu Myle Ott Naman Goyal Jingfei Du Mandar Joshi Danqi Chen Omer Levy Mike Lewis Luke Zettlemoyer and Veselin Stoyanov. 2019. RoBERTa: A robustly optimized BERT pretraining approach. arxiv:https:\/\/arXiv.org\/abs\/1907.11692"},{"key":"e_1_3_3_1_29_2","first-page":"41","volume-title":"Proceedings of the International Conference on Learning for Text Categorization","author":"McCallum Andrew","year":"1998","unstructured":"Andrew McCallum, Kamal Nigam, et\u00a0al. 1998. A comparison of event models for Naive Bayes text classification. In Proceedings of the International Conference on Learning for Text Categorization. 41\u201348."},{"key":"e_1_3_3_1_30_2","unstructured":"Tomas Mikolov Kai Chen Greg Corrado and Jeffrey Dean. 2013. Efficient estimation of word representations in vector space. arxiv:https:\/\/arXiv.org\/abs\/1301.3781"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","unstructured":"Aishwarya Mujumdar and V Vaidehi. 2019. Diabetes prediction using machine learning algorithms. Procedia Computer Science 165 (2019) 292\u2013299.","DOI":"10.1016\/j.procs.2020.01.047"},{"key":"e_1_3_3_1_32_2","unstructured":"Kevin\u00a0P. Murphy. 2006. Naive Bayes classifiers. University of British Columbia 18 60 (2006) 1\u20138."},{"key":"e_1_3_3_1_33_2","first-page":"1864","volume-title":"Proceedings of the Findings of the Association for Computational Linguistics","author":"Ni Jianmo","year":"2022","unstructured":"Jianmo Ni, Gustavo\u00a0Hernandez Abrego, Noah Constant, Ji Ma, Keith Hall, Daniel Cer, and Yinfei Yang. 2022. Sentence-T5: Scalable sentence encoders from pre-trained text-to-text models. In Proceedings of the Findings of the Association for Computational Linguistics. 1864\u20131874."},{"key":"e_1_3_3_1_34_2","first-page":"691","volume-title":"Proceedings of the International Conference on Emerging Trends in Computing, Communication and Nanotechnology","author":"NirmalaDevi M","year":"2013","unstructured":"M NirmalaDevi, S\u00a0Appavu alias Balamurugan, and U.\u00a0V Swathi. 2013. An amalgam KNN to predict diabetes mellitus. In Proceedings of the International Conference on Emerging Trends in Computing, Communication and Nanotechnology. 691\u2013695."},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"crossref","unstructured":"Lucas Francisco Amaral\u00a0Orosco Pellicer Taynan\u00a0Maier Ferreira and Anna Helena\u00a0Reali Costa. 2023. Data augmentation techniques in natural language processing. Applied Soft Computing 132 (2023) 109803.","DOI":"10.1016\/j.asoc.2022.109803"},{"key":"e_1_3_3_1_36_2","first-page":"1532","volume-title":"Proceedings of the International Conference on Empirical Methods in Natural Language Processing","author":"Pennington Jeffrey","year":"2014","unstructured":"Jeffrey Pennington, Richard Socher, and Christopher\u00a0D. Manning. 2014. GloVe: Global vectors for word representation. In Proceedings of the International Conference on Empirical Methods in Natural Language Processing. 1532\u20131543."},{"key":"e_1_3_3_1_37_2","first-page":"1481","volume-title":"Proceedings of the International Conference on Systems, Man and Cybernetics","author":"Polpinij Jantima","year":"2006","unstructured":"Jantima Polpinij, Anirut Chotthanom, Chumsak Sibunruang, Rapeeporn Chamchong, and Somnuk Puangpronpitag. 2006. Content-based text classifiers for pornographic web filtering. In Proceedings of the International Conference on Systems, Man and Cybernetics. 1481\u20131485."},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"crossref","unstructured":"David\u00a0E. Rumelhart Geoffrey\u00a0E. Hinton and Ronald\u00a0J. Williams. 1986. Learning representations by back-propagating errors. Nature 323 (1986) 533\u2013536.","DOI":"10.1038\/323533a0"},{"key":"e_1_3_3_1_39_2","first-page":"86","volume-title":"Proceedings of the Annual Meeting of the Association for Computational Linguistics","author":"Sennrich Rico","year":"2016","unstructured":"Rico Sennrich, Barry Haddow, and Alexandra Birch. 2016. Improving neural machine translation models with monolingual data. In Proceedings of the Annual Meeting of the Association for Computational Linguistics. 86\u201396."},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"crossref","unstructured":"Kanish Shah Henil Patel Devanshi Sanghvi and Manan Shah. 2020. A comparative analysis of logistic regression random forest and KNN models for the text classification. Augmented Human Research 5 (2020) 1\u201316.","DOI":"10.1007\/s41133-020-00032-0"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"crossref","unstructured":"Karen Sparck\u00a0Jones. 1972. A statistical interpretation of term specificity and its application in retrieval. Journal of Documentation 28 1 (1972) 11\u201321.","DOI":"10.1108\/eb026526"},{"key":"e_1_3_3_1_42_2","first-page":"6382","volume-title":"Proceedings of the International Conference on Empirical Methods in Natural Language Processing and the International Joint Conference on Natural Language Processing","author":"Wei Jason","year":"2019","unstructured":"Jason Wei and Kai Zou. 2019. EDA: Easy data augmentation techniques for boosting performance on text classification tasks. In Proceedings of the International Conference on Empirical Methods in Natural Language Processing and the International Joint Conference on Natural Language Processing. 6382\u20136388."},{"key":"e_1_3_3_1_43_2","volume-title":"2024","author":"Wu I-Chen","unstructured":"I-Chen Wu, Hung-Yi Lee, Yuan-Fu Liao, and Hen-Hsen Huang. TAIDE. 2024. Retrieved Sep. 2, 2024 from https:\/\/taide.tw\/index"},{"key":"e_1_3_3_1_44_2","first-page":"871","volume-title":"Proceedings of the Annual Meeting of the Association for Computational Linguistics","author":"Wu Xing","year":"2022","unstructured":"Xing Wu, Chaochen Gao, Meng Lin, Liangjun Zang, and Songlin Hu. 2022. Text smoothing: Enhance various data augmentation methods on text classification tasks. In Proceedings of the Annual Meeting of the Association for Computational Linguistics. 871\u2013875."},{"key":"e_1_3_3_1_45_2","volume-title":"Proceedings of the International Conference on Learning Representations","author":"Xie Ziang","year":"2017","unstructured":"Ziang Xie, Sida\u00a0I. Wang, Jiwei Li, Daniel L\u00e9vy, Aiming Nie, Dan Jurafsky, and Andrew\u00a0Y. Ng. 2017. Data noising as smoothing in neural network language models. In Proceedings of the International Conference on Learning Representations."},{"key":"e_1_3_3_1_46_2","doi-asserted-by":"crossref","unstructured":"Zhengzheng Xing Jian Pei and Eamonn Keogh. 2010. A brief survey on sequence classification. ACM SIGKDD Explorations Newsletter 12 1 (2010) 40\u201348.","DOI":"10.1145\/1882471.1882478"},{"key":"e_1_3_3_1_47_2","first-page":"347","volume-title":"Proceedings of the International Conference on Advanced Multimedia and Ubiquitous Engineering","author":"Xu Shuo","year":"2017","unstructured":"Shuo Xu, Yan Li, and Zheng Wang. 2017. Bayesian multinomial Na\u00efve Bayes classifier to text classification. In Proceedings of the International Conference on Advanced Multimedia and Ubiquitous Engineering. 347\u2013352."},{"key":"e_1_3_3_1_48_2","first-page":"140","volume-title":"Proceedings of the International Conference on New Trends in Database and Information Systems","author":"Yamoun Lahcen","year":"2023","unstructured":"Lahcen Yamoun, Zahia Guessoum, and Christophe Girard. 2023. Transformers and attention mechanism for website classification and porn detection. In Proceedings of the International Conference on New Trends in Database and Information Systems. 140\u2013149."},{"key":"e_1_3_3_1_49_2","unstructured":"Hongyi Zhang Moustapha Cisse Yann\u00a0N Dauphin and David Lopez-Paz. 2018. Mixup: Beyond empirical risk minimization. arxiv:https:\/\/arXiv.org\/abs\/1710.09412"},{"key":"e_1_3_3_1_50_2","doi-asserted-by":"crossref","unstructured":"Yin Zhang Rong Jin and Zhi-Hua Zhou. 2010. Understanding bag-of-words model: A statistical framework. Machine Learning and Cybernetics 1 (2010) 43\u201352.","DOI":"10.1007\/s13042-010-0001-0"}],"event":{"name":"ICEA 2024: The 2024 International Conference on Intelligent Computing and its Emerging Applicaton","location":"Tokyo Japan","acronym":"ICEA 2024"},"container-title":["Proceedings of the 2024 International Conference on Intelligent Computing and its Emerging Applicaton"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3732437.3732756","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T10:35:58Z","timestamp":1760438158000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3732437.3732756"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,28]]},"references-count":49,"alternative-id":["10.1145\/3732437.3732756","10.1145\/3732437"],"URL":"https:\/\/doi.org\/10.1145\/3732437.3732756","relation":{},"subject":[],"published":{"date-parts":[[2024,11,28]]},"assertion":[{"value":"2025-10-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}