{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T13:30:11Z","timestamp":1785763811714,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":57,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,2,22]]},"DOI":"10.1145\/3773966.3777990","type":"proceedings-article","created":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T17:50:01Z","timestamp":1771264201000},"page":"860-870","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["From Data to Model in Bias: A Statistical Analysis of Political Bias in the C4 Corpus and Its Impact on LLMs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-9539-1875","authenticated-orcid":false,"given":"Jaebeom","family":"You","sequence":"first","affiliation":[{"name":"Graduate School of Data Science, Seoul National University of Science and Technology, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7104-8208","authenticated-orcid":false,"given":"Jaewon","family":"Lee","sequence":"additional","affiliation":[{"name":"Department of AI, Big Data &amp; Management, Kookmin University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2234-190X","authenticated-orcid":false,"given":"Sehun","family":"Lee","sequence":"additional","affiliation":[{"name":"Department of Economics, Korea University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1125-6533","authenticated-orcid":false,"given":"Hyuk-Yoon","family":"Kwon","sequence":"additional","affiliation":[{"name":"Graduate School of Data Science, Seoul National University of Science and Technology, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,2,21]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency. 2199-2208","author":"Baack Stefan","year":"2024","unstructured":"Stefan Baack. 2024. A critical analysis of the largest source for generative ai training data: Common crawl. In Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency. 2199-2208."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.151"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442188.3445922"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.2517-6161.1995.tb02031.x"},{"key":"e_1_3_2_1_5_1","volume-title":"Vinay Uday Prabhu, and Emmanuel Kahembwe","author":"Birhane Abeba","year":"2021","unstructured":"Abeba Birhane, Vinay Uday Prabhu, and Emmanuel Kahembwe. 2021. Multimodal datasets: misogyny, pornography, and malignant stereotypes. arXiv preprint arXiv:2110.01963 (2021)."},{"key":"e_1_3_2_1_6_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems Vol. 33 (2020) 1877-1901."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00710"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing. 17140-17161","author":"Chen Kai","year":"2024","unstructured":"Kai Chen, Zihao He, Jun Yan, Taiwei Shi, and Kristina Lerman. 2024. How Susceptible are Large Language Models to Ideological Manipulation?. In Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing. 17140-17161."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3353056"},{"key":"e_1_3_2_1_10_1","volume-title":"Joint European Conference on Machine Learning and Knowledge Discovery in Databases. Springer, 638-654","author":"Delobelle Pieter","year":"2022","unstructured":"Pieter Delobelle and Bettina Berendt. 2022. Fairdistillation: mitigating stereotyping in language models. In Joint European Conference on Machine Learning and Knowledge Discovery in Databases. Springer, 638-654."},{"key":"e_1_3_2_1_11_1","volume-title":"Qlora: Efficient finetuning of quantized llms. Advances in neural information processing systems","author":"Dettmers Tim","year":"2023","unstructured":"Tim Dettmers, Artidoro Pagnoni, Ari Holtzman, and Luke Zettlemoyer. 2023. Qlora: Efficient finetuning of quantized llms. Advances in neural information processing systems, Vol. 36 (2023), 10088-10115."},{"key":"e_1_3_2_1_12_1","first-page":"371","volume-title":"Proceedings of the AAAI\/ACM Conference on AI, Ethics, and Society","volume":"7","author":"D\u00edaz Mark","year":"2024","unstructured":"Mark D\u00edaz, Sunipa Dev, Emily Reif, Emily Denton, and Vinodkumar Prabhakaran. 2024. SoUnD Framework: Analyzing (So) cial Representation in (Un) structured (D) ata. In Proceedings of the AAAI\/ACM Conference on AI, Ethics, and Society, Vol. 7. 371-383."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.98"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1002"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.656"},{"key":"e_1_3_2_1_16_1","volume-title":"Should ChatGPT be biased? Challenges and risks of bias in large language models. First Monday","author":"Ferrara Emilio","year":"2023","unstructured":"Emilio Ferrara. 2023. Should ChatGPT be biased? Challenges and risks of bias in large language models. First Monday (2023)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.328"},{"key":"e_1_3_2_1_18_1","unstructured":"Leo Gao Stella Biderman Sid Black Laurence Golding Travis Hoppe Charles Foster Jason Phang Horace He Anish Thite Noa Nabeshima et al. 2020. The pile: An 800gb dataset of diverse text for language modeling. arXiv preprint arXiv:2101.00027 (2020)."},{"key":"e_1_3_2_1_19_1","first-page":"4534","article-title":"He is very intelligent, she is very beautiful? on mitigating social biases in language modelling and generation. In Findings of the association for computational linguistics","volume":"2021","author":"Garimella Aparna","year":"2021","unstructured":"Aparna Garimella, Akhash Amarnath, Kiran Kumar, Akash Pramod Yalla, Niyati Chhaya, Balaji Vasan Srinivasan, et al., 2021. He is very intelligent, she is very beautiful? on mitigating social biases in language modelling and generation. In Findings of the association for computational linguistics: ACL-IJCNLP 2021. 4534-4545.","journal-title":"ACL-IJCNLP"},{"key":"e_1_3_2_1_20_1","first-page":"2454","volume-title":"Proceedings of the International AAAI Conference on Web and Social Media","volume":"19","author":"Hagar Nick","year":"2025","unstructured":"Nick Hagar and Jack Bandy. 2025. Practical Datasets for Analyzing LLM Corpora Derived from Common Crawl. In Proceedings of the International AAAI Conference on Web and Social Media, Vol. 19. 2454-2464."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689904.3694702"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.411"},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics, Owen Rambow, Leo Wanner, Marianna Apidianaki, Hend Al-Khalifa","author":"Lee Noah","year":"2025","unstructured":"Noah Lee, Jiwoo Hong, and James Thorne. 2025. Evaluating the Consistency of LLM Evaluators. In Proceedings of the 31st International Conference on Computational Linguistics, Owen Rambow, Leo Wanner, Marianna Apidianaki, Hend Al-Khalifa, Barbara Di Eugenio, and Steven Schockaert (Eds.). Association for Computational Linguistics, Abu Dhabi, UAE, 10650-10659. https:\/\/aclanthology.org\/2025.coling-main.710\/"},{"key":"e_1_3_2_1_24_1","volume-title":"ICLR 2024 Workshop on Reliable and Responsible Foundation Models.","author":"Li Jingling","year":"2024","unstructured":"Jingling Li, Zeyu Tang, Xiaoyu Liu, Peter Spirtes, Kun Zhang, Liu Leqi, and Yang Liu. 2024. Steering llms towards unbiased responses: A causality-guided debiasing framework. In ICLR 2024 Workshop on Reliable and Responsible Foundation Models."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.488"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2021.103654"},{"key":"e_1_3_2_1_27_1","volume-title":"International Conference on Artificial Intelligence and Statistics. PMLR, 1036-1044","author":"Liu Shuo Shuo","year":"2024","unstructured":"Shuo Shuo Liu. 2024. Unified transfer learning in high-dimensional linear regression. In International Conference on Artificial Intelligence and Statistics. PMLR, 1036-1044."},{"key":"e_1_3_2_1_28_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)."},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 2: Short Papers). 182-189","author":"Luccioni Alexandra","year":"2021","unstructured":"Alexandra Luccioni and Joseph Viviano. 2021. What's in the Box? An Analysis of Undesirable Content in the Common Crawl Corpus. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 2: Short Papers). 182-189."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 33rd acm international conference on information and knowledge management. 3922-3926","author":"Lunardi Riccardo","year":"2024","unstructured":"Riccardo Lunardi, David La Barbera, and Kevin Roitero. 2024. The elusiveness of detecting political bias in language models. In Proceedings of the 33rd acm international conference on information and knowledge management. 3922-3926."},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP). 5267-5275","author":"Maudslay Rowan Hall","year":"2019","unstructured":"Rowan Hall Maudslay, Hila Gonen, Ryan Cotterell, and Simone Teufel. 2019. It's All in the Name: Mitigating Gender Bias with Name-Based Counterfactual Data Substitution. In Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP). 5267-5275."},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 1878-1898","author":"Meade Nicholas","year":"2022","unstructured":"Nicholas Meade, Elinor Poole-Dayan, and Siva Reddy. 2022. An Empirical Survey of the Effectiveness of Debiasing Techniques for Pre-trained Language Models. In Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 1878-1898."},{"key":"e_1_3_2_1_33_1","volume-title":"Towards safer pretraining: Analyzing and filtering harmful content in webscale datasets for responsible llms. arXiv preprint arXiv:2505.02009","author":"Mendu Sai Krishna","year":"2025","unstructured":"Sai Krishna Mendu, Harish Yenala, Aditi Gulati, Shanu Kumar, and Parag Agrawal. 2025. Towards safer pretraining: Analyzing and filtering harmful content in webscale datasets for responsible llms. arXiv preprint arXiv:2505.02009 (2025)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.3390\/computers13060141"},{"key":"e_1_3_2_1_35_1","first-page":"79155","article-title":"The refinedweb dataset for falcon llm: Outperforming curated corpora with web data only","volume":"36","author":"Penedo Guilherme","year":"2023","unstructured":"Guilherme Penedo, Quentin Malartic, Daniel Hesslow, Ruxandra Cojocaru, Hamza Alobeidli, Alessandro Cappelli, Baptiste Pannier, Ebtesam Almazrouei, and Julien Launay. 2023. The refinedweb dataset for falcon llm: Outperforming curated corpora with web data only. Advances in Neural Information Processing Systems, Vol. 36 (2023), 79155-79172.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics: Student Research Workshop. 223-228","author":"Qian Yusu","year":"2019","unstructured":"Yusu Qian, Urwa Muaz, Ben Zhang, and Jae Won Hyun. 2019. Reducing Gender Bias in Word-Level Language Models with a Gender-Equalizing Loss Function. In Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics: Student Research Workshop. 223-228."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1057\/s41599-024-03609-x"},{"key":"e_1_3_2_1_38_1","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et al. 2019. Language models are unsupervised multitask learners. OpenAI blog Vol. 1 8 (2019) 9."},{"key":"e_1_3_2_1_39_1","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of machine learning research, Vol. 21, 140 (2020), 1-67.","journal-title":"Journal of machine learning research"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.647"},{"key":"e_1_3_2_1_41_1","first-page":"7346","article-title":"The effect of sampling temperature on problem solving in large language models. In Findings of the association for computational linguistics","volume":"2024","author":"Renze Matthew","year":"2024","unstructured":"Matthew Renze. 2024. The effect of sampling temperature on problem solving in large language models. In Findings of the association for computational linguistics: EMNLP 2024. 7346-7356.","journal-title":"EMNLP"},{"key":"e_1_3_2_1_42_1","first-page":"1","article-title":"Assessing political bias in large language models","volume":"8","author":"Rettenberger Luca","year":"2025","unstructured":"Luca Rettenberger, Markus Reischl, and Mark Schutera. 2025. Assessing political bias in large language models. Journal of Computational Social Science, Vol. 8, 2 (2025), 1-17.","journal-title":"Journal of Computational Social Science"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","first-page":"e0306621","DOI":"10.1371\/journal.pone.0306621","article-title":"The political preferences of LLMs","volume":"19","author":"Rozado David","year":"2024","unstructured":"David Rozado. 2024. The political preferences of LLMs. PloS one, Vol. 19, 7 (2024), e0306621.","journal-title":"PloS one"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00434"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF01068419"},{"key":"e_1_3_2_1_46_1","volume-title":"1st International Workshop on AI Governance (AIGOV) in conjunction with the Thirty-Third International Joint Conference on Artificial Intelligence.","author":"Serouis Ibrahim Mohamed","year":"2024","unstructured":"Ibrahim Mohamed Serouis and Florence S\u00e8des. 2024. Exploring large language models for bias mitigation and fairness. In 1st International Workshop on AI Governance (AIGOV) in conjunction with the Thirty-Third International Joint Conference on Artificial Intelligence."},{"key":"e_1_3_2_1_47_1","volume-title":"Slimpajama-dc: Understanding data combinations for llm training. arXiv preprint arXiv:2309.10818","author":"Shen Zhiqiang","year":"2023","unstructured":"Zhiqiang Shen, Tianhua Tao, Liqun Ma, Willie Neiswanger, Zhengzhong Liu, Hongyi Wang, Bowen Tan, Joel Hestness, Natalia Vassilieva, Daria Soboleva, et al., 2023. Slimpajama-dc: Understanding data combinations for llm training. arXiv preprint arXiv:2309.10818 (2023)."},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the 4th Workshop on Gender Bias in Natural Language Processing (GeBNLP). 112-120","author":"Tal Yarden","year":"2022","unstructured":"Yarden Tal, Inbal Magar, and Roy Schwartz. 2022. Fewer Errors, but More Stereotypes? The Effect of Model Size on Gender Bias. In Proceedings of the 4th Workshop on Gender Bias in Natural Language Processing (GeBNLP). 112-120."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-2067"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.167"},{"key":"e_1_3_2_1_51_1","volume-title":"Chi, and Slav Petrov","author":"Webster Kellie","year":"2020","unstructured":"Kellie Webster, Xuezhi Wang, Ian Tenney, Alex Beutel, Emily Pitler, Ellie Pavlick, Jilin Chen, Ed Chi, and Slav Petrov. 2020. Measuring and reducing gendered correlations in pre-trained models. arXiv preprint arXiv:2010.06032 (2020)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.l2m2-1.16"},{"key":"e_1_3_2_1_53_1","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing. 10703-10727","author":"Yifei Li","year":"2023","unstructured":"Li Yifei, Lyle Ungar, and Jo ao Sedoc. 2023. Conceptor-Aided Debiasing of Large Language Models. In Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing. 10703-10727."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1404"},{"key":"e_1_3_2_1_55_1","volume-title":"Proceedings of the 26th Conference on Computational Natural Language Learning (CoNLL). 27-39","author":"Yoder Michael","year":"2022","unstructured":"Michael Yoder, Lynnette Ng, David West Brown, and Kathleen M Carley. 2022. How Hate Speech Varies by Target Identity: A Computational Analysis. In Proceedings of the 26th Conference on Computational Natural Language Learning (CoNLL). 27-39."},{"key":"e_1_3_2_1_56_1","unstructured":"Yongchao Zhou Andrei Ioan Muresanu Ziwen Han Keiran Paster Silviu Pitis Harris Chan and Jimmy Ba. [n.d.]. Large language models are human-level prompt engineers. In The eleventh international conference on learning representations."},{"key":"e_1_3_2_1_57_1","volume-title":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics. 1651-1661","author":"Zmigrod Ran","year":"2019","unstructured":"Ran Zmigrod, Sabrina J Mielke, Hanna Wallach, and Ryan Cotterell. 2019. Counterfactual Data Augmentation for Mitigating Gender Stereotypes in Languages with Rich Morphology. In Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics. 1651-1661."}],"event":{"name":"WSDM '26:The Nineteenth ACM International Conference on Web Search and Data Mining","location":"Boise ID USA","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGIR ACM Special Interest Group on Information Retrieval","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the Nineteenth ACM International Conference on Web Search and Data Mining"],"original-title":[],"deposited":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T17:56:28Z","timestamp":1771264588000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3773966.3777990"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,21]]},"references-count":57,"alternative-id":["10.1145\/3773966.3777990","10.1145\/3773966"],"URL":"https:\/\/doi.org\/10.1145\/3773966.3777990","relation":{},"subject":[],"published":{"date-parts":[[2026,2,21]]},"assertion":[{"value":"2026-02-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}