{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,14]],"date-time":"2026-08-14T15:12:29Z","timestamp":1786720349958,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,6,3]],"date-time":"2024-06-03T00:00:00Z","timestamp":1717372800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,6,3]]},"DOI":"10.1145\/3630106.3659033","type":"proceedings-article","created":{"date-parts":[[2024,6,5]],"date-time":"2024-06-05T13:14:21Z","timestamp":1717593261000},"page":"2199-2208","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":48,"title":["A Critical Analysis of the Largest Source for Generative AI Training Data: Common Crawl"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2464-7699","authenticated-orcid":false,"given":"Stefan","family":"Baack","sequence":"first","affiliation":[{"name":"Mozilla Foundation, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,6,5]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442188.3445922"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2304.01373"},{"key":"e_1_3_2_1_3_1","volume-title":"Retrieved","author":"Birhane Abeba","year":"2021","unstructured":"Abeba Birhane, Pratyusha Kalluri, Dallas Card, William Agnew, Ravit Dotan, and Michelle Bao. 2021. The Values Encoded in Machine Learning Research. ArXiv210615590 Cs (June 2021). Retrieved November 26, 2021 from http:\/\/arxiv.org\/abs\/2106.15590"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","unstructured":"Abeba Birhane Vinay Prabhu Sang Han and Vishnu Naresh Boddeti. 2023. On Hate Scaling Laws For Data-Swamps. https:\/\/doi.org\/10.48550\/arXiv.2306.13141","DOI":"10.48550\/arXiv.2306.13141"},{"key":"e_1_3_2_1_5_1","volume-title":"Retrieved","author":"Birhane Abeba","year":"2021","unstructured":"Abeba Birhane, Vinay Uday Prabhu, and Emmanuel Kahembwe. 2021. Multimodal datasets: misogyny, pornography, and malignant stereotypes. ArXiv211001963 Cs (October 2021). Retrieved March 17, 2022 from http:\/\/arxiv.org\/abs\/2110.01963"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1037\/13620-004"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","unstructured":"Tom B. Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell Sandhini Agarwal Ariel Herbert-Voss Gretchen Krueger Tom Henighan Rewon Child Aditya Ramesh Daniel M. Ziegler Jeffrey Wu Clemens Winter Christopher Hesse Mark Chen Eric Sigler Mateusz Litwin Scott Gray Benjamin Chess Jack Clark Christopher Berner Sam McCandlish Alec Radford Ilya Sutskever and Dario Amodei. 2020. Language Models are Few-Shot Learners. https:\/\/doi.org\/10.48550\/arXiv.2005.14165","DOI":"10.48550\/arXiv.2005.14165"},{"key":"e_1_3_2_1_8_1","unstructured":"Common Crawl Foundation. Our Mission. Retrieved November 2 2023 from https:\/\/commoncrawl.org\/mission"},{"key":"e_1_3_2_1_9_1","volume-title":"Alejandro Cremades. Retrieved","author":"Cremades Alejandro","year":"2019","unstructured":"Alejandro Cremades. 2019. Gil Elbaz On Google Acquiring His Company And Turning It Into A $15 Billion Business. Alejandro Cremades. Retrieved October 17, 2023 from https:\/\/alejandrocremades.com\/gil-elbaz-on-google-acquiring-his-company-and-turning-it-into-a-15-billion-business\/"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1177\/20539517211044808"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2007.07399"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","unstructured":"Jesse Dodge Maarten Sap Ana Marasovi\u0107 William Agnew Gabriel Ilharco Dirk Groeneveld Margaret Mitchell and Matt Gardner. 2021. Documenting Large Webtext Corpora: A Case Study on the Colossal Clean Crawled Corpus. https:\/\/doi.org\/10.48550\/arXiv.2104.08758","DOI":"10.48550\/arXiv.2104.08758"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","unstructured":"Leo Gao. 2021. An Empirical Exploration in Quality Filtering of Text Data. https:\/\/doi.org\/10.48550\/arXiv.2109.00698","DOI":"10.48550\/arXiv.2109.00698"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2101.00027"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.10256836"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1511\/2015.114.184"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","unstructured":"Hugo Lauren\u00e7on Lucile Saulnier Thomas Wang Christopher Akiki Albert Villanova del Moral Teven Le Scao Leandro Von Werra Chenghao Mou Eduardo Gonz\u00e1lez Ponferrada Huu Nguyen J\u00f6rg Frohberg Mario \u0160a\u0161ko Quentin Lhoest Angelina McMillan-Major Gerard Dupont Stella Biderman Anna Rogers Loubna Ben allal Francesco De Toni Giada Pistilli Olivier Nguyen Somaieh Nikpoor Maraim Masoud Pierre Colombo Javier de la Rosa Paulo Villegas Tristan Thrush Shayne Longpre Sebastian Nagel Leon Weber Manuel Mu\u00f1oz Jian Zhu Daniel Van Strien Zaid Alyafeai Khalid Almubarak Minh Chien Vu Itziar Gonzalez-Dios Aitor Soroa Kyle Lo Manan Dey Pedro Ortiz Suarez Aaron Gokaslan Shamik Bose David Adelani Long Phan Hieu Tran Ian Yu Suhas Pai Jenny Chim Violette Lepercq Suzana Ilic Margaret Mitchell Sasha Alexandra Luccioni and Yacine Jernite. 2023. The BigScience ROOTS Corpus: A 1.6TB Composite Multilingual Dataset. https:\/\/doi.org\/10.48550\/arXiv.2303.03915","DOI":"10.48550\/arXiv.2303.03915"},{"key":"e_1_3_2_1_18_1","volume-title":"Common Crawl And Unlocking Web Archives For Research. Forbes. Retrieved","author":"Leetaru Kalev","year":"2023","unstructured":"Kalev Leetaru. 2017. Common Crawl And Unlocking Web Archives For Research. Forbes. Retrieved February 6, 2023 from https:\/\/www.forbes.com\/sites\/kalevleetaru\/2017\/09\/28\/common-crawl-and-unlocking-web-archives-for-research\/"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2105.02732"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1177\/1461444815608807"},{"key":"e_1_3_2_1_21_1","volume-title":"Retrieved","year":"2023","unstructured":"McKinsey. 2023. What is generative AI? Retrieved October 13, 2023 from https:\/\/www.mckinsey.com\/featured-insights\/mckinsey-explainers\/what-is-generative-ai"},{"key":"e_1_3_2_1_22_1","volume-title":"The tricky truth about how generative AI uses your data. Vox. Retrieved","author":"Morrison Sara","year":"2023","unstructured":"Sara Morrison. 2023. The tricky truth about how generative AI uses your data. Vox. Retrieved December 4, 2023 from https:\/\/www.vox.com\/technology\/2023\/7\/27\/23808499\/ai-openai-google-meta-data-privacy-nope"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.4324\/9781003341185"},{"key":"e_1_3_2_1_24_1","volume-title":"Retrieved","author":"Nagel Sebastian","year":"2019","unstructured":"Sebastian Nagel. 2019. commoncrawl vs archive.org etc. Common Crawl mailing list. Retrieved October 25, 2023 from https:\/\/groups.google.com\/g\/common-crawl\/c\/RBFAn0o55cY\/m\/68qiLwZMBAAJ"},{"key":"e_1_3_2_1_25_1","volume-title":"Retrieved","author":"Nagel Sebastian","year":"2022","unstructured":"Sebastian Nagel. 2022. Questions about using Common Crawl for another Hugging Face project. Common Crawl mailing list. Retrieved October 25, 2023 from https:\/\/groups.google.com\/g\/common-crawl\/c\/BgPvP6HB2n0\/m\/P-Nw5YoJAQAJ"},{"key":"e_1_3_2_1_26_1","volume-title":"Common Crawl: Data Collection and Use Cases for NLP. In HPLT & NLPL Winter School on Large-Scale Language Modeling and Neural Machine Translation with Web Data. Retrieved","author":"Nagel Sebastian","year":"2023","unstructured":"Sebastian Nagel. 2023. Common Crawl: Data Collection and Use Cases for NLP. In HPLT & NLPL Winter School on Large-Scale Language Modeling and Neural Machine Translation with Web Data. Retrieved August 3, 2023 from http:\/\/nlpl.eu\/skeikampen23\/nagel.230206.pdf"},{"key":"e_1_3_2_1_27_1","volume-title":"Algorithms of oppression: how search engines reinforce racism","author":"Noble Safiya Umoja","unstructured":"Safiya Umoja Noble. 2018. Algorithms of oppression: how search engines reinforce racism. New York university press, New York."},{"key":"e_1_3_2_1_28_1","volume-title":"Retrieved","author":"Orr Will","year":"2023","unstructured":"Will Orr. 2023. 9 Ways To See A Dataset: Datasets as sociotechnical artifacts \u2014 The case of \u201cColossal Cleaned Common Crawl\u201d (C4). Retrieved January 16, 2024 from https:\/\/knowingmachines.org\/publications\/9-ways-to-see\/essays\/c4"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","unstructured":"Will Orr and Kate Crawford. 2023. The social construction of datasets: On the practices processes and challenges of dataset creation for machine learning. (2023). https:\/\/doi.org\/10.31235\/osf.io\/8c9uh","DOI":"10.31235\/osf.io"},{"key":"e_1_3_2_1_30_1","volume-title":"Retrieved","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeff Wu, Xu Jiang, Diogo Almeida, Carroll L. Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John Schulman, Jacob Hilton, Fraser Kelton, Luke Miller, Maddie Simens, Amanda Askell, Peter Welinder, Paul Christiano, Jan Leike, and Ryan Lowe. 2022. Training language models to follow instructions with human feedback. arXiv.org. Retrieved June 22, 2023 from https:\/\/arxiv.org\/abs\/2203.02155v1"},{"key":"e_1_3_2_1_31_1","volume-title":"Retrieved","author":"Owens Trevor","year":"2014","unstructured":"Trevor Owens. 2014. Machine Scale Analysis of Digital Collections: An Interview with Lisa Green of Common Crawl \u2013 Coffeehouse. Retrieved October 17, 2023 from https:\/\/coffeehouse.dataone.org\/2014\/01\/29\/machine-scale-analysis-of-digital-collections-an-interview-with-lisa-green-of-common-crawl\/"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","unstructured":"Guilherme Penedo Quentin Malartic Daniel Hesslow Ruxandra Cojocaru Alessandro Cappelli Hamza Alobeidli Baptiste Pannier Ebtesam Almazrouei and Julien Launay. 2023. The RefinedWeb Dataset for Falcon LLM: Outperforming Curated Corpora with Web Data and Web Data Only. https:\/\/doi.org\/10.48550\/arXiv.2306.01116","DOI":"10.48550\/arXiv.2306.01116"},{"key":"e_1_3_2_1_33_1","volume-title":"Manning","author":"Pennington Jeffrey","year":"2014","unstructured":"Jeffrey Pennington, Richard Socher, and Christopher D. Manning. 2014. GloVe: Global vectors for word representation. In Empirical methods in natural language processing (EMNLP), 2014. 1532\u20131543. Retrieved from http:\/\/www.aclweb.org\/anthology\/D14-1162"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","unstructured":"Vinay Uday Prabhu and Abeba Birhane. 2020. Large image datasets: A pyrrhic win for computer vision? https:\/\/doi.org\/10.48550\/arXiv.2006.16923","DOI":"10.48550\/arXiv.2006.16923"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1910.10683"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.03522"},{"key":"e_1_3_2_1_37_1","volume-title":"Forbes. Retrieved","author":"Rogers Bruce","year":"2014","unstructured":"Bruce Rogers. 2014. Gil Elbaz Builds Factual To Be The World's Data Steward. Forbes. Retrieved October 17, 2023 from https:\/\/www.forbes.com\/sites\/brucerogers\/2014\/05\/29\/gil-elbaz-builds-factual-to-be-the-worlds-data-steward\/"},{"key":"e_1_3_2_1_38_1","volume-title":"Washington Post. Retrieved","author":"Schaul Kevin","year":"2023","unstructured":"Kevin Schaul, Szu Yu Chen, and Nitasha Tiku. 2023. Inside the secret list of websites that make AI like ChatGPT sound smart. Washington Post. Retrieved June 12, 2023 from https:\/\/www.washingtonpost.com\/technology\/interactive\/2023\/ai-chatbot-learning\/"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","unstructured":"Morgan Klaus Scheuerman Emily Denton and Alex Hanna. 2021. Do Datasets Have Politics? Disciplinary Values in Computer Vision Dataset Development. https:\/\/doi.org\/10.1145\/3476058","DOI":"10.1145\/3476058"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588433"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar Aurelien Rodriguez Armand Joulin Edouard Grave and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. https:\/\/doi.org\/10.48550\/arXiv.2302.13971","DOI":"10.48550\/arXiv.2302.13971"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention Is All You Need. https:\/\/doi.org\/10.48550\/arXiv.1706.03762","DOI":"10.48550\/arXiv.1706.03762"},{"key":"e_1_3_2_1_43_1","volume-title":"The Verge. Retrieved","author":"Vincent James","year":"2023","unstructured":"James Vincent. 2023. OpenAI co-founder on company's past approach to openly sharing research: \u201cWe were wrong.\u201d The Verge. Retrieved April 26, 2024 from https:\/\/www.theverge.com\/2023\/3\/15\/23640180\/openai-gpt-4-launch-closed-research-ilya-sutskever-interview"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","unstructured":"Guillaume Wenzek Marie-Anne Lachaux Alexis Conneau Vishrav Chaudhary Francisco Guzm\u00e1n Armand Joulin and Edouard Grave. 2019. CCNet: Extracting High Quality Monolingual Datasets from Web Crawl Data. https:\/\/doi.org\/10.48550\/arXiv.1911.00359","DOI":"10.48550\/arXiv.1911.00359"},{"key":"e_1_3_2_1_45_1","volume-title":"Noema. Retrieved","author":"Williams Adrienne","year":"2022","unstructured":"Adrienne Williams, Milagros Miceli, and Timnit Gebru. 2022. The Exploited Labor Behind Artificial Intelligence. Noema. Retrieved January 20, 2023 from https:\/\/www.noemamag.com\/the-exploited-labor-behind-artificial-intelligence"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2211.05100"},{"key":"e_1_3_2_1_47_1","volume-title":"And Why Open Web Crawls Matter To Developing Big Data Expertise. DATAVERSITY. Retrieved","author":"Zaino Jennifer","year":"2012","unstructured":"Jennifer Zaino. 2012. Common Crawl Founder Gil Elbaz Speaks About New Relationship With Amazon, Semantic Web Projects Using Its Corpus, And Why Open Web Crawls Matter To Developing Big Data Expertise. DATAVERSITY. Retrieved October 17, 2023 from https:\/\/dev.dataversity.net\/common-crawl-founder-gil-elbaz-speaks-about-new-relationship-with-amazon-semantic-web-projects-using-its-corpus-and-why-open-web-crawls-matter-to-developing-big-data-expertise\/"},{"key":"e_1_3_2_1_48_1","volume-title":"Retrieved","author":"Qs Meta Llama","year":"2024","unstructured":"Meta Llama FAQs. Meta Llama Troubleshooting & FAQ. Retrieved April 26, 2024 from https:\/\/llama.meta.com\/faq\/"}],"event":{"name":"FAccT '24: The 2024 ACM Conference on Fairness, Accountability, and Transparency","location":"Rio de Janeiro Brazil","acronym":"FAccT '24"},"container-title":["The 2024 ACM Conference on Fairness, Accountability, and Transparency"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3630106.3659033","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3630106.3659033","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T23:57:07Z","timestamp":1750291027000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3630106.3659033"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,3]]},"references-count":48,"alternative-id":["10.1145\/3630106.3659033","10.1145\/3630106"],"URL":"https:\/\/doi.org\/10.1145\/3630106.3659033","relation":{},"subject":[],"published":{"date-parts":[[2024,6,3]]},"assertion":[{"value":"2024-06-05","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}