{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T15:26:27Z","timestamp":1742916387132,"version":"3.40.3"},"publisher-location":"Cham","reference-count":38,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031790287"},{"type":"electronic","value":"9783031790294"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-79029-4_13","type":"book-chapter","created":{"date-parts":[[2025,1,29]],"date-time":"2025-01-29T22:25:13Z","timestamp":1738189513000},"page":"185-199","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Aroeira: A Curated Corpus for\u00a0the\u00a0Portuguese Language with\u00a0a\u00a0Large Number of\u00a0Tokens"],"prefix":"10.1007","author":[{"given":"Thiago","family":"Lira","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fl\u00e1vio","family":"Ca\u00e7\u00e3o","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cinthia","family":"Souza","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jo\u00e3o","family":"Valentini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Edson","family":"Bollis","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Otavio","family":"Oliveira","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Renato","family":"Almeida","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Marcio","family":"Magalh\u00e3es","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Katia","family":"Poloni","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andre","family":"Oliveira","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lucas","family":"Pellicer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,30]]},"reference":[{"key":"13_CR1","doi-asserted-by":"publisher","unstructured":"Abadji, J., Ortiz\u00a0Su\u00e1rez, P.J., Romary, L., Sagot, B.: Ungoliant: an optimized pipeline for the generation of a very large-scale multilingual web corpus (2021). https:\/\/doi.org\/10.14618\/IDS-PUB-10468","DOI":"10.14618\/IDS-PUB-10468"},{"key":"13_CR2","unstructured":"Abadji, J., Suarez, P.O., Romary, L., Sagot, B.: Towards a cleaner document-oriented multilingual crawled corpus, January 2022"},{"key":"13_CR3","doi-asserted-by":"publisher","unstructured":"Agrawal, A., Singh, S.: Corpus complexity matters in pretraining language models, pp. 257\u2013263, January 2023. https:\/\/doi.org\/10.18653\/v1\/2023.sustainlp-1.20","DOI":"10.18653\/v1\/2023.sustainlp-1.20"},{"key":"13_CR4","unstructured":"Almeida, T.S., Abonizio, H., Nogueira, R., Pires, R.: Sabi$$\\backslash $$\u2019a-2: a new generation of Portuguese large language models. arXiv preprint arXiv:2403.09887 (2024)"},{"key":"13_CR5","doi-asserted-by":"crossref","unstructured":"Barbaresi, A.: Trafilatura: a web scraping library and command-line tool for text discovery and extraction. In: Proceedings of the Joint Conference of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing: System Demonstrations, pp. 122\u2013131. Association for Computational Linguistics (2021). https:\/\/aclanthology.org\/2021.acl-demo.15","DOI":"10.18653\/v1\/2021.acl-demo.15"},{"key":"13_CR6","doi-asserted-by":"publisher","unstructured":"Blodgett, S.L., Barocas, S., Daum\u00e9\u00a0III, H., Wallach, H.: Language (technology) is power: a critical survey of \u201cbias\u201d in NLP. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics. Association for Computational Linguistics (2020). https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.485","DOI":"10.18653\/v1\/2020.acl-main.485"},{"key":"13_CR7","unstructured":"Bommasani, R., et\u00a0al.: On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258 (2021)"},{"key":"13_CR8","unstructured":"Brown, T., et\u00a0al.: Language models are few-shot learners. In: Larochelle, H., Ranzato, M., Hadsell, R., Balcan, M., Lin, H. (eds.) Advances in Neural Information Processing Systems, vol.\u00a033, pp. 1877\u20131901. Curran Associates, Inc. (2020). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/file\/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf"},{"key":"13_CR9","unstructured":"Carmo, D., Piau, M., Campiotti, I., Nogueira, R., Lotufo, R.: PTT5: pretraining and validating the T5 model on Brazilian Portuguese data. arXiv preprint arXiv:2008.09144 (2020)"},{"key":"13_CR10","unstructured":"Computer, T.: RedPajama: an open source recipe to reproduce LLaMa training dataset (2023). https:\/\/github.com\/togethercomputer\/RedPajama-Data"},{"key":"13_CR11","unstructured":"Crespo, M.C.R.M., et\u00a0al.: Carolina: a general corpus of contemporary Brazilian Portuguese with provenance, typology and versioning information, March 2023"},{"key":"13_CR12","unstructured":"Finardi, P., Viegas, J.D., Ferreira, G.T., Mansano, A.F., Carid\u00e1, V.F.: BERTa$$\\backslash $$\u2019u: Ita$$\\backslash $$\u2019u BERT for digital customer service. arXiv preprint arXiv:2101.12015 (2021)"},{"key":"13_CR13","doi-asserted-by":"publisher","unstructured":"Gallegos, I.O., et\u00a0al.: Bias and fairness in large language models: a survey. Comput. Linguist. 1\u201379 (2024). https:\/\/doi.org\/10.1162\/coli_a_00524","DOI":"10.1162\/coli_a_00524"},{"key":"13_CR14","unstructured":"Gao, L., et\u00a0al.: The pile: an 800 GB dataset of diverse text for language modeling, December 2020"},{"key":"13_CR15","unstructured":"Garcia, G.L., et\u00a0al.: Introducing bode: a fine-tuned large language model for Portuguese prompt-based task. arXiv preprint arXiv:2401.02909 (2024)"},{"key":"13_CR16","doi-asserted-by":"crossref","unstructured":"Garimella, A., Mihalcea, R., Amarnath, A.: Demographic-aware language model fine-tuning as a bias mitigation technique. In: He, Y., et\u00a0al. (eds.) Proceedings of the 2nd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics and the 12th International Joint Conference on Natural Language Processing (Volume 2: Short Papers), pp. 311\u2013319. Association for Computational Linguistics, Online only, November 2022. https:\/\/aclanthology.org\/2022.aacl-short.38","DOI":"10.18653\/v1\/2022.aacl-short.38"},{"key":"13_CR17","unstructured":"Habernal, I., Zayed, O., Gurevych, I.: C4Corpus: multilingual web-size corpus with free license. In: Calzolari, N., et\u00a0al. (eds.) Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC 2016), pp. 914\u2013922. European Language Resources Association (ELRA), Portoro\u017e, Slovenia, May 2016. https:\/\/aclanthology.org\/L16-1146"},{"key":"13_CR18","unstructured":"Hoffmann, J., et\u00a0al.: An empirical analysis of compute-optimal large language model training. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems, vol.\u00a035, pp. 30016\u201330030. Curran Associates, Inc. (2022). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/c1e2faff6f588870935f114ebe04a3e5-Paper-Conference.pdf"},{"key":"13_CR19","unstructured":"Jiang, A.Q., et\u00a0al.: Mistral 7B. arXiv preprint arXiv:2310.06825 (2023)"},{"key":"13_CR20","doi-asserted-by":"crossref","unstructured":"Joulin, A., Grave, E., Bojanowski, P., Mikolov, T.: Bag of tricks for efficient text classification. In: Proceedings of the 15th Conference of the European Chapter of the Association for Computational Linguistics: Volume 2, Short Papers, pp. 427\u2013431. Association for Computational Linguistics, April 2017","DOI":"10.18653\/v1\/E17-2068"},{"key":"13_CR21","unstructured":"Kenton, J.D.M.W.C., Toutanova, L.K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of naacL-HLT, vol.\u00a01, p.\u00a02 (2019)"},{"key":"13_CR22","unstructured":"Larcher, C., Piau, M., Finardi, P., Gengo, P., Esposito, P., Carid\u00e1, V.: Cabrita: closing the gap for Foreign languages. arXiv preprint arXiv:2308.11878 (2023)"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"Liu, Y., Cao, J., Liu, C., Ding, K., Jin, L.: Datasets for large language models: a comprehensive survey. arXiv preprint arXiv:2402.18041 (2024)","DOI":"10.21203\/rs.3.rs-3996137\/v1"},{"key":"13_CR24","doi-asserted-by":"publisher","unstructured":"Mei, K., Fereidooni, S., Caliskan, A.: Bias against 93 stigmatized groups in masked language models and downstream sentiment classification tasks. In: 2023 ACM Conference on Fairness, Accountability, and Transparency, FAccT 2023. ACM, June 2023. https:\/\/doi.org\/10.1145\/3593013.3594109","DOI":"10.1145\/3593013.3594109"},{"key":"13_CR25","doi-asserted-by":"publisher","unstructured":"Nadeem, M., Bethke, A., Reddy, S.: StereoSet: measuring stereotypical bias in pretrained language models. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers). Association for Computational Linguistics (2021). https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.416","DOI":"10.18653\/v1\/2021.acl-long.416"},{"key":"13_CR26","doi-asserted-by":"publisher","unstructured":"Nangia, N., Vania, C., Bhalerao, R., Bowman, S.R.: Crows-pairs: a challenge dataset for measuring social biases in masked language models. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP). Association for Computational Linguistics (2020). https:\/\/doi.org\/10.18653\/v1\/2020.emnlp-main.154","DOI":"10.18653\/v1\/2020.emnlp-main.154"},{"key":"13_CR27","doi-asserted-by":"crossref","unstructured":"Overwijk, A., Xiong, C., Callan, J.: ClueWeb22: 10 billion web documents with rich information. In: Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 3360\u20133362 (2022)","DOI":"10.1145\/3477495.3536321"},{"key":"13_CR28","doi-asserted-by":"crossref","unstructured":"Pires, R., Abonizio, H., Almeida, T.S., Nogueira, R.: Sabi\u00e1: Portuguese large language models. In: Brazilian Conference on Intelligent Systems, pp. 226\u2013240 (2023)","DOI":"10.1007\/978-3-031-45392-2_15"},{"key":"13_CR29","unstructured":"Rae, J.W., et\u00a0al.: Scaling language models: methods, analysis & insights from training gopher, December 2021"},{"key":"13_CR30","unstructured":"Raffel, C., et\u00a0al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(140), 1\u201367 (2020). http:\/\/jmlr.org\/papers\/v21\/20-074.html"},{"issue":"2","key":"13_CR31","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1017\/s0305000900012885","volume":"14","author":"B Richards","year":"1987","unstructured":"Richards, B.: Type\/token ratios: what do they really tell us? J. Child Lang. 14(2), 201\u2013209 (1987). https:\/\/doi.org\/10.1017\/s0305000900012885","journal-title":"J. Child Lang."},{"key":"13_CR32","doi-asserted-by":"crossref","unstructured":"Shin, S., et al.: On the effect of pretraining corpora on in-context learning by a large-scale language model (2022)","DOI":"10.18653\/v1\/2022.naacl-main.380"},{"key":"13_CR33","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"403","DOI":"10.1007\/978-3-030-61377-8_28","volume-title":"Intelligent Systems","author":"F Souza","year":"2020","unstructured":"Souza, F., Nogueira, R., Lotufo, R.: BERTimbau: pretrained BERT models for Brazilian Portuguese. In: Cerri, R., Prati, R.C. (eds.) BRACIS 2020. LNCS (LNAI), vol. 12319, pp. 403\u2013417. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-61377-8_28"},{"key":"13_CR34","unstructured":"Virtanen, A., et\u00a0al.: Multilingual is not enough: BERT for Finnish, December 2019"},{"key":"13_CR35","unstructured":"Wagner\u00a0Filho, J.A., et\u00a0al.: The brWaC corpus: a new open resource for Brazilian Portuguese. In: Calzolari, N., et\u00a0al. (eds.) Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC 2018). European Language Resources Association (ELRA), Miyazaki, Japan, May 2018. https:\/\/aclanthology.org\/L18-1686"},{"key":"13_CR36","unstructured":"Xu, L., Zhang, X., Dong, Q.: CLUECorpus2020: a large-scale Chinese corpus for pre-training language model. arXiv preprint arXiv:2003.01355 (2020)"},{"key":"13_CR37","unstructured":"Youmans, G.: Measuring lexical style and competence: the type-token vocabulary curve. Style 24(4), 584\u2013599 (1990). http:\/\/www.jstor.org\/stable\/42946163"},{"key":"13_CR38","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1016\/j.aiopen.2021.06.001","volume":"2","author":"S Yuan","year":"2021","unstructured":"Yuan, S., et al.: WuDaoCorpora: a super large-scale Chinese corpora for pre-training language models. AI Open 2, 65\u201368 (2021)","journal-title":"AI Open"}],"container-title":["Lecture Notes in Computer Science","Intelligent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-79029-4_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,29]],"date-time":"2025-01-29T22:25:25Z","timestamp":1738189525000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-79029-4_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031790287","9783031790294"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-79029-4_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"30 January 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"Our Aroeira corpus is available for download in the Hugging Face repository:  and is under the CC-BY-NC 4.0 license.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Data Availability"}},{"value":"BRACIS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brazilian Conference on Intelligent Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Bel\u00e9m do Par\u00e1","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brazil","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 November 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 November 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"34","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"bracis2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}