{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T01:21:54Z","timestamp":1784856114247,"version":"3.55.0"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030613761","type":"print"},{"value":"9783030613778","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-61377-8_28","type":"book-chapter","created":{"date-parts":[[2020,10,15]],"date-time":"2020-10-15T19:04:06Z","timestamp":1602788646000},"page":"403-417","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":403,"title":["BERTimbau: Pretrained BERT Models for Brazilian Portuguese"],"prefix":"10.1007","author":[{"given":"F\u00e1bio","family":"Souza","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rodrigo","family":"Nogueira","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5652-0852","authenticated-orcid":false,"given":"Roberto","family":"Lotufo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2020,10,13]]},"reference":[{"key":"28_CR1","unstructured":"Akbik, A., Blythe, D., Vollgraf, R.: Contextual string embeddings for sequence labeling. In: COLING 2018, 27th International Conference on Computational Linguistics, pp. 1638\u20131649 (2018)"},{"key":"28_CR2","unstructured":"Baly, F., Hajj, H., et al.: Arabert: transformer-based model for Arabic language understanding. In: Proceedings of the 4th Workshop on Open-Source Arabic Corpora and Processing Tools, with a Shared Task on Offensive Language Detection, pp. 9\u201315 (2020)"},{"key":"28_CR3","unstructured":"Castro, P., Felix, N., Soares, A.: Contextual representations and semi-supervised named entity recognition for Portuguese language, September 2019"},{"key":"28_CR4","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1007\/978-3-319-99722-3_9","volume-title":"Computational Processing of the Portuguese Language","author":"PV Quinta de Castro","year":"2018","unstructured":"Quinta de Castro, P.V., F\u00e9lix Felipe da Silva, N., da Silva Soares, A.: Portuguese named entity recognition using LSTM-CRF. In: Villavicencio, A., et al. (eds.) PROPOR 2018. LNCS (LNAI), vol. 11122, pp. 83\u201392. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-319-99722-3_9"},{"key":"28_CR5","unstructured":"Ca\u00f1ete, J., Chaperon, G., Fuentes, R., P\u00e9rez, J.: Spanish pre-trained BERT model and evaluation data. In: To Appear in PML4DC at ICLR 2020 (2020)"},{"key":"28_CR6","doi-asserted-by":"crossref","unstructured":"Delobelle, P., Winters, T., Berendt, B.: RobBERT: a Dutch RoBERTa-based language model. arXiv preprint arXiv:2001.06286 (2020)","DOI":"10.18653\/v1\/2020.findings-emnlp.292"},{"key":"28_CR7","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"28_CR8","unstructured":"Fonseca, E., Alvarenga, J.P.R.: Wide and deep transformers applied to semantic relatedness and textual entailment. In: Oliveira et al. [22], pp. 68\u201376. http:\/\/ceur-ws.org\/Vol-2583\/"},{"key":"28_CR9","doi-asserted-by":"crossref","unstructured":"Gururangan, S., Marasovi\u0107, A., Swayamdipta, S., Lo, K., Beltagy, I., Downey, D., Smith, N.A.: Don\u2019t stop pretraining: adapt language models to domains and tasks. arXiv preprint arXiv:2004.10964 (2020)","DOI":"10.18653\/v1\/2020.acl-main.740"},{"key":"28_CR10","doi-asserted-by":"crossref","unstructured":"Howard, J., Ruder, S.: Universal language model fine-tuning for text classification. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 328\u2013339 (2018)","DOI":"10.18653\/v1\/P18-1031"},{"key":"28_CR11","unstructured":"Kaplan, J., et al.: Scaling laws for neural language models. arXiv preprint arXiv:2001.08361 (2020)"},{"key":"28_CR12","doi-asserted-by":"crossref","unstructured":"Kudo, T., Richardson, J.: SentencePiece: a simple and language independent subword tokenizer and detokenizer for neural text processing. arXiv preprint arXiv:1808.06226 (2018)","DOI":"10.18653\/v1\/D18-2012"},{"key":"28_CR13","unstructured":"Kuratov, Y., Arkhipov, M.: Adaptation of deep bidirectional multilingual transformers for russian language. arXiv preprint arXiv:1905.07213 (2019)"},{"key":"28_CR14","unstructured":"Lafferty, J.D., McCallum, A., Pereira, F.C.N.: Conditional random fields: probabilistic models for segmenting and labeling sequence data. In: Proceedings of the Eighteenth International Conference on Machine Learning, ICML 2001, p. 282\u2013289. Morgan Kaufmann Publishers Inc., San Francisco (2001)"},{"key":"28_CR15","unstructured":"Lample, G., Ballesteros, M., Subramanian, S., Kawakami, K., Dyer, C.: Neural architectures for named entity recognition. arXiv preprint arXiv:1603.01360 (2016). http:\/\/arxiv.org\/abs\/1603.01360, version 3"},{"key":"28_CR16","unstructured":"Lan, Z., Chen, M., Goodman, S., Gimpel, K., Sharma, P., Soricut, R.: Albert: a lite BERT for self-supervised learning of language representations. arXiv preprint arXiv:1909.11942 (2019)"},{"key":"28_CR17","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"28_CR18","unstructured":"Liu, Y., et al.: Roberta: a robustly optimized BERT pretraining approach. arXiv preprint arXiv:1907.11692 (2019)"},{"key":"28_CR19","doi-asserted-by":"crossref","unstructured":"Martin, L., et al.: CamemBERT: a tasty French language model. arXiv preprint arXiv:1911.03894 (2019)","DOI":"10.18653\/v1\/2020.acl-main.645"},{"key":"28_CR20","unstructured":"Mikolov, T., Chen, K., Corrado, G., Dean, J.: Efficient estimation of word representations in vector space. arXiv preprint arXiv:1301.3781 (2013)"},{"key":"28_CR21","doi-asserted-by":"crossref","unstructured":"Nguyen, D.Q., Nguyen, A.T.: PhoBERT: pre-trained language models for Vietnamese. arXiv preprint arXiv:2003.00744 (2020)","DOI":"10.18653\/v1\/2020.findings-emnlp.92"},{"key":"28_CR22","unstructured":"Oliveira, H.G., Real, L., Fonseca, E. (eds.): Proceedings of the ASSIN 2 Shared Task: Evaluating Semantic Textual Similarity and Textual Entailment in Portuguese, Extended Semantic Web Conference. No. 2583 in CEUR Workshop Proceedings (2020). http:\/\/ceur-ws.org\/Vol-2583\/"},{"key":"28_CR23","doi-asserted-by":"publisher","unstructured":"Pennington, J., Socher, R., Manning, C.: Glove: global vectors for word representation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543. Association for Computational Linguistics, Doha, Qatar, October 2014. https:\/\/doi.org\/10.3115\/v1\/D14-1162. https:\/\/www.aclweb.org\/anthology\/D14-1162","DOI":"10.3115\/v1\/D14-1162"},{"key":"28_CR24","doi-asserted-by":"crossref","unstructured":"Peters, M., et al.: Deep contextualized word representations. In: Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers), pp. 2227\u20132237 (2018)","DOI":"10.18653\/v1\/N18-1202"},{"key":"28_CR25","doi-asserted-by":"crossref","unstructured":"Peters, M.E., Ruder, S., Smith, N.A.: To tune or not to tune? adapting pretrained representations to diverse tasks. In: Proceedings of the 4th Workshop on Representation Learning for NLP (RepL4NLP-2019), pp. 7\u201314 (2019)","DOI":"10.18653\/v1\/W19-4302"},{"key":"28_CR26","unstructured":"Polignano, M., Basile, P., de Gemmis, M., Semeraro, G., Basile, V.: AlBERTo: Italian BERT language understanding model for NLP challenging tasks based on Tweets. In: Proceedings of the Sixth Italian Conference on Computational Linguistics (CLiC-it 2019), vol. 2481. CEUR (2019). https:\/\/www.scopus.com\/inward\/record.uri?eid=2-s2.0-85074851349&partnerID=40&md5=7abed946e06f76b3825ae5e294ffac14"},{"key":"28_CR27","unstructured":"Radford, A., Narasimhan, K., Salimans, T., Sutskever, I.: Improving language understanding with unsupervised learning. Tech. rep. OpenAI (2018)"},{"issue":"8","key":"28_CR28","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I.: Language models are unsupervised multitask learners. OpenAI Blog 1(8), 9 (2019)","journal-title":"OpenAI Blog"},{"key":"28_CR29","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. arXiv preprint arXiv:1910.10683 (2019)"},{"key":"28_CR30","doi-asserted-by":"publisher","unstructured":"Real, L., Fonseca, E., Gon\u00e7alo Oliveira, H.: The ASSIN 2 shared task: a quick overview, pp. 406\u2013412 (02 2020). https:\/\/doi.org\/10.1007\/978-3-030-41505-1_39","DOI":"10.1007\/978-3-030-41505-1_39"},{"key":"28_CR31","unstructured":"Rodrigues, R., da Silva, J., Castro, P., Felix, N., Soares, A.: Multilingual transformer ensembles for Portuguese natural language tasks. In: Oliveira et al. [22], pp. 27\u201338. http:\/\/ceur-ws.org\/Vol-2583\/"},{"key":"28_CR32","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1007\/978-3-030-41505-1_23","volume-title":"Computational Processing of the Portuguese Language","author":"RC Rodrigues","year":"2020","unstructured":"Rodrigues, R.C., Rodrigues, J., de Castro, P.V.Q., da Silva, N.F.F., Soares, A.: Portuguese language models and word embeddings: evaluating on semantic similarity tasks. In: Quaresma, P., Vieira, R., Alu\u00edsio, S., Moniz, H., Batista, F., Gon\u00e7alves, T. (eds.) PROPOR 2020. LNCS (LNAI), vol. 12037, pp. 239\u2013248. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-41505-1_23"},{"key":"28_CR33","unstructured":"Rodrigues, R., Couto, P., Rodrigues, I.: IPR: the semantic textual similarity and recognizing textual entailment systems. In: Oliveira et al. [22], pp. 39\u201347. http:\/\/ceur-ws.org\/Vol-2583\/"},{"key":"28_CR34","unstructured":"Santos, C.N.D., Guimaraes, V.: Boosting named entity recognition with neural character embeddings. arXiv preprint arXiv:1505.05008 (2015). https:\/\/arxiv.org\/abs\/1505.05008, version 2"},{"key":"28_CR35","unstructured":"Santos, D., Seco, N., Cardoso, N., Vilela, R.: HAREM: an advanced NER evaluation contest for Portuguese (2006)"},{"key":"28_CR36","doi-asserted-by":"crossref","unstructured":"Santos, J., Consoli, B., dos Santos, C., Terra, J., Collonini, S., Vieira, R.: Assessing the impact of contextual embeddings for Portuguese named entity recognition. In: 8th Brazilian Conference on Intelligent Systems, BRACIS, Bahia, Brazil, 15\u201318 October, pp. 437\u2013442 (2019)","DOI":"10.1109\/BRACIS.2019.00083"},{"key":"28_CR37","unstructured":"Santos, J., Terra, J., Consoli, B.S., Vieira, R.: Multidomain contextual embeddings for named entity recognition. In: IberLEF@SEPLN (2019)"},{"key":"28_CR38","doi-asserted-by":"crossref","unstructured":"Schuster, M., Nakajima, K.: Japanese and Korean voice search. In: 2012 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5149\u20135152. IEEE (2012)","DOI":"10.1109\/ICASSP.2012.6289079"},{"key":"28_CR39","doi-asserted-by":"publisher","unstructured":"Sennrich, R., Haddow, B., Birch, A.: Neural machine translation of rare words with subword units. In: Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), Berlin, Germany, pp. 1715\u20131725. Association for Computational Linguistics, August 2016. https:\/\/doi.org\/10.18653\/v1\/P16-1162. https:\/\/www.aclweb.org\/anthology\/P16-1162","DOI":"10.18653\/v1\/P16-1162"},{"key":"28_CR40","doi-asserted-by":"publisher","unstructured":"Speer, R.: ftfy. Zenodo (2019). https:\/\/doi.org\/10.5281\/zenodo.2591652, version 5.5","DOI":"10.5281\/zenodo.2591652"},{"issue":"4","key":"28_CR41","doi-asserted-by":"publisher","first-page":"415","DOI":"10.1177\/107769905303000401","volume":"30","author":"WL Taylor","year":"1953","unstructured":"Taylor, W.L.: \u201ccloze procedure\u201d: a new tool for measuring readability. Journalism Q. 30(4), 415\u2013433 (1953)","journal-title":"Journalism Q."},{"key":"28_CR42","unstructured":"Tjong, E.F., Sang, K., Veenstra, J.: Representing text chunks. In: Ninth Conference of the European Chapter of the Association for Computational Linguistics. Association for Computational Linguistics, Bergen, Norway, June 1999. https:\/\/www.aclweb.org\/anthology\/E99-1023"},{"key":"28_CR43","unstructured":"Tjong Kim Sang, E.F., De Meulder, F.: Introduction to the CoNLL-2003 shared task: Language-independent named entity recognition. In: Proceedings of the Seventh Conference on Natural Language Learning at HLT-NAACL 2003, pp. 142\u2013147 (2003). https:\/\/www.aclweb.org\/anthology\/W03-0419"},{"key":"28_CR44","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, pp. 5998\u20136008 (2017)"},{"key":"28_CR45","unstructured":"Vries, W.D., Cranenburgh, A.V., Bisazza, A., Caselli, T., Noord, G.V., Nissim, M.: BERTje: a Dutch BERT model. arXiv preprint arXiv:1912.09582, December 2019"},{"key":"28_CR46","unstructured":"Wagner Filho, J., Wilkens, R., Idiart, M., Villavicencio, A.: The BRWAC corpus: a new open resource for Brazilian Portuguese, May 2018"},{"key":"28_CR47","unstructured":"Wu, Y., et al.: Google\u2019s neural machine translation system: bridging the gap between human and machine translation. arXiv preprint arXiv:1609.08144 (2016). http:\/\/arxiv.org\/abs\/1609.08144, version 2"},{"key":"28_CR48","unstructured":"Yang, Z., Dai, Z., Yang, Y., Carbonell, J., Salakhutdinov, R., Le, Q.V.: XLNet: generalized autoregressive pretraining for language understanding. arXiv preprint arXiv:1906.08237 (2019)"}],"container-title":["Lecture Notes in Computer Science","Intelligent Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-61377-8_28","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,23]],"date-time":"2022-11-23T13:51:45Z","timestamp":1669211505000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-61377-8_28"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030613761","9783030613778"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-61377-8_28","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"13 October 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"BRACIS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brazilian Conference on Intelligent Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Rio Grande","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brazil","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"bracis2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www2.sbc.org.br\/bracis2020\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"JEMS","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"228","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"91","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"40% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3,5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Due to the Corona pandemic BRACIS 2020 was held as a virtual event.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}