{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,17]],"date-time":"2026-03-17T23:22:17Z","timestamp":1773789737016,"version":"3.50.1"},"reference-count":38,"publisher":"Oxford University Press (OUP)","license":[{"start":{"date-parts":[[2022,8,13]],"date-time":"2022-08-13T00:00:00Z","timestamp":1660348800000},"content-version":"vor","delay-in-days":224,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000057","name":"National Institute of General Medical Sciences","doi-asserted-by":"publisher","award":["R01GM126558"],"award-info":[{"award-number":["R01GM126558"]}],"id":[{"id":"10.13039\/100000057","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,8,13]]},"abstract":"<jats:title>Abstract<\/jats:title>\n               <jats:p>Large volumes of publications are being produced in biomedical sciences nowadays with ever-increasing speed. To deal with the large amount of unstructured text data, effective natural language processing (NLP) methods need to be developed for various tasks such as document classification and information extraction. BioCreative Challenge was established to evaluate the effectiveness of information extraction methods in biomedical domain and facilitate their development as a community-wide effort. In this paper, we summarize our work and what we have learned from the latest round, BioCreative Challenge VII, where we participated in all five tracks. Overall, we found three key components for achieving high performance across a variety of NLP tasks: (1) pre-trained NLP models; (2) data augmentation strategies and (3) ensemble modelling. These three strategies need to be tailored towards the specific tasks at hands to achieve high-performing baseline models, which are usually good enough for practical applications. When further combined with task-specific methods, additional improvements (usually rather small) can be achieved, which might be critical for winning competitions.<\/jats:p>\n               <jats:p>Database URL: https:\/\/doi.org\/10.1093\/database\/baac066<\/jats:p>","DOI":"10.1093\/database\/baac066","type":"journal-article","created":{"date-parts":[[2022,8,13]],"date-time":"2022-08-13T04:56:56Z","timestamp":1660366616000},"source":"Crossref","is-referenced-by-count":9,"title":["Pre-trained models, data augmentation, and ensemble learning for biomedical information extraction and document classification"],"prefix":"10.1093","volume":"2022","author":[{"given":"Arslan","family":"Erdengasileng","sequence":"first","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qing","family":"Han","sequence":"additional","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tingting","family":"Zhao","sequence":"additional","affiliation":[{"name":"Department of Geography, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6415-1439","authenticated-orcid":false,"given":"Shubo","family":"Tian","sequence":"additional","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Sui","sequence":"additional","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keqiao","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wanjing","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jian","family":"Wang","sequence":"additional","affiliation":[{"name":"Cloudmedx Inc , Palo Alto, CA 94301, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ting","family":"Hu","sequence":"additional","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"Pan","sequence":"additional","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1693-0889","authenticated-orcid":false,"given":"Yuan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7429-7615","authenticated-orcid":false,"given":"Jinfeng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Statistics, Florida State University , Tallahassee, FL 32306, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"286","published-online":{"date-parts":[[2022,8,13]]},"reference":[{"key":"2022081304564975400_R1","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1016\/j.cell.2008.06.029","article-title":"Seeking a new biology through text mining","volume":"134","author":"Rzhetsky","year":"2008","journal-title":"Cell"},{"key":"2022081304564975400_R2","doi-asserted-by":"publisher","DOI":"10.1186\/gb-2008-9-s2-s6","article-title":"Introducing meta-services for biomedical information extraction","volume":"9 Suppl 2","author":"Leitner","year":"2008","journal-title":"Genome Biol."},{"key":"2022081304564975400_R3","doi-asserted-by":"publisher","DOI":"10.1186\/1471-2105-8-50","article-title":"BioInfer: a corpus for information extraction in the biomedical domain","volume":"8","author":"Pyysalo","year":"2007","journal-title":"BMC Bioinform."},{"key":"2022081304564975400_R4","doi-asserted-by":"publisher","first-page":"450","DOI":"10.1504\/IJDMB.2013.054232","article-title":"PIMiner: a web tool for extraction of protein interactions from biomedical literature","volume":"7","author":"Chowdhary","year":"2013","journal-title":"Int. J. Data Min. Bioinform."},{"key":"2022081304564975400_R5","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0034480","article-title":"Context-specific protein network miner\u2014an online system for exploring context-specific protein interaction networks from the literature","volume":"7","author":"Chowdhary","year":"2012","journal-title":"PLoS One"},{"key":"2022081304564975400_R6","doi-asserted-by":"publisher","first-page":"747","DOI":"10.1093\/bioinformatics\/bts010","article-title":"IMID: integrated molecular interaction database","volume":"28","author":"Balaji","year":"2012","journal-title":"Bioinformatics"},{"key":"2022081304564975400_R7","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0021474","article-title":"Integrated bio-entity network: a system for biological knowledge discovery","volume":"6","author":"Bell","year":"2011","journal-title":"PLoS One"},{"key":"2022081304564975400_R8","doi-asserted-by":"publisher","first-page":"1536","DOI":"10.1093\/bioinformatics\/btp245","article-title":"Bayesian inference of protein-protein interactions from biological literature","volume":"25","author":"Chowdhary","year":"2009","journal-title":"Bioinformatics"},{"key":"2022081304564975400_R9","doi-asserted-by":"publisher","DOI":"10.1186\/s12864-020-07185-7","article-title":"Triage of documents containing protein interactions affected by mutations using an NLP based machine learning approach","volume":"21","author":"Qu","year":"2020","journal-title":"BMC Genom."},{"key":"2022081304564975400_R10","doi-asserted-by":"publisher","DOI":"10.1093\/database\/bay138","article-title":"Extracting chemical-protein interactions from literature using sentence structure analysis and feature engineering","volume":"bay138","author":"Lung","year":"2019","journal-title":"Database (Oxford)"},{"key":"2022081304564975400_R11","first-page":"1292","article-title":"Extraction of protein-protein interactions using natural language processing based pattern matching","author":"Yu","year":"2017"},{"key":"2022081304564975400_R12","first-page":"130","article-title":"Mining protein interactions affected by mutations using a NLP based machine learning approach","author":"Qu","year":"2017"},{"key":"2022081304564975400_R13","article-title":"Extracting chemical-protein interactions from literature","author":"Lung","year":"2017"},{"key":"2022081304564975400_R14","doi-asserted-by":"publisher","first-page":"132","DOI":"10.1093\/bib\/bbv024","article-title":"Community challenges in biomedical text mining over 10 years: success, failure and the future","volume":"17","author":"Huang","year":"2016","journal-title":"Brief. Bioinformatics"},{"key":"2022081304564975400_R15","doi-asserted-by":"publisher","DOI":"10.1186\/1471-2105-12-S8-S1","article-title":"Overview of the BioCreative III workshop","volume":"12","author":"Arighi","year":"2011","journal-title":"BMC Bioinform."},{"key":"2022081304564975400_R16","doi-asserted-by":"publisher","DOI":"10.1186\/gb-2008-9-s2-s1","article-title":"Evaluation of text-mining systems for biology: overview of the Second BioCreative community challenge","volume":"9 Suppl 2","author":"Krallinger","year":"2008","journal-title":"Genome Biol."},{"key":"2022081304564975400_R17","doi-asserted-by":"publisher","DOI":"10.1186\/1471-2105-6-S1-S1","article-title":"Overview of BioCreAtIvE: critical assessment of information extraction for biology","volume":"6 Suppl 1","author":"Hirschman","year":"2005","journal-title":"BMC Bioinform."},{"key":"2022081304564975400_R18","article-title":"LitCoin Natural Language Processing (NLP) Challenge","year":"2021"},{"key":"2022081304564975400_R19","article-title":"Overview of DrugProt BioCreative VII track: quality evaluation and large scale text mining of drug-gene\/protein relations","author":"Miranda","year":"2021"},{"key":"2022081304564975400_R20","article-title":"The overview of the NLM-Chem BioCreative VII track full-text chemical identification and indexing in PubMed articles","author":"Leaman","year":"2021"},{"key":"2022081304564975400_R21","article-title":"VII - Task 3: automatic extraction of medication names in tweets","author":"Weissenbacher","year":"2021"},{"key":"2022081304564975400_R22","article-title":"Overview of the BioCreative VII LitCovid track: multi-label topic classification for COVID-19 literature annotation","author":"Chen","year":"2021"},{"key":"2022081304564975400_R23","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-021-00875-1","article-title":"NLM-Chem, a new resource for chemical entity recognition in PubMed full text literature","volume":"8","author":"Islamaj","year":"2021","journal-title":"Sci. Data"},{"key":"2022081304564975400_R24","article-title":"The chemical corpus of the NLM-Chem BioCreative VII track full-text chemical identification and indexing in PubMed articles","author":"Islamaj","year":"2021"},{"key":"2022081304564975400_R25","doi-asserted-by":"publisher","first-page":"193","DOI":"10.1038\/d41586-020-00694-1","article-title":"Keep up with the latest coronavirus research","volume":"579","author":"Chen","year":"2020","journal-title":"Nature"},{"key":"2022081304564975400_R26","doi-asserted-by":"publisher","first-page":"D1534","DOI":"10.1093\/nar\/gkaa952","article-title":"LitCovid: an open database of COVID-19 literature","volume":"49","author":"Chen","year":"2021","journal-title":"Nucleic Acids Res."},{"key":"2022081304564975400_R27","article-title":"Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"2022081304564975400_R28","doi-asserted-by":"crossref","DOI":"10.1093\/bioinformatics\/btz682","article-title":"Biobert: pre-trained biomedical language representation model for biomedical text mining","author":"Lee","year":"2019"},{"key":"2022081304564975400_R29","article-title":"Domain-specific language model pretraining for biomedical natural language processing","author":"Gu","year":"2020"},{"key":"2022081304564975400_R30","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/D19-1410","article-title":"Sentence-BERT: sentence embeddings using Siamese BERT-networks","author":"Reimers","year":"2019"},{"key":"2022081304564975400_R31","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","author":"Raffel","year":"2020"},{"key":"2022081304564975400_R32","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/W19-5006","article-title":"Transfer learning in biomedical natural language processing: an evaluation of BERT and ELMo on ten benchmarking datasets","author":"Peng","year":"2019"},{"key":"2022081304564975400_R33","first-page":"3615","article-title":"SciBERT: a pretrained language model for scientific text","author":"Beltagy","year":"2019"},{"key":"2022081304564975400_R34","article-title":"RoBERTa: a robustly optimized BERT pretraining approach","author":"Liu","year":"2019"},{"key":"2022081304564975400_R35","first-page":"72","article-title":"Publicly available clinical BERT embeddings","author":"Alsentzer","year":"2019"},{"key":"2022081304564975400_R36","doi-asserted-by":"crossref","DOI":"10.1101\/2021.10.27.466183","article-title":"A BERT-based hybrid system for chemical identification and indexing in full-text articles","author":"Erdengasileng","year":"2021"},{"key":"2022081304564975400_R37","doi-asserted-by":"publisher","DOI":"10.1186\/1471-2105-9-402","article-title":"Abbreviation definition identification based on automatic precision estimates","volume":"9","author":"Sohn","year":"2008","journal-title":"BMC Bioinform."},{"key":"2022081304564975400_R38","doi-asserted-by":"publisher","first-page":"W587","DOI":"10.1093\/nar\/gkz389","article-title":"PubTator central: automated concept annotation for biomedical full text articles","volume":"47","author":"Wei","year":"2019","journal-title":"Nucleic Acids Res."}],"container-title":["Database"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/academic.oup.com\/database\/article-pdf\/doi\/10.1093\/database\/baac066\/45409320\/baac066.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/academic.oup.com\/database\/article-pdf\/doi\/10.1093\/database\/baac066\/45409320\/baac066.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,13]],"date-time":"2022-08-13T04:56:59Z","timestamp":1660366619000},"score":1,"resource":{"primary":{"URL":"https:\/\/academic.oup.com\/database\/article\/doi\/10.1093\/database\/baac066\/6664140"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,1,1]]},"references-count":38,"URL":"https:\/\/doi.org\/10.1093\/database\/baac066","relation":{},"ISSN":["1758-0463"],"issn-type":[{"value":"1758-0463","type":"electronic"}],"subject":[],"published-other":{"date-parts":[[2022,1,1]]},"published":{"date-parts":[[2022,1,1]]},"article-number":"baac066"}}