{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T00:56:20Z","timestamp":1781312180836,"version":"3.54.1"},"reference-count":34,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,4,9]],"date-time":"2026-04-09T00:00:00Z","timestamp":1775692800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100006319","name":"National Center For Scientific and Technical Research","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100006319","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001665","name":"French National Research Agency","doi-asserted-by":"publisher","award":["20-PCPA-0002"],"award-info":[{"award-number":["20-PCPA-0002"]}],"id":[{"id":"10.13039\/501100001665","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.eswa.2026.132345","type":"journal-article","created":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T16:50:48Z","timestamp":1775494248000},"page":"132345","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Enhancing language models with selective masking for thematic and misinformation classification in a One Health context"],"prefix":"10.1016","volume":"322","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-8333-0958","authenticated-orcid":false,"given":"Youssef","family":"Mahdoubi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0038-2988","authenticated-orcid":false,"given":"Najlae","family":"Idrissi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3272-8568","authenticated-orcid":false,"given":"Mathieu","family":"Roche","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9028-681X","authenticated-orcid":false,"given":"Sarah","family":"Valentin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"1","key":"10.1016\/j.eswa.2026.132345_bib0001","doi-asserted-by":"crossref","first-page":"45","DOI":"10.1016\/S0306-4573(02)00021-3","article-title":"An information-theoretic perspective of TF-IDF measures","volume":"39","author":"Aizawa","year":"2003","journal-title":"Information Processing & Management"},{"issue":"1","key":"10.1016\/j.eswa.2026.132345_bib0002","doi-asserted-by":"crossref","first-page":"46","DOI":"10.1186\/s40537-023-00727-2","article-title":"A survey on deep learning tools dealing with data scarcity: Definitions, challenges, solutions, tips, and applications","volume":"10","author":"Alzubaidi","year":"2023","journal-title":"Journal of Big Data"},{"key":"10.1016\/j.eswa.2026.132345_bib0003","series-title":"Actes de la 31\u00e8me conf\u00e9rence sur le traitement automatique des langues naturelles (TALN)","first-page":"283","article-title":"Adaptation des mod\u00e8les de langue \u00e0 des domaines de sp\u00e9cialit\u00e9 par un masquage s\u00e9lectif fond\u00e9 sur le genre et les caract\u00e9ristiques th\u00e9matiques","volume":"vol. 1","author":"Belfathi","year":"2024"},{"key":"10.1016\/j.eswa.2026.132345_bib0004","series-title":"Proceedings of EMNLP-IJCNLP 2019","first-page":"3615","article-title":"SciBERT: A pretrained language model for scientific text","author":"Beltagy","year":"2019"},{"key":"10.1016\/j.eswa.2026.132345_bib0005","series-title":"Natural language processing and information systems","first-page":"271","article-title":"Could keyword masking strategy improve language model?","author":"Borovikova","year":"2023"},{"key":"10.1016\/j.eswa.2026.132345_bib0006","series-title":"Proceedings of the 34th international conference on neural information processing systems","article-title":"Language models are few-shot learners","author":"Brown","year":"2020"},{"key":"10.1016\/j.eswa.2026.132345_bib0007","unstructured":"Clark, K., Luong, M.-T., Le, Q. V., & Manning, C. D. (2020). Electra: Pre-training text encoders as discriminators rather than generators. 10.48550\/arXiv.2003.10555."},{"key":"10.1016\/j.eswa.2026.132345_bib0008","unstructured":"Cui, L., & Lee, D. (2020). CoAID: COVID-19 healthcare misinformation dataset. 10.48550\/arXiv.2006.00885."},{"key":"10.1016\/j.eswa.2026.132345_bib0009","doi-asserted-by":"crossref","unstructured":"D\u2019Arcy, C. J., Eastburn, D. M., & Schumann, G. L. (2001). Illustrated glossary of plant pathology. The Plant Health Instructor. 10.1094\/PHI-I-2001-0219-01.","DOI":"10.1094\/PHI-I-2001-0219-01"},{"key":"10.1016\/j.eswa.2026.132345_bib0010","series-title":"Proceedings of NAACL-HLT 2019","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.eswa.2026.132345_bib0011","series-title":"Proceedings of the 2024 conference on empirical methods in natural language processing","first-page":"1107","article-title":"A survey on in-context learning","author":"Dong","year":"2024"},{"key":"10.1016\/j.eswa.2026.132345_bib0012","series-title":"Proceedings of the 2024 joint international conference on computational linguistics, language resources and evaluation (LREC-COLING 2024)","first-page":"10058","article-title":"Language models for text classification: Is in-context learning enough?","author":"Edwards","year":"2024"},{"key":"10.1016\/j.eswa.2026.132345_bib0013","unstructured":"FalgunPatel19 (2022). Medical text dataset - cancer document classification [dataset]. Kaggle. Retrieved from https:\/\/www.kaggle.com\/datasets\/falgunipatel19\/biomedical-text-publication-classification. Accessed September 26, 2025."},{"key":"10.1016\/j.eswa.2026.132345_bib0014","series-title":"Proceedings of the 2020 conference on empirical methods in natural language processing (EMNLP)","first-page":"6966","article-title":"Train no evil: Selective masking for task-guided pre-training","author":"Gua","year":"2020"},{"key":"10.1016\/j.eswa.2026.132345_bib0015","series-title":"Proceedings of ACL 2020","first-page":"8342","article-title":"Don\u2019t stop pretraining: Adapt language models to domains and tasks","author":"Gururangan","year":"2020"},{"key":"10.1016\/j.eswa.2026.132345_bib0016","series-title":"Proceedings of the tenth ACM SIGKDD international conference on knowledge discovery and data mining (KDD\u201904), Seattle, WA, USA","article-title":"Mining and summarizing customer reviews","author":"Hu","year":"2004"},{"key":"10.1016\/j.eswa.2026.132345_bib0017","doi-asserted-by":"crossref","first-page":"50809","DOI":"10.1109\/ACCESS.2024.3385855","article-title":"Enhancing financial sentiment analysis ability of language model via targeted numerical change-related masking","volume":"12","author":"Jung","year":"2024","journal-title":"IEEE Access"},{"issue":"4","key":"10.1016\/j.eswa.2026.132345_bib0018","doi-asserted-by":"crossref","first-page":"1234","DOI":"10.1093\/bioinformatics\/btz682","article-title":"BioBERT: A pre-trained biomedical language representation model for biomedical text mining","volume":"36","author":"Lee","year":"2020","journal-title":"Bioinformatics"},{"key":"10.1016\/j.eswa.2026.132345_bib0019","series-title":"Proceedings of the 34th international conference on neural information processing systems","article-title":"Retrieval-augmented generation for knowledge-intensive NLP tasks","author":"Lewis","year":"2020"},{"key":"10.1016\/j.eswa.2026.132345_bib0020","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., & Stoyanov, V. (2019). RoBERTa: A robustly optimized BERT pretraining approach. 10.48550\/arXiv.1907.11692."},{"key":"10.1016\/j.eswa.2026.132345_bib0021","doi-asserted-by":"crossref","unstructured":"McInnes, L., & Healy, J. (2018). Umap: Uniform manifold approximation and projection for dimension reduction,. 10.48550\/arXiv.1802.03426.","DOI":"10.21105\/joss.00861"},{"key":"10.1016\/j.eswa.2026.132345_bib0022","series-title":"Findings of the association for computational linguistics: EMNLP 2024","first-page":"6829","article-title":"Few-shot clinical entity recognition in English, French and Spanish: masked language models outperform generative model prompting","author":"Naguib","year":"2024"},{"key":"10.1016\/j.eswa.2026.132345_sbref0023","series-title":"A dictionary of biomedicine","author":"Oxford Reference","year":"2010"},{"key":"10.1016\/j.eswa.2026.132345_bib0024","series-title":"Advances in neural information processing systems (neurIPS)","first-page":"8026","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume":"vol. 32","author":"Paszke","year":"2019"},{"key":"10.1016\/j.eswa.2026.132345_bib0025","series-title":"Proceedings of the 16th conference of the european chapter of the association for computational linguistics (EACL)","first-page":"1977","article-title":"Boosting low-resource biomedical QA via entity-aware masking strategies","author":"Pergola","year":"2021"},{"key":"10.1016\/j.eswa.2026.132345_bib0026","series-title":"One health","first-page":"1","article-title":"Chapter 1-an introduction to the concept of one health","author":"Prata","year":"2022"},{"key":"10.1016\/j.eswa.2026.132345_bib0027","unstructured":"Rasamoelina, H., Veerapa-Mangroo, L. P., Bedja, S. A., & Roche, M. (2023). Mots-cl\u00e9s pour PADI-web mis en place dans l\u2019oc\u00e9an indien [dataset]. 10.18167\/DVN1\/E7WMAO."},{"key":"10.1016\/j.eswa.2026.132345_bib0028","series-title":"Proceedings of the European chapter of the association for computational linguistics","first-page":"255","article-title":"Exploiting cloze-questions for few-shot text classification and natural language inference","author":"Schick","year":"2020"},{"key":"10.1016\/j.eswa.2026.132345_bib0029","series-title":"Proc. of the 2021 conference on empirical methods in natural language processing","first-page":"4980","article-title":"Improving and simplifying pattern exploiting training","author":"Tam","year":"2021"},{"key":"10.1016\/j.eswa.2026.132345_bib0030","unstructured":"The Devastator (2023). Pubmed article summarization [dataset]. Kaggle. Retrieved from https:\/\/www.kaggle.com\/datasets\/thedevastator\/pubmed-article-summarization-dataset. Accessed September 23, 2025."},{"key":"10.1016\/j.eswa.2026.132345_bib0031","doi-asserted-by":"crossref","DOI":"10.1016\/j.onehlt.2021.100357","article-title":"Padi-web 3.0: A new framework for extracting and disseminating fine-grained information from the news for animal disease surveillance","volume":"13","author":"Valentin","year":"2021","journal-title":"One Health"},{"issue":"3","key":"10.1016\/j.eswa.2026.132345_bib0032","doi-asserted-by":"crossref","DOI":"10.1145\/3386252","article-title":"Generalizing from a few examples: A survey on few-shot learning","volume":"53","author":"Wang","year":"2020","journal-title":"ACM Computing Survey"},{"key":"10.1016\/j.eswa.2026.132345_bib0033","series-title":"Proceedings of the 17th conference of the european chapter of the ACL (EACL)","first-page":"2985","article-title":"Should you mask 15% in masked language modeling?","author":"Wettig","year":"2023"},{"key":"10.1016\/j.eswa.2026.132345_bib0034","series-title":"Proceedings of EMNLP 2020: System demonstrations","first-page":"38","article-title":"Transformers: State-of-the-art natural language processing","author":"Wolf","year":"2020"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426012583?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426012583?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T00:05:42Z","timestamp":1781309142000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426012583"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":34,"alternative-id":["S0957417426012583"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132345","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Enhancing language models with selective masking for thematic and misinformation classification in a One Health context","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132345","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"132345"}}