{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T18:04:08Z","timestamp":1784052248789,"version":"3.55.0"},"reference-count":75,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T00:00:00Z","timestamp":1783641600000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100002241","name":"Japan Science and Technology Agency","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002241","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2027,1]]},"DOI":"10.1016\/j.eswa.2026.133599","type":"journal-article","created":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T23:24:34Z","timestamp":1783639474000},"page":"133599","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PB","title":["RED: Retrieval-enhanced knowledge distillation for smaller language models"],"prefix":"10.1016","volume":"332","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-4188-6626","authenticated-orcid":false,"given":"Xinbai","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4020-9100","authenticated-orcid":false,"given":"Shaowen","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9371-1340","authenticated-orcid":false,"given":"Shoko","family":"Wakamiya","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0201-3609","authenticated-orcid":false,"given":"Eiji","family":"Aramaki","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133599_bib0001","series-title":"The twelfth international conference on learning representations","article-title":"Generalized knowledge distillation for auto-regressive language models","author":"Agarwal","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0002","unstructured":"Asai, A., Wu, Z., Wang, Y., Sil, A., & Hajishirzi, H. (a). Self-RAG: Learning to retrieve, generate, and critique through self-reflection. In The twelfth international conference on learning representations."},{"key":"10.1016\/j.eswa.2026.133599_bib0003","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133599_bib0004","doi-asserted-by":"crossref","unstructured":"Chen, D., Song, S., Yu, Q., Li, Z., Wang, W., Xiong, F., & Tang, B. (2024). Grimoire is all you need for enhancing large language models. arXiv: 2401.03385.","DOI":"10.21203\/rs.3.rs-3845612\/v1"},{"key":"10.1016\/j.eswa.2026.133599_bib0005","series-title":"Findings of the association for computational linguistics: EMNLP 2023","first-page":"6805","article-title":"Mcc-kd: Multi-cot consistent knowledge distillation","author":"Chen","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0006","series-title":"Proceedings of the 2023 conference on empirical methods in natural language processing","first-page":"6325","article-title":"Beyond factuality: A comprehensive evaluation of large language models as knowledge generators","author":"Chen","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0007","series-title":"Proceedings of the ACM web conference 2022","first-page":"2778","article-title":"Knowprompt: Knowledge-aware prompt-tuning with synergistic optimization for relation extraction","author":"Chen","year":"2022"},{"key":"10.1016\/j.eswa.2026.133599_bib0008","series-title":"Proceedings of the 2023 conference on empirical methods in natural language processing","first-page":"12318","article-title":"Uprise: Universal prompt retrieval for improving zero-shot evaluation","author":"Cheng","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0009","first-page":"43780","article-title":"Lift yourself up: Retrieval-augmented text generation with self-memory","volume":"36","author":"Cheng","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133599_bib0010","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"4794","article-title":"On the efficacy of knowledge distillation","author":"Cho","year":"2019"},{"key":"10.1016\/j.eswa.2026.133599_bib0011","series-title":"Proceedings of the 2024 conference on empirical methods in natural language processing","first-page":"1419","article-title":"Lookback lens: Detecting and mitigating contextual hallucinations in large language models using only attention maps","author":"Chuang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0012","series-title":"The twelfth international conference on learning representations","article-title":"Dola: Decoding by contrasting layers improves factuality in large language models","author":"Chuang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0013","unstructured":"Chung, H. W., Hou, L., Longpre, S., Zoph, B., Tay, Y., Fedus, W., Li, E., Wang, X., Dehghani, M., Brahma, S., Webson, A., Gu, S. S., Dai, Z., Suzgun, M., Chen, X., Chowdhery, A., Narang, S., Mishra, G., Yu, A., Zhao, V., Huang, Y., Dai, A., Yu, H., Petrov, S., Chi, E. H., Dean, J., Devlin, J., Roberts, A., Zhou, D., Le, Q. V., & Wei, J. (2022). Scaling instruction-finetuned language models. 10.48550\/ARXIV.2210.11416."},{"key":"10.1016\/j.eswa.2026.133599_bib0014","unstructured":"Dai, Z., Zhao, V. Y., Ma, J., Luan, Y., Ni, J., Lu, J., Bakalov, A., Guu, K., Hall, K., & Chang, M.-W. (n.d). Promptagator: Few-shot dense retrieval from 8 examples. In The eleventh international conference on learning representations."},{"key":"10.1016\/j.eswa.2026.133599_bib0015","unstructured":"Devlin, J. (2018). Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv: 1810.04805."},{"key":"10.1016\/j.eswa.2026.133599_bib0016","series-title":"Proceedings of the 2024 conference on empirical methods in natural language processing","first-page":"1107","article-title":"A survey on in-context learning","author":"Dong","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0017","unstructured":"Dubey, A., Jauhri, A., Pandey, A., Kadian, A., Al-Dahle, A., Letman, A., Mathur, A., Schelten, A., Yang, A., Fan, A. et al. (2024). The llama 3 herd of models. arXiv: 2407.21783."},{"issue":"6","key":"10.1016\/j.eswa.2026.133599_bib0018","doi-asserted-by":"crossref","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","article-title":"Knowledge distillation: A survey","volume":"129","author":"Gou","year":"2021","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.133599_bib0019","series-title":"The twelfth international conference on learning representations","article-title":"MiniLLM: Knowledge distillation of large language models","author":"Gu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0020","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"13766","article-title":"Optimal transport for unsupervised hallucination detection in neural machine translation","author":"Guerreiro","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0021","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"7793","article-title":"Boosting graph neural networks via adaptive knowledge distillation","volume":"vol. 37","author":"Guo","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0022","series-title":"Findings of the association for computational linguistics: EMNLP 2021","first-page":"4536","article-title":"Klmo: Knowledge graph enhanced pretrained language model with fine-grained relationships","author":"He","year":"2021"},{"key":"10.1016\/j.eswa.2026.133599_bib0023","series-title":"Proceedings of the 2024 conference on empirical methods in natural language processing","first-page":"18573","article-title":"Enhanced hallucination detection in neural machine translation through simple detector aggregation","author":"Himmi","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0024","unstructured":"Hinton, G. (2015). Distilling the knowledge in a neural network. arXiv: 1503.02531."},{"key":"10.1016\/j.eswa.2026.133599_bib0025","unstructured":"Ho, N., Schmid, L., & Yun, S.-Y. (2022). Large language models are reasoning teachers. arXiv: 2212.10071."},{"key":"10.1016\/j.eswa.2026.133599_bib0026","series-title":"Findings of the association for computational linguistics: ACL 2023","first-page":"8003","article-title":"Distilling step-by-step! outperforming larger language models with less training data and smaller model sizes","author":"Hsieh","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0027","series-title":"Proceedings of the 60th annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"2225","article-title":"Knowledgeable prompt-tuning: Incorporating knowledge into prompt verbalizer for text classification","author":"Hu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133599_bib0028","series-title":"Proceedings of the 62nd annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"1429","article-title":"Enhancing large language models in coding through multi-perspective self-consistency","author":"Huang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0029","series-title":"Proceedings of the 13th international joint conference on natural language processing and the 3rd conference of the asia-pacific chapter of the association for computational linguistics (volume 1: Long papers)","first-page":"1012","article-title":"Retrieval augmented generation with rich answer encoding","author":"Huang","year":"2023"},{"issue":"12","key":"10.1016\/j.eswa.2026.133599_bib0030","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3571730","article-title":"Survey of hallucination in natural language generation","volume":"55","author":"Ji","year":"2023","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/j.eswa.2026.133599_bib0031","doi-asserted-by":"crossref","unstructured":"Jiang, Z., Xu, F. F., Gao, L., Sun, Z., Liu, Q., Dwivedi-Yu, J., Yang, Y., Callan, J., & Neubig, G. (2023). Active retrieval augmented generation. In The 2023 conference on empirical methods in natural language processing.","DOI":"10.18653\/v1\/2023.emnlp-main.495"},{"issue":"14","key":"10.1016\/j.eswa.2026.133599_bib0032","doi-asserted-by":"crossref","first-page":"6421","DOI":"10.3390\/app11146421","article-title":"What disease does this patient have? a large-scale open domain question answering dataset from medical exams","volume":"11","author":"Jin","year":"2021","journal-title":"Applied Sciences"},{"key":"10.1016\/j.eswa.2026.133599_bib0033","series-title":"Proceedings of the 2019 conference on empirical methods in natural language processing and the 9th international joint conference on natural language processing (EMNLP-IJCNLP)","article-title":"PubmedQA: A dataset for biomedical research question answering","author":"Jin","year":"2019"},{"key":"10.1016\/j.eswa.2026.133599_bib0034","doi-asserted-by":"crossref","first-page":"48573","DOI":"10.52202\/075280-2109","article-title":"Knowledge-augmented reasoning distillation for small language models in knowledge-intensive tasks","volume":"36","author":"Kang","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133599_bib0035","unstructured":"Ko, J., Kim, S., Chen, T., & Yun, S.-Y. (2024). Distillm: Towards streamlined distillation for large language models. arXiv: 2402.03898."},{"key":"10.1016\/j.eswa.2026.133599_bib0036","series-title":"Proceedings of the 2024 joint international conference on computational linguistics, language resources and evaluation (LREC-COLING 2024)","first-page":"10657","article-title":"Llmr: Knowledge distillation with a large language model-induced reward","author":"Li","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0037","series-title":"Proceedings of the 2024 joint international conference on computational linguistics, language resources and evaluation (LREC-COLING 2024)","first-page":"8804","article-title":"Improving faithfulness of large language models in summarization via sliding generation and self-consistency","author":"Li","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0038","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113112","article-title":"Cross-domain recommendation via knowledge distillation","volume":"311","author":"Li","year":"2025","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.133599_bib0039","unstructured":"Lin, X. V., Chen, X., Chen, M., Shi, W., Lomeli, M., James, R., Rodriguez, P., Kahn, J., Szilvasy, G., Lewis, M. et al. (n.d.). Ra-dit: Retrieval-augmented dual instruction tuning. In The twelfth international conference on learning representations."},{"key":"10.1016\/j.eswa.2026.133599_bib0040","series-title":"Proceedings of the 2024 conference of the north american chapter of the association for computational linguistics: Human language technologies (volume 1: Long papers)","first-page":"6748","article-title":"Mind\u2019s mirror: Distilling self-evaluation capability and comprehensive thinking from large language models","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0041","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"2901","article-title":"K-bert: Enabling language representation with knowledge graph","volume":"vol. 34","author":"Liu","year":"2020"},{"key":"10.1016\/j.eswa.2026.133599_bib0042","unstructured":"Liu, Y. (2019). Roberta: A robustly optimized bert pretraining approach. 364. arXiv: 1907.11692."},{"key":"10.1016\/j.eswa.2026.133599_bib0043","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.126395","article-title":"Consistency knowledge distillation based on similarity attribute graph guidance","volume":"269","author":"Ma","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.133599_bib0044","series-title":"Proceedings of the 2022 conference on empirical methods in natural language processing","first-page":"1754","article-title":"Enhancing self-consistency and performance of pre-trained language models through natural language inference","author":"Mitchell","year":"2022"},{"key":"10.1016\/j.eswa.2026.133599_bib0045","unstructured":"Peng, B., Galley, M., He, P., Cheng, H., Xie, Y., Hu, Y., Huang, Q., Liden, L., Yu, Z., Chen, W. et al. (2023). Check your facts and try again: Improving large language models with external knowledge and automated feedback. arXiv: 2302.12813."},{"key":"10.1016\/j.eswa.2026.133599_bib0046","series-title":"Proceedings of the 2019 conference on empirical methods in natural language processing and the 9th international joint conference on natural language processing (EMNLP-IJCNLP)","first-page":"43","article-title":"Knowledge enhanced contextual word representations","author":"Peters","year":"2019"},{"key":"10.1016\/j.eswa.2026.133599_bib0047","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113048","article-title":"A good teacher learns while teaching: Heterogeneous architectural knowledge distillation for fast MRI reconstruction","volume":"311","author":"Qiu","year":"2025","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.133599_bib0048","series-title":"Findings of the association for computational linguistics: EMNLP 2023","first-page":"9248","article-title":"Enhancing retrieval-augmented large language models with iterative retrieval-generation synergy","author":"Shao","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0049","series-title":"Proceedings of the 28th international conference on computational linguistics","first-page":"3660","article-title":"CoLAKE: Contextualized language and knowledge embedding","author":"Sun","year":"2020"},{"key":"10.1016\/j.eswa.2026.133599_bib0050","unstructured":"Sun, Y., Wang, S., Li, Y., Feng, S., Chen, X., Zhang, H., Tian, X., Zhu, D., Tian, H., & Wu, H. (2019). Ernie: Enhanced representation through knowledge integration. arXiv: 1904.09223."},{"key":"10.1016\/j.eswa.2026.133599_bib0051","series-title":"Proceedings of the 2024 conference of the north american chapter of the association for computational linguistics: Human language technologies (volume 1: Long papers)","first-page":"2327","article-title":"Found in the middle: Permutation self-consistency improves listwise ranking in large language models","author":"Tang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0052","unstructured":"G. Team, Riviere, M., Pathak, S., Sessa, P. G., Hardin, C., Bhupatiraju, S., Hussenot, L., Mesnard, T., Shahriari, B., Ram\u00e9, A. et al. (2024). Gemma 2: Improving open language models at a practical size, 2024. 1(3). https:\/\/arxiv.org\/abs\/2408.00118."},{"key":"10.1016\/j.eswa.2026.133599_bib0053","doi-asserted-by":"crossref","unstructured":"Tian, Y., Han, Y., Chen, X., Wang, W., & Chawla, N. V. (2024). Beyond answers: Transferring reasoning capabilities to smaller LLMs using multi-teacher knowledge distillation.","DOI":"10.1145\/3701551.3703577"},{"key":"10.1016\/j.eswa.2026.133599_bib0054","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1186\/s12859-015-0564-6","article-title":"An overview of the BIOASQ large-scale biomedical semantic indexing and question answering competition","volume":"16","author":"Tsatsaronis","year":"2015","journal-title":"BMC Bioinformatics"},{"key":"10.1016\/j.eswa.2026.133599_bib0055","series-title":"Proceedings of the 62nd annual meeting of the association for computational linguistics (volume 2: Short papers)","first-page":"287","article-title":"Soft self-consistency improves language models agents","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0056","unstructured":"Wang, L., Chen, H., Yang, N., Huang, X., Dou, Z., & Wei, F. (2025). Chain-of-retrieval augmented generation. arXiv: 2501.14342."},{"key":"10.1016\/j.eswa.2026.133599_bib0057","unstructured":"Wang, R., Tang, D., Duan, N., Wei, Z., Huang, X., Cao, G., Jiang, D., Zhou, M. et al. (2020). K-adapter: Infusing knowledge into pre-trained models with adapters. arXiv: 2002.01808."},{"key":"10.1016\/j.eswa.2026.133599_bib0058","doi-asserted-by":"crossref","first-page":"176","DOI":"10.1162\/tacl_a_00360","article-title":"Kepler: A unified model for knowledge embedding and pre-trained language representation","volume":"9","author":"Wang","year":"2021","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"10.1016\/j.eswa.2026.133599_bib0059","series-title":"Proceedings of the 2023 conference on empirical methods in natural language processing","first-page":"15361","article-title":"Hallucination detection for generative large language models by Bayesian sequential estimation","author":"Wang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0060","unstructured":"Wang, X., Yang, Q., Qiu, Y., Liang, J., He, Q., Gu, Z., Xiao, Y., & Wang, W. (2023b). Knowledgpt: Enhancing large language models with retrieval and storage access on knowledge bases. arXiv: 2308.11761."},{"key":"10.1016\/j.eswa.2026.133599_bib0061","series-title":"Proceedings of the 62nd annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"1966","article-title":"M-RAG: Reinforcing large language model performance through retrieval-augmented generation with multiple partitions","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0062","series-title":"Crowdsourcing multiple choice science questions","author":"Welbl","year":"2017"},{"key":"10.1016\/j.eswa.2026.133599_bib0063","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"8449","article-title":"AD-KD: Attribution-driven knowledge distillation for language model compression","author":"Wu","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0064","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"8449","article-title":"Ad-kd: Attribution-driven knowledge distillation for language model compression","author":"Wu","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0065","unstructured":"Xiong, W., Du, J., Wang, W. Y., & Stoyanov, V. (2019). Pretrained encyclopedia: Weakly supervised knowledge-pretrained language model. arXiv: 1912.09637."},{"key":"10.1016\/j.eswa.2026.133599_bib0066","series-title":"Proceedings of the 62nd annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"133","article-title":"Unsupervised information refinement training of large language models for retrieval-augmented generation","author":"Xu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0067","unstructured":"Xu, X., Li, M., Tao, C., Shen, T., Cheng, R., Li, J., Xu, C., Tao, D., & Zhou, T. (2024b). A survey on knowledge distillation of large language models. arXiv: 2402.13116."},{"key":"10.1016\/j.eswa.2026.133599_bib0068","series-title":"Proceedings of the 23rd workshop on biomedical natural language processing","first-page":"155","article-title":"KG-rank: Enhancing large language models for medical QA with knowledge graphs and ranking techniques","author":"Yang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0069","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.126808","article-title":"Multi-teacher knowledge distillation for debiasing recommendation with uniform data","volume":"273","author":"Yang","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.133599_bib0070","series-title":"Proceedings of the conference on empirical methods in natural language processing. conference on empirical methods in natural language processing","first-page":"1767","article-title":"Knowledge injected prompt based fine-tuning for multi-label few-shot icd coding","volume":"vol. 2022","author":"Yang","year":"2022"},{"key":"10.1016\/j.eswa.2026.133599_bib0071","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"6451","article-title":"Knowledge prompt-tuning for sequential recommendation","author":"Zhai","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0072","series-title":"Findings of the association for computational linguistics: EMNLP 2023","first-page":"15445","article-title":"SAC3: Reliable hallucination detection in black-box language models via semantic-aware cross-check consistency","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133599_bib0073","series-title":"Proceedings of the 57th annual meeting of the association for computational linguistics","first-page":"1441","article-title":"ERNIE: Enhanced language representation with informative entities","author":"Zhang","year":"2019"},{"key":"10.1016\/j.eswa.2026.133599_bib0074","series-title":"Proceedings of the 2024 joint international conference on computational linguistics, language resources and evaluation (LREC-COLING 2024)","first-page":"14103","article-title":"Revisiting the self-consistency challenges in multi-choice question formats for large language model evaluation","author":"Zhou","year":"2024"},{"key":"10.1016\/j.eswa.2026.133599_bib0075","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2023.126488","article-title":"Knowledge-based BERT word embedding fine-tuning for emotion recognition","volume":"552","author":"Zhu","year":"2023","journal-title":"Neurocomputing"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426025078?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426025078?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T17:06:31Z","timestamp":1784048791000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426025078"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,1]]},"references-count":75,"alternative-id":["S0957417426025078"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133599","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2027,1]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"RED: Retrieval-enhanced knowledge distillation for smaller language models","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133599","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"133599"}}