{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T07:47:49Z","timestamp":1784274469494,"version":"3.55.0"},"reference-count":144,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276029"],"award-info":[{"award-number":["62276029"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Beijing Academy of Artificial Intelligence"},{"name":"CCF-ZhipuAI Large Model Fund","award":["202217"],"award-info":[{"award-number":["202217"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Knowl. Data Eng."],"published-print":{"date-parts":[[2024,4]]},"DOI":"10.1109\/tkde.2023.3310002","type":"journal-article","created":{"date-parts":[[2023,8,30]],"date-time":"2023-08-30T17:27:48Z","timestamp":1693416468000},"page":"1413-1430","source":"Crossref","is-referenced-by-count":169,"title":["A Survey of Knowledge Enhanced Pre-Trained Language Models"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0993-8495","authenticated-orcid":false,"given":"Linmei","family":"Hu","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology, Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5691-6849","authenticated-orcid":false,"given":"Zeyi","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Computer Science, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5538-3190","authenticated-orcid":false,"given":"Ziwang","family":"Zhao","sequence":"additional","affiliation":[{"name":"School of Computer Science, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8907-3526","authenticated-orcid":false,"given":"Lei","family":"Hou","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Technology, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1476-0273","authenticated-orcid":false,"given":"Liqiang","family":"Nie","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Harbin Institute of Technology (Shenzhen), Shenzhen, Guangdong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6244-0664","authenticated-orcid":false,"given":"Juanzi","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Technology, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref2","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018","journal-title":"OpenAI Blog"},{"key":"ref3","first-page":"140:1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1112"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00254"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.586"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1250"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.136"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00342"},{"key":"ref10","article-title":"ERNIE 3.0: Large-scale knowledge enhanced pre-training for language understanding and generation","author":"Sun","year":"2021"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i7.16796"},{"key":"ref12","article-title":"Knowledge enhanced pretrained language models: A compreshensive survey","author":"Wei","year":"2021"},{"key":"ref13","article-title":"A survey of knowledge-intensive NLP with pre-trained language models","author":"Yin","year":"2022"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-1202"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1031"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2021.3090866"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"key":"ref18","first-page":"13 042","article-title":"Unified language model pre-training for natural language understanding and generation","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Dong"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.346"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.20"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.576"},{"key":"ref25","first-page":"1","article-title":"Differentiable prompt makes pre-trained language models better few-shot learners","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-demo.10"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2022.11.003"},{"key":"ref28","article-title":"SentiPrompt: Sentiment knowledge enhanced prompt-tuning for aspect-based analysis","author":"Li","year":"2021"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.158"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3511998"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3511921"},{"key":"ref32","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","volume":"33","author":"Brown"},{"key":"ref33","article-title":"PaLM: Scaling language modeling with pathways","author":"Chowdhery","year":"2022"},{"key":"ref34","article-title":"A survey of large language models","author":"Zhao","year":"2023"},{"key":"ref35","article-title":"GPT-4 technical report","year":"2023"},{"key":"ref36","article-title":"Emergent abilities of large language models","author":"Wei","year":"2022"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11651"},{"key":"ref38","article-title":"Few-shot learning with retrieval augmented language models","author":"Izacard","year":"2022"},{"key":"ref39","article-title":"Check your facts and try again: Improving large language models with external knowledge and automated feedback","author":"Peng","year":"2023"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/3512467"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.118"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.423"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.374"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i15.17592"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/383"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/567"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.399"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.262"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.57"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.228"},{"key":"ref51","first-page":"1","article-title":"Generalization through memorization: Nearest neighbor language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Khandelwal"},{"key":"ref52","first-page":"3929","article-title":"Retrieval augmented language model pre-training","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Guu"},{"key":"ref53","first-page":"1","article-title":"Improving neural language models with a continuous cache","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Grave"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.190"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.138"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-17120-8_11"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.226"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-naacl.115"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.113"},{"key":"ref60","first-page":"1","article-title":"Knowledge-in-context: Towards knowledgeable semi-parametric language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Pan"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1093\/bioinformatics\/btz682"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1371"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.447"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1145\/3447772"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3150080"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2017.2754499"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2021.3090253"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3159539"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3218850"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3142056"},{"key":"ref71","article-title":"ERNIE: Enhanced representation through knowledge integration","author":"Sun","year":"2019"},{"key":"ref72","first-page":"1","article-title":"Pretrained encyclopedia: Weakly supervised knowledge-pretrained language model","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Xiong"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.206"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.260"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21425"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.523"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1139"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1005"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.400"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2021.06.004"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21417"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00360"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.207"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i03.5681"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.327"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.288"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.121"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.820"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.384"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531997"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.372"},{"key":"ref92","first-page":"1","article-title":"GreaseLM: Graph reasoning enhanced language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/tkde.2021.3079836"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.110"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5446"},{"key":"ref96","first-page":"3261","article-title":"SuperGLUE: A stickier benchmark for general-purpose language understanding systems","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Wang"},{"key":"ref97","article-title":"BERT is not a knowledge base (yet): Factual knowledge versus. name-based reasoning in unsupervised QA","author":"P\u00f6rner","year":"2019"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1004"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1514"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1009"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00141"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.3115\/1119176.1119195"},{"key":"ref103","first-page":"4149","article-title":"CommonsenseQA: A question answering challenge targeting commonsense knowledge","volume-title":"Proc. Conf. North Amer. Chapter Assoc. Comput. Linguistics - Hum. Lang. Technol.","author":"Talmor"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1260"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00276"},{"key":"ref106","first-page":"1533","article-title":"Semantic parsing on freebase from question-answer pairs","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process.","author":"Berant"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1147"},{"key":"ref108","first-page":"1631","article-title":"Recursive deep models for semantic compositionality over a sentiment treebank","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process.","author":"Socher"},{"key":"ref109","first-page":"1","article-title":"Wizard of Wikipedia: Knowledge-powered conversational agents","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dinan"},{"key":"ref110","first-page":"9459","article-title":"Retrieval-augmented generation for knowledge-intensive NLP tasks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lewis"},{"key":"ref111","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.249"},{"key":"ref112","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21351"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.340"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-emnlp.27"},{"key":"ref115","first-page":"2206","article-title":"Improving language models by retrieving from trillions of tokens","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Borgeaud"},{"key":"ref116","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.39"},{"key":"ref117","first-page":"1","article-title":"Sequential latent knowledge selection for knowledge-grounded dialogue","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kim"},{"key":"ref118","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.15"},{"key":"ref119","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.74"},{"key":"ref120","first-page":"248","article-title":"Generating commonsense explanation by extracting bridge concepts from reasoning paths","volume-title":"Proc. Conf. Asia-Pacific Chapter Assoc. Comput. Linguistics, Int. Joint Conf. Natural Lang. Process.","author":"Ji"},{"key":"ref121","article-title":"Graph-based multi-hop reasoning for long text generation","author":"Zhao","year":"2020"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00302"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.59"},{"key":"ref124","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.95"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.54"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.223"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.149"},{"key":"ref128","article-title":"Neuro-symbolic procedural planning with commonsense prompting","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lu"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1076"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.165"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/11578.003.0023"},{"key":"ref132","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1393"},{"key":"ref133","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N16-1098"},{"key":"ref134","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K16-1028"},{"key":"ref135","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1206"},{"key":"ref136","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.36"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.200"},{"key":"ref138","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.semeval-1.39"},{"key":"ref139","first-page":"1","article-title":"KB-VLP: Knowledge based vision and language pretraining","volume-title":"Proc. Int. Conf. Mach. Learn. Workshop","author":"Chen"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16431"},{"key":"ref141","article-title":"Three scenarios for continual learning","author":"Van de Ven","year":"2019"},{"key":"ref142","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.220"},{"key":"ref143","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2021.08.002"},{"key":"ref144","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"}],"container-title":["IEEE Transactions on Knowledge and Data Engineering"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/69\/10462568\/10234662.pdf?arnumber=10234662","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,11]],"date-time":"2024-03-11T19:17:07Z","timestamp":1710184627000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10234662\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4]]},"references-count":144,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tkde.2023.3310002","relation":{},"ISSN":["1041-4347","1558-2191","2326-3865"],"issn-type":[{"value":"1041-4347","type":"print"},{"value":"1558-2191","type":"electronic"},{"value":"2326-3865","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,4]]}}}