{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T17:04:53Z","timestamp":1783530293765,"version":"3.55.0"},"reference-count":202,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000083","name":"U.S. National Science Foundation","doi-asserted-by":"publisher","award":["2309760"],"award-info":[{"award-number":["2309760"]}],"id":[{"id":"10.13039\/100000083","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000083","name":"U.S. National Science Foundation","doi-asserted-by":"publisher","award":["2317117"],"award-info":[{"award-number":["2317117"]}],"id":[{"id":"10.13039\/100000083","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1109\/tai.2024.3428519","type":"journal-article","created":{"date-parts":[[2024,7,15]],"date-time":"2024-07-15T15:27:55Z","timestamp":1721057275000},"page":"5928-5948","source":"Crossref","is-referenced-by-count":41,"title":["A Survey on Symbolic Knowledge Distillation of Large Language Models"],"prefix":"10.1109","volume":"5","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9712-0265","authenticated-orcid":false,"given":"Kamal","family":"Acharya","sequence":"first","affiliation":[{"name":"Security and Optimization for Networked Globe Laboratory (SONG Lab), Department of Information Systems, University of Maryland, Baltimore County, Baltimore, MD, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6757-105X","authenticated-orcid":false,"given":"Alvaro","family":"Velasquez","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Colorado, Boulder, CO, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2631-9223","authenticated-orcid":false,"given":"Houbing Herbert","family":"Song","sequence":"additional","affiliation":[{"name":"Security and Optimization for Networked Globe Laboratory (SONG Lab), Department of Information Systems, University of Maryland, Baltimore County, Baltimore, MD, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1250"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.341"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.655"},{"key":"ref4","article-title":"Reinforced self-training (rest) for language modeling","author":"Gulcehre","year":"2023"},{"key":"ref5","article-title":"A survey of large language models","author":"Zhao","year":"2023"},{"issue":"2","key":"ref6","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3605943","article-title":"Recent advances in natural language processing via large pre-trained language models: A survey","volume":"56","author":"Min","year":"2023","journal-title":"ACM Comput. Surv."},{"key":"ref7","article-title":"Large language models: A comprehensive survey of its applications, challenges, limitations, and future prospects","author":"Hadi","year":"2023","journal-title":"Authorea Preprints"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3641289"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.411"},{"key":"ref10","article-title":"ChatGPT for good? On opportunities and challenges of large language models for education","volume-title":"Learn. Individual Differences","volume":"103","author":"Kasneci","year":"2023"},{"key":"ref11","article-title":"A review on language models as knowledge bases","author":"AlKhamissi","year":"2022"},{"key":"ref12","article-title":"Language models as or for knowledge bases","author":"Razniewski","year":"2021"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.67"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3639372"},{"key":"ref15","article-title":"Aligning large language models with human: A survey","author":"Wang","year":"2023"},{"key":"ref16","article-title":"Instruction tuning for large language models: A survey","author":"Zhang","year":"2023"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00704"},{"key":"ref18","article-title":"Trustworthy LLMs: A survey and guideline for evaluating large language models\u2019 alignment","author":"Liu","year":"2023"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE-FoSE59343.2023.00008"},{"key":"ref20","article-title":"Siren\u2019s song in the AI ocean: A survey on hallucination in large language models","author":"Zhang","year":"2023"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/BigData59044.2023.10386743"},{"key":"ref22","article-title":"Large language models for robotics: A survey","author":"Zeng","year":"2023"},{"key":"ref23","article-title":"Large language models for information retrieval: A survey","author":"Zhu","year":"2023"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4020-6710-5_3"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1002\/j.1538-7305.1951.tb01366.x"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/365153.365168"},{"key":"ref27","article-title":"Procedures as a representation for data in a computer program for understanding natural language","author":"Winograd","year":"1971"},{"key":"ref28","first-page":"151","article-title":"A stochastic approach to parsing","volume-title":"Proc. Coling 1986 Volume 1 11th Int. Conf. Comput. Linguistics","author":"Sampson","year":"1986"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.3115\/112405.112427"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/5.880083"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref32","first-page":"1","article-title":"A neural probabilistic language model","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"13","author":"Bengio","year":"2000"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482005"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/1150402.1150464"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2765695"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.521"},{"key":"ref37","first-page":"1","article-title":"BinaryConnect: Training deep neural networks with binary weights during propagations","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Courbariaux","year":"2015"},{"key":"ref38","first-page":"1","article-title":"Structured transforms for small-footprint deep learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Sindhwani","year":"2015"},{"key":"ref39","first-page":"1","article-title":"Learning both weights and connections for efficient neural network","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Han","year":"2015"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2857824"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.15"},{"key":"ref42","first-page":"1","article-title":"Exploiting linear structure within convolutional networks for efficient evaluation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"27","author":"Denton","year":"2014"},{"key":"ref43","article-title":"A survey of model compression and acceleration for deep neural networks","author":"Cheng","year":"2017"},{"key":"ref44","article-title":"Distilling the knowledge in a neural network","author":"Hinton","year":"2015"},{"key":"ref45","first-page":"16","article-title":"The era of cognitive systems: An inside look at IBM Watson and how it works","volume":"1","author":"High","year":"2012","journal-title":"IBM Corporation, Redbooks"},{"key":"ref46","article-title":"Efficient estimation of word representations in vector space","author":"Mikolov","year":"2013"},{"key":"ref47","first-page":"1","article-title":"Sequence to sequence learning with neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"27","author":"Sutskever","year":"2014"},{"key":"ref48","first-page":"1532","article-title":"GloVe: Global vectors for word representation","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process. (EMNLP)","author":"Pennington","year":"2014"},{"key":"ref49","article-title":"FitNets: Hints for thin deep nets","author":"Romero","year":"2014"},{"key":"ref50","article-title":"Paying more attention to attention: Improving the performance of convolutional neural networks via attention transfer","author":"Zagoruyko","year":"2016"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1264"},{"key":"ref52","first-page":"1","article-title":"Attention is all you need","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Vaswani","year":"2017"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.754"},{"key":"ref54","article-title":"Paraphrasing complex network: Network compression via factor transfer","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Kim","year":"2018"},{"key":"ref55","first-page":"2227","article-title":"Deep contextualized word representations","volume-title":"Proc. Conf. North Amer. Chapter Assoc. Comput. Linguistics: Human Lang. Technol.","volume":"1","author":"Peters","year":"2018"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-2029"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5446"},{"key":"ref58","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"ref59","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"key":"ref60","article-title":"Release strategies and the social impacts of language models","author":"Solaiman","year":"2019"},{"issue":"1","key":"ref61","first-page":"5485","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00201"},{"key":"ref63","article-title":"Contrastive representation distillation","author":"Tian","year":"2019"},{"key":"ref64","article-title":"Revisit knowledge distillation: A teacher-free framework","author":"Yuan","year":"2019"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00454"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5963"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00285"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2016.41"},{"key":"ref69","first-page":"5714","article-title":"Self-supervised label augmentation via input transformations","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lee","year":"2020"},{"key":"ref70","article-title":"Explaining sequence-level knowledge distillation as data-augmentation for neural machine translation","author":"Gordon","year":"2019"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33011190"},{"key":"ref72","article-title":"Dataset distillation","author":"Wang","year":"2018"},{"key":"ref73","article-title":"Flexible dataset distillation: Learn labels instead of images","author":"Bohdal","year":"2020"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2021.3054609"},{"key":"ref75","article-title":"GShard: Scaling giant models with conditional computation and automatic sharding","author":"Lepikhin","year":"2020"},{"key":"ref76","first-page":"5547","article-title":"Glam: Efficient scaling of language models with mixture-of-experts","volume-title":"Int. Conf. Mach. Learn.","author":"Du","year":"2022"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.281"},{"key":"ref78","article-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020"},{"key":"ref79","article-title":"Training compute-optimal large language models","author":"Hoffmann","year":"2022"},{"key":"ref80","article-title":"Will we run out of data? An analysis of the limits of scaling datasets in machine learning","author":"Villalobos","year":"2022"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.354"},{"key":"ref82","article-title":"Language models in the loop: Incorporating prompting into weak supervision","author":"Smith","year":"2022"},{"key":"ref83","article-title":"Code alpaca: An instruction-following LLaMA model for code generation","author":"Chaudhary","year":"2023"},{"key":"ref84","article-title":"WizardMath: Empowering mathematical reasoning for large language models via reinforced evol-instruct","author":"Luo","year":"2023"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.183"},{"key":"ref86","article-title":"Lion: Adversarial distillation of closed-source large language model","author":"Jiang","year":"2023"},{"key":"ref87","article-title":"Self-rewarding language models","author":"Yuan","year":"2024"},{"key":"ref88","article-title":"mT5: A massively multilingual pre-trained text-to-text transformer","author":"Xue","year":"2020"},{"key":"ref89","article-title":"Finetuned language models are zero-shot learners","author":"Wei","year":"2021"},{"key":"ref90","article-title":"LaMDA: Language models for dialog applications","author":"Thoppilan","year":"2022"},{"key":"ref91","first-page":"3843","article-title":"Solving quantitative reasoning problems with language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Lewkowycz","year":"2022"},{"key":"ref92","first-page":"1","article-title":"Ul2: Unifying language learning paradigms","volume-title":"Proc. 11th Int. Conf. Learn. Representations","author":"Tay","year":"2023"},{"issue":"240","key":"ref93","first-page":"1","article-title":"PaLM: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2023","journal-title":"J. Mach. Learn. Res."},{"key":"ref94","article-title":"Scaling instruction-finetuned language models","author":"Chung","year":"2022"},{"issue":"8","key":"ref95","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"ref96","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Brown","year":"2020"},{"key":"ref97","article-title":"Evaluating large language models trained on code","author":"Chen","year":"2021"},{"key":"ref98","article-title":"WebGPT: Browser-assisted question-answering with human feedback","author":"Nakano","year":"2021"},{"key":"ref99","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Ouyang","year":"2022"},{"key":"ref100","article-title":"GPT-4 technical report","author":"OpenAI","year":"2023"},{"key":"ref101","article-title":"GPT-J-6B: A 6 billion parameter autoregressive language model","author":"Wang","year":"2021"},{"key":"ref102","article-title":"GPT-Neo: large scale autoregressive language modeling with mesh-tensorflow","author":"Black","year":"2021"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.bigscience-1.9"},{"key":"ref104","article-title":"Scaling language models: Methods, analysis & insights from training Gopher","author":"Rae","year":"2021"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1126\/science.abq1158"},{"key":"ref106","article-title":"Improving alignment of dialogue agents via targeted human judgements","author":"Glaese","year":"2022"},{"key":"ref107","article-title":"Galactica: A large language model for science","author":"Taylor","year":"2022"},{"key":"ref108","article-title":"Opt: Open pre-trained transformer language models","author":"Zhang","year":"2022"},{"key":"ref109","article-title":"OPT-IML: Scaling language model instruction meta learning through the lens of generalization","author":"Iyer","year":"2022"},{"key":"ref110","article-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref111","article-title":"Multitask prompted training enables zero-shot task generalization","author":"Sanh","year":"2021"},{"key":"ref112","article-title":"BLOOM: A 176B-parameter open-access multilingual language model","author":"Workshop","year":"2022"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.891"},{"key":"ref114","article-title":"ERNIE 2.0: A continual pre-training framework for language understanding","author":"Sun","year":"2019"},{"key":"ref115","article-title":"Ernie 3.0: Large-scale knowledge enhanced pre-training for language understanding and generation","author":"Sun","year":"2021"},{"key":"ref116","article-title":"ERNIE 3.0 Titan: Exploring larger-scale knowledge enhanced pre-training for language understanding and generation","author":"Wang","year":"2021"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01453-z"},{"key":"ref118","first-page":"1","article-title":"Learning efficient object detection models with knowledge distillation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Chen","year":"2017"},{"key":"ref119","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00363"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2013.50"},{"key":"ref121","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i8.16865"},{"key":"ref122","article-title":"Like what you like: Knowledge distill via neuron selectivity transfer","author":"Huang","year":"2017"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01252-6_17"},{"key":"ref124","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00143"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013779"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_21"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/158"},{"key":"ref128","article-title":"Graph-based knowledge distillation by multi-head attention network","author":"Lee","year":"2019"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00241"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-28954-6_10"},{"key":"ref131","first-page":"1","article-title":"A unified approach to interpreting model predictions","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Lundberg","year":"2017"},{"key":"ref132","article-title":"BART: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension","author":"Lewis","year":"2019"},{"key":"ref133","article-title":"BLOOM: A 176b-parameter open-access multilingual language model","author":"Scao","year":"2022"},{"key":"ref134","article-title":"GLM-130B: An open bilingual pre-trained model","author":"Zeng","year":"2022"},{"key":"ref135","article-title":"Transcending scaling laws with 0.1% extra compute","author":"Tay","year":"2022"},{"key":"ref136","article-title":"Scaling laws and interpretability of learning from repeated data","author":"Hernandez","year":"2022"},{"key":"ref137","article-title":"Alignment of language agents","author":"Kenton","year":"2021"},{"key":"ref138","article-title":"Formal mathematics statement curriculum learning","author":"Polu","year":"2022"},{"key":"ref139","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.388"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.71"},{"key":"ref141","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00410"},{"key":"ref142","doi-asserted-by":"crossref","first-page":"7811","DOI":"10.18653\/v1\/2020.acl-main.698","article-title":"Negated and misprimed probes for pretrained language models: Birds can talk, but cannot fly","volume-title":"Proc. 58th Annu. Meeting Assoc. Comput. Linguistics","author":"Kassner","year":"2020"},{"key":"ref143","article-title":"Modifying memories in transformer models","author":"Zhu","year":"2020"},{"key":"ref144","article-title":"Knowledge neurons in pretrained transformers","author":"Dai","year":"2021"},{"key":"ref145","article-title":"Editing factual knowledge in language models","author":"Cao","year":"2021"},{"key":"ref146","article-title":"Do language models have beliefs? Methods for detecting, updating, and visualizing model beliefs","author":"Hase","year":"2021"},{"key":"ref147","article-title":"Fast model editing at scale","author":"Mitchell","year":"2021"},{"key":"ref148","article-title":"Towards continual knowledge learning of language models","author":"Jang","year":"2021"},{"key":"ref149","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.conll-1.45"},{"key":"ref150","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/537"},{"key":"ref151","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.9"},{"key":"ref152","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.317"},{"key":"ref153","first-page":"22231","article-title":"Measuring systematic generalization in neural proof generation with transformers","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Gontier","year":"2020"},{"key":"ref154","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21371"},{"key":"ref155","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.110"},{"key":"ref156","article-title":"Analysing mathematical reasoning abilities of neural models","author":"Saxton","year":"2019"},{"key":"ref157","first-page":"1","article-title":"Commonsense reasoning with implicit knowledge in natural language","volume-title":"Proc. 3rd Conf. Automated Knowl. Base Construction","author":"Banerjee","year":"2021"},{"key":"ref158","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.598"},{"key":"ref159","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1487"},{"key":"ref160","article-title":"Attention is not explanation","author":"Jain","year":"2019"},{"key":"ref161","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1002"},{"key":"ref162","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1282"},{"key":"ref163","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN52387.2021.9533563"},{"key":"ref164","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.446"},{"key":"ref165","first-page":"17359","article-title":"Locating and editing factual associations in GPT","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Meng","year":"2022"},{"key":"ref166","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.492"},{"key":"ref167","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.75"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.153"},{"key":"ref169","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.internlp-1.1"},{"key":"ref170","article-title":"Wt5?! training text-to-text models to explain their predictions","author":"Narang","year":"2020"},{"key":"ref171","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.382"},{"key":"ref172","doi-asserted-by":"publisher","DOI":"10.3233\/faia220218"},{"key":"ref173","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1011"},{"key":"ref174","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.366"},{"key":"ref175","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013027"},{"key":"ref176","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.80"},{"key":"ref177","article-title":"I2D2: Inductive knowledge distillation with neurologic and self-imitation","author":"Bhagavatula","year":"2022"},{"key":"ref178","article-title":"Specializing smaller language models towards multi-step reasoning","author":"Fu","year":"2023"},{"key":"ref179","article-title":"Localized symbolic knowledge distillation for visual commonsense models","author":"Park","year":"2023"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"ref181","article-title":"Stanford Alpaca: An instruction-following LLaMA model","author":"Taori","year":"2023"},{"key":"ref182","article-title":"WizardLM: Empowering large language models to follow complex instructions","author":"Xu","year":"2023"},{"key":"ref183","article-title":"Vicuna: An open-source chatbot impressing GPT-4 with 90%* ChatGPT quality","author":"Chiang","year":"2023"},{"key":"ref184","article-title":"Koala: A dialogue model for academic research","author":"Geng","year":"2023"},{"key":"ref185","article-title":"Ask me anything: A simple strategy for prompting language models","author":"Arora","year":"2022"},{"key":"ref186","article-title":"QAmeleon: Multilingual QA with only 5 examples","author":"Agrawal","year":"2022"},{"key":"ref187","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.507"},{"key":"ref188","article-title":"Orca: Progressive learning from complex explanation traces of GPT-4","author":"Mukherjee","year":"2023"},{"key":"ref189","article-title":"Orca 2: Teaching small language models how to reason","author":"Mitra","year":"2023"},{"key":"ref190","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2024.3351798"},{"key":"ref191","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2023.3311428"},{"key":"ref192","article-title":"Transfer from imprecise and abstract models to autonomous technologies (tiamat)","volume-title":"Defense Advanced Research Projects Agency (DARPA) Program Solicitation","author":"Velasquez","year":"2023"},{"key":"ref193","article-title":"Lawyer LLaMA technical report","author":"Huang","year":"2023"},{"key":"ref194","article-title":"ChatLaw: Open-source legal large language model with integrated external knowledge bases","author":"Cui","year":"2023"},{"key":"ref195","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.725"},{"key":"ref196","doi-asserted-by":"publisher","DOI":"10.7759\/cureus.40895"},{"key":"ref197","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3615285"},{"key":"ref198","article-title":"Darwin series: Domain specific large language models for natural science","author":"Xie","year":"2023"},{"key":"ref199","article-title":"SciGLM: Training scientific language models with self-reflective instruction annotation and tuning","author":"Zhang","year":"2024"},{"key":"ref200","article-title":"OmniQuant: Omnidirectionally calibrated quantization for large language models","author":"Shao","year":"2023"},{"key":"ref201","article-title":"PB-LLM: Partially binarized large language models","author":"Shang","year":"2023"},{"key":"ref202","first-page":"55006","article-title":"LIMA: Less is more for alignment","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Zhou","year":"2023"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9078688\/10794552\/10597596.pdf?arnumber=10597596","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T01:09:25Z","timestamp":1755911365000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10597596\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":202,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tai.2024.3428519","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12]]}}}