{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T02:06:17Z","timestamp":1779501977164,"version":"3.53.1"},"reference-count":69,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T00:00:00Z","timestamp":1779494400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T00:00:00Z","timestamp":1779494400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Empir Software Eng"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1007\/s10664-026-10880-w","type":"journal-article","created":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T01:55:53Z","timestamp":1779501353000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Exploring and improving knowledge distillation for pre-trained code models"],"prefix":"10.1007","volume":"31","author":[{"given":"Weifeng","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ruifeng","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongyan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying","family":"Fu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Min","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9538-9121","authenticated-orcid":false,"given":"Meng","family":"Yan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,23]]},"reference":[{"key":"10880_CR1","doi-asserted-by":"crossref","unstructured":"Al Debeyan F, Hall T, Madeyski L (2025) Emerging results in using explainable AI to improve software vulnerability prediction. In: Proceedings of the 33rd ACM international conference on the foundations of software engineering, pp 561\u2013565","DOI":"10.1145\/3696630.3728499"},{"key":"10880_CR2","doi-asserted-by":"crossref","unstructured":"An Y, Zhao X, Yu T, Tang M, Wang J (2024). Fluctuationbased adaptive structured pruning for large language models. In: Proceedings of the AAAI conference on artificial intelligence. Vol 38. 10, pp 10865\u201310873","DOI":"10.1609\/aaai.v38i10.28960"},{"key":"10880_CR3","unstructured":"Ba J, Caruana R (2014) Do deep nets really need to be deep? Adv Neural Inf Process Syst 27"},{"key":"10880_CR4","doi-asserted-by":"crossref","unstructured":"Cheng H, Zhang M, Shi JQ (2024) A survey on deep neural network pruning: taxonomy, comparison, analysis, and recommendations. IEEE Transactions on Pattern Analysis and Machine Intelligence","DOI":"10.1109\/TPAMI.2024.3447085"},{"key":"10880_CR5","doi-asserted-by":"crossref","unstructured":"Cho JH, Hariharan B (2019) On the efficacy of knowledge distillation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4794\u20134802","DOI":"10.1109\/ICCV.2019.00489"},{"key":"10880_CR6","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2018) Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"10880_CR7","unstructured":"Dohmke T, Iansiti M, Richards GL (2023) Sea change in software development: economic and productivity analysis of the AI-powered developer lifecycle. https:\/\/api.semanticscholar.org\/CorpusID:259261950"},{"key":"10880_CR8","doi-asserted-by":"crossref","unstructured":"Dvivedi SS, Vijay V, Rahul Pujari SL, Lodh S, Kumar D (2024). A comparative analysis of large language models for code documentation generation. In: Proceedings of the 1st ACM international conference on AI-powered software, pp 65\u201373","DOI":"10.1145\/3664646.3664765"},{"key":"10880_CR9","doi-asserted-by":"publisher","unstructured":"Feng Z, Guo D, Tang D, Duan N, Feng X, Gong M, Shou L, Qin B, Liu T, Jiang D, Zhou M (2020) CodeBERT: a pre-trained model for programming and natural languages. In: Cohn T, He Y, Liu Y (eds) Findings of the association for computational linguistics: EMNLP 2020. Online: Association for Computational Linguistics, pp 1536\u20131547. https:\/\/doi.org\/10.18653\/v1\/2020.findings-emnlp.139","DOI":"10.18653\/v1\/2020.findings-emnlp.139"},{"key":"10880_CR10","doi-asserted-by":"crossref","unstructured":"Feng S, Suo W, Wu Y, Zou D, Liu Y, Jin H (2024) Machine learning is all you need: a simple token-based approach for effective code clone detection. In: Proceedings of the IEEE\/ACM 46th international conference on software engineering, pp 1\u201313","DOI":"10.1145\/3597503.3639114"},{"key":"10880_CR11","doi-asserted-by":"publisher","first-page":"1061","DOI":"10.1162\/tacl_a_00413","volume":"9","author":"P Ganesh","year":"2021","unstructured":"Ganesh P, Chen Y, Lou X, Khan MA, Yang Y, Sajjad H, Nakov P, Chen D, Winslett M (2021) Compressing large-scale transformer-based models: a case study on bert. Trans Assoc Comput Linguist 9:1061\u20131080","journal-title":"Trans Assoc Comput Linguist"},{"key":"10880_CR12","unstructured":"GitHub Copilot Community (2023) GitHab. GitHub Copilot Community. https:\/\/github.com\/orgs\/community\/"},{"key":"10880_CR13","unstructured":"Gong Y, Liu L, Yang M, Bourdev L (2014) Compressing deep convolutional networks using vector quantization. arXiv preprint arXiv:1412.6115"},{"key":"10880_CR14","doi-asserted-by":"crossref","unstructured":"Gordon MA, Duh K, Andrews N (2020). Compressing bert: studying the effects of weight pruning on transfer learning. arXiv preprint arXiv:2002.08307","DOI":"10.18653\/v1\/2020.repl4nlp-1.18"},{"issue":"6","key":"10880_CR15","doi-asserted-by":"publisher","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","volume":"129","author":"J Gou","year":"2021","unstructured":"Gou J, Baosheng Yu, Maybank SJ, Tao D (2021) Knowledge distillation: a survey. Int J Comput Vision 129(6):1789\u20131819","journal-title":"Int J Comput Vision"},{"key":"10880_CR16","unstructured":"Guo D, Ren S, Lu S, Feng Z, Tang D, Liu S, Zhou L, Duan N, Svyatkovskiy A, Fu S et al (2020). Graphcodebert: pretraining code representations with data flow. arXiv preprint arXiv:2009.08366"},{"key":"10880_CR17","unstructured":"Han S, Mao H, Dally WJ (2015a) Deep compression: compressing deep neural networks with pruning, trained quantization and huffman coding. arXiv preprint arXiv:1510.00149"},{"key":"10880_CR18","unstructured":"Han S, Pool J, Tran J, Dally W (2015b) Learning both weights and connections for efficient neural network. Adv Neural Inf Process Syst 28"},{"key":"10880_CR19","unstructured":"Hinton G (2015) Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531"},{"key":"10880_CR20","unstructured":"Huang Y, Li Y, Wu W, Zhang J, Lyu MR (2023) Do not give away my secrets: Uncovering the privacy issue of neural code completion tools. arXiv preprint arXiv:2309.07639"},{"key":"10880_CR21","doi-asserted-by":"crossref","unstructured":"Huang L, Sun W, Yan M (2025) Iterative Generation of Adversarial Example for Deep Code Models. In: 2025 IEEE\/ACM 47th international conference on software engineering (ICSE). IEEE Computer Society, pp 623\u2013623","DOI":"10.1109\/ICSE55347.2025.00086"},{"key":"10880_CR22","unstructured":"Hui B, Yang J, Cui Z, Yang J, Liu D, Zhang L, Liu T, Zhang J, Yu B, Dang K et al (2024) Qwen2. 5-Coder Technical Report. arXiv preprint arXiv:2409.12186"},{"key":"10880_CR23","unstructured":"Husain H, Wu H-H, Gazit T, Allamanis M, Brockschmidt M (2019) Codesearchnet challenge: evaluating the state of semantic code search. arXiv preprint arXiv:1909.09436"},{"key":"10880_CR24","doi-asserted-by":"publisher","unstructured":"Jiao X, Yin Y, Shang L, Jiang X, Chen X, Li L, Wang F, Liu Q (2020) TinyBERT: distilling BERT for natural language understanding. In: Cohn T, He Y, Liu Y (eds) Findings of the association for computational linguistics: EMNLP 2020. Online: Association for Computational Linguistics, pp.4163\u20134174. https:\/\/doi.org\/10.18653\/v1\/2020.findingsemnlp.372","DOI":"10.18653\/v1\/2020.findingsemnlp.372"},{"key":"10880_CR25","unstructured":"Kanade A, Maniatis P, Balakrishnan G, Shi K (2020). Learning and evaluating contextual embedding of source code. In: International conference on machine learning. PMLR, pp 5110\u20135121"},{"key":"10880_CR26","doi-asserted-by":"publisher","unstructured":"Karmakar A, Robbes R (2021) What do pre-trained code models know about code? In: 2021 36th IEEE\/ACM international conference on automated software engineering (ASE), pp 1332\u20131336. https:\/\/doi.org\/10.1109\/ASE51524.2021.9678927","DOI":"10.1109\/ASE51524.2021.9678927"},{"key":"10880_CR27","first-page":"62414","volume":"36","author":"A Kuzmin","year":"2023","unstructured":"Kuzmin A, Nagel M, Van Baalen M, Behboodi A, Blankevoort T (2023) Pruning vs quantization: which is better? Adv Neural Inf Process Syst 36:62414\u201362427","journal-title":"Adv Neural Inf Process Syst"},{"issue":"3","key":"10880_CR28","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3506695","volume":"31","author":"L Liao","year":"2022","unstructured":"Liao L, Li H, Shang W, Ma L (2022) An empirical study of the impact of hyperparameter tuning and model optimization on the performance properties of deep neural networks. ACM Trans Softw Eng Methodol (TOSEM) 31(3):1\u201340","journal-title":"ACM Trans Softw Eng Methodol (TOSEM)"},{"key":"10880_CR29","doi-asserted-by":"publisher","unstructured":"Lin B, Wang S, Liu Z, Liu Y, Xia X, Mao X (2023) CCT5: a code-change-oriented pre-trained model. In: Proceedings of the 31st ACM joint European software engineering conference and symposium on the foundations of software engineering. ESEC\/FSE 2023. San Francisco, CA, USA: Association for Computing Machinery, pp 1509\u2013 1521. https:\/\/doi.org\/10.1145\/3611643.3616339. isbn: 9798400703270","DOI":"10.1145\/3611643.3616339"},{"key":"10880_CR30","doi-asserted-by":"crossref","unstructured":"Liu Y, Sheng L, Shao J, Yan J, Xiang S, Pan C (2018) Multi-label image classification via knowledge distillation from weakly-supervised detection. In: CoRR","DOI":"10.1145\/3240508.3240567"},{"key":"10880_CR31","doi-asserted-by":"crossref","unstructured":"Lo D (2023) Trustworthy and synergistic artificial intelligence for software engineering: Vision and roadmaps. In: 2023 IEEE\/ACM international conference on software engineering: future of software engineering (ICSE-FoSE). IEEE, pp 69\u201385","DOI":"10.1109\/ICSE-FoSE59343.2023.00010"},{"key":"10880_CR32","unstructured":"Lu S, Guo D, Ren S, Huang J, Svyatkovskiy A, Blanco A, Clement CB, Drain D, Jiang D, Tang D, Li G, Zhou L, Shou L, Zhou L, Tufano M, Gong M, Zhou M, Duan N, Sundaresan N, Deng SK, Fu S, Liu S (2021) CodeXGLUE: a machine learning benchmark dataset for code understanding and generation. arXiv:2102.04664"},{"key":"10880_CR33","doi-asserted-by":"crossref","unstructured":"Luo Q, Ye Y, Liang S, Zhang Z, Qin Y, Lu Y, Wu Y, Cong X, Lin Y, Zhang Y et al. (2024). Repoagent: an llm-powered open-source framework for repository-level code documentation generation. arXiv preprint arXiv:2402.16667","DOI":"10.18653\/v1\/2024.emnlp-demo.46"},{"key":"10880_CR34","doi-asserted-by":"crossref","unstructured":"Mastropaolo A, Scalabrino S, Cooper N, Palacio DN, Poshyvanyk D, Oliveto R, Bavota G (2021) Studying the usage of text-to-text transfer transformer to support code-related tasks. In: 2021 IEEE\/ACM 43rd international conference on software engineering (ICSE). IEEE, pp 336\u2013347","DOI":"10.1109\/ICSE43902.2021.00041"},{"key":"10880_CR35","doi-asserted-by":"crossref","unstructured":"Mirzadeh SI, Farajtabar M, Li A, Levine N, Matsukawa A, Ghasemzadeh H (2020) Improved knowledge distillation via teacher assistant. In: Proceedings of the AAAI conference on artificial intelligence. Vol 34. 04, pp 5191\u20135198","DOI":"10.1609\/aaai.v34i04.5963"},{"key":"10880_CR36","unstructured":"Nijkamp E, Pang B, Hayashi H, Tu L, Wang H, Zhou Y, Savarese S, Xiong C (2022). Codegen: an open large language model for code with multi-turn program synthesis. arXiv preprint arXiv:2203.13474"},{"key":"10880_CR37","doi-asserted-by":"publisher","unstructured":"Niu C, Li C, Luo B, Ng V (2022) Deep Learning Meets Software Engineering: A Survey on Pre-Trained Models of Source Code. In: De Raedt L (ed) Proceedings of the thirty-first international joint conference on artificial intelligence, IJCAI-22. Survey Track. International Joint Conferences on Artificial Intelligence Organization, pp 5546\u20135555. https:\/\/doi.org\/10.24963\/ijcai.2022\/775","DOI":"10.24963\/ijcai.2022\/775"},{"key":"10880_CR38","doi-asserted-by":"publisher","unstructured":"Niu C, Li C, Ng V, Chen D, Ge J, Luo B (2023a) An empirical comparison of pre-trained models of source code. In: 2023 IEEE\/ACM 45th International Conference on Software Engineering (ICSE), pp 2136\u20132148. https:\/\/doi.org\/10.1109\/ICSE48619.2023.00180.","DOI":"10.1109\/ICSE48619.2023.00180."},{"key":"10880_CR39","unstructured":"Niu L, Mirza S, Maradni Z, P\u00f6pper C (2023b) CodexLeaks: privacy leaks from code generation language models in GitHub copilot. In: 32nd USENIX Security Symposium (USENIX Security 23), pp 2133\u20132150"},{"key":"10880_CR40","unstructured":"Phuong M, Lampert C (2019) Towards understanding knowledge distillation. In: International conference on machine learning. PMLR, pp 5142\u2013 5151"},{"key":"10880_CR41","unstructured":"Romero A, Ballas N, Kahou SE, Chassang A, Gatta C, Bengio Y (2014) Fitnets: hints for thin deep nets. arXiv preprint arXiv:1412.6550"},{"key":"10880_CR42","unstructured":"Sanh V, Debut L, Chaumond J, Wolf T (2019) DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter. arXiv preprint arXiv:1910.01108"},{"key":"10880_CR43","doi-asserted-by":"crossref","unstructured":"Shi J, Yang Z, Xu B, Kang HJ, Lo D (2022) Compressing pre-trained models of code into 3 mb. In: Proceedings of the 37th IEEE\/ACM international conference on automated software engineering, pp 1\u201312","DOI":"10.1145\/3551349.3556964"},{"key":"10880_CR44","unstructured":"Shi J, Yang Z, Kang HJ, Xu B, He J, Lo D (2023). Smaller, faster, greener: compressing pre-trained code models via surrogateassisted optimization. arXiv preprint arXiv:2309.04076"},{"key":"10880_CR45","doi-asserted-by":"publisher","unstructured":"Sun Z, Yu H, Song X, Liu R, Yang Y, Zhou D (2020) MobileBERT: a Compact Task-Agnostic BERT for Resource-Limited Devices. In: Jurafsky D, Chai J, Schluter N, Tetreault J (eds) Proceedings of the 58th annual meeting of the association for computational linguistics. Online: Association for Computational Linguistics, pp 2158\u20132170. https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.195","DOI":"10.18653\/v1\/2020.acl-main.195"},{"key":"10880_CR46","doi-asserted-by":"crossref","unstructured":"Svajlenko J, Islam JF, Keivanloo I, Roy CK, MiaMM (2014) Towards a big data curated benchmark of inter-project code clones. In: 2014 IEEE international conference on software maintenance and evolution. IEEE, pp 476\u2013480","DOI":"10.1109\/ICSME.2014.77"},{"key":"10880_CR47","doi-asserted-by":"crossref","unstructured":"Svyatkovskiy A, Lee S, Hadjitofi A, Riechert M, Franco JV, Allamanis M (2021) Fast and memory-efficient neural code completion. In: 2021 IEEE\/ACM 18th international conference on mining software repositories (MSR). IEEE, pp 329\u2013340","DOI":"10.1109\/MSR52588.2021.00045"},{"key":"10880_CR48","unstructured":"Tang R, Lu Y, Liu L, Mou L, Vechtomova O, Lin J (2019) Distilling task-specific knowledge from bert into simple neural networks. arXiv preprint arXiv:1903.12136"},{"key":"10880_CR49","first-page":"19749","volume":"36","author":"SY Tew","year":"2023","unstructured":"Tew SY, Boley M, Schmidt D (2023) Bayes beats cross validation: efficient and accurate ridge regression via expectation maximization. Adv Neural Inf Process Syst 36:19749\u201319768","journal-title":"Adv Neural Inf Process Syst"},{"key":"10880_CR50","doi-asserted-by":"crossref","unstructured":"Wang Y, Wang W, Joty S, Hoi SCH (2021) Codet5: identifieraware unified pre-trained encoder-decoder models for code understanding and generation. arXiv preprint arXiv:2109.00859","DOI":"10.18653\/v1\/2021.emnlp-main.685"},{"key":"10880_CR51","doi-asserted-by":"crossref","unstructured":"Wang Y, Le H, Gotmare AD, Bui NDQ, Li J, Hoi SCH (2023) Codet5+: open code large language models for code understanding and generation. arXiv preprint arXiv:2305.07922","DOI":"10.18653\/v1\/2023.emnlp-main.68"},{"key":"10880_CR52","doi-asserted-by":"crossref","unstructured":"Wang H, Tang Z, Tan SH, Wang J, Liu Y, Fang H, Xia C, Wang Z (2024) Combining structured static code information and dynamic symbolic traces for software vulnerability prediction. In: Proceedings of the IEEE\/ACM 46th international conference on software engineering, pp 1\u201313","DOI":"10.1145\/3597503.3639212"},{"key":"10880_CR53","doi-asserted-by":"crossref","unstructured":"Wei X, Gonugondla SK, Wang S, Ahmad W, Ray B, Qian H, Li X, Kumar V, Wang Z, Tian Y et al (2023) Towards greener yet powerful code generation via quantization: an empirical study. In: Proceedings of the 31st ACM joint european software engineering conference and symposium on the foundations of software engineering, pp 224\u2013 236","DOI":"10.1145\/3611643.3616302"},{"key":"10880_CR54","doi-asserted-by":"crossref","unstructured":"Xia M, Zhong Z, Chen D (2022) Structured pruning learns compact and accurate models. arXiv preprint arXiv:2204.00408","DOI":"10.18653\/v1\/2022.acl-long.107"},{"key":"10880_CR55","unstructured":"Xiao G, Lin J, Seznec M, Wu H, Demouth J, Han S (2023).SmoothQuant: accurate and efficient post-training quantization for large language models. In: Krause A, Brunskill E, Cho K, Engelhardt B, Sabato S, Scarlett J (eds) Proceedings of the 40th international conference on machine learning. Vol 202. Proceedings of Machine Learning Research. PMLR, pp 38087\u201338099. https:\/\/proceedings.mlr.press\/v202\/xiao23c.html"},{"key":"10880_CR56","doi-asserted-by":"publisher","unstructured":"Xu C, Zhou W, Ge T, Wei F, Zhou M (2020) BERT-ofTheseus: compressing BERT by progressive module replacing. In: Webber B, Cohn T, He Y, Liu Y (eds) Proceedings of the 2020 conference on empirical methods in natural language processing (EMNLP). Online: Association for Computational Linguistics, pp 7859\u20137869. https:\/\/doi.org\/10.18653\/v1\/2020.emnlp-main.633","DOI":"10.18653\/v1\/2020.emnlp-main.633"},{"key":"10880_CR57","doi-asserted-by":"crossref","unstructured":"Xu FF, Alon U, Neubig G, Hellendoorn VJ (2022) A systematic evaluation of large language models of code. In: Proceedings of the 6th ACM SIGPLAN international symposium on machine programming, pp 1\u201310","DOI":"10.1145\/3520312.3534862"},{"key":"10880_CR58","unstructured":"Xu X, Li M, Tao C, Shen T, Cheng R, Li J, Xu C, Tao D, Zhou T (2024) A survey on knowledge distillation of large language models. arXiv preprint arXiv:2402.13116"},{"key":"10880_CR59","unstructured":"Yang Z, Zhao Z, Wang C, Shi J, Kim D, Han D, Lo D (2023) What do code models memorize? An empirical study on large language models of code 10. arXiv preprint arXiv:2308.09932"},{"issue":"2","key":"10880_CR60","doi-asserted-by":"publisher","first-page":"296","DOI":"10.1109\/TSE.2023.3347898","volume":"50","author":"Y Yang","year":"2024","unstructured":"Yang Y, Xing H, Gao Z, Chen J, Ni C, Xia X, Lo D (2024) Federated learning for software engineering: A case study of code clone detection and defect prediction. IEEE Trans Softw Eng 50(2):296\u2013321","journal-title":"IEEE Trans Softw Eng"},{"key":"10880_CR61","unstructured":"Yang A, Li A, Yang B, Zhang B, Hui B, Zheng B, Yu B, Gao C, Huang C, Lv C, Zheng C, Liu D, Zhou F, Huang F, Hu F, Ge H, Wei H, Lin H, Tang J, Yang J, Tu J, Zhang J, Yang J, Yang J, Zhou J, Zhou J, Lin J, Dang K, Bao K, Yang K, Yu L, Deng L, Li M, Xue M, Li M, Zhang P, Wang P, Zhu Q, Men R, Gao R, Liu S, Luo S, Li T, Tang T, Yin W, Ren X, Wang X, Zhang X, Ren X, Fan Y, Su Y, Zhang Y, Zhang Y, Wan Y, Liu Y, Wang Z, Cui Z, Zhang Z, Zhou Z, Qiu Z (2025) Qwen3 Technical Report. arXiv preprint arXiv:2505.09388"},{"key":"10880_CR62","first-page":"27168","volume":"35","author":"Z Yao","year":"2022","unstructured":"Yao Z, Aminabadi RY, Zhang M, Xiaoxia W, Li C, He Y (2022) Zeroquant: efficient and affordable post-training quantization for large-scale transformers. Adv Neural Inf Process Syst 35:27168\u201327183","journal-title":"Adv Neural Inf Process Syst"},{"key":"10880_CR63","doi-asserted-by":"publisher","first-page":"1626","DOI":"10.1109\/TASLP.2021.3071662","volume":"29","author":"JW Yoon","year":"2021","unstructured":"Yoon JW, Lee H, Kim HY, Cho WI, Kim NS (2021) TutorNet: towards flexible knowledge distillation for end-to-end speech recognition. IEEE\/ACM Trans Audio Speech Lang Process 29:1626\u20131638","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"10880_CR64","doi-asserted-by":"publisher","unstructured":"Zadeh AH, Edo I, Awad OM, Moshovos A (2020) GOBO: quantizing attention-based NLP models for low latency and energy efficient inference. In: 2020 53rd Annual IEEE\/ACM international symposium on microarchitecture (MICRO), pp 811\u2013824. https:\/\/doi.org\/10.1109\/MICRO50266.2020.00071","DOI":"10.1109\/MICRO50266.2020.00071"},{"key":"10880_CR65","doi-asserted-by":"publisher","unstructured":"Zeng Z, Tan H, Zhang H, Li J, Zhang Y, Zhang L (2022) An extensive study on pre-trained models for program understanding and generation. In: Proceedings of the 31st ACM SIGSOFT international symposium on software testing and analysis. ISSTA 2022. Virtual, South Korea: Association for Computing Machinery, pp 39\u201351. https:\/\/doi.org\/10.1145\/3533767.3534390. isbn: 9781450393799","DOI":"10.1145\/3533767.3534390"},{"key":"10880_CR66","doi-asserted-by":"publisher","unstructured":"Zhang W, Hou L, Yin Y, Shang L, Chen X, Jiang X, Liu Q (2020). TernaryBERT: distillation-aware Ultra-low Bit BERT. In: Webber B, Cohn T, He Y, Liu Y (eds) Proceedings of the 2020 conference on empirical methods in natural language processing (EMNLP). Online: Association for Computational Linguistics, pp 509\u2013521. https:\/\/doi.org\/10.18653\/v1\/2020.emnlp-main.37","DOI":"10.18653\/v1\/2020.emnlp-main.37"},{"key":"10880_CR67","unstructured":"Zhang Z, Chen C, Liu B, Liao C, Gong Z, Yu H, Li J, Wang R (2023) survey on language models for code. arXiv preprint arXiv:2311.07989"},{"key":"10880_CR68","unstructured":"Zhou Y, Liu S, Siow J, Du X, Liu Y (2019) Devign: effective vulnerability identification by learning comprehensive program semantics via graph neural networks. Adv Neural Inf Process Syst 32"},{"key":"10880_CR69","doi-asserted-by":"publisher","first-page":"1556","DOI":"10.1162\/tacl_a_00704","volume":"12","author":"X Zhu","year":"2024","unstructured":"Zhu X, Li J, Liu Y, Ma C, Wang W (2024) A survey on model compression for large language models. Trans Assoc Comput Linguist 12:1556\u20131577","journal-title":"Trans Assoc Comput Linguist"}],"container-title":["Empirical Software Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-026-10880-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10664-026-10880-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-026-10880-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T01:56:04Z","timestamp":1779501364000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10664-026-10880-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,23]]},"references-count":69,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,11]]}},"alternative-id":["10880"],"URL":"https:\/\/doi.org\/10.1007\/s10664-026-10880-w","relation":{},"ISSN":["1382-3256","1573-7616"],"issn-type":[{"value":"1382-3256","type":"print"},{"value":"1573-7616","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,23]]},"assertion":[{"value":"17 December 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 May 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declared that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}},{"value":"This declaration is not applicable for this study.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"This declaration is not applicable for this study.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed Consent"}},{"value":"This declaration is not applicable for this study.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Clinical Trial Number"}}],"article-number":"151"}}