{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T05:16:54Z","timestamp":1788239814306,"version":"build-2803163510"},"reference-count":301,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T00:00:00Z","timestamp":1779408000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,8,6]],"date-time":"2026-08-06T00:00:00Z","timestamp":1785974400000},"content-version":"vor","delay-in-days":76,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"DOI":"10.1007\/s10462-026-11538-1","type":"journal-article","created":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T08:15:38Z","timestamp":1779437738000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["On-device large language models: a survey of model compression and system optimization"],"prefix":"10.1007","volume":"59","author":[{"given":"Wanyi","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junhao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiwei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yufan","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianyi","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shengxian","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenxu","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Andi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenyue","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Minxuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinyu","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoshuai","family":"Hao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yinan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yichen","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuwei","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhao","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mengke","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanbiao","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiwu","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jungong","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yike","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,22]]},"reference":[{"key":"11538_CR1","unstructured":"Abdin M, Aneja J et\u00a0al (2024) Phi-3 technical report: A highly capable language model locally on your phone"},{"key":"11538_CR2","unstructured":"Agarwal R, Vieillard N, Zhou Y, Stanczyk P, Ramos S, Geist M, Bachem O (2024) Learning from self-generated mistakes, on-policy distillation of language models"},{"key":"11538_CR3","doi-asserted-by":"crossref","unstructured":"Ainslie J, Lee-Thorp J, de Jong M, Zemlyanskiy Y, Lebron F, Sanghai S (2023) GQA: Training generalized multi-query transformer models from multi-head checkpoints. In The 2023 Conference on Empirical Methods in Natural Language Processing","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"11538_CR4","doi-asserted-by":"publisher","first-page":"478","DOI":"10.1038\/s44287-024-00053-6","volume":"1","author":"L Ale","year":"2024","unstructured":"Ale L, Zhang N, King SA, Chen D (2024) Empowering generative AI through mobile edge computing. Nat Rev Electric Eng 1:478\u2013486","journal-title":"Nat Rev Electric Eng"},{"key":"11538_CR5","unstructured":"Ali AA, Katz S, Wolf L, Titov I (2025) Detecting and pruning prominent but detrimental neurons in large language models. In Second Conference on Language Modeling"},{"key":"11538_CR6","unstructured":"An Y, Zhao X, Yu T, Tang M, Wang J (2024) Fluctuation-based adaptive structured pruning for large language models. In Proceedings of the Thirty-Eighth AAAI Conference on Artificial Intelligence and Thirty-Sixth Conference on Innovative Applications of Artificial Intelligence and Fourteenth Symposium on Educational Advances in Artificial Intelligence, AAAI'24\/IAAI'24\/EAAI'24. AAAI Press"},{"key":"11538_CR7","unstructured":"Ansel J, Kirisame M, Vozna I, Hai T, Roesch J, Heller S, Scherer M, Smith C, Zou J, Sulsky A et\u00a0al (2024) Pytorch 2.0: Torchdynamo\/inductor. In International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS)"},{"key":"11538_CR8","doi-asserted-by":"crossref","unstructured":"Anshumann MA, Zaidi AK, Ahn J, Kwon T, Lee K, Lee H, Lee J (2025) Accelerating knowledge distillation in llms, Sparse logit sampling","DOI":"10.18653\/v1\/2025.acl-long.885"},{"key":"11538_CR9","doi-asserted-by":"crossref","unstructured":"Arya M, Simmhan Y (2025) Understanding the performance and power of llm inferencing on edge accelerators","DOI":"10.1109\/IPDPSW66978.2025.00173"},{"key":"11538_CR10","doi-asserted-by":"crossref","unstructured":"Ashkboos S, Mohtashami A, Croci ML, Li B, Cameron P, Jaggi M, Alistarh D, Hoefler T, Hensman J (2024) Quarot: Outlier-free 4-bit inference in rotated LLMs. In The Thirty-eighth Annual Conference on Neural Information Processing Systems","DOI":"10.52202\/079017-3180"},{"key":"11538_CR11","doi-asserted-by":"crossref","unstructured":"Bai G, Li Y, Ling C, Kim K, Zhao L (2024) Sparsellm: Towards global pruning of pre-trained language models. In Douwe Kiela, Faisal Ladhak, Denis Paperno, Anna Rogers, and Avirup Sil, editors, Advances in Neural Information Processing Systems, vol 37, pp 52213\u201352234. Curran Associates, Inc.","DOI":"10.52202\/079017-1468"},{"key":"11538_CR12","unstructured":"Banbury C, Reddi VJ, Torelli P, Holleman J, Jeffries N, Kiraly C, Montino P, Kanter D, Ahmed S, Pau D, Thakker U, Torrini A, Warden P, Cordaro J, Di Guglielmo G, Duarte J, Gibellini S, Parekh V, Tran H, Wenxu N, Xuesong X, Tran N (2021) Mlperf tiny benchmark"},{"key":"11538_CR13","doi-asserted-by":"crossref","unstructured":"Banitalebi-Dehkordi A, Vedula N, Pei J, Xia F, Wang L, Zhang Y (2021) A general framework of collaborative edge-cloud AI, Auto-split","DOI":"10.1145\/3447548.3467078"},{"key":"11538_CR14","doi-asserted-by":"crossref","unstructured":"Bhuiyan SB, Adib MSH, Bhuiyan MA, Kabir MR, Farazi M, Rahman S, Mohammed N (2025) Z-pruner: Post-training pruning of large language models for efficiency without retraining","DOI":"10.1109\/AICCSA66935.2025.11315381"},{"key":"11538_CR15","unstructured":"Biderman S, Schoelkopf H, Sutawika L, Gao L, Tow J, Abbasi B, Aji AF, Ammanamanchi PS, Black S, Clive J, DiPofi A, Etxaniz J, Fattori B, Forde JZ, Foster C, Hsu J, Jaiswal M, Lee WY, Li H, Lovering C, Muennighoff N, Pavlick E, Phang J, Skowron A, Tan S, Tang X, Wang KA, Winata GI, Yvon F, Zou A (2024) Lessons from the trenches on reproducible evaluation of language models"},{"key":"11538_CR16","unstructured":"Bondarenko Y, Del Chiaro R, Nagel M (2024) Low-rank quantization-aware training for llms"},{"key":"11538_CR17","doi-asserted-by":"crossref","unstructured":"Brandon W, Mishra M, Nrusimha A, Panda R, Kelly JR (2024) Reducing transformer key-value cache size with cross-layer attention","DOI":"10.52202\/079017-2758"},{"key":"11538_CR18","unstructured":"Brown TB, Mann B, Ryder N, Subbiah M, Kaplan J, Dhariwal P, Neelakantan P, Shyam P, Sastry G, Askell A et\u00a0al (2020) Language models are few-shot learners. In Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"11538_CR19","unstructured":"B\u00fcy\u00fckaky\u00fcz K (2024) Olora: Orthonormal low-rank adaptation of large language models"},{"key":"11538_CR20","unstructured":"Cai Z, Zhang Y, Gao B, Liu Y, Li Y, Liu T, Keming L, Xiong W, Dong Y, Junjie H, Xiao W (2025) Dynamic kv cache compression based on pyramidal information funneling, Pyramidkv"},{"key":"11538_CR21","unstructured":"Chavan A, Magazine R, Kushwaha S, Debbah M, Gupta D (2024) A survey on current challenges and way forward, Faster and lighter llms"},{"key":"11538_CR22","unstructured":"Chen Y, Cheng B, Han J, Zhang Y, Li Y, Zhang S (2025) DLP: Dynamic layerwise pruning in large language models. In Forty-second International Conference on Machine Learning"},{"key":"11538_CR23","unstructured":"Chen T, Ding T, Yadav B, Zharkov I (2023) and Luming Liang. Efficient large language model structured pruning and knowledge recovery, Lorashear"},{"key":"11538_CR24","doi-asserted-by":"crossref","unstructured":"Chen H, Siyue W, Quan X, Wang R, Yan M, Zhang J (2023) Multi-cot consistent knowledge distillation, Mcc-kd","DOI":"10.18653\/v1\/2023.findings-emnlp.454"},{"key":"11538_CR25","doi-asserted-by":"crossref","unstructured":"Chen X, Huang H, Gao Y, Wang Y, Zhao J, Ding K (2024) Learning to maximize mutual information for chain-of-thought distillation","DOI":"10.18653\/v1\/2024.findings-acl.409"},{"key":"11538_CR26","doi-asserted-by":"crossref","unstructured":"Chen H, Saha A, Hoi S, Joty S (2024) Empowering open-sourced llms with adaptive learning for code generation, Personalised distillation","DOI":"10.18653\/v1\/2023.emnlp-main.417"},{"key":"11538_CR27","doi-asserted-by":"crossref","unstructured":"Chen J, Zhao X, Zheng H, Li X, Xiang S, Guo H (2024) Robust knowledge distillation based on feature variance against backdoored teacher model","DOI":"10.1016\/j.asoc.2024.111907"},{"key":"11538_CR28","unstructured":"Chen H et\u00a0al (2025) Rotpruner: Llm pruning in rotated space. In International Conference on Learning Representations (ICLR)"},{"key":"11538_CR29","unstructured":"Chen J et al (2024) Longlora: Efficient fine-tuning of long-context llms"},{"key":"11538_CR30","doi-asserted-by":"crossref","unstructured":"Chen D, Liu N, Zhu Y, Che Z, Ma R, Zhang F, Mou X, Chang Y, Tang J (2024) Epsd: Early pruning with self-distillation for efficient model compression. Proc AAAI Conf Artif Intell 38:11258\u201311266","DOI":"10.1609\/aaai.v38i10.29004"},{"key":"11538_CR31","unstructured":"Chen T, Moreau T, Jiang Z et al (2018) Tvm: An automated end-to-end optimizing compiler for deep learning. In OSDI"},{"key":"11538_CR32","unstructured":"Chen T, Zheng L, Yan E et al (2018) Learning to optimize tensor programs. In NeurIPS"},{"key":"11538_CR33","unstructured":"Chen T, Xu B, Zhang C, Guestrin C (2016) Training deep nets with sublinear memory cost"},{"key":"11538_CR34","doi-asserted-by":"crossref","unstructured":"Chen M, Shao W, Peng X, Wang J, Gao P, Zhang K, Luo P (2025) Efficient quantization-aware training for large language models, Efficientqat","DOI":"10.18653\/v1\/2025.acl-long.498"},{"key":"11538_CR35","doi-asserted-by":"crossref","unstructured":"Cho Y, Kim S, Jeon D, Lee K, Lee B, No A (2025) Assigning distinct roles to quantized and low-rank matrices toward optimal weight decomposition. Preprint at arXiv:2506.02077","DOI":"10.18653\/v1\/2025.findings-acl.746"},{"key":"11538_CR36","unstructured":"Choi J, Wang Z, Venkataramani S, Chuang PIJ, Srinivasan V, Gopalakrishnan K (2018) Parameterized clipping activation for quantized neural networks, Pact"},{"key":"11538_CR37","unstructured":"Choromanski K et al (2020) Performer: Rethinking attention with performers. In NeurIPS"},{"key":"11538_CR38","doi-asserted-by":"crossref","unstructured":"Couturier C, Mastorakis S, Shen H, Rajmohan S, R\u00fchle V (2025) Semantic caching of contextual summaries for efficient question-answering with language models","DOI":"10.1109\/ICCCN69946.2026.11662998"},{"key":"11538_CR39","doi-asserted-by":"crossref","unstructured":"Cui X, Zhu M, Qin Y, Xie L, Zhou W, Li H (2025) Multi-level optimal transport for universal cross-tokenizer knowledge distillation on language models","DOI":"10.1109\/TPAMI.2026.3728858"},{"key":"11538_CR40","doi-asserted-by":"crossref","unstructured":"Dai W, Berleant D (2019) Benchmarking contemporary Deep Learning hardware and frameworks: A survey of qualitative metrics. In IEEE CogMI","DOI":"10.1109\/CogMI48466.2019.00029"},{"key":"11538_CR41","unstructured":"Daliang X, Zhang H, Yang L, Liu R, Xu M, Liu X, Huang G (2024) Fast on-device llm inference with npus"},{"key":"11538_CR42","doi-asserted-by":"crossref","unstructured":"Dao T, Fu D, Ermon S, Rudra A, Re C (2022) Flashattention: Fast and memory-efficient exact attention with io-awareness. In Advances in Neural Information Processing Systems (NeurIPS)","DOI":"10.52202\/068431-1189"},{"key":"11538_CR43","unstructured":"Dao T et al (2022) Monarch: Expressive structured matrices for efficient transformers. In ICML"},{"key":"11538_CR44","unstructured":"Dao T, Fu D, Zhao T et al (2023) Flashattention-2: Faster attention with better parallelism and work partitioning. In Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"11538_CR45","doi-asserted-by":"crossref","unstructured":"Dasgupta S, Cohn T, Baldwin T (2023) Cost-effective distillation of large language models. In Rogers A, Boyd-Graber J, Okazaki N (eds) Findings of the Association for Computational Linguistics: ACL 2023, Toronto, Canada. pp 7346\u20137354. Association for Computational Linguistics","DOI":"10.18653\/v1\/2023.findings-acl.463"},{"key":"11538_CR46","unstructured":"DeepSeek-AI, Daya\u00a0Guo et\u00a0al (2025) Deepseek-r1: Incentivizing reasoning capability in LLMS via reinforcement learning"},{"key":"11538_CR47","doi-asserted-by":"crossref","unstructured":"Dettmers T, Pagnoni A, Holtzman A, Zettlemoyer L (2023) QLoRA: Efficient finetuning of quantized LLMs. In Thirty-seventh Conference on Neural Information Processing Systems","DOI":"10.52202\/075280-0441"},{"key":"11538_CR48","doi-asserted-by":"crossref","unstructured":"Dettmers T, Lewis M, Belkada Y, Zettlemoyer L (2022) Llm.int8(): 8-bit matrix multiplication for transformers at scale","DOI":"10.52202\/068431-2198"},{"key":"11538_CR49","unstructured":"Dettmers T, Svirschevski R, Egiazarian V, Kuznedelev D, Frantar E, Ashkboos S, Borzunov A, Hoefler T, Alistarh D (2023) A sparse-quantized representation for near-lossless llm weight compression, Spqr"},{"key":"11538_CR50","volume-title":"Gemini nano (aicore on-device)","author":"A Developers","year":"2025","unstructured":"Developers A (2025) Gemini nano (aicore on-device). Technical report, Google LLC"},{"key":"11538_CR51","doi-asserted-by":"crossref","unstructured":"Di Palo F, Singhi P, Fadlallah B (2024) Performance-guided llm knowledge distillation for efficient text classification at scale","DOI":"10.18653\/v1\/2024.emnlp-main.215"},{"key":"11538_CR52","doi-asserted-by":"crossref","unstructured":"Ding B, Qin C, Liu L, Chia Y-K, Joty S, Li B, Bing B (2023) Is gpt-3 a good data annotator?","DOI":"10.18653\/v1\/2023.acl-long.626"},{"key":"11538_CR53","doi-asserted-by":"crossref","unstructured":"Ding Y, Yu CH, Zheng B, Liu Y, Wang Y, Pekhimenko G (2023) Hidet: Task-mapping programming paradigm for deep learning tensor programs. In Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2, page 370\u2013384. ACM","DOI":"10.1145\/3575693.3575702"},{"key":"11538_CR54","unstructured":"Du H, Feng X, Ma J, Wang M, Tao S, Zhong Y, Li Y-F, Wang H (2024) Towards proactive interactions for in-vehicle conversational assistants utilizing large language models. In Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, IJCAI '24"},{"key":"11538_CR55","doi-asserted-by":"crossref","unstructured":"Du D, Zhang Y, Cao S, Guo J, CaoT, Chu X, Xu N (2024) BitDistiller: Unleashing the potential of sub-4-bit LLMs via self-distillation. In Lun-Wei Ku, Andre Martins, and Vivek Srikumar, editors, Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp 102\u2013116, Bangkok, Thailand, August 2024. Association for Computational Linguistics","DOI":"10.18653\/v1\/2024.acl-long.7"},{"key":"11538_CR56","unstructured":"Egiazarian V, Panferov A, Kuznedelev D, Frantar E, Babenko A, Alistarh D (2024) Extreme compression of large language models via additive quantization"},{"key":"11538_CR300","unstructured":"eLLM Authors (2025) Elastic memory management framework for efficient llm serving"},{"key":"11538_CR57","unstructured":"European Union (2016) Gdpr (regulation eu 2016\/679). Technical report, Official Journal of the European Union"},{"key":"11538_CR58","doi-asserted-by":"crossref","unstructured":"Fan Q, Huang H, He R (2025) Breaking the low-rank dilemma of linear attention. In Proceedings of the Computer Vision and Pattern Recognition Conference, pp 25271\u201325280","DOI":"10.1109\/CVPR52734.2025.02353"},{"key":"11538_CR59","unstructured":"Fang L, Chen Y, Zhong W, Ma P (2024) Bayesian knowledge distillation: A Bayesian perspective of distillation with uncertainty quantification. In Ruslan Salakhutdinov, Zico Kolter, Katherine Heller, Adrian Weller, Nuria Oliver, Jonathan Scarlett, and Felix Berkenkamp, editors, Proceedings of the 41st International Conference on Machine Learning, volume 235 of Proceedings of Machine Learning Research, pp 12935\u201312956"},{"key":"11538_CR60","unstructured":"Feng K, Li C, Zhang X, Yuan Y, Wang G, Zhou J (2024) Keypoint-based progressive chain-of-thought distillation for llms"},{"key":"11538_CR61","doi-asserted-by":"crossref","unstructured":"Feng Q, Li W, Lin T, Chen X (2024) Distilling cross-modal alignment knowledge for mobile vision-language model, Align-kd","DOI":"10.1109\/CVPR52734.2025.00395"},{"key":"11538_CR62","unstructured":"Frantar E, Ashkboos S, Hoefler T, Alistarh D (2023) Gptq: Accurate posttraining quantization for generative pre-trained transformers"},{"key":"11538_CR63","unstructured":"Frantar E, Alistarh D (2023) SparseGPT: Massive language models can be accurately pruned in one-shot. In Krause A, Brunskill E, Cho K, Engelhardt B, Sabato S, Scarlett J (eds) Proceedings of the 40th International Conference on Machine Learning, volume 202 of Proceedings of Machine Learning Research, pp 10323\u201310337"},{"key":"11538_CR64","unstructured":"Gale T, Elsen E, Hooker S (2019) The state of sparsity in deep neural networks"},{"key":"11538_CR65","unstructured":"Gao J, Pi R, Lin Y, Xu H, Ye J, Wu Z, Zhang W, Liang X, Li Z, Kong L (2023) Self-guided noise-free data generation for efficient zero-shot learning"},{"key":"11538_CR66","doi-asserted-by":"crossref","unstructured":"Gao S, Lin C-H, Hua T, Zheng T, Shen Y, Jin H, Hsu Y-C (2024) Disp-llm: Dimension-independent structural pruning for large language models. In Globerson A, Mackey L, Belgrave D, Fan A, Paquet U, Tomczak J, Zhang C (eds) Advances in Neural Information Processing Systems, vol 37, pp 72219\u201372244. Curran Associates, Inc.","DOI":"10.52202\/079017-2305"},{"key":"11538_CR67","unstructured":"Gemma Team and et\u00a0al (2024) Morgane\u00a0Riviere. Gemma 2: Improving open language models at a practical size"},{"key":"11538_CR68","unstructured":"Georganas E et al (2025) Pushing the envelope of llm inference on ai-pc (1\u20132 bit kernels)"},{"key":"11538_CR69","unstructured":"Gerganov G, et al. (2023) llama.cpp: Port of facebook's llama model in c\/c++. https:\/\/github.com\/ggerganov\/llama.cpp. Accessed 3 Oct 2025"},{"key":"11538_CR70","doi-asserted-by":"crossref","unstructured":"Gholami A et al (2022) A survey of quantization methods for efficient Neural Network inference. ACM Computing Surveys","DOI":"10.1201\/9781003162810-13"},{"key":"11538_CR71","doi-asserted-by":"crossref","unstructured":"Gong G, Wang J, Xu J, Xiang D, Zhang Z, Shen L, Zhang Y, JunhuaShu J, ZhaolongXing Z, Chen Z, Liu P, Zhang K (2025) Beyond logits: Aligning feature dynamics for effective knowledge distillation. In Wanxiang Che, Joyce Nabende, Ekaterina Shutova, and Mohammad Taher Pilehvar, editors, Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp 23067\u201323077, Vienna, Austria, July 2025. Association for Computational Linguistics","DOI":"10.18653\/v1\/2025.acl-long.1125"},{"key":"11538_CR72","doi-asserted-by":"crossref","unstructured":"Gu Y, Khadem A, Umesh S, Liang N, Servot X, Mutlu O, Iyer R, Das R (2025) Pim is all you need: A cxl-enabled gpu-free system for large language model inference. In Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2, ASPLOS '25, pp 862\u2013881. ACM","DOI":"10.1145\/3676641.3716267"},{"key":"11538_CR73","unstructured":"Guinan S, Li Shen L, Yin SL, Yang Y, Geiping J (2025) Large language model pruning through layer cutting and stitching, Gptailor"},{"key":"11538_CR74","unstructured":"Guo S, Jiahang X, Zhang LL, Yang M (2023) Compresso, Structured pruning with collaborative prompting learns compact large language models"},{"key":"11538_CR75","unstructured":"Guo J, Wu J, Wang Z, Liu J, Yang G, Ding Y, Gong R, Qin H, Liu X (2024) Compressing large language models by joint sparsification and quantization. In Forty-first International Conference on Machine Learning"},{"key":"11538_CR76","unstructured":"Guo H, Greengard P, Xing EP, Kim Y (2023) Lq-lora: Low-rank plus quantized matrix decomposition for efficient language model finetuning. Preprint at arXiv:2311.12023"},{"key":"11538_CR77","unstructured":"Guo J, Chen X, Tang Y, Wang Y (2025) SlimLLM: Accurate structured pruning for large language models. In Forty-second International Conference on Machine Learning, 2025. ICML 2025 poster"},{"key":"11538_CR78","unstructured":"Han S, Mao H, Dally WJ (2016) Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding"},{"key":"11538_CR79","unstructured":"Han A, Li J, Huang W, Hong M, Takeda A, Jawanpuria P, Mishra B (2025) Sltrain: a sparse plus low-rank approach for parameter and memory efficient pretraining. In Proceedings of the 38th International Conference on Neural Information Processing Systems, NIPS '24, Red Hook, NY, USA, 2025. Curran Associates Inc."},{"key":"11538_CR80","doi-asserted-by":"crossref","unstructured":"Hao Z, Guo J, Han K, Hu H, Xu C, Wang Y (2023) Vanillakd: Revisit the power of vanilla knowledge distillation from small scale to large scale","DOI":"10.52202\/075280-0444"},{"key":"11538_CR81","doi-asserted-by":"crossref","unstructured":"Hao Z, Jiang H, Jiang S, Ren J, Cao T (2024) Hybrid slm and llm for edge-cloud collaborative inference. In Proceedings of the Workshop on Edge and Mobile Foundation Models, EdgeFM '24, pp 36\u201341, New York, NY, USA, 2024. Association for Computing Machinery","DOI":"10.1145\/3662006.3662067"},{"key":"11538_CR82","unstructured":"Hayou S, Ghosh N, Yu B (2024) Lora+: efficient low rank adaptation of large models. In Proceedings of the 41st International Conference on Machine Learning, ICML'24. JMLR.org"},{"key":"11538_CR83","unstructured":"He Z, Ribeiro MT, Khani F (2023) Finding and fixing model weaknesses, Targeted data generation"},{"key":"11538_CR84","doi-asserted-by":"crossref","unstructured":"He Z, Zhang T, Lee RB (2021) Attacking and protecting data privacy in edge\u2013cloud collaborative inference systems. IEEE Internet Things J 8(12):9706\u20139716","DOI":"10.1109\/JIOT.2020.3022358"},{"key":"11538_CR85","unstructured":"He J, Lin H (2025) Olica: Efficient structured pruning of large language models without retraining. In Forty-second International Conference on Machine Learning"},{"key":"11538_CR86","unstructured":"Hendrycks D, Burns C, Basart S, Zou A, Mazeika M, Song D, Steinhardt J (2021) Measuring massive multitask language understanding"},{"key":"11538_CR87","unstructured":"Hinton G, Vinyals O, Dean J (2015) Distilling the knowledge in a neural network"},{"key":"11538_CR88","unstructured":"Hong K, Dai G, Xu J, Mao Q, Li X, Liu J, Chen K, Dong Y, Wang Y (2024) Flashdecoding++: Faster large language model inference on gpus"},{"key":"11538_CR89","unstructured":"Hooper C et al (2024) Kvquant: Towards 10m-token context. In Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"11538_CR90","unstructured":"Horv\u00e1th S, Laskaridis S, Rajput S, Wang H (2024) Maestro: uncovering low-rank structures via trainable decomposition. In Proceedings of the 41st International Conference on Machine Learning, ICML'24. JMLR.org"},{"key":"11538_CR91","unstructured":"Hou B, Chen Q, Wang J, Yin G, Wang C, Du N, Pang R, Chang S, Lei T (2025) Instruction-following pruning for large language models. In Forty-second International Conference on Machine Learning"},{"key":"11538_CR92","doi-asserted-by":"crossref","unstructured":"Hsieh C-Y, Li C-L, Yeh C-K, Nakhost H, Fujii Y, Ratner A, Krishna R, Lee C-Y, Pfister T (2023) Distilling step-by-step! outperforming larger language models with less training data and smaller model sizes","DOI":"10.18653\/v1\/2023.findings-acl.507"},{"key":"11538_CR93","unstructured":"Hu H, Yuan X (2025) Spap: Structured pruning via alternating optimization and penalty methods"},{"key":"11538_CR94","unstructured":"Hu EJ, Shen Y, Wallis P, Allen-Zhu Z, Li Y, Wang S, Wang L, Chen W (2022) LoRA: Low-rank adaptation of large language models. In International Conference on Learning Representations"},{"key":"11538_CR95","doi-asserted-by":"crossref","unstructured":"Hu J, Li H, Zhang Y, Wang Z, Zhou S, Zhang X, Shum H-Y, Jiang D (2024) Multi-matrix factorization attention. Preprint at arXiv:2412.19255","DOI":"10.18653\/v1\/2025.findings-acl.1288"},{"key":"11538_CR96","unstructured":"Hu X, Chen Z, Yang D, Xu Z, Xu C, Yuan Z, Zhou S, Yu S (2025) Moequant: Enhancing quantization for mixture-of-experts large language models via expert-balanced sampling and affinity guidance"},{"key":"11538_CR97","doi-asserted-by":"crossref","unstructured":"Huang W, Cheng A, Wang Y (2025) Mitigating catastrophic forgetting in large language models with forgetting-aware pruning, 2025","DOI":"10.18653\/v1\/2025.emnlp-main.1108"},{"key":"11538_CR98","doi-asserted-by":"crossref","unstructured":"Huang M, Shen A, Li K, Peng H, Li B, Yupeng S, Yu H (2025) A highly efficient cpu-fpga heterogeneous edge accelerator for large language models, Edgellm","DOI":"10.1109\/TCSI.2025.3546256"},{"key":"11538_CR99","unstructured":"International Telecommunication Union (ITU-T) (2003) G.114: One-way transmission time. Technical report, ITU"},{"key":"11538_CR100","doi-asserted-by":"crossref","unstructured":"Jang S, Morabito R (2025) Edge-first language model inference: Models, metrics, and tradeoffs","DOI":"10.1109\/ICDCSW63273.2025.00058"},{"key":"11538_CR101","unstructured":"Ji Y, Cao Y, Liu J (2023) Pruning large language models via accuracy predictor"},{"key":"11538_CR102","unstructured":"Ji Y, Saratchandran H, Gordon C, Zhang Z, Lucey S (2024) Efficient learning with sine-activated low-rank matrices. Preprint at arXiv:2403.19243"},{"key":"11538_CR103","doi-asserted-by":"crossref","unstructured":"Jia C (2024) Adversarial moment-matching distillation of large language models","DOI":"10.52202\/079017-3562"},{"key":"11538_CR104","unstructured":"Jiang W, Fu A, Zhang Y (2025) Improved methods for model pruning and knowledge distillation"},{"key":"11538_CR105","doi-asserted-by":"crossref","unstructured":"Jiao X, Yin Y, Shang L, Jiang X, Chen X, Li L, Wang F, Liu Q (2020) Distilling bert for natural language understanding, Tinybert","DOI":"10.18653\/v1\/2020.findings-emnlp.372"},{"key":"11538_CR106","volume-title":"Antoni Viros i Martin, and Kaoutar El Maghraoui","author":"T Joshi","year":"2025","unstructured":"Joshi T, Saini H, Dhillon N (2025) Antoni Viros i Martin, and Kaoutar El Maghraoui. Unlocking long-context efficiency in deployed inference, Paged attention meets flexattention"},{"key":"11538_CR107","doi-asserted-by":"crossref","unstructured":"Jung S, Yoon S, Kim DG, Lee H (2025) Token-wise distillation via fine-grained divergence control, Todi","DOI":"10.18653\/v1\/2025.emnlp-main.409"},{"issue":"2","key":"11538_CR108","doi-asserted-by":"publisher","first-page":"586","DOI":"10.3390\/app15020586","volume":"15","author":"C Kachris","year":"2025","unstructured":"Kachris C (2025) A survey on hardware accelerators for large language models. Appl Sci 15(2):586","journal-title":"Appl Sci"},{"key":"11538_CR109","doi-asserted-by":"crossref","unstructured":"Ke W, Li Z, Li D, Tian L, Barsoum E (2025) Dl-qat: Weight-decomposed low-rank quantization-aware training for large language models. Preprint at arXiv:2504.09223","DOI":"10.18653\/v1\/2024.emnlp-industry.10"},{"key":"11538_CR110","unstructured":"Kim T et\u00a0al (2025) Guidedquant: LLM quantization via guided optimization"},{"key":"11538_CR111","unstructured":"Kim S et\u00a0al (2025) RESQ: Mixed-precision quantization of LLMS. In International Conference on Machine Learning (ICML)"},{"key":"11538_CR112","unstructured":"Kim D et al (2023) rslora: A rank stabilization scaling factor for lora"},{"key":"11538_CR113","unstructured":"Kim S, Hooper C, Gholami A, Dong Z, Li X, Shen S, Mahoney MW, Keutzer K (2024) Dense-and-sparse quantization, Squeezellm"},{"key":"11538_CR114","unstructured":"Kim S, Gholami A, Yao Z, Mahoney MW, Keutzer K (2021) Integer-only bert quantization, I-bert"},{"key":"11538_CR115","unstructured":"Ko J, Kim S, Chen T (2024) and Se-Young Yun. Towards streamlined distillation for large language models, Distillm"},{"key":"11538_CR116","doi-asserted-by":"crossref","unstructured":"Kobayashi S, Akram Y, von Oswald J (2024) Weight decay induces low-rank attention layers. In The Thirty-eighth Annual Conference on Neural Information Processing Systems","DOI":"10.52202\/079017-0146"},{"key":"11538_CR117","doi-asserted-by":"crossref","unstructured":"Kweon S, Kim J, Kwak H, Cha D, Yoon H, Kim K, Yang J, Won S, Choi E (2024) An llm benchmark for real-world clinical practice using discharge summaries, Ehrnoteqa","DOI":"10.52202\/079017-3958"},{"key":"11538_CR118","doi-asserted-by":"crossref","unstructured":"Kwon W, Li Z, Zhuang S, Sheng Y, Zheng L, Yu CH, Gonzalez JE, Zhang H, Stoica I (2023) Efficient memory management for large language model serving with pagedattention","DOI":"10.1145\/3600006.3613165"},{"key":"11538_CR119","unstructured":"Lan Z et al (2020) Albert: A lite bert for self-supervised learning of language representations. In ICLR"},{"key":"11538_CR120","doi-asserted-by":"crossref","unstructured":"Laskaridis S, Katevas K, Minto L, Haddadi H (2024) Melting point: Mobile evaluation of language transformers. In Proceedings of the 30th Annual International Conference on Mobile Computing and Networking, ACM MobiCom '24, page 890\u2013907, New York, NY, USA, 2024. Association for Computing Machinery","DOI":"10.1145\/3636534.3690668"},{"key":"11538_CR121","unstructured":"Lattner C, Pienaar J, Amini M et al (2020) Mlir: A compiler infrastructure for the end of moore's law. In PLDI"},{"key":"11538_CR122","unstructured":"Le Q et\u00a0al (2025) Probe pruning: Dynamic pruning via model-probing. In International Conference on Learning Representations (ICLR), Poster"},{"key":"11538_CR123","unstructured":"Le Q, Diao E, Wang Z, Wang X, Ding J, Yang L, Anwar A (2025) Probe pruning: Accelerating LLMs through dynamic pruning via model-probing. In The Thirteenth International Conference on Learning Representations"},{"key":"11538_CR124","doi-asserted-by":"crossref","unstructured":"Lee J, Kim M, Baek S, Hwang SJ, Sung W, Choi J (2024) Enhancing computation efficiency in large language models through weight and activation quantization","DOI":"10.18653\/v1\/2023.emnlp-main.910"},{"key":"11538_CR125","unstructured":"Li M, Zhou F, Song X (2025) Bi-directional logits difference loss for large language model distillation, Bild"},{"key":"11538_CR126","unstructured":"Li Y, Gu Y, Dong L, Wang D, Cheng Y, Wei F (2025) Direct preference knowledge distillation for large language models"},{"key":"11538_CR127","doi-asserted-by":"crossref","unstructured":"Li LH, Hessel J, Yu Y, Ren X, Chang K-W, Choi Y (2024) Symbolic chain-of-thought distillation: Small models can also \"think\" step-by-step","DOI":"10.18653\/v1\/2023.acl-long.150"},{"key":"11538_CR128","unstructured":"Li C, Chen Q, Li L, Wang C, Li Y, Chen Z, Zhang Y (2024) Mixed distillation helps smaller language model better reasoning"},{"key":"11538_CR129","doi-asserted-by":"crossref","unstructured":"Li Z, Li X, Xinyi F, Zhang X, Wang W, Chen S, Yang J (2024) Unsupervised prompt distillation for vision-language models, Promptkd","DOI":"10.1109\/CVPR52733.2024.02513"},{"key":"11538_CR130","unstructured":"Li Y, Yu Y, Liang C, Karampatziakis N, He P, Chen W, Zhao T (2024) Loftq: LoRA-fine-tuning-aware quantization for large language models. In The Twelfth International Conference on Learning Representations"},{"key":"11538_CR131","unstructured":"Li M, Lin Y, Zhang Z, Cai T, Li X, Guo J, Xie E, Meng C, Zhu J-Y, Han S (2024) Svdquant: Absorbing outliers by low-rank components for 4-bit diffusion models. Preprint at arXiv:2411.05007"},{"key":"11538_CR132","doi-asserted-by":"crossref","unstructured":"Li Y, Chengbin D (2025) Optimizing quantized diffusion models via distillation with cross-timestep error correction. Proc AAAI Conf Artif Intell 39:18530\u201318538","DOI":"10.1609\/aaai.v39i17.34039"},{"key":"11538_CR133","unstructured":"Li J, Jiaming X, Huang S, Chen Y, Li W, Liu J, Lian Y, Pan J, Ding L, Zhou H, Wang Y, Dai G (2025) A comprehensive hardware perspective, Large language model inference acceleration"},{"key":"11538_CR134","unstructured":"Li et al (2025) Lemix: Unified scheduling for multi-gpu training \\& inference"},{"key":"11538_CR135","doi-asserted-by":"crossref","unstructured":"Li X, Lu Z, Cai D, Ma X, Xu M (2024) Large language models on mobile devices: Measurements, analysis, and insights. In Proceedings of the Workshop on Edge and Mobile Foundation Models, EdgeFM '24, page 1\u20136, New York, NY, USA, 2024. Association for Computing Machinery","DOI":"10.1145\/3662006.3662059"},{"key":"11538_CR136","doi-asserted-by":"crossref","unstructured":"Li SS, Balachandran V, Feng S, Ilgen JS, Pierson E, Koh PW, Tsvetkov Y (2024) Mediq: Question-asking LLMs and a benchmark for reliable interactive clinical reasoning. In The Thirty-eighth Annual Conference on Neural Information Processing Systems","DOI":"10.52202\/079017-0908"},{"key":"11538_CR137","unstructured":"Li S, Ning X, Wang L, Liu T, Shi X, Yan S, Dai G, Yang H, Wang Y (2024) Evaluating quantized large language models"},{"key":"11538_CR138","doi-asserted-by":"crossref","unstructured":"Liang Y, Shi H, Shao H, Wang Z (2025) Accelerating long-context llm inference via algorithm-hardware co-design, Accllm","DOI":"10.1109\/TVLSI.2026.3658524"},{"key":"11538_CR139","unstructured":"Liao H, He S, Yao X, Zhang Y, Liu K, Zhao J (2025) Advancing small language models for complex reasoning tasks, Neural-symbolic collaborative distillation"},{"key":"11538_CR140","doi-asserted-by":"crossref","unstructured":"Lim EH, Chai TY, Muniandy MA-P, Yong TF, Ooi BY, Lin J-M (2023) Edge computing and AI for IoT: Opportunities and challenges. In 2023 International Conference on Consumer Electronics - Taiwan (ICCE-Taiwan), pp 357\u2013358","DOI":"10.1109\/ICCE-Taiwan58799.2023.10226787"},{"key":"11538_CR141","unstructured":"Lin M, Wang X, Ni L, Li Y, Zhize W, Jin P, Zhang Y (2025) Dense low-rank adaptation of large language models, Denselora"},{"key":"11538_CR142","doi-asserted-by":"crossref","unstructured":"Lin J, Tang J, Tang H, Yang S, Chen W-M, Wang W-C, Han S (2024) Awq: Activation-aware weight quantization for llm compression and acceleration. In Proceedings of the Machine Learning and Systems Conference (MLSys)","DOI":"10.1145\/3714983.3714987"},{"key":"11538_CR143","doi-asserted-by":"crossref","unstructured":"Ling G, Wang Z, Yan Y, Liu Q (2024) Slimgpt: Layer-wise structured pruning for large language models. In A. Globerson, L. Mackey, D. Belgrave, A. Fan, U. Paquet, J. Tomczak, and C. Zhang, editors, Advances in Neural Information Processing Systems, volume 37, pages 107112\u2013107137. Curran Associates, Inc.","DOI":"10.52202\/079017-3401"},{"key":"11538_CR144","unstructured":"Liu Y, Ning J, Xia S, Gao X, Qiang N, Ge B, Han J, Hu X (2025) Pruning large language models by identifying and preserving functional networks"},{"key":"11538_CR145","doi-asserted-by":"crossref","unstructured":"Liu J, Zhang C, Guo J, Zhang Y, Que H, Deng K, Bai Z, Liu J, Zhang G, Wang J, Yanan W, Liu C, Wenbo S, Wang J, Lin Q, Zheng B (2024) Distilling domain knowledge for efficient large language models, Ddk","DOI":"10.52202\/079017-3119"},{"key":"11538_CR146","unstructured":"Liu L, Zhang M (2025) Being strong progressively! enhancing knowledge distillation of large language models through a curriculum learning framework"},{"key":"11538_CR147","unstructured":"Liu X et al (2024) Bayesian lora: Bayesian low-rank adaptation"},{"key":"11538_CR148","unstructured":"Liu S-Y et al (2024) Dora: Weight-decomposed low-rank adaptation"},{"key":"11538_CR149","doi-asserted-by":"crossref","unstructured":"Liu Z, Oguz B, Zhao C, Chang E, Stock P, Mehdad Y, Shi Y, Krishnamoorthi R, Chandra V (2023) Data-free quantization aware training for large language models, Llm-qat","DOI":"10.18653\/v1\/2024.findings-acl.26"},{"key":"11538_CR150","unstructured":"Liu H, et~al (2025) Rap: Runtime-adaptive pruning for llm inference"},{"key":"11538_CR151","unstructured":"Liu Z, Yuan J, Jin H, Zhong S, Xu Z, Braverman V, Chen B, Hu X (2024) Kivi: Plug-and-play 2bit kv cache quantization with streaming asymmetric quantization. In Proceedings of the 41st International Conference on Machine Learning (ICML 2024), 2024"},{"key":"11538_CR152","unstructured":"Liu H, Zaharia M, Abbeel P (2023) Ring attention with blockwise transformers for near-infinite context"},{"key":"11538_CR153","unstructured":"Liu Z, Yuan J, Jin H, Zhong SH, Xu S, Braverman V, Chen B, Hu X (2024) Kivi: a tuning-free asymmetric 2bit quantization for kv cache. In Proceedings of the 41st International Conference on Machine Learning, ICML'24. JMLR.org"},{"key":"11538_CR154","unstructured":"Liu Z, Zhao C, Iandola F, Lai C, Tian Y, Fedorov I, Xiong Y, Chang E, Shi Y, Krishnamoorthi R, Lai L, Chandra V (2024) Mobilellm: optimizing sub-billion parameter language models for on-device use cases. In Proceedings of the 41st International Conference on Machine Learning, ICML\u201924. JMLR.org"},{"key":"11538_CR155","unstructured":"Liu H, Tian C, Li Q, Li L, Wei X (2025) Runtime adaptive pruning for llm inference"},{"key":"11538_CR156","doi-asserted-by":"crossref","unstructured":"Luo X, Lu X, Jiang Z, Zhou SK (2025) Icp: Immediate compensation pruning for mid-to-high sparsity. In 2025 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 9487\u20139496","DOI":"10.1109\/CVPR52734.2025.00886"},{"key":"11538_CR157","doi-asserted-by":"crossref","unstructured":"Luo Z, Can X, Zhao P, Sun Q, Geng X, Wenxiang H, Tao C, Ma J, Lin Q, Jiang D (2025) Empowering code large language models with evol-instruct, Wizardcoder","DOI":"10.18653\/v1\/2025.findings-acl.1009"},{"key":"11538_CR158","unstructured":"Luo H et al (2024) Krona: Kronecker-factorized adapters for efficient tuning. In CVPR"},{"key":"11538_CR159","doi-asserted-by":"crossref","unstructured":"Lupart S, Aliannejadi M, Kanoulas E (2025) Llm knowledge distillation for efficient sparse retrieval in conversational search, Disco","DOI":"10.1145\/3726302.3729966"},{"key":"11538_CR160","unstructured":"Ma Z, Yuan Q, Zhang L, Zhou D (2025) Slow tuning and low-entropy masking for safe chain-of-thought distillation"},{"key":"11538_CR161","doi-asserted-by":"crossref","unstructured":"Ma Z, Cao A, Yang F, Gong Y, Wei X (2025) Curriculum dataset distillation","DOI":"10.1109\/TIP.2025.3579228"},{"key":"11538_CR162","unstructured":"Ma S, Wang H, Ma L, Wang L, Wang W, Huang S, Dong L, Wang R, Xue J, Wei F (2024) The era of 1-bit llms: All large language models are in 1.58 bits"},{"key":"11538_CR163","doi-asserted-by":"crossref","unstructured":"Ma X, Fang G, Wang X (2023) Llm-pruner: On the structural pruning of large language models. In Oh A, Naumann T, Globerson A, Saenko K, Hardt M, Levine S (eds) Advances in Neural Information Processing Systems, vol 36, pp 21702\u201321720. Curran Associates, Inc.","DOI":"10.52202\/075280-0950"},{"key":"11538_CR164","unstructured":"Malekar J, Chandarana P, Amin MH, Elbtity ME, Zand R (2025) Pim-llm: A high-throughput hybrid pim architecture for 1-bit llms"},{"key":"11538_CR165","unstructured":"Mehta S, Sekhavat MH, Cao Q, Horton M, Jin Y, Sun C, Mirzadeh I, Najibi M, Belenko D, Zatloukal P, Rastegari M (2024) Openelm: An efficient language model family with open training and inference framework"},{"key":"11538_CR166","doi-asserted-by":"crossref","unstructured":"Meng F, Wang Z, Zhang M (2024) PiSSA: Principal singular values and singular vectors adaptation of large language models. In The Thirty-eighth Annual Conference on Neural Information Processing Systems","DOI":"10.52202\/079017-3846"},{"issue":"12","key":"11538_CR167","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3578938","volume":"55","author":"G Menghani","year":"2023","unstructured":"Menghani G (2023) Efficient deep learning: A survey on making deep learning models smaller, faster, and better. ACM Comput Surv 55(12):1\u201337","journal-title":"ACM Comput Surv"},{"issue":"4","key":"11538_CR168","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1109\/MIC.2024.3383758","volume":"28","author":"T Meuser","year":"2024","unstructured":"Meuser T, Lov\u00e9n L, Bhuyan M, Patil SG, Dustdar S, Aral A et al (2024) Revisiting edge AI: opportunities and challenges. IEEE Internet Comput 28(4):49\u201359","journal-title":"IEEE Internet Comput"},{"key":"11538_CR169","unstructured":"Micikevicius P, Narang S, Alben J, Diamos G, Elsen E, Garcia D, Ginsburg B, Houston M, Venkatesh G, Wu H, Kuchaiev O (2018) Mixed precision training"},{"key":"11538_CR170","unstructured":"Mitra A, Del Corro L, Mahajan S, Codas A, Simoes C, Agarwal S, Chen X, Razdaibiedina A, Jones E, Aggarwal K, Palangi H, Zheng G, Rosset C, Khanpour H, Awadallah A (2023) Orca 2: Teaching small language models how to reason"},{"key":"11538_CR171","doi-asserted-by":"crossref","unstructured":"Morabito R, Jang S (2025) Smaller, smarter, closer: The edge of collaborative generative AI","DOI":"10.1109\/MIC.2025.3575493"},{"key":"11538_CR172","unstructured":"Mukherjee S, Mitra A, Jawahar G, Agarwal S, Palangi H, Awadallah A (2023) Orca: Progressive learning from complex explanation traces of gpt-4"},{"key":"11538_CR173","doi-asserted-by":"crossref","unstructured":"Muralidharan S, Sreenivas ST, Joshi R, Chochowski M, Patwary M, Shoeybi M, Catanzaro B, Kautz J, Molchanov P (2024) Compact language models via pruning and knowledge distillation","DOI":"10.52202\/079017-1299"},{"key":"11538_CR174","unstructured":"Nagel M, Amjad RA, van Baalen M, Louizos C, Blankevoort T (2020) Up or down? adaptive rounding for post-training quantization"},{"key":"11538_CR175","doi-asserted-by":"crossref","unstructured":"Nahshan Y, Chmiel B, Baskin C, Zheltonozhskii E, Bronstein AM, Mendelson A, Ron B (2020) Loss aware post-training quantization","DOI":"10.1007\/s10994-021-06053-z"},{"key":"11538_CR176","doi-asserted-by":"crossref","unstructured":"Ning J, Zheng C, Yang T (2025) Dssd: Efficient edge-device llm deployment and collaborative inference via distributed split speculative decoding. In International Conference on Machine Learning (ICML)","DOI":"10.1109\/WCSP68525.2025.1010651"},{"key":"11538_CR177","unstructured":"ONNX Community (2025) ONNX format specification"},{"key":"11538_CR178","unstructured":"OpenAI (2023) Gpt-4 Technical Report"},{"key":"11538_CR179","unstructured":"Padarha S (2025) Enhancing reasoning capabilities in slms with reward guided dataset distillation"},{"key":"11538_CR180","unstructured":"Pan J, Li G (2025) A survey of llm inference systems"},{"key":"11538_CR181","unstructured":"Pang J, Cai T (2025) Stabilizing quantization-aware training by implicit-regularization on hessian matrix"},{"key":"11538_CR182","unstructured":"Park J et al (2024) Qdylora: Quantized dynamic low-rank adaptation"},{"key":"11538_CR183","unstructured":"Park S, Jeon S, Lee C, Jeon S, Kim B-S, Lee J (2025) Perspectives on optimization and efficiency, A survey on inference engines for large language models"},{"key":"11538_CR184","doi-asserted-by":"crossref","unstructured":"Park S, Bae J, Kwon B, Kim M, Kim B, Kwon SJ, Kang S, Lee D (2025) Unifying uniform and binary-coding quantization for accurate compression of large language models. In Wanxiang Che, Joyce Nabende, Ekaterina Shutova, and Mohammad Taher Pilehvar, editors, Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pages 28468\u201328488, Vienna, Austria, July 2025. Association for Computational Linguistics","DOI":"10.18653\/v1\/2025.acl-long.1382"},{"key":"11538_CR185","unstructured":"Peng H, Lv X, Bai Y, Yao Z, Zhang J, Hou L, Li J (2024) A design space exploration, Pre-training distillation for large language models"},{"key":"11538_CR186","unstructured":"Qin J, Tan J, Zhang K, Cai X, Wang W (2025) Mask-based llm pruning for layer-wise uniform structures, Maskprune"},{"key":"11538_CR187","doi-asserted-by":"crossref","unstructured":"Qu X, Aponte D, Banbury C, Robinson DP, Ding T, Koishida K, Zharkov I, Chen T (2025) Automatic joint structured pruning and quantization for efficient neural network training and compression. In Proceedings of the Computer Vision and Pattern Recognition Conference, pp 15234\u201315244","DOI":"10.1109\/CVPR52734.2025.01419"},{"key":"11538_CR188","doi-asserted-by":"crossref","unstructured":"Rajbhandari S, Rasley J, Ruwase O, He Y (2020) Zero: Memory optimizations toward training trillion parameter models. In SC20: International Conference for High Performance Computing, Networking, Storage and Analysis, pp 1\u201316","DOI":"10.1109\/SC41405.2020.00024"},{"key":"11538_CR189","doi-asserted-by":"crossref","unstructured":"Raje A, Askin B, Jhunjhunwala D, Tran G (2025) Multi-head low-rank adaptation for federated fine-tuning, Ravan","DOI":"10.52202\/085713-1622"},{"key":"11538_CR190","doi-asserted-by":"crossref","unstructured":"Ramesh SK, Sengupta A, Chakraborty T (2025) On the generalization vs fidelity paradox in knowledge distillation","DOI":"10.18653\/v1\/2025.findings-acl.923"},{"key":"11538_CR191","unstructured":"Reddi VJ, Kanter D, Mattson P, Duke J, Nguyen T, Chukka R, Shiring K, Tan K-S, Charlebois M, Chou W, El-Khamy M, Hong J, St. John T, Trinh C, Buch M, Mazumder M, Markovic R, Atta T, Cakir F, Charkhabi M, Chen X, Chiang C-M, Dexter D, Heo T, Schmuelling G, Shabani M, Zika D (2022) Mlperf mobile inference benchmark"},{"key":"11538_CR192","unstructured":"Roesch J, Lyubomirsky S, Kirisame M, Weber L, Pollock J, Vega L, Jiang Z, Chen T, Moreau T, Tatlock Z (2019) A high-level compiler for deep learning, Relay"},{"key":"11538_CR193","doi-asserted-by":"crossref","unstructured":"Saha R et al (2024) Caldera: Compressing llms using low rank and low precision decomposition","DOI":"10.52202\/079017-2823"},{"issue":"1","key":"11538_CR194","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1109\/MC.2017.9","volume":"50","author":"M Satyanarayanan","year":"2017","unstructured":"Satyanarayanan M (2017) The emergence of edge computing. IEEE Comput 50(1):30\u201339","journal-title":"IEEE Comput"},{"key":"11538_CR195","doi-asserted-by":"crossref","unstructured":"Shah J, Bikshandi G, Zhang Y, Thakkar V, Ramani P, Dao T (2024) Flashattention-3: Fast and accurate attention with asynchrony and low-precision. In The Thirty-eighth Annual Conference on Neural Information Processing Systems","DOI":"10.52202\/079017-2193"},{"key":"11538_CR196","unstructured":"Shangyu W, Hongchao D, Xiong Y, Chen S, Kuo T-W, Guan N (2025) and Chun Jason Xue. Robust llm inference via evolutionary pruning, Evop"},{"key":"11538_CR197","doi-asserted-by":"crossref","unstructured":"Shao J, Zhou X, Feng S, Hou B, Lai R, Jin J, Lin W, Masuda M, Yu CH, Chen T (2022) Tensor program optimization with probabilistic programs","DOI":"10.52202\/068431-2593"},{"key":"11538_CR198","unstructured":"Shao C, Li T, Chenhao P, Fengli X, Li Y (2025) Reinforcing large language model for anonymizing user-generated text, Agentstealth"},{"key":"11538_CR199","unstructured":"Shao W, Chen M, Zhang Z, Xu P, Zhao L, Li Z, Zhang K, Gao P, Qiao Y, Luo P (2024) Omniquant: Omnidirectionally calibrated quantization for large language models"},{"key":"11538_CR200","doi-asserted-by":"crossref","unstructured":"Shao H, Liu B, Qian Y (2024) One-shot sensitivity-aware mixed sparsity pruning for large language models. In ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp 11296\u201311300","DOI":"10.1109\/ICASSP48485.2024.10445737"},{"key":"11538_CR201","unstructured":"Shen Z, Qin Z, Huang Z, Chen H, Jiaqi H, Zhuang Y, Chen G, Zhao J, Lu G (2025) Merge-of-thought distillation"},{"key":"11538_CR202","doi-asserted-by":"crossref","unstructured":"Shen X, Dong P, Lei L, Kong Z, Li Z, Lin M, Chao W, Wang D (2025) Activation-guided quantization for faster inference of llms on the edge, Agile-quant","DOI":"10.1609\/aaai.v38i17.29860"},{"issue":"5","key":"11538_CR203","doi-asserted-by":"publisher","first-page":"637","DOI":"10.1109\/JIOT.2016.2579198","volume":"3","author":"W Shi","year":"2016","unstructured":"Shi W, Cao J, Zhang Q, Li Y, Xu L (2016) Edge computing: Vision and challenges. IEEE Internet Things J 3(5):637\u2013646","journal-title":"IEEE Internet Things J"},{"key":"11538_CR204","doi-asserted-by":"crossref","unstructured":"Shi Y, Di S, Chen Q, Xie W (2025) Enhancing video-llm reasoning via agent-of-thoughts distillation","DOI":"10.1109\/CVPR52734.2025.00797"},{"key":"11538_CR205","doi-asserted-by":"crossref","unstructured":"Shridhar K, Stolfo A, Sachan M (2023) Distilling reasoning capabilities into smaller language models","DOI":"10.18653\/v1\/2023.findings-acl.441"},{"key":"11538_CR206","unstructured":"Shu F, Liao Y, Zhuo L, Chenning X, Zhang L, Zhang G, Shi H, Chen L, Zhong T, He W, Siming F, Li H, Li B, Zhelun Yu, Liu S, Li H, Jiang H (2024) Making llava tiny via moe knowledge distillation, Llava-mod"},{"key":"11538_CR207","doi-asserted-by":"crossref","unstructured":"Song J, Chen Y, Ye J, Song M (2022) Spot-adaptive knowledge distillation. IEEE Trans Image Process 31:3359\u20133370","DOI":"10.1109\/TIP.2022.3170728"},{"key":"11538_CR208","unstructured":"Steven K, Jeffrey LE, Appuswamy R, Modha DS (2020) Learned step size quantization, McKinstry, Deepika Bablani"},{"key":"11538_CR209","doi-asserted-by":"crossref","unstructured":"Sun Z, Yu H, Song X, Liu R, Yang Y, Zhou D (2020) Mobilebert: a compact task-agnostic bert for resource-limited devices","DOI":"10.18653\/v1\/2020.acl-main.195"},{"key":"11538_CR210","unstructured":"Sun Y et\u00a0al (2025) Flatquant: Flatness matters for LLM quantization. In International Conference on Machine Learning (ICML)"},{"key":"11538_CR211","unstructured":"Sun Y, Dong L, Huang S, Ma S, Xia Y, Xue J, Wang J, Wei F (2024) A successor to transformer for large language models, Retentive network"},{"key":"11538_CR212","unstructured":"Sun M, Liu Z, Bair A, and Kolter Z (2024) A simple and effective pruning approach for large language models. In Kim B, Yue Y, Chaudhuri S, Fragkiadaki K, Khan M, Sun Y (eds) International Conference on Representation Learning, vol 2024, pp 4942\u20134964"},{"key":"11538_CR213","unstructured":"Sundrani S et al (2025) Low-rank compression via differentiable rank selection (llrc), 2025. ICLR 2025 under review"},{"key":"11538_CR214","unstructured":"Sung Y-L, Ma X, Read T, Song A, Tu Q, Zhang J, Tegta C, de With PHN (2025) Sinq: Spherical integer quantization for large language models. Preprint at arXiv:2509.22944"},{"key":"11538_CR215","doi-asserted-by":"crossref","unstructured":"Suwannaphong T, Jovan F, Craddock I, McConville R (2024) Optimising tinyml with quantization and distillation of transformer and mamba models for indoor localisation on edge devices","DOI":"10.1038\/s41598-025-94205-9"},{"key":"11538_CR216","doi-asserted-by":"crossref","unstructured":"Suzgun M, Scales N, Sch{\\\"a}rli N, Gehrmann S, Tay Y, Chung HW, Chowdhery A, Le QV, Chi EH, Zhou D, Wei J (2022) Challenging big-bench tasks and whether chain-of-thought can solve them","DOI":"10.18653\/v1\/2023.findings-acl.824"},{"key":"11538_CR217","unstructured":"Tang J, Chen S, Gong C (2024) Hybrid data-free knowledge distillation"},{"key":"11538_CR218","unstructured":"Touvron H, Lavril M, Izacard G, Martinet X, Lachaux M, Lacroix T, Rozi\u00e8re B, Goyal N, Hambro E, Azhar F et\u00a0al (2023) Llama: Open and efficient foundation language models"},{"key":"11538_CR219","unstructured":"Tseng A, Chee J, Sun Q, Kuleshov V (2024) and Christopher De Sa. Even better llm quantization with hadamard incoherence and lattice codebooks, Quip"},{"key":"11538_CR220","unstructured":"U.S. Department of Health and Human Services (HHS) (2025) Summary of the hipaa privacy rule. Technical report, HHS.gov"},{"key":"11538_CR301","unstructured":"van Breugel B, Bondarenko Y, Whatmough P, Nagel M (2025) Fptquant: Function-preserving transforms for LLM quantization"},{"key":"11538_CR221","doi-asserted-by":"crossref","unstructured":"Waheed A, Kadaoui K, Abdul-Mageed M (2024) To distill or not to distill? On the robustness of robust knowledge distillation","DOI":"10.18653\/v1\/2024.acl-long.680"},{"key":"11538_CR222","unstructured":"Wan F, Huang X, Cai D, Bi W, Shi S, Quan X (2024) Knowledge z"},{"key":"11538_CR223","unstructured":"Wang W, Chen W, Luo Y, Long Y, Lin Z, Zhang L, Lin B, Cai D (2024) and Xiaofei He. Model compression and efficient inference for large language models, a survey"},{"key":"11538_CR224","doi-asserted-by":"crossref","unstructured":"Wang P, Wang Z, Li Z, Gao Y, Yin B, Ren X (2023) Self-consistent chain-of-thought distillation","DOI":"10.18653\/v1\/2023.acl-long.304"},{"key":"11538_CR225","doi-asserted-by":"crossref","unstructured":"Wang S, Liu Y, Xu Y, Zhu C, Zeng M (2021) Want to reduce labeling cost? gpt-3 can help","DOI":"10.18653\/v1\/2021.findings-emnlp.354"},{"key":"11538_CR226","doi-asserted-by":"crossref","unstructured":"Wang Y, Zhang D, Wenren H, Wang Y, Li Y (2025) Ekd4rec: Ensemble knowledge distillation from llm-based models to traditional sequential recommenders. In Proceedings of the 33rd ACM International Conference on Information and Knowledge Management, New York, NY, USA, 2025. Association for Computing Machinery","DOI":"10.1145\/3701716.3715527"},{"key":"11538_CR227","doi-asserted-by":"crossref","unstructured":"Wang Y, Li P, Sun M, Liu Y (2023) Self-knowledge guided retrieval augmentation for large language models","DOI":"10.18653\/v1\/2023.findings-emnlp.691"},{"key":"11538_CR228","doi-asserted-by":"crossref","unstructured":"Wang X, Cui J, Suzuki Y, Fukumoto F (2025) Rationale distillation for llm-based recommendation, Rdrec","DOI":"10.18653\/v1\/2024.acl-short.6"},{"key":"11538_CR229","unstructured":"Wang Z, Liang J, He R, Wang Z, Tan T (2025) LoRA-pro: Are low-rank adapters properly optimized? In The Thirteenth International Conference on Learning Representations"},{"key":"11538_CR230","unstructured":"Wang S et al (2020) Linformer: Self-attention with linear complexity. In ICML"},{"key":"11538_CR231","doi-asserted-by":"crossref","unstructured":"Wang TW, Wang T et al (2022) Efficientvlm: Fast and accurate vision-language models via knowledge distillation and modal-adaptive pruning (2022). Preprint at arXiv:2210.07795","DOI":"10.18653\/v1\/2023.findings-acl.873"},{"key":"11538_CR232","doi-asserted-by":"crossref","unstructured":"Wang M, Zhao Y, Liu J, Chen J, Zhuang C, Jinjie G, Guo R, Zhao X (2024) Large multimodal model compression via iterative efficient pruning and distillation. Companion Proc ACM Web Conf 2024:235\u2013244","DOI":"10.1145\/3589335.3648321"},{"key":"11538_CR233","unstructured":"Wang J, Ebert J, Filatov O, Kesselheim S (2025) Memory and bandwidth are all you need for fully sharded data parallel"},{"key":"11538_CR234","unstructured":"Wang J, Li Z, Li L, He F, Lin L, Lai Y, Li Y, Zeng X, Guo Y (2025) Ip-safe knowledge transfer via local-cloud collaboration, Principle-guided verilog optimization"},{"key":"11538_CR235","unstructured":"Wang P, Chen Q, He X, Cheng J (2020) Towards accurate post-training network quantization via bit-split and stitching. In Hal Daum\\'{e} III and Aarti Singh (ed), Proceedings of the 37th International Conference on Machine Learning, volume 119 of Proceedings of Machine Learning Research, pp 9847\u20139856. PMLR"},{"key":"11538_CR236","unstructured":"Wei X, Gong R, Li Y, Liu X, Yu F (2023) Randomly dropping quantization for extremely low-bit post-training quantization, Qdrop"},{"key":"11538_CR237","doi-asserted-by":"crossref","unstructured":"Wei X, Zhang Y, Li Y, Zhang X, Gong R, Guo J, Liu X (2023) Outlier suppression+: Accurate quantization of large language models by equivalent and optimal shifting and scaling","DOI":"10.18653\/v1\/2023.emnlp-main.102"},{"key":"11538_CR238","unstructured":"Weir N, Mishra BD, Weller O, Tafjord O, Hornstein S, Sabol A, Jansen P, Van Durme B, Clark P (2024) Distilling a model's topical knowledge for grounded question answering, From models to microtheories"},{"key":"11538_CR239","doi-asserted-by":"crossref","unstructured":"Wen Y, Li Z, Du W, Mou L (2023) f-divergence minimization for sequence-level knowledge distillation","DOI":"10.18653\/v1\/2023.acl-long.605"},{"key":"11538_CR240","unstructured":"Wu T, Tao C, Wang J, Yang R, Zhao Z, Wong N (2024) Rethinking kullback-leibler divergence in knowledge distillation for large language models"},{"key":"11538_CR241","unstructured":"Wu T, Wang L, Wen Z, Zhang X, Duan J, Zhang X, Zuo J (2025) Accelerating edge inference for distributed moe models with latency-optimized expert placement"},{"key":"11538_CR242","unstructured":"Xia M, Gao T, Zeng Z, Chen D (2024) Sheared llama: Accelerating language model pre-training via structured pruning. In Kim B, Yue Y, Chaudhuri S, Fragkiadaki K, Khan M, Sun Y (eds) International Conference on Representation Learning, volume 2024, pp 5385\u20135409"},{"key":"11538_CR243","unstructured":"Xiao G et al (2024) StreamingLLM: Efficient streaming language models with attention sinks. In ICLR"},{"key":"11538_CR244","unstructured":"Xiao G, Lin J, Seznec M, Hao W, Demouth J, Han S (2024) Accurate and efficient post-training quantization for large language models, Smoothquant"},{"key":"11538_CR245","unstructured":"Xie H, Yao Y, Ban Y, Huang Z, Wang D, Wu Z, Su H, Wang C, Song S (2025) Mitigating spurious correlations between question and answer via chain-of-thought correctness perception distillation"},{"key":"11538_CR246","doi-asserted-by":"crossref","unstructured":"Xiong Y et al (2021) Nystr{\\\"o}mformer: A nystr{\\\"o}m-based algorithm for approximating self-attention. In AAAI","DOI":"10.1609\/aaai.v35i16.17664"},{"key":"11538_CR247","unstructured":"Xu Y, Jie Z, Dong H, Wang L, Lu X, Zhou A, Saha A, Xiong C, Sahoo D (2025) Think: Thinner key cache by query-driven pruning. In The Thirteenth International Conference on Learning Representations"},{"key":"11538_CR248","unstructured":"Xu Y et al (2024) Relora: High-rank training through low-rank updates. In ICLR"},{"key":"11538_CR249","unstructured":"Xu Y, Xie L, Gu X, Chen X, Chang H, Zhang H, Chen Z, Zhang X, Tian Q (2023) Qa-lora: Quantization-aware low-rank adaptation of large language models. Preprint at arXiv:2309.14717"},{"key":"11538_CR250","doi-asserted-by":"crossref","unstructured":"Xu Z, Zhang Y, Xie E, Zhao Z, Guo Y, Wong K-YK, Li Z, Zhao H (2025) Drivegpt4-v2: Harnessing llm capabilities for enhanced closed-loop autonomous driving. In Proceedings of CVPR","DOI":"10.1109\/CVPR52734.2025.01609"},{"key":"11538_CR251","doi-asserted-by":"crossref","unstructured":"Xu K, Shao X, Tian Y, Yang S, Zhang X (2024) Autompq: Automatic mixed-precision neural network search via few-shot quantization adapter. IEEE Transactions on Emerging Topics in Computational Intelligence, pp 1\u201313","DOI":"10.1109\/TETCI.2024.3394679"},{"key":"11538_CR252","doi-asserted-by":"crossref","unstructured":"Yang C, Yu X, Yang H, An Z, Yu C, Huang L, Xu Y (2025) Multi-teacher knowledge distillation with reinforcement learning for visual recognition","DOI":"10.1609\/aaai.v39i9.32990"},{"key":"11538_CR253","doi-asserted-by":"crossref","unstructured":"Yang E-h, Ye L (2024) Markov knowledge distillation: Make nasty teachers trained by self-undermining knowledge distillation fully distillable. In Leonardis A, Ricci E, Roth S, Russakovsky O, Sattler T, Varol G (eds) Computer Vision \u2013 ECCV 2024, pages 154\u2013171, Cham, 2025. Springer Nature Switzerland","DOI":"10.1007\/978-3-031-73024-5_10"},{"key":"11538_CR254","doi-asserted-by":"crossref","unstructured":"Yang M, Chen Y, Liu Y, Shi L (2024) Distillseq: A framework for safety alignment testing in large language models using knowledge distillation. In Proceedings of the 33rd ACM SIGSOFT International Symposium on Software Testing and Analysis, ISSTA '24, pp 578\u2013589. ACM","DOI":"10.1145\/3650212.3680304"},{"key":"11538_CR255","doi-asserted-by":"crossref","unstructured":"Yang C, Sui Y, Xiao J, Huang L, Gong Y, Duan Y, Jia W, Yin M, Cheng Y, Yuan B (2024) Moe-i2: Compressing mixture of experts models through inter-expert pruning and intra-expert low-rank decomposition. Preprint at arXiv:2411.01016","DOI":"10.18653\/v1\/2024.findings-emnlp.612"},{"key":"11538_CR256","unstructured":"Yang Z, Hu Y, Sun S, Ji W (2025) Ec2moe: Adaptive end-cloud pipeline collaboration enabling scalable mixture-of-experts inference"},{"key":"11538_CR257","doi-asserted-by":"crossref","unstructured":"Yang B, He L, Ling N, Yan Z, Xing G, Shuai X, Ren X, Jiang X (2023) Leveraging foundation model for open-set learning on the edge, Edgefm","DOI":"10.1145\/3625687.3625793"},{"key":"11538_CR258","unstructured":"Yang A et\u00a0al (2025) Anfeng\u00a0Li. Qwen3 technical report"},{"key":"11538_CR259","doi-asserted-by":"crossref","unstructured":"Yang Y, Cao Z, Zhao H (2024) Laco: Large language model pruning via layer collapse. In: Al-Onaizan Y, Bansal M, Chen Y-N (eds) Findings of the Association for Computational Linguistics: EMNLP 2024, pp 6401\u20136417, Miami, Florida, USA, Nov 2024. Association for Computational Linguistics","DOI":"10.18653\/v1\/2024.findings-emnlp.372"},{"key":"11538_CR260","unstructured":"Yang M, Lin S, Li C, Chang X (2025) Let LLM tell what to prune and how much to prune. In Forty-second International Conference on Machine Learning"},{"key":"11538_CR261","doi-asserted-by":"crossref","unstructured":"Yao Z, Aminabadi RY, Zhang M, Wu X, Li C, He Y (2022) Zeroquant: Efficient and affordable post-training quantization for large-scale transformers","DOI":"10.52202\/068431-1970"},{"key":"11538_CR262","unstructured":"Yao Z, Dong Z, Zheng Z, Gholami A, Yu J, Tan E, Wang L, Huang Q, Wang Y, Mahoney MW, Keutzer K (2021) Hawqv3: Dyadic neural network quantization"},{"key":"11538_CR263","doi-asserted-by":"crossref","unstructured":"Ye R, Tang M (2025) One-for-all pruning: a universal model for customized compression of large language models","DOI":"10.18653\/v1\/2025.findings-acl.132"},{"key":"11538_CR264","doi-asserted-by":"crossref","unstructured":"Yu X, Qingyang W, Yu L, Yu Z (2024) An empirically optimized approach to align language models, Lions","DOI":"10.18653\/v1\/2024.emnlp-main.496"},{"key":"11538_CR265","unstructured":"Yu XW, Zheng ZW, Zhang M (2025) Truncation-aware singular value decomposition for large language model compression, Svd-llm"},{"key":"11538_CR266","doi-asserted-by":"crossref","unstructured":"Yu H, Zhou Y, Chen B, Yang Z, Li S, Li Y, Jianxin W (2025) Treasures in discarded weights for llm quantization. Proc AAAI Conf Artif Intell 39(21):22218\u201322226","DOI":"10.1609\/aaai.v39i21.34376"},{"key":"11538_CR267","doi-asserted-by":"crossref","unstructured":"Yuan J, Liu H, Zhong S, Chuang Y-N, Li S, Wang G, Le D, Jin H, Chaudhary V, Xu Z, Liu Z, Hu X (2024) KV cache compression, but what must we give in return? a comprehensive benchmark of long context capable approaches. In: Al-Onaizan Y, Bansal M, Chen Y-N (eds) Findings of the Association for Computational Linguistics: EMNLP 2024, pp 4623\u20134648, Miami, Florida, USA, November 2024. Association for Computational Linguistics","DOI":"10.18653\/v1\/2024.findings-emnlp.266"},{"key":"11538_CR268","unstructured":"Yuan Z, Niu L, Liu J, Liu W, Wang X, Shang Y, Sun G, Qiang W, Jiaxiang W, Wu B (2023) Reorder-based post-training quantization for large language models, Rptq"},{"key":"11538_CR269","unstructured":"Yue X, Xingwei Q, Zhang G, Yao F, Huang W, Huan Sun YS, Chen W (2023) Building math generalist models through hybrid instruction tuning, Mammoth"},{"key":"11538_CR270","unstructured":"Yue Y, Yuan Z, Duanmu H, Zhou S, Jianlong W, Nie L (2024) Quantizing weight and key\/value cache for large language models gains more, Wkvquant"},{"key":"11538_CR271","unstructured":"Yuxian G, Dong L, Wei F, Huang M (2024) Knowledge distillation of large language models, Minillm"},{"key":"11538_CR272","doi-asserted-by":"crossref","unstructured":"Zandieh A, Daliri M, Han I (2024) Qjl: 1-bit quantized jl transform for kv cache quantization with zero overhead","DOI":"10.1609\/aaai.v39i24.34773"},{"key":"11538_CR273","unstructured":"Zangrando E, Rinaldi F, Tudisco F et al (2025) debora: Efficient bilevel optimization-based low-rank adaptation. In The Thirteenth International Conference on Learning Representations. International Conference on Learning Representations"},{"key":"11538_CR274","doi-asserted-by":"crossref","unstructured":"Zeng S, Liu J, Dai G, Yang X, Fu T, Wang H, Ma W, Sun H, Li S, Huang Z, Dai Y, Li J, Wang Z, Zhang R, Wen K, Ning X, Wang Y (2024) Flightllm: Efficient large language model inference with a complete mapping flow on fpgas","DOI":"10.1145\/3626202.3637562"},{"key":"11538_CR275","doi-asserted-by":"crossref","unstructured":"Zhan Y-L, Lu Z-Y, Sun H, Gao Z-F (2024) Over-parameterized student model via tensor decomposition boosted knowledge distillation. In The Thirty-eighth Annual Conference on Neural Information Processing Systems","DOI":"10.52202\/079017-2218"},{"key":"11538_CR276","doi-asserted-by":"crossref","unstructured":"Zhang S, Zhang X, Sun Z, Chen Y, Xu J (2024) Dual-space knowledge distillation for large language models","DOI":"10.18653\/v1\/2024.emnlp-main.1010"},{"key":"11538_CR277","unstructured":"Zhang S, Song Z, He K (2024) Neural collapse inspired knowledge distillation"},{"key":"11538_CR278","doi-asserted-by":"crossref","unstructured":"Zhang C, Li Q, Song D, Ye Z, Gao Y, Hu Y (2025) Towards the law of capacity gap in distilling language models","DOI":"10.18653\/v1\/2025.acl-long.1097"},{"key":"11538_CR279","unstructured":"Zhang Q et al (2023) Adalora: Adaptive low-rank adaptation"},{"key":"11538_CR280","unstructured":"Zhang R et al (2024) Autolora: Automatic rank search for peft"},{"key":"11538_CR281","unstructured":"Zhang K et al (2023) Lplr: Matrix compression via randomized low rank and low precision factorization. In NeurIPS (Poster)"},{"key":"11538_CR282","unstructured":"Zhang C, Cheng J, Constantinides GA, Zhao Y (2024) Lqer: Low-rank quantization error reconstruction for llms. Preprint at arXiv:2402.02446"},{"key":"11538_CR283","doi-asserted-by":"crossref","unstructured":"Zhang M et al (2024) Loraprune: Structured pruning meets low-rank. In ACL Findings","DOI":"10.18653\/v1\/2024.findings-acl.178"},{"key":"11538_CR284","doi-asserted-by":"crossref","unstructured":"Zhang J et al (2025) Heterogeneous parallel acceleration for edge intelligence systems: Challenges and solutions. IEEE Consumer Electronics Magazine","DOI":"10.1109\/MCE.2024.3456468"},{"key":"11538_CR285","doi-asserted-by":"crossref","unstructured":"Zhang T, Yi J, Xu Z, Shrivastava A (2024) Kv cache is 1 bit per channel: efficient large language model inference with coupled quantization","DOI":"10.52202\/079017-0109"},{"key":"11538_CR286","unstructured":"Zhang H, Jia A, Bu W, Cai Y, Sheng K, Chen H, He X (2025) Flexq: Efficient post-training int6 quantization for llm serving via algorithm-system co-design"},{"key":"11538_CR287","unstructured":"Zhang T, Shrivastava A (2024) Leanquant: Accurate and scalable large language model quantization with loss-error-aware grid, 2024"},{"key":"11538_CR288","unstructured":"Zhang Y, Zhao L, Lin M, Yunyun S, Yao Y, Han X, Tanner J, Liu S, Ji R (2024) Dynamic sparse no training: Training-free fine-tuning for sparse llms. In Kim B, Yue Y, Chaudhuri S, Fragkiadaki K, Khan M, Sun Y (eds) International Conference on Representation Learning, vol 2024, pp 249\u2013264"},{"key":"11538_CR289","doi-asserted-by":"crossref","unstructured":"Zhao Y, Singh P, Bhathena H, Ramos B, Joshi A, Gadiyaram S, Sharma S (2024) Optimizing LLM based retrieval augmented generation pipelines in the financial domain. In: Yang Y, Davani A, Sil A, Kumar A (eds) Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 6: Industry Track), pp 279\u2013294, Mexico City, Mexico, June 2024. Association for Computational Linguistics","DOI":"10.18653\/v1\/2024.naacl-industry.23"},{"key":"11538_CR290","unstructured":"Zhao B, Hajishirzi H, Cao Q (2024) APT: Adaptive pruning and tuning pretrained language models for efficient training and inference. In Salakhutdinov R, Kolter Z, Heller K, Weller A, Oliver N, Scarlett J, Berkenkamp F (eds) Proceedings of the 41st International Conference on Machine Learning, volume 235 of Proceedings of Machine Learning Research, pp 60812\u201360831"},{"key":"11538_CR291","unstructured":"Zheng L, Yan E, Hu Y et al (2021) Ansor: Generating high-performance tensor programs for deep learning. In MLSys"},{"key":"11538_CR292","doi-asserted-by":"crossref","unstructured":"Zheng L, Chiang WL, Sheng Y, Zhuang S, Wu Z, Zhuang Y, Lin Z, Li Z, Li D, Xing EP, Zhang H, Gonzalez JE, Stoica I (2023) Judging llm-as-a-judge with mt-bench and chatbot arena. In Proceedings of the 37th International Conference on Neural Information Processing Systems, NIPS '23, Red Hook, NY, USA. Curran Associates Inc","DOI":"10.52202\/075280-2020"},{"key":"11538_CR293","unstructured":"Zhong Y, Zhao J, Zhou Y (2024) Low-rank interconnected adaptation across layers. Preprint at arXiv:2407.09946, 2024"},{"key":"11538_CR294","unstructured":"Zhong Y, Liu S, Chen J, Hu J, Zhu Y, Liu X, Jin X, Zhang H (2024) Distserve: disaggregating prefill and decoding for goodput-optimized large language model serving. In Proceedings of the 18th USENIX Conference on Operating Systems Design and Implementation, OSDI'24, USA, 2024. USENIX Association"},{"key":"11538_CR295","doi-asserted-by":"crossref","unstructured":"Zhou Y, Ai W (2024) Teaching-assistant-in-the-loop: Improving knowledge distillation from imperfect teacher models in low-budget scenarios","DOI":"10.18653\/v1\/2024.findings-acl.17"},{"key":"11538_CR296","doi-asserted-by":"crossref","unstructured":"Zhou J, Zhu K, Wu J (2025) All you need in knowledge distillation is a tailored coordinate system","DOI":"10.1609\/aaai.v39i21.34457"},{"key":"11538_CR297","unstructured":"Zhou W, Yu Z, Liu X, Yang J, Xiao R, Wang T, Tang C, Lv J (2025) Precision neural network quantization via learnable adaptive modules"},{"key":"11538_CR298","unstructured":"Zhu P, Yang T (2025) Ce-lslm: Efficient large-small language model inference and communication via cloud-edge collaboration"},{"key":"11538_CR299","unstructured":"Zi B, Qi X, Wang L, Wang J, Wong K-F, Zhang L (2024) Fine-tuning high-rank parameters with the delta of low-rank matrices, Delta-loRA"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-026-11538-1","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-026-11538-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-026-11538-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T04:26:40Z","timestamp":1788236800000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-026-11538-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,22]]},"references-count":301,"journal-issue":{"issue":"9","published-online":{"date-parts":[[2026,9]]}},"alternative-id":["11538"],"URL":"https:\/\/doi.org\/10.1007\/s10462-026-11538-1","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-7975734\/v1","asserted-by":"object"}]},"ISSN":["1573-7462"],"issn-type":[{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,22]]},"assertion":[{"value":"29 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no conflict of interest.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"191"}}