{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T00:32:32Z","timestamp":1777422752950,"version":"3.51.4"},"reference-count":109,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2026,2,12]],"date-time":"2026-02-12T00:00:00Z","timestamp":1770854400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,12]],"date-time":"2026-02-12T00:00:00Z","timestamp":1770854400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Front. Comput. Sci."],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1007\/s11704-025-50814-1","type":"journal-article","created":{"date-parts":[[2026,2,12]],"date-time":"2026-02-12T08:46:36Z","timestamp":1770885996000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["A study and formal framework of the composability of LLM compression techniques"],"prefix":"10.1007","volume":"20","author":[{"given":"Gansen","family":"Hu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhaoguo","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,12]]},"reference":[{"key":"50814_CR1","unstructured":"Zhou Z, Ning X, Hong K, Fu T, Xu J, Li S, Lou Y, Wang L, Yuan Z, Li X, Yan S, Dai G, Zhang X P, Dong Y, Wang Y. A survey on efficient inference for large language models. 2024, arXiv preprint arXiv: 2404.14294"},{"key":"50814_CR2","doi-asserted-by":"publisher","first-page":"1556","DOI":"10.1162\/tacl_a_00704","volume":"12","author":"X Zhu","year":"2024","unstructured":"Zhu X, Li J, Liu Y, Ma C, Wang W. A survey on model compression for large language models. Transactions of the Association for Computational Linguistics, 2024, 12: 1556\u20131577","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"50814_CR3","unstructured":"Wang W, Chen W, Luo Y, Long Y, Lin Z, Zhang L, Lin B, Cai D, He X. Model compression and efficient inference for large language models: a survey. 2024, arXiv preprint arXiv: 2402.09748"},{"key":"50814_CR4","unstructured":"DeepSeek-AI. DeepSeek-R1: incentivizing reasoning capability in LLMs via reinforcement learning. 2025, arXiv preprint arXiv: 2501.12948"},{"key":"50814_CR5","doi-asserted-by":"publisher","first-page":"590","DOI":"10.1145\/3694715.3695964","volume-title":"Proceedings of the 30th ACM SIGOPS Symposium on Operating Systems Principles","author":"Y Song","year":"2024","unstructured":"Song Y, Mi Z, Xie H, Chen H. PowerInfer: fast large language model serving with a consumer-grade GPU. In: Proceedings of the 30th ACM SIGOPS Symposium on Operating Systems Principles. 2024, 590\u2013606"},{"key":"50814_CR6","first-page":"1970","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"Z Yao","year":"2022","unstructured":"Yao Z, Aminabadi R Y, Zhang M, Wu X, Li C, He Y. ZeroQuant: efficient and affordable post-training quantization for large-scale transformers. In: Proceedings of the 36th International Conference on Neural Information Processing Systems. 2022, 1970"},{"key":"50814_CR7","first-page":"37524","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"X Wu","year":"2023","unstructured":"Wu X, Li C, Aminabadi R Y, Yao Z, He Y. Understanding Int4 quantization for language models: latency speedup, composability, and failure cases. In: Proceedings of the 40th International Conference on Machine Learning. 2023, 37524\u201337539"},{"key":"50814_CR8","first-page":"441","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"T Dettmers","year":"2023","unstructured":"Dettmers T, Pagnoni A, Holtzman A, Zettlemoyer L. QLORA: efficient finetuning of quantized LLMs. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. 2023, 441"},{"key":"50814_CR9","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1201\/9781003162810-13","volume-title":"Low-Power Computer Vision: Improve the Efficiency of Artificial Intelligence","author":"A Gholami","year":"2022","unstructured":"Gholami A, Kim S, Dong Z, Yao Z, Mahoney M W, Keutzer K. A survey of quantization methods for efficient neural network inference. In: Thiruvathukal G K, Lu Y H, Kim J, Chen Y, Chen B, eds. Low-Power Computer Vision: Improve the Efficiency of Artificial Intelligence. New York: Chapman and Hall\/CRC, 2022, 291\u2013326"},{"issue":"11","key":"50814_CR10","doi-asserted-by":"publisher","first-page":"3064","DOI":"10.1109\/JSAC.2024.3431516","volume":"42","author":"P Han","year":"2024","unstructured":"Han P, Shi X, Huang J. FedAL: black-box federated knowledge distillation enabled by adversarial learning. IEEE Journal on Selected Areas in Communications, 2024, 42(11): 3064\u20133077","journal-title":"IEEE Journal on Selected Areas in Communications"},{"key":"50814_CR11","unstructured":"Xu X, Li M, Tao C, Shen T, Cheng R, Li J, Xu C, Tao D, Zhou T. A survey on knowledge distillation of large language models. 2024, arXiv preprint arXiv: 2402.13116"},{"key":"50814_CR12","first-page":"87","volume-title":"Proceedings of the Seventh Annual Conference on Machine Learning and Systems","author":"J Lin","year":"2024","unstructured":"Lin J, Tang J, Tang H, Yang S, Chen W M, Wang W C, Xiao G, Dang X, Gan C, Han S. AWQ: activation-aware weight quantization for on-device LLM compression and acceleration. In: Proceedings of the Seventh Annual Conference on Machine Learning and Systems. 2024, 87\u2013100"},{"key":"50814_CR13","first-page":"196","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"J Chee","year":"2023","unstructured":"Chee J, Cai Y, Kuleshov V, De Sa C. QuIP: 2-bit quantization of large language models with guarantees. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. 2023, 196"},{"key":"50814_CR14","first-page":"22137","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Z Liu","year":"2023","unstructured":"Liu Z, Wang J, Dao T, Zhou T, Yuan B, Song Z, Shrivastava A, Zhang C, Tian Y, Re C, Chen B. Deja vu: contextual sparsity for efficient LLMs at inference time. In: Proceedings of the 40th International Conference on Machine Learning. 2023, 22137\u201322176"},{"key":"50814_CR15","volume-title":"Proceedings of the 7th International Conference on Learning Representations","author":"J Frankle","year":"2019","unstructured":"Frankle J, Carbin M. The lottery ticket hypothesis: finding sparse, trainable neural networks. In: Proceedings of the 7th International Conference on Learning Representations. 2019"},{"key":"50814_CR16","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"M Sun","year":"2024","unstructured":"Sun M, Liu Z, Bair A, Kolter J Z. A simple and effective pruning approach for large language models. In: Proceedings of the 12th International Conference on Learning Representations. 2024"},{"key":"50814_CR17","volume-title":"International conference on machine learning","author":"E Frantar","year":"2023","unstructured":"Frantar E, Alistarh D. Sparsegpt: Massive language models can be accurately pruned in one-shot. In: International conference on machine learning. 2023"},{"key":"50814_CR18","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"Y Zhang","year":"2024","unstructured":"Zhang Y, Bai H, Lin H, Zhao J, Hou L, Cannistraci C V. Plug-and-play: an efficient post-training pruning method for large language models. In: Proceedings of the 12th International Conference on Learning Representations. 2024"},{"key":"50814_CR19","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1145\/3410463.3414654","volume-title":"Proceedings of the ACM International Conference on Parallel Architectures and Compilation Techniques","author":"Z Wang","year":"2020","unstructured":"Wang Z. SparseRT: accelerating unstructured sparsity on GPUs for deep learning inference. In: Proceedings of the ACM International Conference on Parallel Architectures and Compilation Techniques. 2020, 31\u201342"},{"key":"50814_CR20","doi-asserted-by":"crossref","unstructured":"Wang Z. SparseRT: accelerating unstructured sparsity on GPUs for deep learning inference. 2020, arXiv preprint arXiv: 2008.11849","DOI":"10.1145\/3410463.3414654"},{"key":"50814_CR21","first-page":"1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"P Okanovic","year":"2024","unstructured":"Okanovic P, Kwasniewski G, Labini P S, Besta M, Vella F, Hoefler T. High performance unstructured SpMM computation using tensor cores. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis. 2024, 1\u201314"},{"key":"50814_CR22","first-page":"950","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"X Ma","year":"2023","unstructured":"Ma X, Fang G, Wang X. LLM-pruner: on the structural pruning of large language models. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. 2023, 950"},{"key":"50814_CR23","first-page":"2305","volume-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems","author":"S Gao","year":"2024","unstructured":"Gao S, Lin C H, Hua T, Zheng T, Shen Y, Jin H, Hsu Y C. DISP-LLM: dimension-independent structural pruning for large language models. In: Proceedings of the 38th International Conference on Neural Information Processing Systems. 2024, 2305"},{"key":"50814_CR24","unstructured":"Cheng H, Zhang M, Shi J Q. MINI-LLM: memory-efficient structured pruning for large language models. 2024, arXiv preprint arXiv: 2407.11681"},{"key":"50814_CR25","unstructured":"Dong H, Chen B, Chi Y. Prompt-prompted adaptive structured pruning for efficient LLM generation. 2024, arXiv preprint arXiv: 2404.01365"},{"key":"50814_CR26","first-page":"10865","volume-title":"Proceedings of the 38th AAAI Conference on Artificial Intelligence","author":"Y An","year":"2024","unstructured":"An Y, Zhao X, Yu T, Tang M, Wang J. Fluctuation-based adaptive structured pruning for large language models. In: Proceedings of the 38th AAAI Conference on Artificial Intelligence. 2024, 10865\u201310873"},{"key":"50814_CR27","unstructured":"Dery L, Kolawole S, Kagy J F, Smith V, Neubig G, Talwalkar A. Everybody Prune now: structured pruning of LLMs with only forward passes. 2024, arXiv preprint arXiv: 2402.05406"},{"key":"50814_CR28","volume-title":"Accelerating inference with sparsity using the NVIDIA ampere architecture and NVIDIA TensorRT","author":"J Pool","year":"2021","unstructured":"Pool J, Sawarkar A, Rodge J. Accelerating inference with sparsity using the NVIDIA ampere architecture and NVIDIA TensorRT. See developer.nvidia.com\/blog\/accelerating-inference-with-sparsity-using-ampere-and-tensorrt\/website, 2021"},{"key":"50814_CR29","unstructured":"Yu J, Huang T. AutoSlim: towards one-shot architecture search for channel numbers. 2019, arXiv preprint arXiv: 1903.11728"},{"key":"50814_CR30","volume-title":"Proceedings of the 8th International Conference on Learning Representations","author":"H Cai","year":"2020","unstructured":"Cai H, Gan C, Wang T, Zhang Z, Han S. Once-for-all: train one network and specialize it for efficient deployment. In: Proceedings of the 8th International Conference on Learning Representations. 2020"},{"issue":"11","key":"50814_CR31","doi-asserted-by":"publisher","first-page":"4069","DOI":"10.1109\/TCAD.2024.3444695","volume":"43","author":"Z Chen","year":"2024","unstructured":"Chen Z, Jia C, Hu M, Xie X, Li A, Chen M. FlexFL: heterogeneous federated learning via APoZ-guided flexible pruning in uncertain scenarios. IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems, 2024, 43(11): 4069\u20134080","journal-title":"IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems"},{"key":"50814_CR32","unstructured":"Lin C, Tang J, Yang S, Wang H, Tang T, Tian B, Stoica I, Han S, Gao M. Twilight: adaptive attention sparsity with hierarchical top-p pruning. 2025, arXiv preprint arXiv: 2502.02770"},{"key":"50814_CR33","unstructured":"Liu Z, Li C, Xiao S, Li C, Lian D, Shao Y. Matryoshka re-ranker: a flexible re-ranking architecture with configurable depth and width. 2025, arXiv preprint arXiv: 2501.16302"},{"key":"50814_CR34","unstructured":"Guo Y. A survey on methods and theories of quantized neural networks. 2018, arXiv preprint arXiv: 1808.04752"},{"key":"50814_CR35","unstructured":"Ma S, Wang H, Ma L, Wang L, Wang W, Huang S, Dong L, Wang R, Xue J, Wei F. The era of 1-bit LLMs: all large language models are in 1.58 bits. 2024, arXiv preprint arXiv: 2402.17764"},{"key":"50814_CR36","unstructured":"Frantar E, Ashkboos S, Hoefler T, Alistarh D. GPTQ: accurate post-training quantization for generative pre-trained transformers. 2022, arXiv preprint arXiv: 2210.17323"},{"issue":"1","key":"50814_CR37","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1007\/s44267-024-00070-x","volume":"2","author":"W Huang","year":"2024","unstructured":"Huang W, Zheng X, Ma X, Qin H, Lv C, Chen H, Luo J, Qi X, Liu X, Magno M. An empirical study of LLaMA3 quantization: from LLMs to MLLMs. Visual Intelligence, 2024, 2(1): 36","journal-title":"Visual Intelligence"},{"key":"50814_CR38","doi-asserted-by":"publisher","first-page":"1107","DOI":"10.18653\/v1\/2024.emnlp-main.64","volume-title":"Proceedings of 2024 Conference on Empirical Methods in Natural Language Processing","author":"Q Dong","year":"2024","unstructured":"Dong Q, Li L, Dai D, Zheng C, Ma J, Li R, Xia H, Xu J, Wu Z, Chang B, Sun X, Li L, Sui Z. A survey on in-context learning. In: Proceedings of 2024 Conference on Empirical Methods in Natural Language Processing. 2024, 1107\u20131128"},{"key":"50814_CR39","first-page":"10675","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"Z Wang","year":"2021","unstructured":"Wang Z. Zero-shot knowledge distillation from a decision-based black-box model. In: Proceedings of the 38th International Conference on Machine Learning. 2021, 10675\u201310685"},{"key":"50814_CR40","first-page":"15424","volume-title":"Proceedings of the 35th AAAI Conference on Artificial Intelligence","author":"D Wang","year":"2021","unstructured":"Wang D, Zhang S, Wang L. Deep epidemiological modeling by black-box knowledge distillation: an accurate deep learning model for COVID-19. In: Proceedings of the 35th AAAI Conference on Artificial Intelligence. 2021, 15424\u201315430"},{"key":"50814_CR41","first-page":"10421","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Y Fu","year":"2023","unstructured":"Fu Y, Peng H, Ou L, Sabharwal A, Khot T. Specializing smaller language models towards multi-step reasoning. In: Proceedings of the 40th International Conference on Machine Learning. 2023, 10421\u201310430"},{"key":"50814_CR42","first-page":"196","volume-title":"Proceedings of the 17th European Conference on Computer Vision","author":"D Nguyen","year":"2022","unstructured":"Nguyen D, Gupta S, Do K, Venkatesh S. Black-box few-shot knowledge distillation. In: Proceedings of the 17th European Conference on Computer Vision. 2022, 196\u2013211"},{"key":"50814_CR43","first-page":"8003","volume-title":"Proceedings of the Findings of the Association for Computational Linguistics","author":"C Y Hsieh","year":"2023","unstructured":"Hsieh C Y, Li C L, Yeh C K, Nakhost H, Fujii Y, Ratner A, Krishna R, Lee C Y, Pfister T. Distilling step-by-step! Outperforming larger language models with less training data and smaller model sizes. In: Proceedings of the Findings of the Association for Computational Linguistics. 2023, 8003\u20138017"},{"key":"50814_CR44","first-page":"14852","volume-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics","author":"N Ho","year":"2023","unstructured":"Ho N, Schmid L, Yun S Y. Large language models are reasoning teachers. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics. 2023, 14852\u201314882"},{"key":"50814_CR45","first-page":"2791","volume-title":"Proceedings of 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","author":"S Min","year":"2022","unstructured":"Min S, Lewis M, Zettlemoyer L, Hajishirzi H. MetaICL: learning to learn in context. In: Proceedings of 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. 2022, 2791\u20132809"},{"key":"50814_CR46","unstructured":"Huang Y, Chen Y, Yu Z, McKeown K. In-context learning distillation: transferring few-shot learning ability of pre-trained language models. 2022, arXiv preprint arXiv: 2212.10670"},{"key":"50814_CR47","first-page":"14596","volume-title":"Proceedings of the Findings of the Association for Computational Linguistics","author":"J Hong","year":"2024","unstructured":"Hong J, Tu Q, Chen C, Xing G, Zhang J, Yan R. CycleAlign: iterative distillation from black-box LLM to white-box models for better human alignment. In: Proceedings of the Findings of the Association for Computational Linguistics. 2024, 14596\u201314609"},{"key":"50814_CR48","doi-asserted-by":"crossref","unstructured":"Timiryasov I, Tastet J L. Baby llama: knowledge distillation from an ensemble of teachers trained on a small dataset with no performance penalty. 2023, arXiv preprint arXiv: 2308.02019","DOI":"10.18653\/v1\/2023.conll-babylm.24"},{"key":"50814_CR49","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"Y Gu","year":"2024","unstructured":"Gu Y, Dong L, Wei F, Huang M. MiniLLM: knowledge distillation of large language models. In: Proceedings of the 12th International Conference on Learning Representations. 2024"},{"key":"50814_CR50","unstructured":"Agarwal R, Vieillard N, Stanczyk P, Ramos S, Geist M, Bachem O. GKD: generalized knowledge distillation for auto-regressive sequence models. 2023, arXiv preprint arXiv: 2306.13649"},{"key":"50814_CR51","first-page":"20852","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"C Liang","year":"2023","unstructured":"Liang C, Zuo S, Zhang Q, He P, Chen W, Zhao T. Less is more: task-aware layer-wise distillation for language model compression. In: Proceedings of the 40th International Conference on Machine Learning. 2023, 20852\u201320867"},{"key":"50814_CR52","doi-asserted-by":"publisher","DOI":"10.1145\/3699518","volume-title":"ACM Transactions on Intelligent Systems and Technology","author":"C Yang","year":"2024","unstructured":"Yang C, Zhu Y, Lu W, Wang Y, Chen Q, Gao C, Yan B, Chen Y. Survey on knowledge distillation for large language models: methods, evaluation, and application. ACM Transactions on Intelligent Systems and Technology, 2024, doi: https:\/\/doi.org\/10.1145\/3699518"},{"issue":"11","key":"50814_CR53","doi-asserted-by":"publisher","first-page":"2212","DOI":"10.1080\/03081087.2016.1267104","volume":"65","author":"N Kishore Kumar","year":"2017","unstructured":"Kishore Kumar N, Schneider J. Literature survey on low rank approximation of matrices. Linear and Multilinear Algebra, 2017, 65(11): 2212\u20132244","journal-title":"Linear and Multilinear Algebra"},{"key":"50814_CR54","volume-title":"Proceedings of the 10th International Conference on Learning Representations","author":"E J Hu","year":"2022","unstructured":"Hu E J, Shen Y, Wallis P, Allen-Zhu Z, Li Y, Wang S, Wang L, Chen W. LoRA: low-rank adaptation of large language models. In: Proceedings of the 10th International Conference on Learning Representations. 2022"},{"key":"50814_CR55","first-page":"3013","volume-title":"Proceedings of the Findings of the Association for Computational Linguistics","author":"M Zhang","year":"2024","unstructured":"Zhang M, Chen H, Shen C, Yang Z, Ou L, Yu X, Zhuang B. LoRAPrune: structured pruning meets low-rank parameter-efficient fine-tuning. In: Proceedings of the Findings of the Association for Computational Linguistics. 2024, 3013\u20133026"},{"key":"50814_CR56","first-page":"20336","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Y Li","year":"2023","unstructured":"Li Y, Yu Y, Zhang Q, Liang C, He P, Chen W, Zhao T. LoSparse: structured compression of large language models based on low-rank and sparse approximation. In: Proceedings of the 40th International Conference on Machine Learning. 2023, 20336\u201320350"},{"key":"50814_CR57","unstructured":"Chen T, Ding T, Yadav B, Zharkov I, Liang L. LoRAShear: efficient large language model structured pruning and knowledge recovery. 2023, arXiv preprint arXiv: 2310.18356"},{"key":"50814_CR58","volume-title":"Proceedings of the 13th International Conference on Learning Representations","author":"W Huang","year":"2025","unstructured":"Huang W, Zhang Y, Zheng X, Liu Y, Lin J, Yao Y, Ji R. Dynamic low-rank sparse adaptation for large language models. In: Proceedings of the 13th International Conference on Learning Representations. 2025"},{"key":"50814_CR59","first-page":"826","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"R Saha","year":"2023","unstructured":"Saha R, Srivastava V, Pilanci M. Matrix compression via randomized low rank and low precision factorization. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. 2023, 826"},{"key":"50814_CR60","unstructured":"Yao Z, Wu X, Li C, Youn S, He Y. ZeroQuant-v2: exploring posttraining quantization in LLMs from comprehensive study to low rank compensation. 2023, arXiv preprint arXiv: 2303.08302"},{"key":"50814_CR61","unstructured":"Wu X, Yao Z, He Y. ZeroQuant-FP: a leap forward in LLMs posttraining W4A8 quantization using floating-point formats. 2023, arXiv preprint arXiv: 2307.09782"},{"key":"50814_CR62","unstructured":"Li S, Chen J, Han X, Bai J. NutePrune: efficient progressive pruning with numerous teachers for large language models. 2024, arXiv preprint arXiv: 2402.09773"},{"key":"50814_CR63","unstructured":"Zhu H, Shen C. SDMPrune: self-distillation MLP pruning for efficient large language models. 2025, arXiv preprint arXiv: 2506.11120"},{"key":"50814_CR64","first-page":"1299","volume-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems","author":"S Muralidharan","year":"2024","unstructured":"Muralidharan S, Sreenivas S T, Joshi R, Chochowski M, Patwary M, Shoeybi M, Catanzaro B, Kautz J, Molchanov P. Compact language models via pruning and knowledge distillation. In: Proceedings of the 38th International Conference on Neural Information Processing Systems. 2024, 1299"},{"key":"50814_CR65","first-page":"13899","volume-title":"Proceedings of the Findings of the Association for Computational Linguistics","author":"T Wang","year":"2023","unstructured":"Wang T, Zhou W, Zeng Y, Zhang X. EfficientVLM: fast and accurate vision-language models via knowledge distillation and modal-adaptive pruning. In: Proceedings of the Findings of the Association for Computational Linguistics. 2023, 13899\u201313913"},{"key":"50814_CR66","first-page":"429","volume-title":"Proceedings of the 48th Annual Computers, Software, and Applications Conference (COMPSAC)","author":"C Y Chiu","year":"2024","unstructured":"Chiu C Y, Hong D Y, Liu P, Wu J J. Effective compression of language models by combining pruning and knowledge distillation. In: Proceedings of the 48th Annual Computers, Software, and Applications Conference (COMPSAC). 2024, 429\u2013438"},{"key":"50814_CR67","unstructured":"Zafrir O, Larey A, Boudoukh G, Shen H, Wasserblat M. Prune once for all: sparse pre-trained language models. 2021, arXiv preprint arXiv: 2111.05754"},{"key":"50814_CR68","first-page":"24167","volume-title":"Proceedings of the 39th AAAI Conference on Artificial Intelligence","author":"W Huang","year":"2025","unstructured":"Huang W, Hu Y, Jian G, Zhu J, Chen J. Pruning large language models with semi-structural adaptive sparse training. In: Proceedings of the 39th AAAI Conference on Artificial Intelligence. 2025, 24167\u201324175"},{"key":"50814_CR69","doi-asserted-by":"crossref","unstructured":"Fan T, Ma G, Song Y, Fan L, Chen K, Yang Q. PPC-GPT: federated task-specific compression of large language models via pruning and chain-of-thought distillation. 2025, arXiv preprint arXiv: 2502.15857","DOI":"10.18653\/v1\/2025.emnlp-main.747"},{"key":"50814_CR70","unstructured":"Thangarasa V, Venkatesh G, Lasby M, Sinnadurai N, Lie S. Self-data distillation for recovering quality in pruned large language models. 2024, arXiv preprint arXiv: 2410.09982"},{"key":"50814_CR71","unstructured":"Sreenivas S T, Muralidharan S, Joshi R, Chochowski M, Mahabaleshwarkar A S, Shen G, Zeng J, Chen Z, Suhara Y, Diao S, Yu C, Chen W C, Ross H, Olabiyi O, Aithal A, Kuchaiev O, Korzekwa D, Molchanov P, Patwary M, Shoeybi M, Kautz J, Catanzaro B. LLM pruning and distillation in practice: the minitron approach. 2024, arXiv preprint arXiv: 2408.11796"},{"key":"50814_CR72","first-page":"467","volume-title":"Proceedings of the Findings of the Association for Computational Linguistics","author":"Z Liu","year":"2024","unstructured":"Liu Z, Oguz B, Zhao C, Chang E, Stock P, Mehdad Y, Shi Y, Krishnamoorthi R, Chandra V. LLM-QAT: data-free quantization aware training for large language models. In: Proceedings of the Findings of the Association for Computational Linguistics. 2024, 467\u2013484"},{"key":"50814_CR73","first-page":"1329","volume-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics","author":"J O\u2019Neill","year":"2023","unstructured":"O\u2019Neill J, Dutta S. Self-distilled quantization: achieving high compression rates in transformer-based language models. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics. 2023, 1329\u20131339"},{"key":"50814_CR74","unstructured":"Chen T, Li Z, Xu W, Zhu Z, Li D, Tian L, Barsoum E, Wang P, Cheng J. TernaryLLM: ternarized large language model. 2024, arXiv preprint arXiv: 2406.07177"},{"key":"50814_CR75","volume-title":"Proceedings of the 12th International Conference on Learning Representation","author":"Y Xu","year":"2024","unstructured":"Xu Y, Xie L, Gu X, Chen X, Chang H, Zhang H, Chen Z, Zhang X, Tian Q. QA-LoRA: quantization-aware low-rank adaptation of large language models. In: Proceedings of the 12th International Conference on Learning Representation. 2024"},{"key":"50814_CR76","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"Y Li","year":"2024","unstructured":"Li Y, Yu Y, Liang C, He P, Karampatziakis N, Chen W, Zhao T. LoftQ: LoRA-fine-tuning-aware quantization for large language models. In: Proceedings of the 12th International Conference on Learning Representations. 2024"},{"key":"50814_CR77","first-page":"2002","volume-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics","author":"H Jeon","year":"2025","unstructured":"Jeon H, Kim Y, Kim J J. L4Q: parameter efficient quantization-aware fine-tuning on large language models. In: Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics. 2025, 2002\u20132024"},{"key":"50814_CR78","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"J Liu","year":"2024","unstructured":"Liu J, Gong R, Wei X, Dong Z, Cai J, Zhuang B. QLLM: accurate and efficient low-bitwidth quantization for large language models. In: Proceedings of the 12th International Conference on Learning Representations. 2024"},{"key":"50814_CR79","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"J Guo","year":"2024","unstructured":"Guo J, Wu J, Wang Z, Liu J, Yang G, Ding Y, Gong R, Qin H, Liu X. Compressing large language models by joint sparsification and quantization. In: Proceedings of the 41st International Conference on Machine Learning. 2024"},{"key":"50814_CR80","unstructured":"Wu, X., Li, C., Aminabadi, R.Y., Yao, Z. and He, Y. Understanding INT4 Quantization for Transformer Models: Latency Speedup, Composability, and Failure Cases. 2023, arXiv preprint arXiv: 2301.12017"},{"key":"50814_CR81","first-page":"4276","volume-title":"Proceedings of the Findings of the Association for Computational Linguistics","author":"C Zhou","year":"2025","unstructured":"Zhou C, Zhou Y, Wang Y, Han S, Qiao Q, Li H. QPruner: probabilistic decision quantization for structured pruning in large language models. In: Proceedings of the Findings of the Association for Computational Linguistics. 2025, 4276\u20134286"},{"key":"50814_CR82","unstructured":"Almazrouei E, Alobeidli H, Alshamsi A, Cappelli A, Cojocaru R, Debbah M, Goffinet \u00c9, Hesslow D, Launay J, Malartic Q, Mazzotta D, Noune B, Pannier B, Penedo G. The falcon series of open language models. 2023, arXiv preprint arXiv: 2311.16867"},{"key":"50814_CR83","unstructured":"Grattafiori A, Dubey A, Jauhri A, Pandey A, Kadian A, et al. The llama 3 herd of models. 2024, arXiv preprint arXiv: 2407.21783"},{"key":"50814_CR84","first-page":"7021","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"L Liu","year":"2021","unstructured":"Liu L, Zhang S, Kuang Z, Zhou A, Xue J H, Wang X, Chen Y, Yang W, Liao Q, Zhang W. Group fisher pruning for practical network compression. In: Proceedings of the 38th International Conference on Machine Learning. 2021, 7021\u20137032"},{"key":"50814_CR85","first-page":"655","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems","author":"A Peste","year":"2021","unstructured":"Peste A, Iofinova E, Vladu A, Alistarh D. AC\/DC: alternating compressed\/decompressed training of deep neural networks. In: Proceedings of the 35th International Conference on Neural Information Processing Systems. 2021, 655"},{"key":"50814_CR86","first-page":"2943","volume-title":"Proceedings of the 37th International Conference on Machine Learning","author":"U Evci","year":"2020","unstructured":"Evci U, Gale T, Menick J, Castro P S, Elsen E. Rigging the lottery: making all tickets winners. In: Proceedings of the 37th International Conference on Machine Learning. 2020, 2943\u20132952"},{"key":"50814_CR87","unstructured":"Merity S, Xiong C, Bradbury J, et al. Pointer sentinel mixture models. 2016, arXiv preprint arXiv:1609.07843"},{"key":"50814_CR88","volume-title":"Proceedings of the 13th International Conference on Learning Representations","author":"S B Harma","year":"2025","unstructured":"Harma S B, Chakraborty A, Kostenok E, Mishin D, Ha D, Falsafi B, Jaggi M, Liu M, Oh Y, Subramanian S, Yazdanbakhsh A. Effective interplay between sparsity and quantization: from theory to practice. In: Proceedings of the 13th International Conference on Learning Representations. 2025"},{"key":"50814_CR89","unstructured":"Leofante F, Narodytska N, Pulina L, Tacchella A. Automated verification of neural networks: advances, challenges and perspectives. 2018, arXiv preprint arXiv: 1805.09938"},{"issue":"4","key":"50814_CR90","doi-asserted-by":"publisher","first-page":"364","DOI":"10.1109\/TSE.2002.995426","volume":"28","author":"G J Holzmann","year":"2002","unstructured":"Holzmann G J, Smith M H. An automated verification method for distributed systems software based on model extraction. IEEE Transactions on Software Engineering, 2002, 28(4): 364\u2013377","journal-title":"IEEE Transactions on Software Engineering"},{"key":"50814_CR91","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1007\/978-3-319-15579-1_2","volume-title":"Proceedings of the 9th International Conference on Language and Automata Theory and Applications","author":"A Farzan","year":"2015","unstructured":"Farzan A, Heizmann M, Hoenicke J, Kincaid Z, Podelski A. Automated program verification. In: Proceedings of the 9th International Conference on Language and Automata Theory and Applications. 2015, 25\u201346"},{"key":"50814_CR92","first-page":"53","volume-title":"Proceedings of the 11th International School on Formal Methods for the Design of Computer, Communication and Software Systems","author":"V Forejt","year":"2011","unstructured":"Forejt V, Kwiatkowska M, Norman G, Parker D. Automated verification techniques for probabilistic systems. In: Proceedings of the 11th International School on Formal Methods for the Design of Computer, Communication and Software Systems. 2011, 53\u2013113"},{"key":"50814_CR93","doi-asserted-by":"publisher","first-page":"305","DOI":"10.1007\/978-3-319-10575-8_11","volume-title":"Handbook of Model Checking","author":"C Barrett","year":"2018","unstructured":"Barrett C, Tinelli C. Satisfiability modulo theories. In: Clarke E M, Henzinger T A, Veith H, Bloem R, eds. Handbook of Model Checking. Cham: Springer, 2018, 305\u2013343"},{"key":"50814_CR94","first-page":"337","volume-title":"Proceedings of the 14th International Conference on Tools and Algorithms for the Construction and Analysis of Systems","author":"L de Moura","year":"2008","unstructured":"de Moura L, Bj\u00f8rner N. Z3: an efficient SMT solver. In: Proceedings of the 14th International Conference on Tools and Algorithms for the Construction and Analysis of Systems. 2008, 337\u2013340"},{"key":"50814_CR95","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1145\/3514221.3526125","volume-title":"Proceedings of 2022 International Conference on Management of Data","author":"Z Wang","year":"2022","unstructured":"Wang Z, Zhou Z, Yang Y, Ding H, Hu G, Ding D, Tang C, Chen H, Li J. WeTune: automatic discovery and verification of query rewrite rules. In: Proceedings of 2022 International Conference on Management of Data. 2022, 94\u2013107"},{"key":"50814_CR96","first-page":"338","volume-title":"Proceedings of the 24th ACM SIGSOFT International Symposium on Foundations of Software Engineering","author":"A Gurfinkel","year":"2016","unstructured":"Gurfinkel A, Shoham S, Meshman Y. SMT-based verification of parameterized systems. In: Proceedings of the 24th ACM SIGSOFT International Symposium on Foundations of Software Engineering. 2016, 338\u2013348"},{"issue":"1","key":"50814_CR97","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1145\/1328897.1328461","volume":"43","author":"S Lahiri","year":"2008","unstructured":"Lahiri S, Qadeer S. Back to the future: revisiting precise program verification using SMT solvers. ACM SIGPLAN Notices, 2008, 43(1): 171\u2013182","journal-title":"ACM SIGPLAN Notices"},{"issue":"4","key":"50814_CR98","doi-asserted-by":"publisher","first-page":"957","DOI":"10.1109\/TSE.2011.59","volume":"38","author":"L Cordeiro","year":"2012","unstructured":"Cordeiro L, Fischer B, Marques-Silva J. SMT-based bounded model checking for embedded ANSI-C software. IEEE Transactions on Software Engineering, 2012, 38(4): 957\u2013974","journal-title":"IEEE Transactions on Software Engineering"},{"key":"50814_CR99","series-title":"Notes for the Summer School on Formal Techniques","volume-title":"Applications of SMT solvers to program verification","author":"N Bj\u00f8rner","year":"2014","unstructured":"Bj\u00f8rner N, de Moura L. Applications of SMT solvers to program verification. Notes for the Summer School on Formal Techniques, 2014"},{"key":"50814_CR100","first-page":"337","volume-title":"International conference on Tools and Algorithms for the Construction and Analysis of Systems","author":"L De Moura","year":"2008","unstructured":"De Moura L, Bj\u00f8rner N. Z3: An efficient SMT solver. In: International conference on Tools and Algorithms for the Construction and Analysis of Systems. 2008. 337\u2013340"},{"key":"50814_CR101","first-page":"7873","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"F Tung","year":"2018","unstructured":"Tung F, Mori G. CLIP-Q: deep network compression learning by in-parallel pruning-quantization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2018, 7873\u20137882"},{"key":"50814_CR102","unstructured":"Mishra A, Latorre J A, Pool J, Stosic D, Stosic D, Venkatesh G, Yu C, Micikevicius P. Accelerating sparse deep neural networks. 2021, arXiv preprint arXiv: 2104.08378"},{"key":"50814_CR103","first-page":"7780","volume-title":"Proceedings of the 35th AAAI Conference on Artificial Intelligence","author":"P Hu","year":"2021","unstructured":"Hu P, Peng X, Zhu H, Aly M M S, Lin J. OPQ: compressing deep neural networks with one-shot pruning-quantization. In: Proceedings of the 35th AAAI Conference on Artificial Intelligence. 2021, 7780\u20137788"},{"key":"50814_CR104","doi-asserted-by":"publisher","first-page":"127257","DOI":"10.1016\/j.neucom.2024.127257","volume":"573","author":"S H S Basha","year":"2024","unstructured":"Basha S H S, Farazuddin M, Pulabaigari V, Dubey S R, Mukherjee S. Deep model compression based on the training history. Neurocomputing, 2024, 573: 127257","journal-title":"Neurocomputing"},{"issue":"1","key":"50814_CR105","first-page":"241","volume":"22","author":"T Hoefler","year":"2021","unstructured":"Hoefler T, Alistarh D, Ben-Nun T, Dryden N, Peste A. Sparsity in deep learning: pruning and growth for efficient inference and training in neural networks. The Journal of Machine Learning Research, 2021, 22(1): 241","journal-title":"The Journal of Machine Learning Research"},{"issue":"11","key":"50814_CR106","doi-asserted-by":"publisher","first-page":"7436","DOI":"10.1109\/TPAMI.2021.3117837","volume":"44","author":"Y Han","year":"2022","unstructured":"Han Y, Huang G, Song S, Yang L, Wang H, Wang Y. Dynamic neural networks: a survey. IEEE Transactions on Pattern Analysis and Machine Intelligence, 2022, 44(11): 7436\u20137456","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"50814_CR107","volume-title":"Proceedings of the 9th International Conference on Learning Representations","author":"E S Lubana","year":"2021","unstructured":"Lubana E S, Dick R P. A gradient flow framework for analyzing network pruning. In: Proceedings of the 9th International Conference on Learning Representations. 2021"},{"key":"50814_CR108","first-page":"2778","volume-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems","author":"G Yang","year":"2024","unstructured":"Yang G, He C, Guo J, Wu J, Ding Y, Liu A, Qin H, Ji P, Liu X. LLMCBench: benchmarking large language model compression for efficient deployment. In: Proceedings of the 38th International Conference on Neural Information Processing Systems. 2024, 2778"},{"key":"50814_CR109","unstructured":"Zhao J, Wang M, Zhang M, Shang Y, Liu X, Wang Y, Zhang M, Nie L. Benchmarking post-training quantization in LLMs: comprehensive taxonomy, unified evaluation, and comparative analysis. 2025, arXiv preprint arXiv: 2502.13178"}],"container-title":["Frontiers of Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-025-50814-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11704-025-50814-1","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-025-50814-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,12]],"date-time":"2026-02-12T10:04:46Z","timestamp":1770890686000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11704-025-50814-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,12]]},"references-count":109,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2026,9]]}},"alternative-id":["50814"],"URL":"https:\/\/doi.org\/10.1007\/s11704-025-50814-1","relation":{},"ISSN":["2095-2228","2095-2236"],"issn-type":[{"value":"2095-2228","type":"print"},{"value":"2095-2236","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,12]]},"assertion":[{"value":"10 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 July 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare that they have no competing interests or financial conflicts to disclose.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"2009616"}}