{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T10:04:15Z","timestamp":1784801055747,"version":"3.55.0"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T00:00:00Z","timestamp":1777507200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T00:00:00Z","timestamp":1784764800000},"content-version":"vor","delay-in-days":84,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J. King Saud Univ. Comput. Inf. Sci."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1007\/s44443-026-00724-4","type":"journal-article","created":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T10:35:46Z","timestamp":1777545346000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Layer-wise heterogeneity-guided heterogeneous pruning: An LLM compression method"],"prefix":"10.1007","volume":"38","author":[{"given":"Zhihao","family":"Yu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zuxin","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guowei","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chun","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoqi","family":"Duan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qing","family":"Qian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunhe","family":"Cui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,30]]},"reference":[{"key":"724_CR1","unstructured":"Ashkboos S, Croci M, Nascimento M, Hoefler T, Hensman J (2024) Slicegpt: Compress large language models by deleting rows and columns. In: Kim B, Yue Y, Chaudhuri S, Fragkiadaki K, Khan M, Sun Y (eds) International Conference on Learning Representations, ICLR, vol 2024, pp 11682\u201311701"},{"key":"724_CR2","unstructured":"Agarwal R, Vieillard N, Zhou Y, Stanczyk P, Ramos\u00a0Garea S, Geist M, Bachem O (2024) On-policy distillation of language models: Learning from self-generated mistakes. In: Kim B, Yue Y, Chaudhuri S, Fragkiadaki K, Khan M, Sun Y (eds) International Conference on Learning Representations, ICLR, vol 2024, pp 21246\u201321263"},{"key":"724_CR3","unstructured":"Blalock DW, Ortiz JJG, Frankle J, Guttag JV (2020) What is the state of neural network pruning? In: Dhillon IS, Papailiopoulos DS, Sze V (eds) Proceedings of the Third Conference on Machine Learning and Systems, MLSys. mlsys.org, Austin, TX, USA"},{"key":"724_CR4","first-page":"4396","volume-title":"Advances in Neural Information Processing Systems, Neurips","author":"J Chee","year":"2023","unstructured":"Chee J, Cai Y, Kuleshov V, De Sa CM (2023) Quip: 2-bit quantization of large language models with guarantees. In: Oh A, Naumann T, Globerson A, Saenko K, Hardt M, Levine S (eds) Advances in Neural Information Processing Systems, Neurips, vol 36. Curran Associates Inc, New Orleans, Louisiana, USA, pp 4396\u20134429"},{"key":"724_CR5","unstructured":"Chen Z, Hu X, Yang D, Xu Z, Xu C, Yuan Z, Zhou S, Yu J (2025) Moequant: Enhancing quantization for mixture-of-experts large language models via expert-balanced sampling and affinity guidance. In: Forty-second International Conference on Machine Learning, ICML, Vancouver, BC, Canada"},{"key":"724_CR6","doi-asserted-by":"publisher","unstructured":"Clark C, Lee K, Chang M-W, Kwiatkowski T, Collins M, Toutanova K (2019) Boolq: Exploring the surprising difficulty of natural yes\/no questions. In: Burstein J, Doran C, Solorio T (eds) Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL-HLT, pp 2924\u20132936. Association for Computational Linguistics, Minneapolis, MN, USA. https:\/\/doi.org\/10.18653\/v1\/N19-1300","DOI":"10.18653\/v1\/N19-1300"},{"key":"724_CR7","unstructured":"Egiazarian V, Panferov A, Kuznedelev D, Frantar E, Babenko A, Alistarh D (2024) Extreme compression of large language models via additive quantization. In: Proceedings of the 41st International Conference on Machine Learning, ICML, pp 46795\u201346814. PMLR, Vienna, Austria"},{"key":"724_CR8","unstructured":"Frantar E, Alistarh D (2023) Sparsegpt: Massive language models can be accurately pruned in one-shot. In: Krause A, Brunskill E, Cho K, Engelhardt B, Sabato S, Scarlett J (eds) Proceedings of the 40th International Conference on Machine Learning, ICML, vol 202, pp 10323\u201310337. PMLR, Law Street, San Diego CA"},{"key":"724_CR9","unstructured":"Foret P, Kleiner A, Mobahi H, Neyshabur B (2021) Sharpness-aware minimization for efficiently improving generalization. In: 9th International Conference on Learning Representations, ICLR, Virtual Event, Austria"},{"key":"724_CR10","doi-asserted-by":"publisher","unstructured":"Fang G, Yin H, Muralidharan S, Heinrich G, Pool J, Kautz J, Molchanov P, Wang X (2024) Maskllm: Learnable semi-structured sparsity for large language models. In: Globerson A, Mackey L, Belgrave D, Fan A, Paquet U, Tomczak J, Zhang C (eds) Advances in Neural Information Processing Systems, Neurips, vol 37, pp 7736\u20137758. Curran Associates, Inc., Vancouver, Canada. https:\/\/doi.org\/10.52202\/079017-0248","DOI":"10.52202\/079017-0248"},{"key":"724_CR11","doi-asserted-by":"publisher","unstructured":"Gu Z, Liu L, Chen X, Yi R, Zhang J, Wang Y, Wang C, Shu A, Jiang G, Ma L (2023) Remembering normality: Memory-guided knowledge distillation for unsupervised anomaly detection. In: IEEE\/CVF International Conference on Computer Vision, ICCV, pp 16355\u201316363. IEEE, Paris, France. https:\/\/doi.org\/10.1109\/ICCV51070.2023.01503","DOI":"10.1109\/ICCV51070.2023.01503"},{"issue":"241","key":"724_CR12","first-page":"1","volume":"22","author":"T Hoefler","year":"2021","unstructured":"Hoefler T, Alistarh D, Ben-Nun T, Dryden N, Peste A (2021) Sparsity in deep learning: Pruning and growth for efficient inference and training in neural networks. J Mach Learn Res 22(241):1\u2013124","journal-title":"J Mach Learn Res"},{"key":"724_CR13","doi-asserted-by":"publisher","unstructured":"Huang X, Huang Y-L, Wen Z (2025a) Sola: Leveraging soft activation sparsity and low-rank decomposition for large language model compression. In: Walsh T, Shah J, Kolter Z (eds) the Association for the Advancement of Artificial Intelligence, AAAI, pp 17494\u201317502. AAAI Press, Philadelphia, PA, USA. https:\/\/doi.org\/10.1609\/AAAI.V39I16.33923","DOI":"10.1609\/AAAI.V39I16.33923"},{"key":"724_CR14","unstructured":"Huang W, Liao Y, Liu J, He R, Tan H, Zhang S, Li H, Liu S, Qi X (2025b) Mixture compressor for mixture-of-experts llms gains more. In: The Thirteenth International Conference on Learning Representations, ICLR, Singapore"},{"key":"724_CR15","unstructured":"Hu EJ, Shen Y, Wallis P, Allen-Zhu Z, Li Y, Wang S, Wang L, Chen W (2022) Lora: Low-rank adaptation of large language models. In: International Conference on Learning Representations, ICLR"},{"key":"724_CR16","unstructured":"Ko J, Kim S, Chen T, Yun S-Y (2024) Distillm: Towards streamlined distillation for large language models. In: Forty-first International Conference on Machine Learning, ICML. PMLR, Vienna, Austria"},{"key":"724_CR17","unstructured":"Kurtz M, Kopinsky J, Gelashvili R, Matveev A, Carr J, Goin M, Leiserson W, Moore S, Nell B, Shavit N, Alistarh D (2020) Inducing and exploiting activation sparsity for fast inference on deep neural networks. In: Daum\u00e9\u00a0III H, Singh A (eds) Proceedings of the 37th International Conference on Machine Learning, ICML, vol 119, pp 5533\u20135543. PMLR, Virtual"},{"key":"724_CR18","unstructured":"Keskar NS, Mudigere D, Nocedal J, Smelyanskiy M, Tang PTP (2017) On large-batch training for deep learning: Generalization gap and sharp minima. In: International Conference on Learning Representations, ICLR"},{"key":"724_CR19","unstructured":"Lecun Y, Denker J, Solla S (1989) Optimal brain damage 2:598\u2013605"},{"key":"724_CR20","doi-asserted-by":"publisher","unstructured":"Lee B-K, Kim J, Ro YM (2022) Masking adversarial damage: Finding adversarial saliency for robust and sparse network. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, pp 15105\u201315115. IEEE, New Orleans, LA, USA. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01470","DOI":"10.1109\/CVPR52688.2022.01470"},{"key":"724_CR21","doi-asserted-by":"publisher","unstructured":"Liu J, Kong Z, Zhao P, Yang C, Shen X, Tang H, Yuan G, Niu W, Zhang W, Lin X, Huang D, Wang Y (2025) Toward adaptive large language models: Structured pruning via hybrid-grained weight importance assessment. In: Walsh T, Shah J, Kolter Z (eds) Proceedings of the 39th AAAI Conference on Artificial Intelligence, AAAI, pp 18879\u201318887. AAAI Press. https:\/\/doi.org\/10.1609\/aaai.v39i18.34078","DOI":"10.1609\/aaai.v39i18.34078"},{"key":"724_CR22","unstructured":"Lee J, Park S, Mo S, Ahn S, Shin J (2021) Layer-adaptive sparsity for the magnitude-based pruning. In: International Conference on Learning Representations, ICLR"},{"key":"724_CR23","unstructured":"Li H, Xu Z, Taylor G, Studer C, Goldstein T (2018) Visualizing the loss landscape of neural nets. In: Advances in neural information processing systems, neurips"},{"key":"724_CR24","unstructured":"Magic N (2021) DeepSparse Engine: Sparsity-aware Deep Learning Inference Runtime for CPUs"},{"issue":"146","key":"724_CR25","first-page":"1","volume":"21","author":"J Martens","year":"2020","unstructured":"Martens J (2020) New insights and perspectives on the natural gradient method. J Mach Learn Res 21(146):1\u201376","journal-title":"J Mach Learn Res"},{"key":"724_CR26","doi-asserted-by":"publisher","unstructured":"Mihaylov T, Clark P, Khot T, Sabharwal A (2018) Can a suit of armor conduct electricity? a new dataset for open book question answering. In: Riloff E, Chiang D, Hockenmaier J, Tsujii J (eds) Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, EMNLP, pp 2381\u20132391. Association for Computational Linguistics, Brussels, Belgium. https:\/\/doi.org\/10.18653\/v1\/D18-1260","DOI":"10.18653\/v1\/D18-1260"},{"key":"724_CR27","doi-asserted-by":"crossref","unstructured":"Ma X, Fang G, Wang X (2023) Llm-pruner: On the structural pruning of large language models. In: Oh A, Naumann T, Globerson A, Saenko K, Hardt M, Levine S (eds) Advances in Neural Information Processing Systems, Neurips, vol 36. Curran Associates Inc, New Orleans, Louisiana, USA, pp 21702\u201321720","DOI":"10.52202\/075280-0950"},{"key":"724_CR28","unstructured":"Merity S, Xiong C, Bradbury J, Socher R (2017) Pointer sentinel mixture models. In: International Conference on Learning Representations, ICLR"},{"key":"724_CR29","doi-asserted-by":"publisher","unstructured":"Men X, Xu M, Zhang Q, Yuan Q, Wang B, Lin H, Lu Y, Han X, Chen W (2025) Shortgpt: Layers in large language models are more redundant than you expect. In: Che W, Nabende J, Shutova E, Pilehvar MT (eds) Findings of the Association for Computational Linguistics, ACL, pp 20192\u201320204. Association for Computational Linguistics, Vienna, Austria. https:\/\/doi.org\/10.18653\/v1\/2025.findings-acl.1035","DOI":"10.18653\/v1\/2025.findings-acl.1035"},{"key":"724_CR30","first-page":"47124","volume-title":"Advances in Neural Information Processing Systems, Neurips","author":"S Padmanabhan","year":"2023","unstructured":"Padmanabhan S, Onoe Y, Zhang M, Durrett G, Choi E (2023) Propagating knowledge updates to lms through distillation. In: Oh A, Naumann T, Globerson A, Saenko K, Hardt M, Levine S (eds) Advances in Neural Information Processing Systems, Neurips, vol 36. Curran Associates Inc, New Orleans, Louisiana, USA, pp 47124\u201347142"},{"issue":"9","key":"724_CR31","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3474381","volume":"64","author":"K Sakaguchi","year":"2021","unstructured":"Sakaguchi K, Le Bras R, Bhagavatula C, Choi Y (2021) Winogrande: An adversarial winograd schema challenge at scale. Commun ACM 64(9):99\u2013106. https:\/\/doi.org\/10.1145\/3474381","journal-title":"Commun ACM"},{"key":"724_CR32","unstructured":"Sun M, Liu Z, Bair A, Kolter Z (2024) A simple and effective pruning approach for large language models. In: Kim B, Yue Y, Chaudhuri S, Fragkiadaki K, Khan M, Sun Y (eds) International Conference on Learning Representations, ICLR, vol 2024, pp 4942\u20134964"},{"key":"724_CR33","unstructured":"Song J, Oh K, Kim T, Kim H, Kim Y, Kim J-J (2024) Sleb: Streamlining llms through redundancy verification and elimination of transformer blocks. In: Salakhutdinov R, Kolter Z, Heller K, Weller A, Oliver N, Scarlett J, Berkenkamp F (eds) Proceedings of the 41st International Conference on Machine Learning, ICML, vol 235, pp 46136\u201346155. PMLR, Vienna, Austria"},{"key":"724_CR34","doi-asserted-by":"publisher","unstructured":"Shen X, Zhao P, Gong Y, Kong Z, Zhan Z, Wu Y, Lin M, Wu C, Lin X, Wang Y (2024) Search for efficient large language models. In: Globerson A, Mackey L, Belgrave D, Fan A, Paquet U, Tomczak J, Zhang C (eds) Advances in Neural Information Processing Systems, Neurips, vol 37, pp 139294\u2013139315. Curran Associates, Inc., Vancouver, Canada. https:\/\/doi.org\/10.52202\/079017-4421","DOI":"10.52202\/079017-4421"},{"key":"724_CR35","doi-asserted-by":"publisher","unstructured":"Sun Y, Zheng L, Wang Q, Ye X, Huang Y, Yao P, Liao X, Jin H (2022) Accelerating sparse deep neural network inference using gpu tensor cores. In: IEEE High Performance Extreme Computing Conference, HPEC, pp 1\u20137. https:\/\/doi.org\/10.1109\/HPEC55821.2022.9926300","DOI":"10.1109\/HPEC55821.2022.9926300"},{"key":"724_CR36","doi-asserted-by":"publisher","unstructured":"Wang Y, Chen Y (2026) Temperature-driven category decoupled knowledge distillation with interpretability for model compression. Adv Eng Inf 69:104051. https:\/\/doi.org\/10.1016\/J.AEI.2025.104051","DOI":"10.1016\/J.AEI.2025.104051"},{"key":"724_CR37","doi-asserted-by":"crossref","unstructured":"Wang A, Singh A, Michael J, Hill F, Levy O, Bowman SR (2019) Glue: A multi-task benchmark and analysis platform for natural language understanding. In: 7th International Conference on Learning Representations, ICLR, New Orleans, LA, USA","DOI":"10.18653\/v1\/W18-5446"},{"key":"724_CR38","unstructured":"Xia M, Gao T, Zeng Z, Chen D (2024) Sheared llama: Accelerating language model pre-training via structured pruning. In: Kim B, Yue Y, Chaudhuri S, Fragkiadaki K, Khan M, Sun Y (eds) International Conference on Learning Representations, ICLR, vol 2024, pp 5385\u20135409"},{"key":"724_CR39","unstructured":"Xiong R, Yang Y, He D, Zheng K, Zheng S, Xing C, Zhang H, Lan Y, Wang L, Liu T-Y (2020) On layer normalization in the transformer architecture. In: Proceedings of the 37th International Conference on Machine Learning, ICML, pp 10524\u201310533. PMLR, Virtual Event"},{"issue":"5","key":"724_CR40","doi-asserted-by":"publisher","first-page":"779","DOI":"10.1049\/iet-ipr.2018.6191","volume":"13","author":"W Yang","year":"2019","unstructured":"Yang W, Jin L, Wang S, Cui Z, Chen X, Chen L (2019) Thinning of convolutional neural network with mixed pruning. IET Image Process 13(5):779\u2013784. https:\/\/doi.org\/10.1049\/iet-ipr.2018.6191","journal-title":"IET Image Process"},{"key":"724_CR41","unstructured":"Yang M, Lin S, Li C, Chang X (2025) Let llm tell what to prune and how much to prune. In: Forty-second International Conference on Machine Learning, ICML"},{"key":"724_CR42","doi-asserted-by":"crossref","unstructured":"Zhang Y, Bai H, Lin H, Zhao J, Hou L, Cannistraci CV (2024a) Plug-and-play: An efficient post-training pruning method for large language models. In: Kim B, Yue Y, Chaudhuri S, Fragkiadaki K, Khan M, Sun Y (eds) International Conference on Learning Representations, ICLR, vol 2024, pp 50490\u201350508","DOI":"10.20944\/preprints202310.1487.v2"},{"key":"724_CR43","unstructured":"Zhang C, Cheng J, Constantinides GA, Zhao Y (2024b) Lqer: Low-rank quantization error reconstruction for llms. In: Forty-first International Conference on Machine Learning, ICML"},{"key":"724_CR44","doi-asserted-by":"publisher","unstructured":"Zellers R, Holtzman A, Bisk Y, Farhadi A, Choi Y (2019) Hellaswag: Can a machine really finish your sentence? In: Korhonen A, Traum DR, M\u00e0rquez L (eds) Proceedings of the 57th Conference of the Association for Computational Linguistics, ACL, pp 4791\u20134800. Association for Computational Linguistics, Florence, Italy. https:\/\/doi.org\/10.18653\/v1\/P19-1472","DOI":"10.18653\/v1\/P19-1472"},{"key":"724_CR45","doi-asserted-by":"publisher","unstructured":"Zhong L, Wan F, Chen R, Quan X, Li L (2025) Blockpruner: Fine-grained pruning for large language models. In: Che W, Nabende J, Shutova E, Pilehvar MT (eds) Findings of the Association for Computational Linguistics, ACL, pp 5065\u20135080. Association for Computational Linguistics, Vienna, Austria. https:\/\/doi.org\/10.18653\/v1\/2025.findings-acl.262","DOI":"10.18653\/v1\/2025.findings-acl.262"},{"key":"724_CR46","unstructured":"Zhang G, Wang C, Xu B, Grosse RB (2019) Three mechanisms of weight decay regularization. In: 7th International Conference on Learning Representations, ICLR, New Orleans, LA, USA"}],"container-title":["Journal of King Saud University Computer and Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s44443-026-00724-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44443-026-00724-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44443-026-00724-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T09:32:32Z","timestamp":1784799152000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s44443-026-00724-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,30]]},"references-count":46,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,8]]}},"alternative-id":["724"],"URL":"https:\/\/doi.org\/10.1007\/s44443-026-00724-4","relation":{},"ISSN":["1319-1578","2213-1248"],"issn-type":[{"value":"1319-1578","type":"print"},{"value":"2213-1248","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,30]]},"assertion":[{"value":"29 January 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","label":"Competing interests","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"374"}}