{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T15:52:23Z","timestamp":1782489143822,"version":"3.54.5"},"reference-count":149,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.neunet.2026.109267","type":"journal-article","created":{"date-parts":[[2026,6,21]],"date-time":"2026-06-21T14:36:48Z","timestamp":1782052608000},"page":"109267","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["The landscape of pruning for large language models: A systematic review and unified taxonomy"],"prefix":"10.1016","volume":"204","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-3530-6903","authenticated-orcid":false,"given":"Yuli","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuchuan","family":"Mu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiang","family":"Tong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiulei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neunet.2026.109267_bib0001","series-title":"Proceedings of the 2024 conference on empirical methods in natural language processing, EMNLP 2024, miami, fl, usa, november 12\u201316, 2024","first-page":"19154","article-title":"ShadowLLM: Predictor-based contextual sparsity for large language models","author":"Akhauri","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0002","series-title":"SC22: International conference for high performance computing, networking, storage and analysis, dallas, tx, usa, november 13\u201318, 2022","first-page":"46:1","article-title":"Deepspeed- inference: Enabling efficient inference of transformer models at unprecedented scale","author":"Aminabadi","year":"2022"},{"key":"10.1016\/j.neunet.2026.109267_bib0003","first-page":"10865","article-title":"Fluctuation-based adaptive structured pruning for large language models","author":"An","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0004","series-title":"The twelfth international conference on learning representations, ICLR 2024, vienna, austria, may 7\u201311, 2024","article-title":"SliceGPT: Compress large language models by deleting rows and columns","author":"Ashkboos","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0005","unstructured":"Bai, H., Jian, S., Liang, T., Yin, Y., & Wang, H. (2025). ResSVD: Residual compensated SVD for large language model compression. arXiv preprint arXiv: abs\/2505.20112."},{"key":"10.1016\/j.neunet.2026.109267_bib0006","article-title":"Qwen technical report","volume":"abs\/2309.16609","author":"Bai","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0007","series-title":"Proceedings of the workshop on machine reading for question answering@ACL 2018, melbourne, australia, july 19, 2018","first-page":"60","article-title":"A systematic classification of knowledge, reasoning, and context within the ARC dataset","author":"Boratko","year":"2018"},{"key":"10.1016\/j.neunet.2026.109267_bib0008","article-title":"Fast and effective weight update for pruned large language models","volume":"2024","author":"Boza","year":"2024","journal-title":"Transactions Machine Learning Research"},{"key":"10.1016\/j.neunet.2026.109267_bib0009","series-title":"Practice and experience in advanced research computing 2019: Rise of the machines (learning)","first-page":"1","article-title":"Open compass: Accelerating the adoption of AI in open research","author":"Buitrago","year":"2019"},{"key":"10.1016\/j.neunet.2026.109267_bib0010","article-title":"LoRAShear: Efficient large language model structured pruning and knowledge recovery","volume":"abs\/2310.18356","author":"Chen","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0011","article-title":"A simple linear patch revives layer-pruned large language models","volume":"abs\/2505.24680","author":"Chen","year":"2025","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0012","series-title":"The thirteenth international conference on learning representations, ICLR 2025, singapore, april 24\u201328, 2025","article-title":"Streamlining redundant layers to compress large language models","author":"Chen","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0013","article-title":"Prune&comp: Free lunch for layer-pruned LLMs via iterative pruning with magnitude compensation","volume":"abs\/2507.18212","author":"Chen","year":"2025","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0014","series-title":"Forty-second international conference on machine learning, ICML 2025, vancouver, bc, canada, july 13\u201319, 2025","article-title":"DLP: Dynamic layerwise pruning in large language models","volume":"vol. 267","author":"Chen","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0015","article-title":"MINI-LLM: Memory-efficient structured pruning for large language models","volume":"abs\/2407.11681","author":"Cheng","year":"2024","journal-title":"CoRR"},{"issue":"5","key":"10.1016\/j.neunet.2026.109267_bib0016","doi-asserted-by":"crossref","first-page":"2025","DOI":"10.1109\/TNNLS.2025.3628671","article-title":"Survey on efficient large language models: principles, algorithms, applications, and open issues","volume":"37","author":"Cheng","year":"2026","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.neunet.2026.109267_bib0017","series-title":"Proceedings of the 2019 conference of the north american chapter of the association for computational linguistics: Human language technologies, NAACL-HLT 2019, minneapolis, mn, usa, june 2\u20137, 2019, volume 1 (long and short papers)","first-page":"2924","article-title":"BooLQ: Exploring the surprising difficulty of natural yes\/no questions","author":"Clark","year":"2019"},{"key":"10.1016\/j.neunet.2026.109267_bib0018","article-title":"Beyond size: How gradients shape pruning decisions in large language models","volume":"abs\/2311.04902","author":"Das","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0019","unstructured":"Ding, X., Sun, R., Zhang, Y., Yan, X., Zhou, Y., Huang, K., Fu, S., Aviles-Rivero, A. I., Xie, C., & Zhu, Y. (2025a). A sliding layer merging method for efficient depth-wise pruning in LLMs. arXiv preprint arXiv: abs\/2502.19159."},{"key":"10.1016\/j.neunet.2026.109267_bib0020","unstructured":"Ding, X., Sun, R., Zhang, Y., Yan, X., Zhou, Y., Huang, K., Fu, S., Xie, C., & Zhu, Y. (2025b). DipSVD: Dual-importance protected SVD for efficient LLM compression. arXiv preprint arXiv: abs\/2506.20353."},{"key":"10.1016\/j.neunet.2026.109267_bib0021","series-title":"Forty-first international conference on machine learning, ICML 2024, vienna, austria, july 21\u201327, 2024","article-title":"Pruner-zero: Evolving symbolic pruning metric from scratch for large language models","author":"Dong","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0022","series-title":"Forty-second international conference on machine learning, ICML 2025, vancouver, bc, canada, july 13\u201319, 2025","article-title":"Can compressed llms truly act? an empirical evaluation of agentic capabilities in LLM compression","volume":"vol. 267","author":"Dong","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0023","series-title":"Proceedings of the 63rd annual meeting of the association for computational linguistics (volume 1: Long papers), ACL 2025, vienna, austria, july 27, - august 1, 2025","first-page":"32174","article-title":"Emergent abilities of large language models under continued pre-training for language adaptation","author":"Elhady","year":"2025"},{"issue":"55","key":"10.1016\/j.neunet.2026.109267_bib0024","first-page":"1","article-title":"Neural architecture search: A survey","volume":"20","author":"Elsken","year":"2019","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.neunet.2026.109267_bib0025","series-title":"Proceedings of the twentieth european conference on computer systems, eurosys 2025, rotterdam, the netherlands, 30 march 2025 - 3 april 2025","first-page":"243","article-title":"Spinfer: Leveraging low-level sparsity for efficient large language model inference on GPUs","author":"Fan","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0026","series-title":"Advances in neural information processing systems 38: annual conference on neural information processing systems 2024, neurIPS 2024, vancouver, BC, canada, december 10, - 15, 2024","article-title":"MaskLLM: Learnable semi-structured sparsity for large language models","author":"Fang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0027","unstructured":"Feng, M., Wu, J., Liu, S., Zhang, S., Fang, H., Jin, R., Che, F., Shao, P., Wen, Z., & Tao, J. (2025a). Elder: Getting efficient llms through data-driven regularized layer-wise pruning. arXiv preprint arXiv: abs\/2505.18232."},{"key":"10.1016\/j.neunet.2026.109267_bib0028","unstructured":"Feng, M., Wu, J., Zhang, S., Shao, P., Jin, R., Wen, Z., Tao, J., & Che, F. (2025b). Dress: Data-driven regularized structured streamlining for large language models. arXiv preprint arXiv: abs\/2501.17905."},{"key":"10.1016\/j.neunet.2026.109267_bib0029","series-title":"Advances in Neural Information Processing Systems 28","first-page":"2962","article-title":"Efficient and robust automated machine learning","author":"Feurer","year":"2015"},{"key":"10.1016\/j.neunet.2026.109267_bib0030","unstructured":"Fischer, T., Biemann, C. et al. (2024). Large language models are overparameterized text encoders. arXiv preprint arXiv: abs\/2410.14578."},{"key":"10.1016\/j.neunet.2026.109267_bib0031","series-title":"7th international conference on learning representations, ICLR 2019, new orleans, la, usa, may 6\u20139, 2019","article-title":"The lottery ticket hypothesis: Finding sparse, trainable neural networks","author":"Frankle","year":"2019"},{"key":"10.1016\/j.neunet.2026.109267_bib0032","series-title":"International conference on machine learning, ICML 2023, 23\u201329 july 2023, honolulu, hawaii, USA","first-page":"10323","article-title":"SparseGPT: Massive language models can be accurately pruned in one-shot","volume":"vol. 202","author":"Frantar","year":"2023"},{"key":"10.1016\/j.neunet.2026.109267_bib0033","series-title":"The eleventh international conference on learning representations, ICLR 2023, kigali, rwanda, may 1\u20135, 2023","article-title":"OPTQ: Accurate quantization for generative pre-trained transformers","author":"Frantar","year":"2023"},{"key":"10.1016\/j.neunet.2026.109267_bib0034","unstructured":"Gao, L., Tow, J., Abbasi, B., Biderman, S., Black, S., DiPofi, A., Foster, C., Golding, L., Hsu, J., Le Noac\u2019h, A., Li, H., McDonell, K., Muennighoff, N., Ociepa, C., Phang, J., Reynolds, L., Schoelkopf, H., Skowron, A., Sutawika, L., Tang, E., Thite, A., Wang, B., Wang, K., & Zou, A. (2024). A framework for few-shot language model evaluation. https:\/\/zenodo.org\/records\/12608602. 10.5281\/zenodo.12608602."},{"issue":"2","key":"10.1016\/j.neunet.2026.109267_bib0035","first-page":"155","article-title":"Structure-mapping: A theoretical framework for analogy","volume":"7","author":"Gentner","year":"1983","journal-title":"Cognitive Science"},{"key":"10.1016\/j.neunet.2026.109267_bib0036","unstructured":"Goyal, V., Khan, M., Tirupati, A., Saini, H., Lam, M., & Zhu, K. (2024). Enhancing knowledge distillation for LLMs with response-priming prompting. arXiv preprint arXiv: 2412.17846."},{"key":"10.1016\/j.neunet.2026.109267_bib0037","series-title":"The thirteenth international conference on learning representations, ICLR 2025, singapore, april 24\u201328, 2025","article-title":"The unreasonable ineffectiveness of the deeper layers","author":"Gromov","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0038","series-title":"First conference on language modeling","article-title":"Mamba: Linear-time sequence modeling with selective state spaces","author":"Gu","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0039","series-title":"The twelfth international conference on learning representations, ICLR 2024, vienna, austria, may 7\u201311, 2024","article-title":"MiniLLM: Knowledge distillation of large language models","author":"Gu","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0040","series-title":"Forty-second international conference on machine learning, ICML 2025, vancouver, bc, canada, july 13\u201319, 2025","article-title":"SlimLLM: Accurate structured pruning for large language models","volume":"vol. 267","author":"Guo","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0041","unstructured":"Guo, S., Xu, J., Zhang, L. L., & Yang, M. (2023). Compresso: Structured pruning with collaborative prompting learns compact large language models. arXiv preprint arXiv: 2310.05015."},{"key":"10.1016\/j.neunet.2026.109267_bib0042","article-title":"Dependency-aware semi-structured sparsity of GLU variants in large language models","volume":"2025","author":"Guo","year":"2025","journal-title":"IEEE Transactions on Machine Learning Research"},{"key":"10.1016\/j.neunet.2026.109267_bib0043","series-title":"4Th international conference on learning representations, ICLR 2016, san juan, puerto rico, may 2\u20134, 2016, conference track proceedings","article-title":"Deep compression: Compressing deep neural network with pruning, trained quantization and huffman coding","author":"Han","year":"2016"},{"key":"10.1016\/j.neunet.2026.109267_bib0044","series-title":"Proceedings of international conference on neural networks (ICNN\u201988), san francisco, CA, USA, march 28, - april 1, 1993","first-page":"293","article-title":"Optimal brain surgeon and general network pruning","author":"Hassibi","year":"1993"},{"key":"10.1016\/j.neunet.2026.109267_bib0045","unstructured":"He, S., Sun, G., Shen, Z., & Li, A. (2024). What matters in transformers? not all attention is needed. arXiv preprint arXiv: 2406.15786."},{"issue":"5","key":"10.1016\/j.neunet.2026.109267_bib0046","doi-asserted-by":"crossref","first-page":"2900","DOI":"10.1109\/TPAMI.2023.3334614","article-title":"Structured pruning for deep convolutional neural networks: A survey","volume":"46","author":"He","year":"2024","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.neunet.2026.109267_bib0047","series-title":"Forty-first international conference on machine learning, ICML 2024, vienna, austria, july 21\u201327, 2024","article-title":"Decoding compressed trust: Scrutinizing the trustworthiness of efficient LLMs under compression","author":"Hong","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0048","doi-asserted-by":"crossref","first-page":"1270","DOI":"10.52202\/079017-0040","article-title":"KVQuant: Towards 10 million context length llm inference with kv cache quantization","volume":"37","author":"Hooper","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"9","key":"10.1016\/j.neunet.2026.109267_bib0049","first-page":"5149","article-title":"Meta-learning in neural networks: A survey","volume":"44","author":"Hospedales","year":"2021","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.neunet.2026.109267_bib0050","series-title":"Forty-second international conference on machine learning, ICML 2025, vancouver, bc, canada, july 13\u201319, 2025","article-title":"Instruction-following pruning for large language models","volume":"vol. 267","author":"Hou","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0051","series-title":"The tenth international conference on learning representations, ICLR 2022, virtual event, april 25\u201329, 2022","article-title":"Language model compression with weighted low-rank factorization","author":"Hsu","year":"2022"},{"key":"10.1016\/j.neunet.2026.109267_bib0052","series-title":"The tenth international conference on learning representations, ICLR 2022, virtual event, april 25\u201329, 2022","article-title":"LoRA: Low-rank adaptation of large language models","author":"Hu","year":"2022"},{"key":"10.1016\/j.neunet.2026.109267_bib0053","unstructured":"Huang, H., Song, H.-J., & Pao, H.-K. (2024). Large language model pruning. arXiv preprint arXiv: 2406.00030."},{"key":"10.1016\/j.neunet.2026.109267_bib0054","series-title":"Aaai-25, sponsored by the association for the advancement of artificial intelligence, february 25, - march 4, 2025, philadelphia, pa, USA","first-page":"24167","article-title":"Pruning large language models with semi-structural adaptive sparse training","author":"Huang","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0055","article-title":"In-context learning distillation: Transferring few-shot learning ability of pre-trained language models","volume":"abs\/2212.10670","author":"Huang","year":"2022","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0056","series-title":"Advances in neural information processing systems 36: Annual conference on neural information processing systems 2023, neurIPS 2023, new orleans, LA, USA, december 10, - 16, 2023","article-title":"The emergence of essential sparsity in large pre-trained models: The weights that matter","author":"Jaiswal","year":"2023"},{"key":"10.1016\/j.neunet.2026.109267_bib0057","series-title":"The twelfth international conference on learning representations, ICLR 2024, vienna, austria, may 7\u201311, 2024","article-title":"Compressing LLMs: The truth is rarely pure and never simple","author":"Jaiswal","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0058","unstructured":"Jha, S., Erdogan, L. E., Kim, S., Keutzer, K., & Gholami, A. (2024). Characterizing prompt compression methods for long context inference. arXiv preprint arXiv: 2407.08892."},{"key":"10.1016\/j.neunet.2026.109267_bib0059","article-title":"Pruning large language models via accuracy predictor","volume":"abs\/2309.09507","author":"Ji","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0060","article-title":"Mistral 7b","volume":"abs\/2310.06825","author":"Jiang","year":"2023","journal-title":"CoRR"},{"issue":"5","key":"10.1016\/j.neunet.2026.109267_bib0061","doi-asserted-by":"crossref","first-page":"8091","DOI":"10.1007\/s11042-020-10139-6","article-title":"A review on genetic algorithm: Past, present, and future","volume":"80","author":"Katoch","year":"2021","journal-title":"Multimedia Tools and Applications"},{"key":"10.1016\/j.neunet.2026.109267_bib0062","unstructured":"Kim, B.-K., Kim, G., Kim, T.-H., Castells, T., Choi, S., Shin, J., & Song, H.-K. (2024). Shortened llama: Depth pruning for large language models with comparison of retraining methods. arXiv preprint arXiv: 2402.02834."},{"key":"10.1016\/j.neunet.2026.109267_bib0063","series-title":"Computer vision - ECCV 2022 - 17th european conference, tel aviv, israel, october 23\u201327, 2022, proceedings, part XX","first-page":"651","article-title":"Cprune: Compiler-informed model pruning for efficient target-aware DNN execution","volume":"vol. 13680","author":"Kim","year":"2022"},{"key":"10.1016\/j.neunet.2026.109267_bib0064","series-title":"Findings of the association for computational linguistics: EMNLP 2023, singapore, december 6\u201310, 2023","first-page":"6076","article-title":"NASH: A simple unified framework of structured pruning for accelerating encoder-decoder language models","author":"Ko","year":"2023"},{"key":"10.1016\/j.neunet.2026.109267_bib0065","article-title":"Sparse fine-tuning for inference acceleration of large language models","volume":"abs\/2310.06927","author":"Kurtic","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0066","series-title":"The thirteenth international conference on learning representations, ICLR 2025, singapore, april 24\u201328, 2025","article-title":"Probe Pruning: Accelerating LLMs through dynamic pruning via model-probing","author":"Le","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0067","series-title":"Advances in neural information processing systems 2, [NIPS conference, denver, colorado, USA, november 27\u201330, 1989]","first-page":"598","article-title":"Optimal brain damage","author":"LeCun","year":"1989"},{"key":"10.1016\/j.neunet.2026.109267_bib0068","article-title":"OWQ: Lessons learned from activation outliers for weight quantization in large language models","volume":"abs\/2306.02272","author":"Lee","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0069","unstructured":"Lee, D., Lee, J.-Y., Zhang, G., Tiwari, M., & Mirhoseini, A. (2024). Cats: Contextually-aware thresholding for sparsity in large language models. arXiv preprint arXiv: 2404.08763."},{"key":"10.1016\/j.neunet.2026.109267_bib0070","unstructured":"Leng, Y., & Xiong, D. (2024). Towards understanding multi-task learning (generalization) of llms via detecting and exploring task-specific neurons. arXiv preprint arXiv: 2407.06488."},{"key":"10.1016\/j.neunet.2026.109267_bib0071","article-title":"Pruningbench: A comprehensive benchmark of structural pruning","volume":"abs\/2406.12315","author":"Li","year":"2024","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0072","series-title":"Advances in neural information processing systems 38: Annual conference on neural information processing systems 2024, neurIPS 2024, vancouver, BC, canada, december 10, - 15, 2024","article-title":"Discovering sparsity allocation for layer-wise pruning of large language models","author":"Li","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0073","series-title":"Advances in neural information processing systems 38: Annual conference on neural information processing systems 2024, neurIPS 2024, vancouver, BC, canada, december 10, - 15, 2024","article-title":"Adaptive layer sparsity for large language models via activation correlation assessment","author":"Li","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0074","article-title":"IDEA Prune: An integrated enlarge-and-prune pipeline in generative language model pretraining","volume":"abs\/2503.05920","author":"Li","year":"2025","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0075","article-title":"E-Sparse: Boosting the large language model inference through entropy-based N: M sparsity","volume":"abs\/2310.15929","author":"Li","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0076","doi-asserted-by":"crossref","unstructured":"Li, Z., Jiang, G., Xie, H., Song, L., Lian, D., & Wei, Y. (2024d). Understanding and patching compositional reasoning in llms. arXiv preprint arXiv: 2402.14328.","DOI":"10.18653\/v1\/2024.findings-acl.576"},{"key":"10.1016\/j.neunet.2026.109267_bib0077","series-title":"Proceedings of the seventh annual conference on machine learning and systems, MLSys 2024, santa clara, CA, USA, may 13\u201316, 2024","article-title":"AWQ: Activation-aware weight quantization for on-device LLM compression and acceleration","author":"Lin","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0078","article-title":"Beemanc at the PLABA track of TAC-2024: roberta for task 1 - llama3.1 and gpt-4o for task 2","volume":"abs\/2411.07381","author":"Ling","year":"2024","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0079","series-title":"Proceedings of the 2024 conference on empirical methods in natural language processing, EMNLP 2024, miami, fl, usa, november 12\u201316, 2024","first-page":"17817","article-title":"Pruning via merging: Compressing LLMs via manifold alignment based layer merging","author":"Liu","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0080","article-title":"RAP: Runtime-adaptive pruning for LLM inference","volume":"abs\/2505.17138","author":"Liu","year":"2025","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0081","series-title":"Proceedings of the 2025 conference on empirical methods in natural language processing, EMNLP 2025, suzhou, china, november 4\u20139, 2025","first-page":"26333","article-title":"GRASP: Replace redundant layers with adaptive singular parameters for efficient model compression","author":"Liu","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0082","series-title":"Proceedings of the 2023 conference on empirical methods in natural language processing, EMNLP 2023, singapore, december 6\u201310, 2023","first-page":"592","article-title":"LLM-FP4: 4-Bit floating-point quantized transformers","author":"Liu","year":"2023"},{"key":"10.1016\/j.neunet.2026.109267_bib0083","article-title":"EoRA: Training-free compensation for compressed LLM with eigenspace low-rank approximation","volume":"abs\/2410.21271","author":"Liu","year":"2024","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0084","series-title":"International conference on machine learning, ICML 2023, 23\u201329 july 2023, honolulu, hawaii, USA","first-page":"22137","article-title":"Deja Vu: Contextual sparsity for efficient LLMs at inference time","volume":"vol. 202","author":"Liu","year":"2023"},{"key":"10.1016\/j.neunet.2026.109267_bib0085","series-title":"Advances in neural information processing systems 38: Annual conference on neural information processing systems 2024, neurIPS 2024, vancouver, BC, canada, december 10, - 15, 2024","article-title":"Alphapruning: Using heavy-tailed self regularization theory for improved layer-wise pruning of large language models","author":"Lu","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0086","series-title":"Advances in neural information processing systems 36: Annual conference on neural information processing systems 2023, neurIPS 2023, new orleans, LA, USA, december 10, - 16, 2023","article-title":"Llm-pruner: On the structural pruning of large language models","author":"Ma","year":"2023"},{"key":"10.1016\/j.neunet.2026.109267_bib0087","series-title":"Findings of the association for computational linguistics, ACL 2025, vienna, austria, july 27, - august 1, 2025","first-page":"20192","article-title":"ShortGPT: Layers in large language models are more redundant than you expect","volume":"vol. ACL 2025","author":"Men","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0088","unstructured":"Mi, Z., Kong, Z., Yuan, G., & Huang, S. (2025). Ace: Exploring activation cosine similarity and variance for accurate and calibration-efficient llm pruning. arXiv preprint arXiv: 2505.21987."},{"key":"10.1016\/j.neunet.2026.109267_bib0089","series-title":"Proceedings of the 2018 conference on empirical methods in natural language processing, brussels, belgium, october 31, - november 4, 2018","first-page":"2381","article-title":"Can a suit of armor conduct electricity? a new dataset for open book question answering","author":"Mihaylov","year":"2018"},{"key":"10.1016\/j.neunet.2026.109267_bib0090","series-title":"International joint conference on neural networks, IJCNN 2025, rome, italy, june 30, - july 5, 2025","first-page":"1","article-title":"Efficient llms with AMP: attention heads and MLP pruning","author":"Mugnaini","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0091","series-title":"Advances in neural information processing systems 38: Annual conference on neural information processing systems 2024, neurIPS 2024, vancouver, BC, canada, december 10, - 15, 2024","article-title":"Compact language models via pruning and knowledge distillation","author":"Muralidharan","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0092","unstructured":"NeuralMagic (2024). Neuralmagic nm-vllm inference engine. https:\/\/github.com\/neuralmagic\/nm-vllm."},{"key":"10.1016\/j.neunet.2026.109267_bib0093","article-title":"GPT-4 Technical report","volume":"abs\/2303.08774","author":"OpenAI","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0094","unstructured":"Park, S., Choi, J., Lee, S., & Kang, U. (2024). A comprehensive survey of compression algorithms for language models. arXiv preprint arXiv: 2401.15347."},{"key":"10.1016\/j.neunet.2026.109267_bib0095","series-title":"Findings of the association for computational linguistics: EMNLP 2023, singapore, december 6\u201310, 2023","first-page":"14048","article-title":"RWKV: Reinventing rnns for the transformer era","volume":"vol. EMNLP 2023","author":"Peng","year":"2023"},{"key":"10.1016\/j.neunet.2026.109267_bib0096","unstructured":"Phan, P., Tran, H., & Phan, L. (2024). Distillation contrastive decoding: Improving llms reasoning with contrastive decoding and distillation. arXiv preprint arXiv: 2402.14874."},{"key":"10.1016\/j.neunet.2026.109267_bib0097","unstructured":"Reda, W., Jangda, A., & Chintalapudi, K. (2025). Task specific pruning with LLM-sieve: How many parameters does your task really need?* arXiv preprint arXiv: 2505.18350."},{"key":"10.1016\/j.neunet.2026.109267_bib0098","first-page":"8732","article-title":"WinoGrande: An adversarial winograd schema challenge at scale","author":"Sakaguchi","year":"2020"},{"key":"10.1016\/j.neunet.2026.109267_bib0099","article-title":"2SSP: A two-stage framework for structured pruning of LLMs","volume":"2025","author":"Sandri","year":"2025","journal-title":"IEEE Transactions on Machine Learning in Communications Research"},{"key":"10.1016\/j.neunet.2026.109267_bib0100","series-title":"Proceedings of the workshop on data management for end-to-end machine learning","first-page":"1","article-title":"PQ Bench: Benchmarking pruning and quantization techniques","author":"Schulze","year":"2025"},{"issue":"2","key":"10.1016\/j.neunet.2026.109267_bib0101","doi-asserted-by":"crossref","first-page":"216","DOI":"10.1109\/LSP.2017.2647948","article-title":"Total variation denoising via the moreau envelope","volume":"24","author":"Selesnick","year":"2017","journal-title":"IEEE Signal Processing Letters"},{"key":"10.1016\/j.neunet.2026.109267_bib0102","series-title":"Proceedings of the 2025 conference of the nations of the americas chapter of the association for computational linguistics: Human language technologies, NAACL 2025 - volume 1: Long papers, albuquerque, new mexico, usa, april 29, - may 4, 2025","first-page":"1511","article-title":"CompAct: Compressed activations for memory-efficient LLM training","author":"Shamshoum","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0103","series-title":"IEEE International conference on acoustics, speech and signal processing, ICASSP 2024, seoul, republic of korea, april 14\u201319, 2024","first-page":"11296","article-title":"One-shot sensitivity-aware mixed sparsity pruning for large language models","author":"Shao","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0104","series-title":"The twelfth international conference on learning representations, ICLR 2024, vienna, austria, may 7\u201311, 2024","article-title":"OmniQuant: Omnidirectionally calibrated quantization for large language models","author":"Shao","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0105","article-title":"ReplaceMe: Network simplification via layer pruning and linear transformations","volume":"abs\/2505.02819","author":"Shopkhoev","year":"2025","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0106","series-title":"Proceedings of the 2025 conference of the nations of the americas chapter of the association for computational linguistics: Human language technologies, NAACL 2025 - volume 1: Long papers, albuquerque, new mexico, usa, april 29, - may 4, 2025","first-page":"718","article-title":"FlexiGPT: Pruning and extending large language models with low-rank weight sharing","author":"Smith","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0107","series-title":"Forty-first international conference on machine learning, ICML 2024, vienna, austria, july 21\u201327, 2024","first-page":"46136","article-title":"SLEB: Streamlining llms through redundancy verification and elimination of transformer blocks","volume":"vol. 235","author":"Song","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0108","series-title":"Proceedings of the 31st international conference on computational linguistics, COLING 2025, abu dhabi, uae, january 19\u201324, 2025","first-page":"10298","article-title":"Self-evolution knowledge distillation for LLM-based machine translation","author":"Song","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0109","series-title":"The twelfth international conference on learning representations, ICLR 2024, vienna, austria, may 7\u201311, 2024","article-title":"A simple and effective pruning approach for large language models","author":"Sun","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0110","article-title":"A survey on transformer compression","volume":"abs\/2402.05964","author":"Tang","year":"2024","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0111","article-title":"LLaMA: Open and efficient foundation language models","volume":"abs\/2302.13971","author":"Touvron","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0112","series-title":"7th international conference on learning representations, ICLR 2019, new orleans, la, usa, may 6\u20139, 2019","article-title":"GLUE: A multi-task benchmark and analysis platform for natural language understanding","author":"Wang","year":"2019"},{"key":"10.1016\/j.neunet.2026.109267_bib0113","series-title":"The thirteenth international conference on learning representations, ICLR 2025, singapore, april 24\u201328, 2025","article-title":"Dobi-SVD: Differentiable SVD for LLM compression and some new perspectives","author":"Wang","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0114","article-title":"Model compression and efficient inference for large language models: A survey","volume":"abs\/2402.09748","author":"Wang","year":"2024","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0115","series-title":"Proceedings of the 2025 conference of the nations of the americas chapter of the association for computational linguistics: Human language technologies, NAACL 2025 - volume 1: Long papers, albuquerque, new mexico, usa, april 29, - may 4, 2025","first-page":"4287","article-title":"SVD-LLM V2: Optimizing singular value truncation for large language model compression","author":"Wang","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0116","series-title":"The thirteenth international conference on learning representations, ICLR 2025, singapore, april 24\u201328, 2025","article-title":"SVD-LLM: Truncation-aware singular value decomposition for large language model compression","author":"Wang","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0117","series-title":"Proceedings of the 31st international conference on computational linguistics, COLING 2025, abu dhabi, uae, january 19\u201324, 2025","first-page":"9311","article-title":"CFSP: An efficient structured pruning framework for llms with coarse-to-fine activation information","author":"Wang","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0118","unstructured":"Wang, Z., Zhang, J., Zhao, W., Farnia, F., & Yu, B. (2024b). Moreaupruner: Robust pruning of large language models against weight perturbations. arXiv preprint arXiv: 2406.07017."},{"key":"10.1016\/j.neunet.2026.109267_bib0119","series-title":"Forty-second international conference on machine learning, ICML 2025, vancouver, bc, canada, july 13\u201319, 2025","article-title":"Prompt-based depth pruning of large language models","volume":"vol. 267","author":"Wee","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0120","article-title":"Emergent abilities of large language models","volume":"2022","author":"Wei","year":"2022","journal-title":"IEEE Transactions on Machine Learning Research"},{"issue":"3","key":"10.1016\/j.neunet.2026.109267_bib0121","first-page":"729","article-title":"Reinforcement learning","volume":"12","author":"Wiering","year":"2012","journal-title":"Adaptation, Learning, and Optimization"},{"key":"10.1016\/j.neunet.2026.109267_bib0122","doi-asserted-by":"crossref","first-page":"36637","DOI":"10.52202\/075280-1593","article-title":"The learnability of in-context learning","volume":"36","author":"Wies","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109267_bib0123","article-title":"Flash-LLM: Enabling cost-effective and highly-efficient large generative model inference with unstructured sparsity","volume":"abs\/2309.10285","author":"Xia","year":"2023","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0124","series-title":"International conference on learning representations (ICLR)","article-title":"Sheared LLaMA: Accelerating language model pre-training via structured pruning","author":"Xia","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0125","series-title":"International conference on machine learning, ICML 2023, 23\u201329 july 2023, honolulu, hawaii, USA","first-page":"38087","article-title":"Smoothquant: Accurate and efficient post-training quantization for large language models","volume":"vol. 202","author":"Xiao","year":"2023"},{"key":"10.1016\/j.neunet.2026.109267_bib0126","article-title":"EfficientLLM: Scalable pruning-aware pretraining for architecture-agnostic edge language models","volume":"abs\/2502.06663","author":"Xing","year":"2025","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0127","article-title":"A survey on knowledge distillation of large language models","volume":"abs\/2402.13116","author":"Xu","year":"2024","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0128","series-title":"Findings of the association for computational linguistics: EMNLP 2024, miami, florida, usa, november 12\u201316, 2024","first-page":"15359","article-title":"Beyond perplexity: Multi-dimensional safety evaluation of LLM compression","author":"Xu","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0129","article-title":"Progressive binarization with semi-structured pruning for LLMs","volume":"abs\/2502.01705","author":"Yan","year":"2025","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0130","series-title":"Advances in neural information processing systems 38: Annual conference on neural information processing systems 2024, neurIPS 2024, vancouver, BC, canada, december 10, - 15, 2024","article-title":"LLMCBench: Benchmarking large language model compression for efficient deployment","author":"Yang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0131","article-title":"DASH: Input-aware dynamic layer skipping for efficient LLM inference with markov decision policies","volume":"abs\/2505.17420","author":"Yang","year":"2025","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0132","unstructured":"Yang, R., Wu, T., Wang, J., Hu, P., Wu, Y.-C., Wong, N., & Yang, Y. (2024b). Llm-neo: Parameter efficient knowledge distillation for large language models. arXiv preprint arXiv: 2411.06839."},{"key":"10.1016\/j.neunet.2026.109267_bib0133","series-title":"Findings of the association for computational linguistics: EMNLP 2024, miami, florida, usa, november 12\u201316, 2024","first-page":"6401","article-title":"LaCo: Large language model pruning via layer collapse","author":"Yang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0134","series-title":"Findings of the association for computational linguistics, ACL 2024, bangkok, thailand and virtual meeting, august 11\u201316, 2024","first-page":"1898","article-title":"Can large multimodal models uncover deep semantics behind images?","volume":"vol. ACL 2024","author":"Yang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0135","series-title":"Findings of the association for computational linguistics, ACL 2025, vienna, austria, july 27, - august 1, 2025","first-page":"4321","article-title":"Wanda++: Pruning large language models via regional gradients","author":"Yang","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0136","series-title":"Forty-first international conference on machine learning, ICML 2024, vienna, austria, july 21\u201327, 2024","article-title":"Outlier weighed layerwise sparsity (OWL): a missing secret sauce for pruning LLMs to high sparsity","author":"Yin","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0137","series-title":"Proceedings of the 57th conference of the association for computational linguistics, ACL 2019, florence, italy, july 28- august 2, 2019, volume 1: Long papers","first-page":"4791","article-title":"HellaSwag: Can a machine really finish your sentence?","author":"Zellers","year":"2019"},{"key":"10.1016\/j.neunet.2026.109267_bib0138","series-title":"Findings of the association for computational linguistics, ACL 2024, bangkok, thailand and virtual meeting, august 11\u201316, 2024","first-page":"3013","article-title":"LoRAPrune: Structured pruning meets low-rank parameter-efficient fine-tuning","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0139","series-title":"Findings of the association for computational linguistics, ACL 2024, bangkok, thailand and virtual meeting, august 11\u201316, 2024","first-page":"3013","article-title":"LoRAPrune: Structured pruning meets low-rank parameter-efficient fine-tuning","volume":"vol. ACL 2024","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0140","series-title":"Findings of the association for computational linguistics: NAACL 2024, mexico city, mexico, june 16\u201321, 2024","first-page":"1417","article-title":"Pruning as a domain-specific LLM extractor","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0141","article-title":"OPT: Open pre-trained transformer language models","volume":"abs\/2205.01068","author":"Zhang","year":"2022","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0142","series-title":"The twelfth international conference on learning representations, ICLR 2024, vienna, austria, may 7\u201311, 2024","article-title":"Plug-and-play: An efficient post-training pruning method for large language models","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0143","series-title":"Forty-first international conference on machine learning, ICML 2024, vienna, austria, july 21\u201327, 2024","article-title":"APT: Adaptive pruning and tuning pretrained language models for efficient training and inference","author":"Zhao","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0144","series-title":"Findings of the association for computational linguistics, ACL 2025, vienna, austria, july 27, - august 1, 2025","first-page":"5065","article-title":"BlockPruner: Fine-grained pruning for large language models","author":"Zhong","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0145","series-title":"Findings of the association for computational linguistics: NAACL 2025, albuquerque, new mexico, usa, april 29, - may 4, 2025","first-page":"5781","article-title":"Rankadaptor: Hierarchical rank allocation for efficient fine-tuning pruned LLMs via performance model","author":"Zhou","year":"2025"},{"key":"10.1016\/j.neunet.2026.109267_bib0146","series-title":"Advances in neural information processing systems 38: Annual conference on neural information processing systems 2024, neurIPS 2024, vancouver, BC, canada, december 10, - 15, 2024","article-title":"SIRIUS : Contexual sparisty with correction for efficient llms","author":"Zhou","year":"2024"},{"key":"10.1016\/j.neunet.2026.109267_bib0147","doi-asserted-by":"crossref","first-page":"7103","DOI":"10.52202\/068431-0515","article-title":"Mixture-of-experts with expert choice routing","volume":"35","author":"Zhou","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109267_bib0148","article-title":"Sdmprune: Self-distillation MLP pruning for efficient large language models","volume":"abs\/2506.11120","author":"Zhu","year":"2025","journal-title":"CoRR"},{"key":"10.1016\/j.neunet.2026.109267_bib0149","doi-asserted-by":"crossref","first-page":"1556","DOI":"10.1162\/tacl_a_00704","article-title":"A survey on model compression for large language models","volume":"12","author":"Zhu","year":"2024","journal-title":"Transactions Association Computer Linguistics"}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026007276?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026007276?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T15:32:00Z","timestamp":1782487920000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0893608026007276"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":149,"alternative-id":["S0893608026007276"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109267","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"The landscape of pruning for large language models: A systematic review and unified taxonomy","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109267","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"109267"}}