{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T04:51:56Z","timestamp":1785905516177,"version":"3.56.0"},"reference-count":269,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100002460","name":"Chung-Ang University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002460","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010418","name":"IITP","doi-asserted-by":"publisher","award":["2021-0-01341"],"award-info":[{"award-number":["2021-0-01341"]}],"id":[{"id":"10.13039\/501100010418","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010418","name":"IITP","doi-asserted-by":"publisher","award":["RS-2026-25546026"],"award-info":[{"award-number":["RS-2026-25546026"]}],"id":[{"id":"10.13039\/501100010418","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neucom.2026.134332","type":"journal-article","created":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T23:33:58Z","timestamp":1782257638000},"page":"134332","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Survey on joint compression and fine-tuning of large language models: Methods, toolchains, and open challenges under resource constraints"],"prefix":"10.1016","volume":"698","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-6069-262X","authenticated-orcid":false,"given":"Jihyeon","family":"Kim","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6900-4638","authenticated-orcid":false,"given":"Hanyong","family":"Lee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3757-3510","authenticated-orcid":false,"given":"Jaesung","family":"Lee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"12","key":"10.1016\/j.neucom.2026.134332_bib0005","doi-asserted-by":"crossref","first-page":"6074","DOI":"10.1109\/JBHI.2023.3316750","article-title":"Large AI models in health informatics: applications, challenges, and the future","volume":"27","author":"Qiu","year":"2023","journal-title":"IEEE J. Biomed. Health Inform."},{"key":"10.1016\/j.neucom.2026.134332_bib0010","series-title":"Proceedings of the 37th Annual Conference on Neural Information Processing Systems, December 10-16, 2023, New Orleans, Louisiana, USA","article-title":"FinGPT: instruction tuning benchmark for open-source large language models in financial datasets","author":"Wang","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0015","doi-asserted-by":"crossref","DOI":"10.1016\/j.lindif.2023.102274","article-title":"ChatGPT for good? On opportunities and challenges of large language models for education","volume":"103","author":"Kasneci","year":"2023","journal-title":"Learn. Individ. Differ."},{"key":"10.1016\/j.neucom.2026.134332_bib0020","author":"Grattafior"},{"key":"10.1016\/j.neucom.2026.134332_bib0025","series-title":"Proceedings of the 40th International Conference on Machine Learning, 202, July 23-29, 2023","first-page":"38087","article-title":"SmoothQuant: accurate and efficient post-training quantization for large language models","author":"Xiao","year":"2023"},{"issue":"2","key":"10.1016\/j.neucom.2026.134332_bib0030","doi-asserted-by":"crossref","first-page":"29","DOI":"10.1109\/MM.2021.3061394","article-title":"NVIDIA A100 tensor core GPU: performance and innovation","volume":"41","author":"Choquette","year":"2021","journal-title":"IEEE Micro"},{"key":"10.1016\/j.neucom.2026.134332_bib0035","series-title":"Proceedings of the 2024 USENIX Annual Technical Conference, July 10-12, 2024","first-page":"699","article-title":"Quant-LLM: accelerating the serving of large language models via FP6-centric algorithm-system co-design on modern GPUs","author":"Xia","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0040","doi-asserted-by":"crossref","first-page":"1556","DOI":"10.1162\/tacl_a_00704","article-title":"A survey on model compression for large language models","volume":"12","author":"Zhu","year":"2024","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"10.1016\/j.neucom.2026.134332_bib0045","doi-asserted-by":"crossref","DOI":"10.3389\/frobt.2025.1518965","article-title":"A survey of model compression techniques: past, present, and future","volume":"12","author":"Liu","year":"2025","journal-title":"Front. Robot. AI"},{"issue":"4","key":"10.1016\/j.neucom.2026.134332_bib0050","doi-asserted-by":"crossref","first-page":"87","DOI":"10.3390\/bdcc9040087","article-title":"LLM fine-tuning: concepts, opportunities, and challenges","volume":"9","author":"Wu","year":"2025","journal-title":"Big Data Cogn. Comput."},{"issue":"8","key":"10.1016\/j.neucom.2026.134332_bib0055","doi-asserted-by":"crossref","first-page":"227","DOI":"10.1007\/s10462-025-11236-4","article-title":"Parameter-efficient fine-tuning in large language models: a survey of methodologies","volume":"58","author":"Wang","year":"2025","journal-title":"Artif. Intell. Rev."},{"key":"10.1016\/j.neucom.2026.134332_bib0060","article-title":"Parameter-efficient fine-tuning for large models: a comprehensive survey","volume":"2024","author":"Han","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134332_bib0065","article-title":"Efficient large language models: a survey","volume":"2024","author":"Wan","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"issue":"5","key":"10.1016\/j.neucom.2026.134332_bib0070","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3744746","article-title":"A comprehensive overview of large language models","volume":"16","author":"Naveed","year":"2025","journal-title":"ACM Trans. Intell. Syst. Technol."},{"issue":"3","key":"10.1016\/j.neucom.2026.134332_bib0075","doi-asserted-by":"crossref","first-page":"2967","DOI":"10.1007\/s10115-024-02310-4","article-title":"Large language models: a survey of their development, capabilities, and applications","volume":"67","author":"Annepaka","year":"2025","journal-title":"Knowl. Inf. Syst."},{"issue":"10","key":"10.1016\/j.neucom.2026.134332_bib0080","first-page":"1","article-title":"Efficient compressing and tuning methods for large language models: a systematic literature review","volume":"57","author":"Kim","year":"2025","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.neucom.2026.134332_bib0085","series-title":"Proceedings of the 31st Conference on Neural Information Processing Systems, December 4-9, 2017, Long Beach, CA, USA","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.neucom.2026.134332_bib0090","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134332_bib0095","series-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, June 2-7, 2019","first-page":"4171","article-title":"BERT: pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.neucom.2026.134332_bib0100","series-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems, December 6-12, 2020, Vancouver, BC, Canada","article-title":"Language models are few-shot learners","author":"Brown","year":"2020"},{"key":"10.1016\/j.neucom.2026.134332_bib0105","series-title":"Text Summarization Branches Out, July 25-26, 2004","first-page":"74","article-title":"ROUGE: a package for automatic evaluation of summaries","author":"Lin","year":"2004"},{"key":"10.1016\/j.neucom.2026.134332_bib0110","series-title":"Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, July 7-12, 2002","first-page":"311","article-title":"BLEU: a method for automatic evaluation of machine translation","author":"Papineni","year":"2002"},{"key":"10.1016\/j.neucom.2026.134332_bib0115","series-title":"Proceedings of the 5th International Conference on Learning Representations, april 24\u201326, 2017","article-title":"Pointer sentinel mixture models","author":"Merity","year":"2017"},{"issue":"2","key":"10.1016\/j.neucom.2026.134332_bib0120","first-page":"313","article-title":"Building a large annotated corpus of english: the penn treebank","volume":"19","author":"Marcus","year":"1993","journal-title":"Comput. Linguist."},{"key":"10.1016\/j.neucom.2026.134332_bib0125","series-title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, November 7-11, 2021","first-page":"1286","article-title":"Documenting large webtext corpora: a case study on the colossal clean crawled corpus","author":"Dodge","year":"2021"},{"key":"10.1016\/j.neucom.2026.134332_bib0130","series-title":"Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), August 7-12, 2016","first-page":"1525","article-title":"The LAMBADA dataset: word prediction requiring a broad discourse context","author":"Paperno","year":"2016"},{"key":"10.1016\/j.neucom.2026.134332_bib0135","series-title":"Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"1601","article-title":"TriviaQA: a large scale distantly supervised challenge dataset for reading comprehension","author":"Joshi","year":"2017"},{"key":"10.1016\/j.neucom.2026.134332_bib0140","series-title":"Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing, November 1-4, 2016","first-page":"2383","article-title":"SQuAD: 100, 000+ questions for machine comprehension of text","author":"Rajpurkar","year":"2016"},{"key":"10.1016\/j.neucom.2026.134332_bib0145","series-title":"Proceedings of the 9th International Conference on Learning Representations, May 3-7, 2021","article-title":"Measuring massive multitask language understanding","author":"Hendrycks","year":"2021"},{"key":"10.1016\/j.neucom.2026.134332_bib0150","series-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems, 37, December 10-15, 2024","first-page":"95266","article-title":"MMLU-pro: a more robust and challenging multi-task language understanding benchmark","author":"Wang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0155","author":"Rein"},{"key":"10.1016\/j.neucom.2026.134332_bib0160","series-title":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, July 28\u2013August 2, 2019","first-page":"4791","article-title":"HellaSwag: can a machine really finish your sentence?","author":"Zellers","year":"2019"},{"issue":"5","key":"10.1016\/j.neucom.2026.134332_bib0165","first-page":"7432","article-title":"PIQA: reasoning about physical commonsense in natural language","volume":"34","author":"Bisk","year":"2020","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.neucom.2026.134332_bib0170","series-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP): System Demonstrations, November 3-7, 2019","first-page":"4463","article-title":"Social IQa: commonsense reasoning about social interactions","author":"Sap","year":"2019"},{"key":"10.1016\/j.neucom.2026.134332_bib0175","series-title":"The 34th AAAI Conference on Artificial Intelligence, February 7\u201312, 2020","first-page":"8732","article-title":"WINOGRANDE: an adversarial winograd schema challenge at scale","author":"Sakaguchi","year":"2020"},{"key":"10.1016\/j.neucom.2026.134332_bib0180","series-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), June 3-5, 2019","first-page":"2924","article-title":"BoolQ: exploring the surprising difficulty of natural Yes\/No questions","author":"Clark","year":"2019"},{"key":"10.1016\/j.neucom.2026.134332_bib0185","series-title":"Proceedings of the 20th International Conference on Learning Representations, May 7-11, 2024","article-title":"MuSR: testing the limits of chain-of-thought with multistep soft reasoning","author":"Sprague","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0190","author":"Cobbe"},{"key":"10.1016\/j.neucom.2026.134332_bib0195","series-title":"Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks, December 5-14, 2021, Virtual event","article-title":"Measuring mathematical problem solving with the MATH dataset","author":"Hendrycks","year":"2021"},{"key":"10.1016\/j.neucom.2026.134332_bib0200","author":"Zhou"},{"key":"10.1016\/j.neucom.2026.134332_bib0205","series-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), May 22-27, 2022","first-page":"3214","article-title":"TruthfulQA: measuring how models MIMIC human falsehoods","author":"Lin","year":"2022"},{"issue":"5","key":"10.1016\/j.neucom.2026.134332_bib0210","first-page":"1","article-title":"Beyond the imitation game: quantifying and extrapolating the capabilities of language models","volume":"2023","author":"Srivastava","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134332_bib0215","series-title":"Findings of the Association for Computational Linguistics: ACL 2023, July 9-14, 2023","first-page":"13003","article-title":"Challenging BIG-bench tasks and whether chain-of-thought can solve them","author":"Suzgun","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0220","series-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems Track on Datasets and Benchmarks, 36, December 10-16, 2023, New Orleans, LA, USA","article-title":"Judging LLM-as-a-Judge with MT-bench and chatbot arena","author":"Zheng","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0225","series-title":"Advances in Neural Information Processing Systems, December 10-16, 2023, New Orleans, LA, USA","first-page":"21702","article-title":"LLM-pruner: on the structural pruning of large language models","author":"Ma","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0230","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"Sheared LLaMA: accelerating language model pre-training via structured pruning","author":"Xia","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0235","series-title":"Proceedings of the 38th AAAI Conference on Artificial Intelligence, February 20\u201427, 2024","article-title":"Fluctuation-based adaptive structured pruning for large language models","author":"An","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0240","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"SliceGPT: compress large language models by deleting rows and columns","author":"Ashkboos","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0245","series-title":"Proceedings of the 38th Annual Conference on Neural Information Processing Systems, December 10-15, 2024, Vancouver, BC, Canada","article-title":"DISP-LLM: dimension-independent structural pruning for large language models","author":"Gao","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0250","series-title":"Proceedings of the 42nd International Conference on Machine Learning, July 13-19, 2025, Vancouver, BC, Canada","article-title":"SlimLLM: accurate structured pruning for large language models","author":"Guo","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0255","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2025, November 4-9, 2025","first-page":"3685","article-title":"PIP: perturbation-based iterative pruning for large language models","author":"Cao","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0260","series-title":"Findings of the Association for Computational Linguistics: ACL 2025, July 27\u2013August 1, 2025","article-title":"BlockPruner: fine-grained pruning for large language models","author":"Zhong","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0265","series-title":"Proceedings of the 40th International Conference on Machine Learning, July 23-29, 2023","article-title":"SparseGPT: massive language models can be accurately pruned in one-shot","author":"Frantar","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0270","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"A simple and effective pruning approach for large language models","author":"Sun","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0275","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"Outlier weighed layerwise sparsity (OWL): a missing secret sauce for pruning LLMs to high sparsity","author":"Yin","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0280","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"Dynamic sparse no training: training-free fine-tuning for sparse LLMs","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0285","series-title":"International Conference on Acoustics, Speech and Signal Processing, April 14-19, 2024","article-title":"One-shot sensitivity-aware mixed sparsity pruning for large language models","author":"Shao","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0290","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"Pruner-zero: evolving symbolic pruning metric from scratch for large language models","author":"Dong","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0295","series-title":"Findings of the Association for Computational Linguistics: ACL 2025, July 27\u2013August 1, 2025","first-page":"4321","article-title":"Wanda++: pruning large language models via regional gradients","author":"Yang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0300","series-title":"International Conference on Acoustics, Speech and Signal Processing, April 6-11, 2025","article-title":"LEP: leveraging local entropy pruning for sparsity in large language models","author":"Chen","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0305","series-title":"Proceedings of the 39th AAAI Conference on Artificial Intelligence, February 25\u2013March 4, 2025","first-page":"24167","article-title":"Pruning large language models with semi-structural adaptive sparse training","author":"Huang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0310","series-title":"2026 IEEE International Conference on Acoustics, Speech and Signal Processing, May 4-8, 2026, Barcelona, Spain","article-title":"Sparsity induction for accurate post-training pruning of large language models","author":"Jiang","year":"2026"},{"key":"10.1016\/j.neucom.2026.134332_bib0315","series-title":"The 14th International Conference on Learning Representations, April 23-27, 2026, Rio de Janeiro, Brazil","article-title":"Learning semi-structured sparsity for LLMs via shared and context-aware hypernetwork","author":"Sun","year":"2026"},{"key":"10.1016\/j.neucom.2026.134332_bib0320","series-title":"Proceedings of the 3rd International Conference on Neural Information Processing Systems, 2, November 27-30, 1989, Denver, CO, USA","first-page":"598","article-title":"Optimal brain damage","author":"LeCun","year":"1989"},{"key":"10.1016\/j.neucom.2026.134332_bib0325","series-title":"Proceedings of the 7th International Conference on Neural Information Processing Systems, November 30\u2013December 3, 1992, Denver, CO, USA","first-page":"263","article-title":"Second order derivatives for network pruning: optimal brain surgeon","author":"Hassibi","year":"1992"},{"key":"10.1016\/j.neucom.2026.134332_bib0330","series-title":"Proceedings of the 5th International Conference on Learning Representations, April 24-26, 2017","article-title":"Pruning filters for efficient ConvNets","author":"Li","year":"2017"},{"key":"10.1016\/j.neucom.2026.134332_bib0335","series-title":"Proceedings of the 29th International Conference on Neural Information Processing Systems, 28, December 7-12, 2015, Montreal, Quebec, Canada","first-page":"1135","article-title":"Learning both weights and connections for efficient neural network","author":"Han","year":"2015"},{"key":"10.1016\/j.neucom.2026.134332_bib0340","series-title":"Proceedings of the 7th International Conference on Learning Representations, May 6-9, 2019","article-title":"The lottery ticket hypothesis: finding sparse, trainable neural networks","author":"Frankle","year":"2019"},{"key":"10.1016\/j.neucom.2026.134332_bib0345","series-title":"Findings of the Association for Computational Linguistics: ACL 2024, August 11-16, 2024","first-page":"467","article-title":"LLM-QAT: data-free quantization aware training for large language models","author":"Liu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0350","series-title":"Proceeding of the 38th Annual Conference on Neural Information Processing Systems, December 10-15, 2024, Vancouver, BC, Canada","article-title":"OneBit: towards extremely low-bit large language models","author":"Xu","year":"2024"},{"issue":"125","key":"10.1016\/j.neucom.2026.134332_bib0355","first-page":"1","article-title":"BitNet: 1-bit pre-training for large language models","volume":"26","author":"Wang","year":"2025","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134332_bib0360","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), July 27\u2013August 1, 2025","article-title":"EfficientQAT: efficient quantization-aware training for large language models","author":"Chen","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0365","series-title":"Proceedings of the 36th Annual Conference on Neural Information Processing Systems, vOL.35, November 28\u2013December 9, 2022, New Orleans, Louisiana, USA","first-page":"30318","article-title":"GPT3.Int8: 8-bit sformers at scale","author":"Dettmers","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0370","series-title":"Proceedings of the 36th Annual Conference on Neural Information Processing Systems, Vol.35, November 28\u2013December 9, 2022, New Orleans, Louisiana, USA","first-page":"27168","article-title":"ZeroQuant: efficient and affordable post-training quantization for large-scale transformers","author":"Yao","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0375","author":"Wu"},{"key":"10.1016\/j.neucom.2026.134332_bib0380","series-title":"Proceedings of the 36th Annual Conference on Neural Information Processing Systems, December 10-16, 2023, New Orleans, LA, USA","article-title":"Quantizable transformers: removing outliers by helping attention heads do nothing","author":"Bondarenko","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0385","series-title":"Proceedings of the 11th International Conference on Learning Representations, May 1-5, 2023","article-title":"OPTQ: accurate quantization for generative pre-trained transformers","author":"Frantar","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0390","series-title":"Proceedings of the 37th Annual Conference on Neural Information Processing Systems, December 10-16, 2023, New Orleans, Louisiana, USA","article-title":"QuIP: 2-bit quantization of large language models with guarantees","author":"Chee","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0395","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"QuIP#: even better LLM quantization with hadamard incoherence and lattice codebooks","author":"Tseng","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0400","series-title":"Proceedings of the 38th Annual Conference on Artificial Intelligence, February 20-27, 2024","first-page":"13355","article-title":"OWQ: outlier-aware weight quantization for efficient fine-tuning and inference of large language models","author":"Lee","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0405","series-title":"Proceedings of the 7th Annual Conference on Machine Learning and Systems, 6, May 13-16, 2024, Santa Clara, CA, USA","first-page":"87","article-title":"AWQ: activation-aware weight quantization for on-device LLM compression and acceleration","author":"Lin","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0410","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"SqueezeLLM: dense-and-sparse quantization","author":"Kim","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0415","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"SpQR: a sparse-quantized representation for near-lossless LLM weight compression","author":"Dettmers","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0420","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"LUT-GEMM: quantized matrix multiplication based on LUTs for efficient inference in large-scale generative language models","author":"Park","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0425","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"OmniQuant: omnidirectionally calibrated quantization for large language models","author":"Shao","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0430","series-title":"Proceedings of the 7th Annual Conference on Machine Learning and Systems, 6, May 13-16, 2024, Santa Clara, CA, USA","first-page":"196","article-title":"Atom: low-bit quantization for efficient and accurate LLM serving","author":"Zhao","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0435","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"Extreme compression of large language models via additive quantization","author":"Egiazarian","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0440","series-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems, December 10-15, 2024, Vancouver, BC, Canada","article-title":"DuQuant: distributing outliers via dual transformation makes stronger quantized LLMs","author":"Lin","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0445","series-title":"Proceedings of the 39th AAAI Conference on Artificial Intelligence, February 25\u2013March 4, 2025","first-page":"22299","article-title":"ABQ-LLM: arbitrary-bit quantized inference acceleration for large language models","author":"Zeng","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0450","series-title":"Proceedings of the 2024 USENIX Conference on USENIX Annual Technical Conference, July 10-12, 2024, Santa Clara, CA, USA","article-title":"Quant-LLM: accelerating the serving of large language models via FP6-centric algorithm-system co-design on modern GPUs","author":"Xia","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0455","author":"Luo"},{"key":"10.1016\/j.neucom.2026.134332_bib0460","author":"Meng"},{"key":"10.1016\/j.neucom.2026.134332_bib0465","series-title":"Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems, March 22-26, 2026","article-title":"M2XFP: a metadata-augmented microscaling data format for efficient low-bit quantization","author":"Hu","year":"2026"},{"key":"10.1016\/j.neucom.2026.134332_bib0470","author":"Zhang"},{"key":"10.1016\/j.neucom.2026.134332_bib0475","unstructured":"C. Bao, X. Yan, Z. Li, G. Qin, G. Yu, Y. Zhang, SOAR: scale optimization for accurate reconstruction in NVFP4 quantization, 2026,. https:\/\/arxiv.org\/abs\/2605.12245 (Accessed: 2 June 2026)."},{"key":"10.1016\/j.neucom.2026.134332_bib0480","unstructured":"Z. Xu, X. Hu, D. Yang, TORQ: two-level orthogonal rotation for MXFP4 quantization, 2026,. https:\/\/arxiv.org\/abs\/2605.19561 (Accessed: 2 June 2026)."},{"issue":"6","key":"10.1016\/j.neucom.2026.134332_bib0485","doi-asserted-by":"crossref","first-page":"2325","DOI":"10.1109\/18.720541","article-title":"Quantization","volume":"44","author":"Gray","year":"1998","journal-title":"IEEE Trans. Inf. Theory"},{"key":"10.1016\/j.neucom.2026.134332_bib0490","series-title":"Advances in Neural Information Processing Systems, 32, December 8-14, 2019","article-title":"Post training 4-bit quantization of convolutional networks for rapid-deployment","author":"Banner","year":"2019"},{"key":"10.1016\/j.neucom.2026.134332_bib0495","series-title":"Proceedings of the 4th International Conference on Learning Representations, May 2-4, 2016, San Juan, Puerto Rico","article-title":"Deep compression: compressing deep neural network with pruning, trained quantization and huffman coding","author":"Han","year":"2016"},{"key":"10.1016\/j.neucom.2026.134332_bib0500","first-page":"1","article-title":"Quantized neural networks: training neural networks with low precision weights and activations","volume":"18","author":"Hubara","year":"2017","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134332_bib0505","series-title":"Proceedings of the 12th International Conference on Knowledge Discovery and Data Mining, August 20-23, 2006","first-page":"535","article-title":"Model compression","author":"Bucila","year":"2006"},{"key":"10.1016\/j.neucom.2026.134332_bib0510","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"MiniLLM: knowledge distillation of large language models","author":"Gu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0515","series-title":"Proceedings of the 40th International Conference on Machine Learning, Vol.202, Proceedings of Machine Learning Research, July 23-29, 2023","first-page":"10421","article-title":"Specializing smaller language models towards multi-step reasoning","author":"Fu","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0520","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, November 12-16, 2024","first-page":"18164","article-title":"Dual-space knowledge distillation for large language models","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0525","series-title":"Proceedings of the 39th AAAI Conference on Artificial Intelligence, February 25\u2013March 4, 2025","first-page":"23724","article-title":"Multi-level optimal transport for universal cross-tokenizer knowledge distillation on language models","author":"Cui","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0530","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), July 27\u2013August 1, 2025","first-page":"22504","article-title":"Towards the law of capacity gap in distilling language models","author":"Zhang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0535","article-title":"Emergent abilities of large language models","volume":"2022","author":"Wei","year":"2022","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134332_bib0540","series-title":"Proceedings of the 36th Annual Conference on Neural Information Processing Systems, 35, November 28\u2013December 9, 2022, New Orleans, Louisiana, USA","first-page":"22199","article-title":"Large language models are zero-shot reasoners","author":"Kojima","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0545","series-title":"Findings of the Association for Computational Linguistics: ACL 2023, July 9-14, 2023","first-page":"8003","article-title":"Distilling step-by-step! Outperforming larger language models with less training data and smaller model sizes","author":"Hsieh","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0550","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), July 9-14, 2023","article-title":"Large language models are reasoning teachers","author":"Ho","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0555","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), July 9-14, 2023","article-title":"SCOTT: self-consistent chain-of-thought distillation","author":"Wang","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0560","series-title":"Proceedings of the 38th Annual Conference on Artificial Intelligence, February 20-27, 2024","first-page":"18591","article-title":"Turning dust into gold: distilling complex reasoning capabilities from LLMs by leveraging negative data","author":"Li","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0565","series-title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), June 16\u201421","article-title":"PaD: program-aided distillation can teach small models reasoning better than chain-of-thought fine-tuning","author":"Zhu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0570","series-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, December 6\u201410, 2023","article-title":"Lion: adversarial distillation of proprietary large language models","author":"Jiang","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0575","series-title":"Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers), March 17-22, 2024","first-page":"944","article-title":"LaMini-LM: a diverse herd of distilled models from large-scale instructions","author":"Wu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0580","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), July 9-14, 2023","article-title":"DISCO: distilling counterfactuals with large language models","author":"Chen","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0585","series-title":"Proceedings of the 20th International Conference on International Conference on Machine Learning, August 21\u201324, 2003","first-page":"720","article-title":"Weighted low-rank approximations","author":"Srebro","year":"2003"},{"key":"10.1016\/j.neucom.2026.134332_bib0590","series-title":"Proceedings of the 13th International Conference on Learning Representations, April 24-28, 2025","article-title":"SVD-LLM: truncation-aware singular value decomposition for large language model compression","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0595","series-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), April 29\u2014May 4, 2025","first-page":"4287","article-title":"SVD-LLM v2: optimizing singular value truncation for large language model compression","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0600","series-title":"Proceedings of the 13th International Conference on Learning Representations, April 24-28, 2025","article-title":"Dobi-SVD: differentiable SVD for LLM compression and some new perspectives","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0605","series-title":"Proceedings of the 12th International Conference on Learning Representations, 2024, May 7-11, 2024","first-page":"17632","article-title":"The truth is in there: improving reasoning in language models with layer-selective rank reduction","author":"Sharma","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0610","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2024, November 12-16, 2024","first-page":"4152","article-title":"Adaptive feature-based low-rank compression of large language models via Bayesian optimization","author":"Ji","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0615","series-title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), June 16\u201421, 2024","article-title":"Adaptive rank selections for low-rank approximation of language models","author":"Gao","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0620","series-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems, 37, December 10-15, 2024","first-page":"17489","article-title":"ESPACE: dimensionality reduction of activations for model compression","author":"Sakr","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0625","series-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, November 4-9, 2025","first-page":"14945","article-title":"FLRC: fine-grained low-rank compressor for efficient LLM inference","author":"Lu","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0630","series-title":"Proceedings of the 13th International Conference on Learning Representations, April 24-28, 2025","article-title":"MoDeGPT: modular decomposition for large language model compression","author":"Lin","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0635","series-title":"Proceedings of the 10th International Conference on Learning Representations, April 25-29, 2022","article-title":"Finetuned language models are zero-shot learners","author":"Wei","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0640","series-title":"Proceedings of the 40th International Conference on Machine Learning, vol.202 of Proceedings of Machine Learning Research, July 23-29, 2023","first-page":"22631","article-title":"The flan collection: designing data and methods for effective instruction tuning","author":"Longpre","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0645","first-page":"70:1","article-title":"Scaling instruction-finetuned language models","volume":"25","author":"Chung","year":"2024","journal-title":"J. Mach. Learn. Res."},{"issue":"6","key":"10.1016\/j.neucom.2026.134332_bib0650","first-page":"7","article-title":"Alpaca: a strong, replicable instruction-following model","volume":"3","author":"Taori","year":"2023","journal-title":"Stanf. Cent. Res. Found. Models"},{"key":"10.1016\/j.neucom.2026.134332_bib0655","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), August 11-16, 2024","first-page":"8187","article-title":"Full parameter fine-tuning for large language models with limited resources","author":"Lv","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0660","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, November 12-16, 2024","first-page":"18266","article-title":"HiFT: a hierarchical full parameter fine-tuning strategy","author":"Liu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0665","series-title":"Proceedings of the 36th Annual Conference on Neural Information Processing Systems, November 28\u2013December 9, 2022, New Orleans, Louisiana, USA","article-title":"Few-shot parameter-efficient fine-tuning is better and cheaper than in-context learning","author":"Liu","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0670","series-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), August 1-6, 2021","first-page":"4884","article-title":"Parameter-efficient transfer learning with diff pruning","author":"Guo","year":"2021"},{"key":"10.1016\/j.neucom.2026.134332_bib0675","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"Random masking finds winning tickets for parameter efficient fine-tuning","author":"Xu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0680","doi-asserted-by":"crossref","first-page":"1767","DOI":"10.1162\/TACL.a.59","article-title":"Step-by-step unmasking for parameter-efficient fine-tuning of large language models","volume":"13","author":"Agarwal","year":"2025","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"10.1016\/j.neucom.2026.134332_bib0685","series-title":"Proceedings of the 36th International Conference on Machine Learning, Vol.97 of Proceedings of Machine Learning Research, June 9\u201315 2019","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","author":"Houlsby","year":"2019"},{"key":"10.1016\/j.neucom.2026.134332_bib0690","series-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, October 20-23, 2020","first-page":"46","article-title":"AdapterHub: a framework for adapting transformers","author":"Pfeiffer","year":"2020"},{"key":"10.1016\/j.neucom.2026.134332_bib0695","series-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, December 6\u201410, 2023","first-page":"5254","article-title":"LLM-adapters: an adapter family for parameter-efficient fine-tuning of large language models","author":"Hu","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0700","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"LLaMA-Adapter: efficient fine-tuning of large language models with zero-initialized attention","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0705","series-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), August 1-6, 2021","first-page":"4582","article-title":"Prefix-tuning: optimizing continuous prompts for generation","author":"Li","year":"2021"},{"key":"10.1016\/j.neucom.2026.134332_bib0710","series-title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, November 7-11, 2021","first-page":"3045","article-title":"The power of scale for parameter-efficient prompt tuning","author":"Lester","year":"2021"},{"key":"10.1016\/j.neucom.2026.134332_bib0715","series-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers), May 22-27, 2022","first-page":"61","article-title":"P-Tuning: prompt tuning can be comparable to fine-tuning across scales and tasks","author":"Liu","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0720","series-title":"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, December 7\u201411, 2022","first-page":"11033","article-title":"XPrompt: exploring the extreme of prompt tuning","author":"Ma","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0725","author":"Wang"},{"key":"10.1016\/j.neucom.2026.134332_bib0730","series-title":"Proceedings of the 36th Annual Conference on Neural Information Processing Systems, Vol.35, November 28\u2013December 9, 2022","first-page":"12991","article-title":"LST: ladder side-tuning for parameter and memory efficient transfer learning","author":"Sung","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0735","series-title":"Proceedings of the 35th International Conference on Machine Learning, Vol.80, Proceedings of Machine Learning Research July 10-15, 2018","first-page":"254","article-title":"Stronger generalization bounds for deep nets via a compression approach","author":"Arora","year":"2018"},{"key":"10.1016\/j.neucom.2026.134332_bib0740","series-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), August 1-6, 2021","first-page":"7319","article-title":"Intrinsic dimensionality explains the effectiveness of language model fine-tuning","author":"Aghajanyan","year":"2021"},{"key":"10.1016\/j.neucom.2026.134332_bib0745","series-title":"Proceedings of the 10th International Conference on Learning Representations, April 25-29, 2022","article-title":"LoRA: low-rank adaptation of large language models","author":"Hu","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0750","series-title":"Proceedings of the 11th International Conference on Learning Representations, May 1-5, 2023","article-title":"Adaptive budget allocation for parameter-efficient fine-tuning","author":"Zhang","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0755","series-title":"Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics, May 2-6, 2023","first-page":"3274","article-title":"DyLoRA: parameter-efficient tuning of pre-trained models using dynamic search-free low-rank adaptation","author":"Valipour","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0760","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"DoRA: weight-decomposed low-rank adaptation","author":"Liu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0765","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"LoRA+: efficient low rank adaptation of large models","author":"Hayou","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0770","series-title":"Proceedings of the 2025 IEEE International Conference on Acoustics, Speech and Signal Processing, April 6-11, 2025, Hyderabad, India","first-page":"1","article-title":"RoRA: efficient fine-tuning of LLM with reliability optimization for rank adaptation","author":"Liu","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0775","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"GaLore: memory-efficient LLM training by gradient low-rank projection","author":"Zhao","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0780","series-title":"Proceedings of the 13th International Conference on Learning Representations, April 24-28, 2025","article-title":"LoRA-Pro: are low-rank adapters properly optimized?","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0785","series-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems, 37, December 10-15, 2024","first-page":"121038","article-title":"PiSSA: principal singular values and singular vectors adaptation of large language models","author":"Meng","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0790","series-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), April 29\u2014May 4, 2025","first-page":"4823","article-title":"MiLoRA: harnessing minor singular components for parameter-efficient LLM finetuning","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0795","author":"He"},{"key":"10.1016\/j.neucom.2026.134332_bib0800","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, November 12-16, 2024","first-page":"7880","article-title":"Mixture-of-subspaces in low-rank adaptation","author":"Wu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0805","series-title":"Proceedings of the 13th International Conference on Learning Representations, 2025, April 24-28, 2025","first-page":"29614","article-title":"HiRA: parameter-efficient hadamard high-rank adaptation for large language models","author":"Huang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0810","author":"Li"},{"key":"10.1016\/j.neucom.2026.134332_bib0815","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), July 27\u2013August 1, 2025","first-page":"14643","article-title":"Flexora: flexible low-rank adaptation for large language models","author":"Wei","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0820","series-title":"Proceedings of the 31st Conference on Neural Information Processing Systems, 30, December 4-9, 2017","article-title":"Deep reinforcement learning from human preferences","author":"Christiano","year":"2017"},{"key":"10.1016\/j.neucom.2026.134332_bib0825","series-title":"Proceedings of the 36th Annual Conference on Neural Information Processing Systems, Vol. 35, November 28\u2013December 9, 2022","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","author":"Ouyang","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0830","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"RLAIF vs. RLHF: scaling reinforcement learning from human feedback with AI feedback","author":"Lee","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0835","author":"Schulman"},{"key":"10.1016\/j.neucom.2026.134332_bib0840","author":"Bai"},{"key":"10.1016\/j.neucom.2026.134332_bib0845","series-title":"Proceedings of the 36th Annual Conference on Neural Information Processing Systems, Vol. 35, November 28\u2013December 9, 2022","first-page":"38176","article-title":"Fine-tuning language models to find agreement among humans with diverse preferences","author":"Bakker","year":"2022"},{"key":"10.1016\/j.neucom.2026.134332_bib0850","series-title":"Proceedings of the 37th Annual Conference on Neural Information Processing Systems, 36, December 10-16, 2023","first-page":"53728","article-title":"Direct preference optimization: your language model is secretly a reward model","author":"Rafailov","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0855","series-title":"Proceeding of the 27th International Conference on Artificial Intelligence and Statistics, Vol.238 of Proceedings of Machine Learning Research, May 2-4, 2024","first-page":"4447","article-title":"A general theoretical paradigm to understand learning from human preferences","author":"Azar","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0860","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"Model alignment as prospect theoretic optimization","author":"Ethayarajh","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0865","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, November 12-16, 2024","first-page":"11170","article-title":"ORPO: monolithic preference optimization without reference model","author":"Hong","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0870","series-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems, 37, December 10-15, 2024","first-page":"124198","article-title":"SimPO: simple preference optimization with a reference-free reward","author":"Meng","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0875","author":"G\u00fcl\u00e7ehre"},{"key":"10.1016\/j.neucom.2026.134332_bib0880","series-title":"Proceedings of the 37th Annual Conference on Neural Information Processing Systems, December 10-16, 2023, New Orleans, Louisiana, USA","article-title":"RRHF: rank responses to align language models with human feedback","author":"Yuan","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0885","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"Nash learning from human feedback","author":"Munos","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0890","author":"Bai"},{"key":"10.1016\/j.neucom.2026.134332_bib0895","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"Self-rewarding language models","author":"Yuan","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0900","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"Self-play fine-tuning converts weak language models to strong language models","author":"Chen","year":"2024"},{"issue":"8081","key":"10.1016\/j.neucom.2026.134332_bib0905","doi-asserted-by":"crossref","first-page":"633","DOI":"10.1038\/s41586-025-09422-z","article-title":"DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning","volume":"645","author":"Guo","year":"2025","journal-title":"Nature"},{"key":"10.1016\/j.neucom.2026.134332_bib0910","author":"Shao"},{"key":"10.1016\/j.neucom.2026.134332_bib0915","series-title":"37th Conference on Neural Information Processing Systems, December 10-16, 2023, New Orleans, LA, USA","article-title":"Fine-tuning language models with just forward passes","author":"Malladi","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0920","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, November 12-16, 2024, Miami, Florida, USA","article-title":"AdaZeta: adaptive zeroth-order tensor-train adaption for memory-efficient large language models fine-tuning","author":"Yang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0925","series-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, November 4-9, 2025, Suzhou, China","article-title":"HELENE: Hessian layer-wise clipping and gradient annealing for accelerating fine-tuning LLM with zeroth-order optimization","author":"Zhao","year":"2025"},{"issue":"2","key":"10.1016\/j.neucom.2026.134332_bib0930","doi-asserted-by":"crossref","first-page":"24","DOI":"10.1145\/3682068","article-title":"Differentially private low-rank adaptation of large language model using federated learning","volume":"16","author":"Liu","year":"2025","journal-title":"ACM Trans. Manag. Inf. Syst."},{"key":"10.1016\/j.neucom.2026.134332_bib0935","series-title":"Findings of the Association for Computational Linguistics: ACL 2025, July 27\u2013August 1, 2025, Vienna, Austria","article-title":"Communication-efficient and tensorized federated fine-tuning of large language models","author":"Ghiasvand","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0940","series-title":"ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), April 14-19, 2024, Seoul, Republic of Korea","article-title":"Towards building the federatedgpt: federated instruction tuning","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0945","series-title":"Findings of the Association for Computational Linguistics: ACL 2024, August 11-16, 2024","first-page":"3013","article-title":"LoRAPrune: structured pruning meets low-rank parameter-efficient fine-tuning","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0950","series-title":"Proceedings of the 30th Asia and South Pacific Design Automation Conference, January 20-23, 2025","first-page":"36","article-title":"Learning to prune and low-rank adaptation for compact language model deployment","author":"Ali","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0955","author":"Lu"},{"key":"10.1016\/j.neucom.2026.134332_bib0960","series-title":"Proceedings of the 39th Annual Conference on Neural Information Processing Systems, December 2-7, 2025, San Diego, CA, USA","article-title":"Simultaneous fine-tuning and pruning of LLMs","author":"Reinecke","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0965","series-title":"Proceedings of the 39th AAAI Conference on Artificial Intelligence, February 25\u2013March 4, 2025","article-title":"PAT: pruning-aware tuning for large language models","author":"Liu","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib0970","series-title":"Proceedings of the 31st International Conference on Computational Linguistics, January 19\u201324 2025","first-page":"5530","article-title":"LoRA-drop: efficient LoRA parameter pruning based on output evaluation","author":"Zhou","year":"2025"},{"issue":"7","key":"10.1016\/j.neucom.2026.134332_bib0975","doi-asserted-by":"crossref","first-page":"14882","DOI":"10.1109\/JIOT.2026.3654102","article-title":"Adaptive pruning for large language models with structural importance awareness","volume":"13","author":"Zheng","year":"2026","journal-title":"IEEE Internet Things J."},{"key":"10.1016\/j.neucom.2026.134332_bib0980","author":"Zhao"},{"key":"10.1016\/j.neucom.2026.134332_bib0985","series-title":"Proceedings of the 37th Annual Conference on Neural Information Processing Systems, 36, December 10-16, 2023","first-page":"10088","article-title":"QLoRA: efficient finetuning of quantized LLMs","author":"Dettmers","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib0990","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"QA-LoRA: quantization-aware low-rank adaptation of large language models","author":"Xu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib0995","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing: Industry Track, November 12-16, 2024","first-page":"712","article-title":"QDyLoRA: quantized dynamic low-rank adaptation for efficient large language model tuning","author":"Rajabzadeh","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1000","series-title":"Proceedings of the 41st International Conference on Machine Learning, July 21-27, 2024","article-title":"Accurate LoRA-Finetuning quantization of LLMs via information retention","author":"Qin","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1005","series-title":"Proceedings of the 12th International Conference on Learning Representations, May 7-11, 2024","article-title":"LoftQ: LoRA-Fine-Tuning-aware quantization for large language models","author":"Li","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1010","series-title":"Proceedings of the 42nd International Conference on Machine Learning, July 13-19, 2025, Vancouver, BC, Canada","article-title":"LowRA: accurate and efficient LoRA fine-tuning of LLMs under 2 bits","author":"Zhou","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1015","series-title":"Proceedings of the 37th Annual Conference on Neural Information Processing Systems, December 10-16, 2023, New Orleans, Louisiana, USA","article-title":"Memory-efficient fine-tuning of compressed large language models via sub-4-bit integer quantization","author":"Kim","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib1020","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2024, November 12-16, 2024","first-page":"13823","article-title":"QEFT: quantization for efficient fine-tuning of LLMs","author":"Lee","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1025","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), August 11-16, 2024","first-page":"1","article-title":"Quantized side tuning: fast and memory-efficient tuning of quantized large language models","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1030","series-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, 4\u20139 November 2025","first-page":"5341","article-title":"QuZO: quantized zeroth-order fine-tuning for large language models","author":"Zhou","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1035","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), July 27\u2013August 1, 2025","first-page":"2002","article-title":"L4Q: parameter efficient quantization-aware fine-tuning on large language models","author":"Jeon","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1040","series-title":"Proceedings of the 42nd International Conference on Machine Learning, 13-19 July 2025, Vancouver, BC, Canada","article-title":"RoSTE: an efficient quantization-aware supervised fine-tuning approach for large language models","author":"Wei","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1045","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), August 11-16, 2024","first-page":"11346","article-title":"Improving conversational abilities of quantized large language models via direct preference alignment","author":"Lee","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1050","author":"Li"},{"key":"10.1016\/j.neucom.2026.134332_bib1055","author":"Gu"},{"key":"10.1016\/j.neucom.2026.134332_bib1060","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2024, November 12-16, 2024","first-page":"6266","article-title":"PromptKD: distilling student-friendly knowledge for generative language models via prompt tuning","author":"Kim","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1065","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, November 12-16, 2024","first-page":"17643","article-title":"Mentor-KD: making small language models better multi-step reasoners","author":"Lee","year":"2024"},{"issue":"9","key":"10.1016\/j.neucom.2026.134332_bib1070","doi-asserted-by":"crossref","first-page":"4835","DOI":"10.1109\/TKDE.2024.3376453","article-title":"PanDa: prompt transfer meets knowledge distillation for efficient model adaptation","volume":"36","author":"Zhong","year":"2024","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"10.1016\/j.neucom.2026.134332_bib1075","author":"Tunstall"},{"key":"10.1016\/j.neucom.2026.134332_bib1080","series-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 4: Student Research Workshop)","first-page":"448","article-title":"Streamlining LLMs: adaptive knowledge distillation for tailored language models","author":"Saxena","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1085","series-title":"Proceedings of the 18th International Conference on Web Search and Data Mining, March 10-14, 2025","first-page":"251","article-title":"Beyond answers: transferring reasoning capabilities to smaller LLMs using multi-teacher knowledge distillation","author":"Tian","year":"2025"},{"issue":"40","key":"10.1016\/j.neucom.2026.134332_bib1090","first-page":"34151","article-title":"RLKD: distilling LLMs\u2019 reasoning via reinforcement learning","volume":"40","author":"Xu","year":"2026","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.neucom.2026.134332_bib1095","series-title":"Proceedings of the 42nd International Conference on Machine Learning, July 13-19, 2025, Vancouver, BC, Canada","article-title":"LIFT the veil for the truth: principal weights emerge after rank reduction for reasoning-focused supervised fine-tuning","author":"Liu","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1100","doi-asserted-by":"crossref","unstructured":"L. Zhang, Z. Lou, Y. Ying, C. Yang, H. Zhou, Efficient fine-tuning of large language models via a low-rank gradient estimator, Appl. Sci. 15 (1) (2025) Article 82.","DOI":"10.3390\/app15010082"},{"key":"10.1016\/j.neucom.2026.134332_bib1105","series-title":"Proceedings of the 13th International Conference on Learning Representations, april 24\u201328, 2025","article-title":"Enhancing zeroth-order fine-tuning for language models with low-rank structures","author":"Chen","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1110","series-title":"Proceedings of the 12th International Conference on Learning Representations, may 7\u201311, 2024","article-title":"LQ-LoRA: low-rank plus quantized matrix decomposition for efficient language model finetuning","author":"Guo","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1115","series-title":"Proceedings of the 41st International Conference on Machine Learning, july 21\u201327, 2024","article-title":"CLAM: unifying finetuning, quantization, and pruning by chaining LLM adapter modules","author":"Velingker","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1120","author":"Sreenivas"},{"key":"10.1016\/j.neucom.2026.134332_bib1125","doi-asserted-by":"crossref","first-page":"1474","DOI":"10.1162\/TACL.a.44","article-title":"Safe pruning LoRA: robust distance-guided pruning for safety alignment in adaptation of LLMs","volume":"13","author":"Ao","year":"2025","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"10.1016\/j.neucom.2026.134332_bib1130","series-title":"The 14th International Conference on Learning Representations, april 23\u201327, 2026, Rio de Janeiro, Brazil","article-title":"QeRL: beyond efficiency - quantization-enhanced reinforcement learning for LLMs","author":"Huang","year":"2026"},{"key":"10.1016\/j.neucom.2026.134332_bib1135","series-title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 3: System Demonstrations), june 16\u201321, 2024, Mexico City, Mexico","article-title":"Newspaper signaling for crisis prediction","author":"Saxena","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1140","series-title":"Proceedings of the 20th European Conference on Computer Systems, March 30\u2013April 3, 2025","article-title":"HybridFlow: a flexible and efficient RLHF framework","author":"Sheng","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1145","series-title":"Proceedings of the 26th Conference on Knowledge Discovery and Data Mining, august 23\u201327, 2020","first-page":"3505","article-title":"DeepSpeed: system optimizations enable training deep learning models with over 100 billion parameters","author":"Rasley","year":"2020"},{"key":"10.1016\/j.neucom.2026.134332_bib1150","author":"Shoeybi"},{"key":"10.1016\/j.neucom.2026.134332_bib1155","series-title":"Proceedings of the 52nd International Conference on Parallel Processing, august 7\u201310, 2023","first-page":"766","article-title":"Colossal-AI: a unified deep learning system for large-scale parallel training","author":"Li","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib1160","series-title":"Proceedings of the 29th Symposium on Operating Systems Principles, october 23\u201326, 2023","first-page":"611","article-title":"Efficient memory management for large language model serving with PagedAttention","author":"Kwon","year":"2023"},{"key":"10.1016\/j.neucom.2026.134332_bib1165","series-title":"The 38th Annual Conference on Neural Information Processing Systems, December 10-15, 2024, Vancouver, Canada","article-title":"SGLang: efficient execution of structured language model programs","author":"Zheng","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1170","series-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, October 20-23, 2020","first-page":"38","article-title":"Transformers: state-of-the-art natural language processing","author":"Wolf","year":"2020"},{"key":"10.1016\/j.neucom.2026.134332_bib1175","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 3: System Demonstrations), August 11-16, 2024","first-page":"400","article-title":"LlamaFactory: unified efficient fine-tuning of 100+ language models","author":"Zheng","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1180","author":"Scao"},{"key":"10.1016\/j.neucom.2026.134332_bib1185","author":"Luo"},{"key":"10.1016\/j.neucom.2026.134332_bib1190","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), August 11-16, 2024, Bangkok, Thailand","article-title":"Mitigating catastrophic forgetting in large language models with self-synthesized rehearsal","author":"Huang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1195","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2024, November 12-16, 2024, Miami, Florida, USA","article-title":"Revisiting catastrophic forgetting in large language model tuning","author":"Li","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1200","series-title":"2024 ACM\/IEEE 51st Annual International Symposium on Computer Architecture (ISCA), June 29\u2013July 3 2024, Buenos Aires, Argentina","article-title":"LLMCompass: enabling efficient hardware design for large language model inference","author":"Zhang","year":"2024"},{"issue":"4","key":"10.1016\/j.neucom.2026.134332_bib1205","doi-asserted-by":"crossref","first-page":"18","DOI":"10.1145\/3744244","article-title":"HAPE: hardware-aware LLM pruning for efficient on-device inference optimization","volume":"30","author":"Zhao","year":"2025","journal-title":"ACM Trans. Des. Autom. Electron. Syst."},{"key":"10.1016\/j.neucom.2026.134332_bib1210","series-title":"2024 IEEE International Symposium on Workload Characterization (IISWC), September 15-17, 2024, Vancouver, BC, Canada","article-title":"Understanding performance implications of LLM inference on CPUs","author":"Na","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1215","first-page":"24","article-title":"Quantization variation: a new perspective on training transformers with low-bit precision","author":"Huang","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134332_bib1220","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1613\/jair.1.18405","article-title":"Incremental learning methodologies for addressing catastrophic forgetting: analysis and experimental evaluation","volume":"83","author":"Serra-Perello","year":"2025","journal-title":"J. Artif. Intell. Res."},{"key":"10.1016\/j.neucom.2026.134332_bib1225","author":"Chen"},{"key":"10.1016\/j.neucom.2026.134332_bib1230","series-title":"AIBC 2025: Proceedings of the 2025 6th International Artificial Intelligence and Blockchain Conference, September 17-19, 2025, Tokyo, Japan","article-title":"Idle consumer GPUs as a complement to enterprise hardware for LLM inference: performance, cost and carbon analysis","author":"Almeida","year":"2025"},{"issue":"2","key":"10.1016\/j.neucom.2026.134332_bib1235","doi-asserted-by":"crossref","first-page":"56","DOI":"10.1145\/3757892.3757900","article-title":"Energy efficient or exhaustive? Benchmarking power consumption of LLM inference engines","volume":"5","author":"Niu","year":"2025","journal-title":"ACM SIGENERGY Energy Inform. Rev."},{"issue":"2","key":"10.1016\/j.neucom.2026.134332_bib1240","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1145\/3788870","article-title":"Sometimes painful but promising: feasibility and trade-offs of on-device language model inference","volume":"25","author":"Abstreiter","year":"2026","journal-title":"ACM Trans. Embed. Comput. Syst."},{"key":"10.1016\/j.neucom.2026.134332_bib1245","series-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems, 37, December 10-15, 2024","first-page":"71768","article-title":"CorDA: context-oriented decomposition adaptation of large language models for task-aware parameter-efficient fine-tuning","author":"Yang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1250","series-title":"Proceedings of the 2025 Annual International ACM SIGIR Conference on Research and Development in Information Retrieval in the Asia Pacific Region, SIGIR-AP 2025, December 7-10, 2025, Xi\u2019an, China","article-title":"ATACompressor: adaptive task-aware compression for efficient long-context processing in LLMs","author":"Li","year":"2025"},{"issue":"8","key":"10.1016\/j.neucom.2026.134332_bib1255","first-page":"1","article-title":"A review on edge large language models: design, execution, and applications","volume":"57","author":"Zheng","year":"2025","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.neucom.2026.134332_bib1260","series-title":"Proceedings of the 57th IEEE\/ACM International Symposium on Microarchitecture, November 2-6, 2024, Austin, Texas, USA","first-page":"1474","article-title":"Cambricon-LLM: a chiplet-based hybrid architecture for on-device inference of 70B LLM","author":"Yu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1265","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102840","article-title":"Federated and edge learning for large language models","volume":"117","author":"Piccialli","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.neucom.2026.134332_bib1270","series-title":"Proceedings of the IEEE International Conference on Computer Communications, May 19-22, 2025","first-page":"1","article-title":"Memory-efficient split federated learning for LLM fine-tuning on heterogeneous mobile devices","author":"Chen","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1275","author":"Xu"},{"key":"10.1016\/j.neucom.2026.134332_bib1280","series-title":"Proceedings of the ACM Web Conference 2026, June 29\u2013July 3, 2026, Dubai, United Arab Emirates","article-title":"Personalized federated fine-tuning for LLMs via data-driven heterogeneous model architectures","author":"Zhang","year":"2026"},{"key":"10.1016\/j.neucom.2026.134332_bib1285","author":"Feng"},{"key":"10.1016\/j.neucom.2026.134332_bib1290","series-title":"ICLR 2025 Workshop on Sparsity in LLMs (SLLM), EXPO, Singapore","article-title":"Brain-inspired sparse training enables transformers and LLMs to perform as fully connected","author":"Zhang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1295","author":"Zhu"},{"key":"10.1016\/j.neucom.2026.134332_bib1300","series-title":"The 13th International Conference on Learning Representations, April 24-28, 2025, EXPO, Singapore","article-title":"DS-LLM: leveraging dynamical systems to enhance both training and inference of large language models","author":"Song","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1305","series-title":"The 39th Annual Conference on Neural Information Processing Systems, December 2-7, 2025, San Diego, USA","article-title":"DynaAct: large language model reasoning with dynamic action spaces","author":"Zhao","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1310","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, November 12-16, 2024, Miami, Florida, USA","article-title":"MAgIC: investigation of large language model powered multi-agent in cognition, adaptability, rationality and collaboration","author":"Xu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1315","series-title":"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024), May 20-25, 2024, Torino, Italia","article-title":"Mixture-of-LoRAs: an efficient multitask tuning method for large language models","author":"Feng","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1320","series-title":"ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), May 4-8, 2026, Barcelona, Spain","article-title":"Let more experts speak: balancing exploration and exploitation in PEFT for mixture-of-experts models","author":"Fang","year":"2026"},{"issue":"7","key":"10.1016\/j.neucom.2026.134332_bib1325","first-page":"3896","article-title":"A survey on mixture of experts in large language models","volume":"37","author":"Cai","year":"2025","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"10.1016\/j.neucom.2026.134332_bib1330","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, November 12-16, 2024","first-page":"784","article-title":"Let the expert stick to his last: expert-specialized fine-tuning for sparse architectural large language models","author":"Wang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134332_bib1335","series-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 3: Industry Track), April 29\u2014May 4, 2025","first-page":"340","article-title":"MoFE: mixture of frozen experts architecture","author":"Seo","year":"2025"},{"key":"10.1016\/j.neucom.2026.134332_bib1340","author":"Sarah"},{"key":"10.1016\/j.neucom.2026.134332_bib1345","series-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems, December 10-15, 2024, Vancouver, BC, Canada","article-title":"Large language model compression with neural architecture search","author":"Sukthanker","year":"2024"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226017303?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226017303?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T04:43:05Z","timestamp":1785904985000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226017303"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":269,"alternative-id":["S0925231226017303"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134332","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Survey on joint compression and fine-tuning of large language models: Methods, toolchains, and open challenges under resource constraints","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134332","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"134332"}}