{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T21:40:59Z","timestamp":1777498859138,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T00:00:00Z","timestamp":1737331200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2314591"],"award-info":[{"award-number":["2314591"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2414603"],"award-info":[{"award-number":["2414603"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2349802"],"award-info":[{"award-number":["2349802"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2342726"],"award-info":[{"award-number":["2342726"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,1,20]]},"DOI":"10.1145\/3658617.3697648","type":"proceedings-article","created":{"date-parts":[[2025,3,4]],"date-time":"2025-03-04T14:23:57Z","timestamp":1741098237000},"page":"36-42","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Learning to Prune and Low-Rank Adaptation for Compact Language Model Deployment"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-5291-0487","authenticated-orcid":false,"given":"Asmer Hamid","family":"Ali","sequence":"first","affiliation":[{"name":"School of Electrical, Computer and Energy Engineering, Arizona State University, Tempe, Arizona, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6823-2700","authenticated-orcid":false,"given":"Fan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Johns Hopkins Univ., Baltimore, MD, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2839-6196","authenticated-orcid":false,"given":"Li","family":"Yang","sequence":"additional","affiliation":[{"name":"University of North Carolina, Charlotte, North Carolina, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7989-6297","authenticated-orcid":false,"given":"Deliang","family":"Fan","sequence":"additional","affiliation":[{"name":"School of Electrical, Computer and Energy Engineering, Arizona State University, Tempe, Arizona, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,3,4]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Lora: Low-rank adaptation of large language models","author":"Hu E. J.","year":"2021","unstructured":"E. J. Hu et al. Lora: Low-rank adaptation of large language models, 2021."},{"key":"e_1_3_2_1_2_1","volume-title":"Gpt-4 technical report","author":"AI","year":"2024","unstructured":"OpenAI et al. Gpt-4 technical report, 2024."},{"key":"e_1_3_2_1_3_1","volume-title":"Gemini: A family of highly capable multimodal models","author":"Team G.","year":"2024","unstructured":"G. Team et al. Gemini: A family of highly capable multimodal models, 2024."},{"key":"e_1_3_2_1_4_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin J.","year":"2019","unstructured":"J. Devlin et al. Bert: Pre-training of deep bidirectional transformers for language understanding, 2019."},{"key":"e_1_3_2_1_5_1","volume-title":"Llama: Open and efficient foundation language models","author":"Touvron H.","year":"2023","unstructured":"H. Touvron et al. Llama: Open and efficient foundation language models, 2023."},{"key":"e_1_3_2_1_6_1","volume-title":"Multilingual machine translation with large language models: Empirical results and analysis","author":"Zhu W.","year":"2023","unstructured":"W. Zhu et al. Multilingual machine translation with large language models: Empirical results and analysis, 2023."},{"key":"e_1_3_2_1_7_1","volume-title":"Bertweet: A pre-trained language model for english tweets","author":"Nguyen D. Q.","year":"2020","unstructured":"D. Q. Nguyen et al. Bertweet: A pre-trained language model for english tweets, 2020."},{"key":"e_1_3_2_1_8_1","volume-title":"Towards efficient generative large language model serving: A survey from algorithms to systems","author":"Miao X.","year":"2023","unstructured":"X. Miao et al. Towards efficient generative large language model serving: A survey from algorithms to systems, 2023."},{"key":"e_1_3_2_1_9_1","volume-title":"A comprehensive overview of large language models","author":"Naveed H.","year":"2024","unstructured":"H. Naveed et al. A comprehensive overview of large language models, 2024."},{"key":"e_1_3_2_1_10_1","volume-title":"Peft: State-of-the-art parameter-efficient fine-tuning methods. https:\/\/github.com\/huggingface\/peft","author":"Mangrulkar S.","year":"2022","unstructured":"S. Mangrulkar et al. Peft: State-of-the-art parameter-efficient fine-tuning methods. https:\/\/github.com\/huggingface\/peft, 2022."},{"key":"e_1_3_2_1_11_1","volume-title":"Scaling down to scale up: A guide to parameter-efficient fine-tuning","author":"Lialin V.","year":"2023","unstructured":"V. Lialin et al. Scaling down to scale up: A guide to parameter-efficient fine-tuning, 2023."},{"key":"e_1_3_2_1_12_1","volume-title":"Chinese-vicuna: A chinese instruction-following llama-based model","author":"Chenghao Fan Z. L.","year":"2023","unstructured":"Z. L. Chenghao Fan et al. Chinese-vicuna: A chinese instruction-following llama-based model. 2023."},{"key":"e_1_3_2_1_13_1","volume-title":"Dora: Weight-decomposed low-rank adaptation","author":"Liu S.-Y.","year":"2024","unstructured":"S.-Y. Liu et al. Dora: Weight-decomposed low-rank adaptation, 2024."},{"key":"e_1_3_2_1_14_1","volume-title":"Sparsegpt: Massive language models can be accurately pruned in one-shot","author":"Frantar E.","year":"2023","unstructured":"E. Frantar et al. Sparsegpt: Massive language models can be accurately pruned in one-shot, 2023."},{"key":"e_1_3_2_1_15_1","volume-title":"Distilbert, a distilled version of bert: smaller, faster, cheaper and lighter","author":"Sanh V.","year":"2020","unstructured":"V. Sanh et al. Distilbert, a distilled version of bert: smaller, faster, cheaper and lighter, 2020."},{"key":"e_1_3_2_1_16_1","volume-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu Y.","year":"2019","unstructured":"Y. Liu et al. Roberta: A robustly optimized bert pretraining approach, 2019."},{"key":"e_1_3_2_1_17_1","volume-title":"Glue: A multi-task benchmark and analysis platform for natural language understanding","author":"Wang A.","year":"2019","unstructured":"A. Wang et al. Glue: A multi-task benchmark and analysis platform for natural language understanding, 2019."},{"key":"e_1_3_2_1_18_1","volume-title":"Llama: Open and efficient foundation language models","author":"Touvron H.","year":"2023","unstructured":"H. Touvron et al. Llama: Open and efficient foundation language models, 2023."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01086"},{"key":"e_1_3_2_1_20_1","volume-title":"Towards efficient visual adaption via structural re-parameterization","author":"Luo G.","year":"2023","unstructured":"G. Luo et al. Towards efficient visual adaption via structural re-parameterization, 2023."},{"key":"e_1_3_2_1_21_1","volume-title":"Towards a unified view of parameter-efficient transfer learning","author":"He J.","year":"2022","unstructured":"J. He et al. Towards a unified view of parameter-efficient transfer learning, 2022."},{"key":"e_1_3_2_1_22_1","volume-title":"Qlora: Efficient finetuning of quantized llms","author":"Dettmers T.","year":"2023","unstructured":"T. Dettmers et al. Qlora: Efficient finetuning of quantized llms, 2023."},{"key":"e_1_3_2_1_23_1","volume-title":"One-for-all: Generalized lora for parameter-efficient fine-tuning","author":"Chavan A.","year":"2023","unstructured":"A. Chavan et al. One-for-all: Generalized lora for parameter-efficient fine-tuning, 2023."},{"key":"e_1_3_2_1_24_1","volume-title":"Longlora: Efficient fine-tuning of long-context large language models","author":"Chen Y.","year":"2024","unstructured":"Y. Chen et al. Longlora: Efficient fine-tuning of long-context large language models, 2024."},{"key":"e_1_3_2_1_25_1","volume-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding","author":"Han S.","year":"2016","unstructured":"S. Han et al. Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding, 2016."},{"key":"e_1_3_2_1_26_1","volume-title":"Lottery tickets in linear models: An analysis of iterative magnitude pruning","author":"Elesedy B.","year":"2021","unstructured":"B. Elesedy et al. Lottery tickets in linear models: An analysis of iterative magnitude pruning, 2021."},{"key":"e_1_3_2_1_27_1","volume-title":"Layer-adaptive sparsity for the magnitude-based pruning","author":"Lee J.","year":"2021","unstructured":"J. Lee et al. Layer-adaptive sparsity for the magnitude-based pruning, 2021."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20222"},{"key":"e_1_3_2_1_29_1","volume-title":"The combinatorial brain surgeon: Pruning weights that cancel one another in neural networks","author":"Yu X.","year":"2022","unstructured":"X. Yu et al. The combinatorial brain surgeon: Pruning weights that cancel one another in neural networks, 2022."},{"key":"e_1_3_2_1_30_1","volume-title":"Advances in Neural Information Processing Systems","author":"LeCun Y.","year":"1989","unstructured":"Y. LeCun et al. Optimal brain damage. In D. Touretzky, editor, Advances in Neural Information Processing Systems, volume 2. Morgan-Kaufmann, 1989."},{"key":"e_1_3_2_1_31_1","first-page":"4860","volume-title":"NIPS'17","author":"Dong X.","year":"2017","unstructured":"X. Dong et al. Learning to prune deep neural networks via layer-wise optimal brain surgeon. NIPS'17, pp. 4860--4874, Red Hook, NY, USA, 2017. Curran Associates Inc."},{"key":"e_1_3_2_1_32_1","volume-title":"The lottery ticket hypothesis for pre-trained bert networks","author":"Chen T.","year":"2020","unstructured":"T. Chen et al. The lottery ticket hypothesis for pre-trained bert networks, 2020."},{"key":"e_1_3_2_1_33_1","volume-title":"Language models are few-shot learners","author":"Brown T. B.","year":"2020","unstructured":"T. B. Brown et al. Language models are few-shot learners, 2020."},{"key":"e_1_3_2_1_34_1","volume-title":"Structured pruning is all you need for pruning cnns at initialization","author":"Cai Y.","year":"2022","unstructured":"Y. Cai et al. Structured pruning is all you need for pruning cnns at initialization, 2022."},{"key":"e_1_3_2_1_35_1","volume-title":"Dsee: Dually sparsity-embedded efficient tuning of pre-trained language models","author":"Chen X.","year":"2023","unstructured":"X. Chen et al. Dsee: Dually sparsity-embedded efficient tuning of pre-trained language models, 2023."},{"key":"e_1_3_2_1_36_1","volume-title":"A simple and effective pruning approach for large language models","author":"Sun M.","year":"2023","unstructured":"M. Sun et al. A simple and effective pruning approach for large language models, 2023."},{"key":"e_1_3_2_1_37_1","volume-title":"Llm-pruner: On the structural pruning of large language models","author":"Ma X.","year":"2023","unstructured":"X. Ma et al. Llm-pruner: On the structural pruning of large language models, 2023."},{"key":"e_1_3_2_1_38_1","volume-title":"Loraprune: Pruning meets low-rank parameter-efficient fine-tuning","author":"Zhang M.","year":"2023","unstructured":"M. Zhang et al. Loraprune: Pruning meets low-rank parameter-efficient fine-tuning, 2023."},{"key":"e_1_3_2_1_39_1","volume-title":"Sheared llama: Accelerating language model pre-training via structured pruning","author":"Xia M.","year":"2024","unstructured":"M. Xia et al. Sheared llama: Accelerating language model pre-training via structured pruning, 2024."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/586"},{"key":"e_1_3_2_1_41_1","volume-title":"Categorical reparameterization with gumbel-softmax","author":"Jang E.","year":"2017","unstructured":"E. Jang et al. Categorical reparameterization with gumbel-softmax, 2017."},{"key":"e_1_3_2_1_42_1","first-page":"2924","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Clark C.","year":"2019","unstructured":"C. Clark et al. Boolq: Exploring the surprising difficulty of natural yes\/no questions. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 2924--2936, Minneapolis, Minnesota, June 2019. Association for Computational Linguistics."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1472"},{"key":"e_1_3_2_1_45_1","volume-title":"Winogrande: An adversarial winograd schema challenge at scale. arXiv preprint arXiv:1907.10641","author":"Sakaguchi K.","year":"2019","unstructured":"K. Sakaguchi et al. Winogrande: An adversarial winograd schema challenge at scale. arXiv preprint arXiv:1907.10641, 2019."},{"key":"e_1_3_2_1_46_1","volume-title":"Think you have solved question answering? try arc, the ai2 reasoning challenge. arXiv preprint arXiv:1803.05457v1","author":"Clark P.","year":"2018","unstructured":"P. Clark et al. Think you have solved question answering? try arc, the ai2 reasoning challenge. arXiv preprint arXiv:1803.05457v1, 2018."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1260"},{"key":"e_1_3_2_1_48_1","volume-title":"Pytorch lightning. https:\/\/github.com\/PyTorchLightning\/pytorch-lightning","author":"Falcon W.","year":"2019","unstructured":"W. Falcon. Pytorch lightning. https:\/\/github.com\/PyTorchLightning\/pytorch-lightning, 2019."},{"key":"e_1_3_2_1_49_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron H.","year":"2023","unstructured":"H. Touvron et al. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971, 2023."},{"key":"e_1_3_2_1_50_1","volume-title":"Lorashear: Efficient large language model structured pruning and knowledge recovery","author":"Chen T.","year":"2023","unstructured":"T. Chen et al. Lorashear: Efficient large language model structured pruning and knowledge recovery, 2023."},{"key":"e_1_3_2_1_51_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Liu Y.","year":"2019","unstructured":"Y. Liu et al. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692, 2019."},{"key":"e_1_3_2_1_52_1","volume-title":"Lora-drop: Efficient lora parameter pruning based on output evaluation","author":"Zhou H.","year":"2024","unstructured":"H. Zhou et al. Lora-drop: Efficient lora parameter pruning based on output evaluation, 2024."},{"key":"e_1_3_2_1_53_1","volume-title":"Dylora: Parameter efficient tuning of pre-trained models using dynamic search-free low-rank adaptation","author":"Valipour M.","year":"2023","unstructured":"M. Valipour et al. Dylora: Parameter efficient tuning of pre-trained models using dynamic search-free low-rank adaptation, 2023."}],"event":{"name":"ASPDAC '25: 30th Asia and South Pacific Design Automation Conference","location":"Tokyo Japan","acronym":"ASPDAC '25","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEICE","IPSJ","IEEE CAS","IEEE CEDA"]},"container-title":["Proceedings of the 30th Asia and South Pacific Design Automation Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658617.3697648","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3658617.3697648","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3658617.3697648","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:49Z","timestamp":1750295869000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658617.3697648"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,20]]},"references-count":53,"alternative-id":["10.1145\/3658617.3697648","10.1145\/3658617"],"URL":"https:\/\/doi.org\/10.1145\/3658617.3697648","relation":{},"subject":[],"published":{"date-parts":[[2025,1,20]]},"assertion":[{"value":"2025-03-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}