{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T23:17:24Z","timestamp":1784330244883,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":102,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T00:00:00Z","timestamp":1743292800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"name":"Hong Kong Research Grants Council","award":["C7004-22G, C6015-23G"],"award-info":[{"award-number":["C7004-22G, C6015-23G"]}]},{"name":"NSFC\/RGC Collaborative Research Scheme","award":["CRS_HKUST601\/24"],"award-info":[{"award-number":["CRS_HKUST601\/24"]}]},{"DOI":"10.13039\/501100006374","name":"Guangzhou Municipal Science and Technology Bureau","doi-asserted-by":"publisher","award":["2024A03J0616"],"award-info":[{"award-number":["2024A03J0616"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Guangzhou Municipality Big Data Intelligence Key Lab","award":["2023A03J0012"],"award-info":[{"award-number":["2023A03J0012"]}]},{"DOI":"10.13039\/501100006374","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62272122"],"award-info":[{"award-number":["62272122"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,3,30]]},"DOI":"10.1145\/3689031.3717481","type":"proceedings-article","created":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T06:25:20Z","timestamp":1742970320000},"page":"243-260","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":18,"title":["SpInfer: Leveraging Low-Level Sparsity for Efficient Large Language Model Inference on GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7492-2069","authenticated-orcid":false,"given":"Ruibo","family":"Fan","sequence":"first","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2478-1512","authenticated-orcid":false,"given":"Xiangrui","family":"Yu","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1952-4544","authenticated-orcid":false,"given":"Peijie","family":"Dong","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4381-0544","authenticated-orcid":false,"given":"Zeyu","family":"Li","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2269-2358","authenticated-orcid":false,"given":"Gu","family":"Gong","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2986-967X","authenticated-orcid":false,"given":"Qiang","family":"Wang","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4585-4152","authenticated-orcid":false,"given":"Wei","family":"Wang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong SAR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9745-4372","authenticated-orcid":false,"given":"Xiaowen","family":"Chu","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), China and The Hong Kong University of Science and Technology, Hong Kong SAR"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,3,30]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Marah Abdin Sam Ade Jacobs Ammar Ahmad Awan Jyoti Aneja Ahmed Awadallah Hany Awadalla Nguyen Bach Amit Bahree Arash Bakhtiari Harkirat Behl et al. Phi-3 technical report: A highly capable language model locally on your phone. arXiv preprint arXiv:2404.14219 2024."},{"key":"e_1_3_2_1_2_1","volume-title":"AMD Instinct MI300 CDNA3 Instruction Set Architecture Reference Guide. Advanced Micro Devices","author":"Devices Advanced Micro","year":"2024","unstructured":"Advanced Micro Devices. AMD Instinct MI300 CDNA3 Instruction Set Architecture Reference Guide. Advanced Micro Devices, Inc., 2024."},{"key":"e_1_3_2_1_3_1","volume-title":"AMD RDNA3 Instruction Set Architecture. Advanced Micro Devices","author":"Devices Advanced Micro","year":"2024","unstructured":"Advanced Micro Devices. AMD RDNA3 Instruction Set Architecture. Advanced Micro Devices, Inc., 2024."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"},{"key":"e_1_3_2_1_5_1","volume-title":"ICLR","author":"Ashkboos Saleh","year":"2024","unstructured":"Saleh Ashkboos, Maximilian L. Croci, Marcelo Gennari do Nascimento, Torsten Hoefler, and James Hensman. SliceGPT: Compress large language models by deleting rows and columns. In ICLR, 2024."},{"key":"e_1_3_2_1_6_1","volume-title":"Quarot: Outlier-free 4-bit inference in rotated llms. arXiv preprint arXiv:2404.00456","author":"Ashkboos Saleh","year":"2024","unstructured":"Saleh Ashkboos, Amirkeivan Mohtashami, Maximilian L Croci, Bo Li, Martin Jaggi, Dan Alistarh, Torsten Hoefler, and James Hensman. Quarot: Outlier-free 4-bit inference in rotated llms. arXiv preprint arXiv:2404.00456, 2024."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3495883"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3559009.3569691"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476182"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3489517.3530508"},{"key":"e_1_3_2_1_11_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Dao Tri","year":"2024","unstructured":"Tri Dao. FlashAttention-2: Faster attention with better parallelism and work partitioning. In International Conference on Learning Representations (ICLR), 2024."},{"key":"e_1_3_2_1_12_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Dao Tri","year":"2022","unstructured":"Tri Dao, Daniel Y. Fu, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. FlashAttention: Fast and memory-efficient exact attention with IO-awareness. In Advances in Neural Information Processing Systems (NeurIPS), 2022."},{"key":"e_1_3_2_1_13_1","volume-title":"Beyond size: How gradients shape pruning decisions in large language models. arXiv preprint arXiv:2311.04902","author":"Das Rocktim Jyoti","year":"2023","unstructured":"Rocktim Jyoti Das, Liqun Ma, and Zhiqiang Shen. Beyond size: How gradients shape pruning decisions in large language models. arXiv preprint arXiv:2311.04902, 2023."},{"key":"e_1_3_2_1_14_1","first-page":"30318","article-title":"8-bit matrix multiplication for transformers at scale","volume":"35","author":"Dettmers Tim","year":"2022","unstructured":"Tim Dettmers, Mike Lewis, Younes Belkada, and Luke Zettlemoyer. Gpt3. int8 (): 8-bit matrix multiplication for transformers at scale. Advances in Neural Information Processing Systems, 35:30318--30332, 2022.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning. PMLR","author":"Dong Peijie","year":"2024","unstructured":"Peijie Dong, Lujun Li, Zhenheng Tang, Xiang Liu, Xinglin Pan, Qiang Wang, and Xiaowen Chu. Pruner-zero: Evolving symbolic pruning metric from scratch for large language models. In Proceedings of the 41st International Conference on Machine Learning. PMLR, 2024. [arXiv: 2406.02924]."},{"key":"e_1_3_2_1_16_1","volume-title":"Bitdistiller: Unleashing the potential of sub-4-bit llms via self-distillation. arXiv preprint arXiv:2402.10631","author":"Du Dayou","year":"2024","unstructured":"Dayou Du, Yijia Zhang, Shijie Cao, Jiaqi Guo, Ting Cao, Xiaowen Chu, and Ningyi Xu. Bitdistiller: Unleashing the potential of sub-4-bit llms via self-distillation. arXiv preprint arXiv:2402.10631, 2024."},{"key":"e_1_3_2_1_17_1","volume-title":"The llama 3 herd of models. arXiv preprint arXiv:2407.21783","author":"Dubey Abhimanyu","year":"2024","unstructured":"Abhimanyu Dubey, Abhinav Jauhri, Abhinav Pandey, Abhishek Kadian, Ahmad Al-Dahle, Aiesha Letman, Akhil Mathur, Alan Schelten, Amy Yang, Angela Fan, et al. The llama 3 herd of models. arXiv preprint arXiv:2407.21783, 2024."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS54959.2023.00057"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651378"},{"key":"e_1_3_2_1_20_1","volume-title":"ICML","author":"Frantar Elias","year":"2023","unstructured":"Elias Frantar and Dan Alistarh. Sparsegpt: Massive language models can be accurately pruned in one-shot. In ICML, 2023."},{"key":"e_1_3_2_1_21_1","volume-title":"Gptq: Accurate post-training quantization for generative pre-trained transformers. arXiv preprint arXiv:2210.17323","author":"Frantar Elias","year":"2022","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. Gptq: Accurate post-training quantization for generative pre-trained transformers. arXiv preprint arXiv:2210.17323, 2022."},{"key":"e_1_3_2_1_22_1","volume-title":"Marlin: Mixed-precision auto-regressive parallel inference on large language models. arXiv preprint arXiv:2408.11743","author":"Frantar Elias","year":"2024","unstructured":"Elias Frantar, Roberto L Castro, Jiale Chen, Torsten Hoefler, and Dan Alistarh. Marlin: Mixed-precision auto-regressive parallel inference on large language models. arXiv preprint arXiv:2408.11743, 2024."},{"key":"e_1_3_2_1_23_1","volume-title":"Megablocks: Efficient sparse training with mixture-of-experts","author":"Gale Trevor","year":"2022","unstructured":"Trevor Gale, Deepak Narayanan, Cliff Young, and Matei Zaharia. Megablocks: Efficient sparse training with mixture-of-experts, 2022."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.5555\/3433701.3433723"},{"key":"e_1_3_2_1_25_1","volume-title":"llama.cpp. https:\/\/github.com\/ggerganov\/llama.cpp","author":"Gerganov Georgi","year":"2023","unstructured":"Georgi Gerganov. llama.cpp. https:\/\/github.com\/ggerganov\/llama.cpp, 2023."},{"key":"e_1_3_2_1_26_1","volume-title":"Block-sparse gpu kernels","author":"Gray Scott","year":"2017","unstructured":"Scott Gray, Alec Radford, and Diederik P Kingma. Block-sparse gpu kernels, 2017."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.5555\/3546258.3546499"},{"key":"e_1_3_2_1_29_1","unstructured":"Connor Holmes Masahiro Tanaka Michael Wyatt Ammar Ahmad Awan Jeff Rasley Samyam Rajbhandari Reza Yazdani Aminabadi Heyang Qin Arash Bakhtiari Lev Kurilenko and Yuxiong He. Deepspeed-fastgen: High-throughput text generation for llms via mii and deepspeed-inference 2024."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3208040.3208062"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3293883.3295712"},{"key":"e_1_3_2_1_32_1","volume-title":"Faster large language model inference on gpus","author":"Hong Ke","year":"2024","unstructured":"Ke Hong, Guohao Dai, Jiaming Xu, Qiuli Mao, Xiuhong Li, Jun Liu, Kangdi Chen, Yuhan Dong, and Yu Wang. Flashdecoding++: Faster large language model inference on gpus, 2024."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00075"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00076"},{"key":"e_1_3_2_1_35_1","volume-title":"Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al. Mixtral of experts. arXiv preprint arXiv:2401.04088","author":"Jiang Albert Q","year":"2024","unstructured":"Albert Q Jiang, Alexandre Sablayrolles, Antoine Roux, Arthur Mensch, Blanche Savary, Chris Bamford, Devendra Singh Chaplot, Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al. Mixtral of experts. arXiv preprint arXiv:2401.04088, 2024."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080246"},{"key":"e_1_3_2_1_37_1","volume-title":"Exploiting intel\u00ae advanced matrix extensions (amx) for large language model inference","author":"Kim Hyungyo","year":"2024","unstructured":"Hyungyo Kim, Gaohan Ye, Nachuan Wang, Amir Yazdanbakhsh, and Nam Sung Kim. Exploiting intel\u00ae advanced matrix extensions (amx) for large language model inference. IEEE Computer Architecture Letters, 2024."},{"key":"e_1_3_2_1_38_1","volume-title":"Sparse fine-tuning for inference acceleration of large language models","author":"Kurtic Eldar","year":"2023","unstructured":"Eldar Kurtic, Denis Kuznedelev, Elias Frantar, Michael Goin, and Dan Alistarh. Sparse fine-tuning for inference acceleration of large language models, 2023."},{"key":"e_1_3_2_1_39_1","first-page":"5533","volume-title":"Proceedings of the 37th International Conference on Machine Learning, volume 119 of Proceedings of Machine Learning Research","author":"Kurtz Mark","year":"2020","unstructured":"Mark Kurtz, Justin Kopinsky, Rati Gelashvili, Alexander Matveev, John Carr, Michael Goin, William Leiserson, Sage Moore, Bill Nell, Nir Shavit, and Dan Alistarh. Inducing and exploiting activation sparsity for fast inference on deep neural networks. In Hal Daum\u00e9 III and Aarti Singh, editors, Proceedings of the 37th International Conference on Machine Learning, volume 119 of Proceedings of Machine Learning Research, pages 5533--5543, Virtual, 13--18 Jul 2020. PMLR."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_41_1","volume-title":"NeurIPS","volume":"2","author":"LeCun Yann","year":"1989","unstructured":"Yann LeCun, John Denker, and Sara Solla. Optimal brain damage. In NeurIPS, volume 2, 1989."},{"key":"e_1_3_2_1_42_1","volume-title":"Cats: Contextually-aware thresholding for sparsity in large language models. arXiv preprint arXiv:2404.08763","author":"Lee Je-Yong","year":"2024","unstructured":"Je-Yong Lee, Donghyun Lee, Genghan Zhang, Mo Tiwari, and Azalia Mirhoseini. Cats: Contextually-aware thresholding for sparsity in large language models. arXiv preprint arXiv:2404.08763, 2024."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.5555\/3571885.3571934"},{"key":"e_1_3_2_1_44_1","first-page":"513","volume-title":"Proceedings of Machine Learning and Systems","volume":"5","author":"Lin Bin","year":"2023","unstructured":"Bin Lin, Ningxin Zheng, Lei Wang, Shijie Cao, Lingxiao Ma, Quanlu Zhang, Yi Zhu, Ting Cao, Jilong Xue, Yuqing Yang, and Fan Yang. Efficient gpu kernels for n:m-sparse weights in deep learning. In D. Song, M. Carbin, and T. Chen, editors, Proceedings of Machine Learning and Systems, volume 5, pages 513--525. Curan, 2023."},{"key":"e_1_3_2_1_45_1","first-page":"87","article-title":"Activation-aware weight quantization for on-device llm compression and acceleration","volume":"6","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Wei-Ming Chen, Wei-Chen Wang, Guangxuan Xiao, Xingyu Dang, Chuang Gan, and Song Han. Awq: Activation-aware weight quantization for on-device llm compression and acceleration. Proceedings of Machine Learning and Systems, 6:87--100, 2024.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_46_1","volume-title":"Qserve: W4a8kv4 quantization and system co-design for efficient llm serving. arXiv preprint arXiv:2405.04532","author":"Lin Yujun","year":"2024","unstructured":"Yujun Lin, Haotian Tang, Shang Yang, Zhekai Zhang, Guangxuan Xiao, Chuang Gan, and Song Han. Qserve: W4a8kv4 quantization and system co-design for efficient llm serving. arXiv preprint arXiv:2405.04532, 2024."},{"key":"e_1_3_2_1_47_1","volume-title":"Training-free activation sparsity in large language models","author":"Liu James","year":"2024","unstructured":"James Liu, Pragaash Ponnusamy, Tianle Cai, Han Guo, Yoon Kim, and Ben Athiwaratkun. Training-free activation sparsity in large language models, 2024."},{"key":"e_1_3_2_1_48_1","volume-title":"Llm-qat: Data-free quantization aware training for large language models. arXiv preprint arXiv:2305.17888","author":"Liu Zechun","year":"2023","unstructured":"Zechun Liu, Barlas Oguz, Changsheng Zhao, Ernie Chang, Pierre Stock, Yashar Mehdad, Yangyang Shi, Raghuraman Krishnamoorthi, and Vikas Chandra. Llm-qat: Data-free quantization aware training for large language models. arXiv preprint arXiv:2305.17888, 2023."},{"key":"e_1_3_2_1_49_1","volume-title":"Spinquant-llm quantization with learned rotations. arXiv preprint arXiv:2405.16406","author":"Liu Zechun","year":"2024","unstructured":"Zechun Liu, Changsheng Zhao, Igor Fedorov, Bilge Soran, Dhruv Choudhary, Raghuraman Krishnamoorthi, Vikas Chandra, Yuandong Tian, and Tijmen Blankevoort. Spinquant-llm quantization with learned rotations. arXiv preprint arXiv:2405.16406, 2024."},{"key":"e_1_3_2_1_50_1","first-page":"22137","volume-title":"International Conference on Machine Learning","author":"Liu Zichang","year":"2023","unstructured":"Zichang Liu, Jue Wang, Tri Dao, Tianyi Zhou, Binhang Yuan, Zhao Song, Anshumali Shrivastava, Ce Zhang, Yuandong Tian, Christopher Re, et al. Deja vu: Contextual sparsity for efficient llms at inference time. In International Conference on Machine Learning, pages 22137--22176. PMLR, 2023."},{"key":"e_1_3_2_1_51_1","volume-title":"Benchmarking and dissecting the nvidia hopper gpu architecture. arXiv preprint arXiv:2402.13499","author":"Luo Weile","year":"2024","unstructured":"Weile Luo, Ruibo Fan, Zeyu Li, Dayou Du, Qiang Wang, and Xiaowen Chu. Benchmarking and dissecting the nvidia hopper gpu architecture. arXiv preprint arXiv:2402.13499, 2024."},{"key":"e_1_3_2_1_52_1","volume-title":"Llm-pruner: On the structural pruning of large language models. Advances in neural information processing systems, 36:21702--21720","author":"Ma Xinyin","year":"2023","unstructured":"Xinyin Ma, Gongfan Fang, and Xinchao Wang. Llm-pruner: On the structural pruning of large language models. Advances in neural information processing systems, 36:21702--21720, 2023."},{"key":"e_1_3_2_1_53_1","volume-title":"GPU Technology Conference","author":"Naumov Maxim","year":"2010","unstructured":"Maxim Naumov, L Chien, Philippe Vandermersch, and Ujval Kapasi. Cusparse library. In GPU Technology Conference, 2010."},{"key":"e_1_3_2_1_54_1","volume-title":"NVIDIA volta gpu architecture whitepaper. https:\/\/images.nvidia.com\/content\/volta-architecture\/pdf\/volta-architecture-whitepaper.pdf","author":"NVIDIA.","year":"2017","unstructured":"NVIDIA. NVIDIA volta gpu architecture whitepaper. https:\/\/images.nvidia.com\/content\/volta-architecture\/pdf\/volta-architecture-whitepaper.pdf, 2017."},{"key":"e_1_3_2_1_55_1","volume-title":"Nvidia a100 tensor core gpu architecture","author":"NVIDIA.","year":"2020","unstructured":"NVIDIA. Nvidia a100 tensor core gpu architecture, 2020."},{"key":"e_1_3_2_1_56_1","volume-title":"https:\/\/github.com\/NVIDIA\/FasterTransformer","author":"Fastertransformer NVIDIA.","year":"2023","unstructured":"NVIDIA. Fastertransformer. https:\/\/github.com\/NVIDIA\/FasterTransformer, 2023."},{"key":"e_1_3_2_1_57_1","volume-title":"NVIDIA CUDA C Programming Guide. https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html","author":"NVIDIA.","year":"2023","unstructured":"NVIDIA. NVIDIA CUDA C Programming Guide. https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html, 2023."},{"key":"e_1_3_2_1_58_1","volume-title":"PTX ISA: CUDA Toolkit documentation. https:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/index.html","author":"NVIDIA.","year":"2023","unstructured":"NVIDIA. PTX ISA: CUDA Toolkit documentation. https:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/index.html, 2023."},{"key":"e_1_3_2_1_59_1","volume-title":"cublas docs. https:\/\/docs.nvidia.com\/cuda\/cublas\/index.html","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. cublas docs. https:\/\/docs.nvidia.com\/cuda\/cublas\/index.html, 2024."},{"key":"e_1_3_2_1_60_1","volume-title":"cusparse library. https:\/\/docs.nvidia.com\/cuda\/cusparse\/index.html","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. cusparse library. https:\/\/docs.nvidia.com\/cuda\/cusparse\/index.html, 2024."},{"key":"e_1_3_2_1_61_1","volume-title":"Nsight compute. https:\/\/developer.nvidia.com\/nsight-compute","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. Nsight compute. https:\/\/developer.nvidia.com\/nsight-compute, 2024."},{"key":"e_1_3_2_1_62_1","volume-title":"Nsight systems. https:\/\/developer.nvidia.com\/nsight-systems","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. Nsight systems. https:\/\/developer.nvidia.com\/nsight-systems, 2024."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640383"},{"key":"e_1_3_2_1_64_1","volume-title":"Maciej Besta, Flavio Vella, and Torsten Hoefler. High performance unstructured spmm computation using tensor cores. arXiv preprint arXiv:2408.11551","author":"Okanovic Patrik","year":"2024","unstructured":"Patrik Okanovic, Grzegorz Kwasniewski, Paolo Sylos Labini, Maciej Besta, Flavio Vella, and Torsten Hoefler. High performance unstructured spmm computation using tensor cores. arXiv preprint arXiv:2408.11551, 2024."},{"key":"e_1_3_2_1_65_1","volume-title":"Gpt-4 technical report","author":"AI.","year":"2023","unstructured":"OpenAI. Gpt-4 technical report, 2023."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627535.3638470"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"e_1_3_2_1_68_1","volume-title":"Mooncake: Kimi's kvcache-centric architecture for llm serving. arXiv preprint arXiv:2407.00079","author":"Qin Ruoyu","year":"2024","unstructured":"Ruoyu Qin, Zheming Li, Weiran He, Mingxing Zhang, Yongwei Wu, Weimin Zheng, and Xinran Xu. Mooncake: Kimi's kvcache-centric architecture for llm serving. arXiv preprint arXiv:2407.00079, 2024."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS61541.2024.00022"},{"key":"e_1_3_2_1_71_1","volume-title":"Flashattention-3: Fast and accurate attention with asynchrony and low-precision","author":"Shah Jay","year":"2024","unstructured":"Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, and Tri Dao. Flashattention-3: Fast and accurate attention with asynchrony and low-precision, 2024."},{"key":"e_1_3_2_1_72_1","first-page":"965","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Sheng Ying","year":"2024","unstructured":"Ying Sheng, Shiyi Cao, Dacheng Li, Banghua Zhu, Zhuohan Li, Danyang Zhuo, Joseph E. Gonzalez, and Ion Stoica. Fairness in serving large language models. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24), pages 965--988, Santa Clara, CA, July 2024. USENIX Association."},{"key":"e_1_3_2_1_73_1","volume-title":"Flexgen: High-throughput generative inference of large language models with a single gpu","author":"Sheng Ying","year":"2023","unstructured":"Ying Sheng, Lianmin Zheng, Binhang Yuan, Zhuohan Li, Max Ryabinin, Daniel Y. Fu, Zhiqiang Xie, Beidi Chen, Clark Barrett, Joseph E. Gonzalez, Percy Liang, Christopher R\u00e9, Ion Stoica, and Ce Zhang. Flexgen: High-throughput generative inference of large language models with a single gpu, 2023."},{"key":"e_1_3_2_1_74_1","volume-title":"Flashsparse: Minimizing computation redundancy for fast sparse matrix multiplications on tensor cores. arXiv preprint arXiv:2412.11007","author":"Shi Jinliang","year":"2024","unstructured":"Jinliang Shi, Shigang Li, Youxuan Xu, Rongtian Fu, Xueying Wang, and Tong Wu. Flashsparse: Minimizing computation redundancy for fast sparse matrix multiplications on tensor cores. arXiv preprint arXiv:2412.11007, 2024."},{"key":"e_1_3_2_1_75_1","volume-title":"Powerinfer: Fast large language model serving with a consumer-grade gpu. arXiv preprint arXiv:2312.12456","author":"Song Yixin","year":"2023","unstructured":"Yixin Song, Zeyu Mi, Haotong Xie, and Haibo Chen. Powerinfer: Fast large language model serving with a consumer-grade gpu. arXiv preprint arXiv:2312.12456, 2023."},{"key":"e_1_3_2_1_76_1","volume-title":"Workshop on Efficient Systems for Foundation Models @ ICML2023","author":"Sun Mingjie","year":"2023","unstructured":"Mingjie Sun, Zhuang Liu, Anna Bair, and J Zico Kolter. A simple and effective pruning approach for large language models. In Workshop on Efficient Systems for Foundation Models @ ICML2023, 2023."},{"key":"e_1_3_2_1_77_1","volume-title":"ICLR","author":"Sun Mingjie","year":"2024","unstructured":"Mingjie Sun, Zhuang Liu, Anna Bair, and J. Zico Kolter. A simple and effective pruning approach for large language models. In ICLR, 2024."},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3217824"},{"key":"e_1_3_2_1_79_1","unstructured":"Vijay Thakkar Pradeep Ramani Cris Cecka Aniket Shivam Honghao Lu Ethan Yan Jack Kosaian Mark Hoemmen Haicheng Wu Andrew Kerr Matt Nicely Duane Merrill Dustyn Blasig Fengqi Qiao Piotr Majcher Paul Springer Markus Hohnerbach Jin Wang and Manish Gupta. CUTLASS January 2023."},{"key":"e_1_3_2_1_80_1","volume-title":"Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Louis Martin, Kevin Stone, Peter Albert, Amjad Almahairi, Yasmine Babaei, Nikolay Bashlykov, Soumya Batra, Prajjwal Bhargava, Shruti Bhosale, et al. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288, 2023."},{"key":"e_1_3_2_1_81_1","volume-title":"Advances in Neural Information Processing Systems","author":"Vaswani A","year":"2017","unstructured":"A Vaswani. Attention is all you need. Advances in Neural Information Processing Systems, 2017."},{"key":"e_1_3_2_1_82_1","first-page":"307","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Wang Lei","year":"2024","unstructured":"Lei Wang, Lingxiao Ma, Shijie Cao, Quanlu Zhang, Jilong Xue, Yining Shi, Ningxin Zheng, Ziming Miao, Fan Yang, Ting Cao, et al. Ladder: Enabling efficient low-precision deep learning computing through hardware-aware tensor transformation. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24), pages 307--323, 2024."},{"key":"e_1_3_2_1_83_1","first-page":"149","volume-title":"2023 USENIX Annual Technical Conference (USENIX ATC 23)","author":"Wang Yuke","year":"2023","unstructured":"Yuke Wang, Boyuan Feng, Zheng Wang, Guyue Huang, and Yufei Ding. TC-GNN: Bridging sparse GNN computation and dense tensor cores on GPUs. In 2023 USENIX Annual Technical Conference (USENIX ATC 23), pages 149--164, 2023."},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498765.1498785"},{"key":"e_1_3_2_1_85_1","volume-title":"Fast distributed inference serving for large language models","author":"Wu Bingyang","year":"2024","unstructured":"Bingyang Wu, Yinmin Zhong, Zili Zhang, Shengyu Liu, Fangyue Liu, Yuanhang Sun, Gang Huang, Xuanzhe Liu, and Xin Jin. Fast distributed inference serving for large language models, 2024."},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.14778\/3626292.3626303"},{"key":"e_1_3_2_1_87_1","first-page":"38087","volume-title":"International Conference on Machine Learning","author":"Xiao Guangxuan","year":"2023","unstructured":"Guangxuan Xiao, Ji Lin, Mickael Seznec, Hao Wu, Julien Demouth, and Song Han. Smoothquant: Accurate and efficient post-training quantization for large language models. In International Conference on Machine Learning, pages 38087--38099. PMLR, 2023."},{"key":"e_1_3_2_1_88_1","volume-title":"Lpvit: Low-power semi-structured pruning for vision transformers. arXiv preprint arXiv:2407.02068","author":"Xu Kaixin","year":"2024","unstructured":"Kaixin Xu, Zhe Wang, Chunyun Chen, Xue Geng, Jie Lin, Xulei Yang, Min Wu, Xiaoli Li, and Weisi Lin. Lpvit: Low-power semi-structured pruning for vision transformers. arXiv preprint arXiv:2407.02068, 2024."},{"key":"e_1_3_2_1_89_1","volume-title":"ICLR","author":"Xu Peng","year":"2024","unstructured":"Peng Xu, Wenqi Shao, Mengzhao Chen, Shitao Tang, Kaipeng Zhang, Peng Gao, Fengwei An, Yu Qiao, and Ping Luo. BESA: Pruning large language models with blockwise parameter-efficient sparsity allocation. In ICLR, 2024."},{"key":"e_1_3_2_1_90_1","volume-title":"Qwen2 technical report. arXiv preprint arXiv:2407.10671","author":"Yang An","year":"2024","unstructured":"An Yang, Baosong Yang, Binyuan Hui, Bo Zheng, Bowen Yu, Chang Zhou, Chengpeng Li, Chengyuan Li, Dayiheng Liu, Fei Huang, et al. Qwen2 technical report. arXiv preprint arXiv:2407.10671, 2024."},{"key":"e_1_3_2_1_91_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-96983-1_48"},{"key":"e_1_3_2_1_92_1","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582047"},{"key":"e_1_3_2_1_93_1","first-page":"521","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Yu Gyeong-In","year":"2022","unstructured":"Gyeong-In Yu, Joo Seong Jeong, Geon-Woo Kim, Soojeong Kim, and Byung-Gon Chun. Orca: A distributed serving system for Transformer-Based generative models. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pages 521--538, Carlsbad, CA, July 2022. USENIX Association."},{"key":"e_1_3_2_1_94_1","volume-title":"Yan Yan, et al. Llm inference unveiled: Survey and roofline model insights. arXiv preprint arXiv:2402.16363","author":"Yuan Zhihang","year":"2024","unstructured":"Zhihang Yuan, Yuzhang Shang, Yang Zhou, Zhen Dong, Chenhao Xue, Bingzhe Wu, Zhikai Li, Qingyi Gu, Yong Jae Lee, Yan Yan, et al. Llm inference unveiled: Survey and roofline model insights. arXiv preprint arXiv:2402.16363, 2024."},{"key":"e_1_3_2_1_95_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS54959.2023.00042"},{"key":"e_1_3_2_1_96_1","volume-title":"Xi Victoria Lin, et al. Opt: Open pre-trained transformer language models. arXiv preprint arXiv:2205.01068","author":"Zhang Susan","year":"2022","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, Shuohui Chen, Christopher Dewan, Mona Diab, Xian Li, Xi Victoria Lin, et al. Opt: Open pre-trained transformer language models. arXiv preprint arXiv:2205.01068, 2022."},{"key":"e_1_3_2_1_97_1","doi-asserted-by":"publisher","DOI":"10.20944\/preprints202310.1487.v2"},{"key":"e_1_3_2_1_98_1","first-page":"196","article-title":"Low-bit quantization for efficient and accurate llm serving","volume":"6","author":"Zhao Yilong","year":"2024","unstructured":"Yilong Zhao, Chien-Yu Lin, Kan Zhu, Zihao Ye, Lequn Chen, Size Zheng, Luis Ceze, Arvind Krishnamurthy, Tianqi Chen, and Baris Kasikci. Atom: Low-bit quantization for efficient and accurate llm serving. Proceedings of Machine Learning and Systems, 6:196--209, 2024.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_99_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613139"},{"key":"e_1_3_2_1_100_1","first-page":"213","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng Ningxin","year":"2022","unstructured":"Ningxin Zheng, Bin Lin, Quanlu Zhang, Lingxiao Ma, Yuqing Yang, Fan Yang, Yang Wang, Mao Yang, and Lidong Zhou. SparTA: Deep-Learning model sparsity via Tensor-with-Sparsity-Attribute. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pages 213--232, 2022."},{"key":"e_1_3_2_1_101_1","first-page":"193","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong Yinmin","year":"2024","unstructured":"Yinmin Zhong, Shengyu Liu, Junda Chen, Jianbo Hu, Yibo Zhu, Xuanzhe Liu, Xin Jin, and Hao Zhang. DistServe: Disaggregating prefill and decoding for goodput-optimized large language model serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24), pages 193--210, Santa Clara, CA, July 2024. USENIX Association."},{"key":"e_1_3_2_1_102_1","volume-title":"A survey on efficient inference for large language models. arXiv preprint arXiv:2404.14294","author":"Zhou Zixuan","year":"2024","unstructured":"Zixuan Zhou, Xuefei Ning, Ke Hong, Tianyu Fu, Jiaming Xu, Shiyao Li, Yuming Lou, Luning Wang, Zhihang Yuan, Xiuhong Li, Shengen Yan, Guohao Dai, Xiao-Ping Zhang, Yuhan Dong, and Yu Wang. A survey on efficient inference for large language models. arXiv preprint arXiv:2404.14294, 2024."}],"event":{"name":"EuroSys '25: Twentieth European Conference on Computer Systems","location":"Rotterdam Netherlands","acronym":"EuroSys '25","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the Twentieth European Conference on Computer Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3689031.3717481","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3689031.3717481","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T11:24:24Z","timestamp":1755775464000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3689031.3717481"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,30]]},"references-count":102,"alternative-id":["10.1145\/3689031.3717481","10.1145\/3689031"],"URL":"https:\/\/doi.org\/10.1145\/3689031.3717481","relation":{},"subject":[],"published":{"date-parts":[[2025,3,30]]},"assertion":[{"value":"2025-03-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}