{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T15:30:08Z","timestamp":1773588608500,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":90,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,22]]},"DOI":"10.1145\/3779212.3790250","type":"proceedings-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T13:55:26Z","timestamp":1773150926000},"page":"2264-2280","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Z\n                    <scp>ip<\/scp>\n                    S\n                    <scp>erv<\/scp>\n                    : Fast and Memory-Efficient LLM Inference with Hardware-Aware Lossless Compression"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7492-2069","authenticated-orcid":false,"given":"Ruibo","family":"Fan","sequence":"first","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2478-1512","authenticated-orcid":false,"given":"Xiangrui","family":"Yu","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1172-9935","authenticated-orcid":false,"given":"Xinglin","family":"Pan","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4381-0544","authenticated-orcid":false,"given":"Zeyu","family":"Li","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2875-0056","authenticated-orcid":false,"given":"Weile","family":"Luo","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2986-967X","authenticated-orcid":false,"given":"Qiang","family":"Wang","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4585-4152","authenticated-orcid":false,"given":"Wei","family":"Wang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9745-4372","authenticated-orcid":false,"given":"Xiaowen","family":"Chu","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China and The Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,3,22]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.5555\/3691938.3691945"},{"key":"e_1_3_2_1_2_1","volume-title":"arXiv preprint arXiv:2310.06825","author":"Mistral AI.","year":"2023","unstructured":"Mistral AI. 2023. Mistral 7B. arXiv preprint arXiv:2310.06825 (2023)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Zeyuan Allen-Zhu and Yuanzhi Li. 2025. Physics of Language Models: Part 3.3 Knowledge Capacity Scaling Laws. In ICLR. OpenReview.net.","DOI":"10.2139\/ssrn.5250617"},{"key":"e_1_3_2_1_4_1","volume-title":"Quarot: Outlier-free 4-bit inference in rotated llms. arXiv preprint arXiv:2404.00456","author":"Ashkboos Saleh","year":"2024","unstructured":"Saleh Ashkboos, Amirkeivan Mohtashami, Maximilian L Croci, Bo Li, Martin Jaggi, Dan Alistarh, Torsten Hoefler, and James Hensman. 2024. Quarot: Outlier-free 4-bit inference in rotated llms. arXiv preprint arXiv:2404.00456 (2024)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731024"},{"key":"e_1_3_2_1_6_1","volume-title":"Tianle Li, Dacheng Li, Banghua Zhu, Hao Zhang, Michael I. Jordan, Joseph E. Gonzalez, and Ion Stoica.","author":"Chiang Wei-Lin","year":"2024","unstructured":"Wei-Lin Chiang, Lianmin Zheng, Ying Sheng, Anastasios Nikolas Angelopoulos, Tianle Li, Dacheng Li, Banghua Zhu, Hao Zhang, Michael I. Jordan, Joseph E. Gonzalez, and Ion Stoica. 2024. Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference. In ICML. OpenReview.net."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2018.00051"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00080"},{"key":"e_1_3_2_1_9_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Dao Tri","year":"2024","unstructured":"Tri Dao. 2024. FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Tri Dao Daniel Y. Fu Stefano Ermon Atri Rudra and Christopher R\u00e9. 2022. FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness. In Advances in Neural Information Processing Systems (NeurIPS).","DOI":"10.52202\/068431-1189"},{"key":"e_1_3_2_1_11_1","volume-title":"Beyond size: How gradients shape pruning decisions in large language models. arXiv preprint arXiv:2311.04902","author":"Das Rocktim Jyoti","year":"2023","unstructured":"Rocktim Jyoti Das, Liqun Ma, and Zhiqiang Shen. 2023. Beyond size: How gradients shape pruning decisions in large language models. arXiv preprint arXiv:2311.04902 (2023)."},{"key":"e_1_3_2_1_12_1","first-page":"30318","article-title":"Gpt3. int8 (): 8-bit matrix multiplication for transformers at scale","volume":"35","author":"Dettmers Tim","year":"2022","unstructured":"Tim Dettmers, Mike Lewis, Younes Belkada, and Luke Zettlemoyer. 2022. Gpt3. int8 (): 8-bit matrix multiplication for transformers at scale. Advances in Neural Information Processing Systems, Vol. 35 (2022), 30318-30332.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Tim Dettmers Artidoro Pagnoni Ari Holtzman and Luke Zettlemoyer. 2023. QLoRA: Efficient Finetuning of Quantized LLMs. In NeurIPS.","DOI":"10.52202\/075280-0441"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning. PMLR. https:\/\/arxiv.org\/abs\/2406","author":"Dong Peijie","year":"2024","unstructured":"Peijie Dong, Lujun Li, Zhenheng Tang, Xiang Liu, Xinglin Pan, Qiang Wang, and Xiaowen Chu. 2024a. Pruner-Zero: Evolving Symbolic Pruning Metric from Scratch for Large Language Models. In Proceedings of the 41st International Conference on Machine Learning. PMLR. https:\/\/arxiv.org\/abs\/2406.02924 [arXiv: 2406.02924]."},{"key":"e_1_3_2_1_15_1","volume-title":"Stbllm: Breaking the 1-bit barrier with structured binary llms. arXiv preprint arXiv:2408.01803","author":"Dong Peijie","year":"2024","unstructured":"Peijie Dong, Lujun Li, Yuedong Zhong, Dayou Du, Ruibo Fan, Yuhan Chen, Zhenheng Tang, Qiang Wang, Wei Xue, Yike Guo, et al., 2024b. Stbllm: Breaking the 1-bit barrier with structured binary llms. arXiv preprint arXiv:2408.01803 (2024)."},{"key":"e_1_3_2_1_16_1","volume-title":"Can Compressed LLMs Truly Act? An Empirical Evaluation of Agentic Capabilities in LLM Compression. arXiv preprint arXiv:2505.19433","author":"Dong Peijie","year":"2025","unstructured":"Peijie Dong, Zhenheng Tang, Xiang Liu, Lujun Li, Xiaowen Chu, and Bo Li. 2025. Can Compressed LLMs Truly Act? An Empirical Evaluation of Agentic Capabilities in LLM Compression. arXiv preprint arXiv:2505.19433 (2025)."},{"key":"e_1_3_2_1_17_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/PCS.2015.7170048"},{"key":"e_1_3_2_1_19_1","volume-title":"Lu Hou, Boxing Chen, Masoud Asgharian, and Vahid Partovi Nia.","author":"Edalati Ali","year":"2025","unstructured":"Ali Edalati, Alireza Ghaffari, Mahsa Ghazvini Nejad, Lu Hou, Boxing Chen, Masoud Asgharian, and Vahid Partovi Nia. 2025. OAC: Output-adaptive Calibration for Accurate Post-training Quantization. In AAAI. AAAI Press, 16453-16461."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2005.6"},{"key":"e_1_3_2_1_21_1","first-page":"243","article-title":"SpInfer","author":"Fan Ruibo","year":"2025","unstructured":"Ruibo Fan, Xiangrui Yu, Peijie Dong, Zeyu Li, Gu Gong, Qiang Wang, Wei Wang, and Xiaowen Chu. 2025. SpInfer: Leveraging Low-Level Sparsity for Efficient Large Language Model Inference on GPUs. In EuroSys. ACM, 243-260.","journal-title":"In EuroSys. ACM"},{"key":"e_1_3_2_1_22_1","unstructured":"Elias Frantar and Dan Alistarh. 2023. SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot. In ICML."},{"key":"e_1_3_2_1_23_1","volume-title":"Gptq: Accurate post-training quantization for generative pre-trained transformers. arXiv preprint arXiv:2210.17323","author":"Frantar Elias","year":"2022","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2022. Gptq: Accurate post-training quantization for generative pre-trained transformers. arXiv preprint arXiv:2210.17323 (2022)."},{"key":"e_1_3_2_1_24_1","first-page":"239","article-title":"MARLIN","author":"Frantar Elias","year":"2025","unstructured":"Elias Frantar, Roberto L. Castro, Jiale Chen, Torsten Hoefler, and Dan Alistarh. 2025. MARLIN: Mixed-Precision Auto-Regressive Parallel Inference on Large Language Models. In PPoPP. ACM, 239-251.","journal-title":"Mixed-Precision Auto-Regressive Parallel Inference on Large Language Models. In PPoPP. ACM"},{"key":"e_1_3_2_1_25_1","first-page":"135","article-title":"ServerlessLLM: Low-Latency Serverless Inference for Large Language Models","author":"Fu Yao","year":"2024","unstructured":"Yao Fu, Leyang Xue, Yeqi Huang, Andrei-Octavian Brabete, Dmitrii Ustiugov, Yuvraj Patel, and Luo Mai. 2024. ServerlessLLM: Low-Latency Serverless Inference for Large Language Models. In OSDI. USENIX Association, 135-153.","journal-title":"OSDI. USENIX Association"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3725843.3756073"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3676641.3716011"},{"key":"e_1_3_2_1_28_1","volume-title":"NeuZip: Memory-Efficient Training and Inference with Dynamic Compression of Neural Networks. CoRR","author":"Hao Yongchang","year":"2065","unstructured":"Yongchang Hao, Yanshuai Cao, and Lili Mou. 2024. NeuZip: Memory-Efficient Training and Inference with Dynamic Compression of Neural Networks. CoRR, Vol. abs\/2410.20650 (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"ZipNN: Lossless Compression for AI Models. CoRR","author":"Hershcovitch Moshik","year":"2024","unstructured":"Moshik Hershcovitch, Andrew Wood, Leshem Choshen, Guy Girmonsky, Roy Leibovitz, Ilias Ennmouri, Michal Malka, Peter Chin, Swaminathan Sundararaman, and Danny Harnik. 2024. ZipNN: Lossless Compression for AI Models. CoRR, Vol. abs\/2411.05239 (2024)."},{"key":"e_1_3_2_1_30_1","volume-title":"Jeff Rasley, Samyam Rajbhandari, Reza Yazdani Aminabadi, Heyang Qin, Arash Bakhtiari, Lev Kurilenko, and Yuxiong He.","author":"Holmes Connor","year":"2024","unstructured":"Connor Holmes, Masahiro Tanaka, Michael Wyatt, Ammar Ahmad Awan, Jeff Rasley, Samyam Rajbhandari, Reza Yazdani Aminabadi, Heyang Qin, Arash Bakhtiari, Lev Kurilenko, and Yuxiong He. 2024. DeepSpeed-FastGen: High-throughput Text Generation for LLMs via MII and DeepSpeed-Inference. arXiv:2401.08671 [cs.PF] https:\/\/arxiv.org\/abs\/2401.08671"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/JRPROC.1952.273898"},{"key":"e_1_3_2_1_32_1","volume-title":"Dissecting the NVIDIA Blackwell Architecture with Microbenchmarks. arXiv preprint arXiv:2507.10789","author":"Jarmusch Aaron","year":"2025","unstructured":"Aaron Jarmusch, Nathan Graddon, and Sunita Chandrasekaran. 2025. Dissecting the NVIDIA Blackwell Architecture with Microbenchmarks. arXiv preprint arXiv:2507.10789 (2025)."},{"key":"e_1_3_2_1_33_1","unstructured":"Jeff Johnson. 2024. DIET-GPU: Efficient Model Inference on GPUs. https:\/\/github.com\/facebookresearch\/dietgpu."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589350"},{"key":"e_1_3_2_1_35_1","volume-title":"Nataraj Jammalamadaka, Jianyu Huang, Hector Yuen, et al.","author":"Kalamkar Dhiraj","year":"2019","unstructured":"Dhiraj Kalamkar, Dheevatsa Mudigere, Naveen Mellempudi, Dipankar Das, Kunal Banerjee, Sasikanth Avancha, Dharma Teja Vooturi, Nataraj Jammalamadaka, Jianyu Huang, Hector Yuen, et al., 2019. A study of BFLOAT16 for deep learning training. arXiv preprint arXiv:1905.12322 (2019)."},{"key":"e_1_3_2_1_36_1","volume-title":"Scaling Laws for Neural Language Models. CoRR","author":"Kaplan Jared","year":"2020","unstructured":"Jared Kaplan, Sam McCandlish, Tom Henighan, Tom B. Brown, Benjamin Chess, Rewon Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei. 2020. Scaling Laws for Neural Language Models. CoRR, Vol. abs\/2001.08361 (2020)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2024.3397747"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3007787.3001172"},{"key":"e_1_3_2_1_39_1","first-page":"611","article-title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","author":"Kwon Woosuk","year":"2023","unstructured":"Woosuk Kwon, Zhuohan Li, Siyuan Zhuang, Ying Sheng, Lianmin Zheng, Cody Hao Yu, Joseph Gonzalez, Hao Zhang, and Ion Stoica. 2023. Efficient Memory Management for Large Language Model Serving with PagedAttention. In SOSP. ACM, 611-626.","journal-title":"SOSP. ACM"},{"key":"e_1_3_2_1_40_1","article-title":"Deep Neural Networks with Dependent Weights: Gaussian Process Mixture Limit, Heavy Tails, Sparsity and Compressibility","volume":"24","author":"Lee Hoil","year":"2023","unstructured":"Hoil Lee, Fadhel Ayed, Paul Jung, Juho Lee, Hongseok Yang, and Francois Caron. 2023. Deep Neural Networks with Dependent Weights: Gaussian Process Mixture Limit, Heavy Tails, Sparsity and Compressibility. J. Mach. Learn. Res., Vol. 24 (2023), 289:1-289:78.","journal-title":"J. Mach. Learn. Res."},{"key":"e_1_3_2_1_41_1","volume-title":"Quantization Meets Reasoning: Exploring LLM Low-Bit Quantization Degradation for Mathematical Reasoning. CoRR","author":"Li Zhen","year":"2025","unstructured":"Zhen Li, Yupeng Su, Runming Yang, Zhongwei Xie, Ngai Wong, and Hongxia Yang. 2025. Quantization Meets Reasoning: Exploring LLM Low-Bit Quantization Degradation for Mathematical Reasoning. CoRR, Vol. abs\/2501.03035 (2025)."},{"key":"e_1_3_2_1_42_1","first-page":"663","article-title":"AlpaServe: Statistical Multiplexing with Model Parallelism for Deep Learning Serving","author":"Li Zhuohan","year":"2023","unstructured":"Zhuohan Li, Lianmin Zheng, Yinmin Zhong, Vincent Liu, Ying Sheng, Xin Jin, Yanping Huang, Zhifeng Chen, Hao Zhang, Joseph E. Gonzalez, and Ion Stoica. 2023. AlpaServe: Statistical Multiplexing with Model Parallelism for Deep Learning Serving. In OSDI. USENIX Association, 663-679.","journal-title":"OSDI. USENIX Association"},{"key":"e_1_3_2_1_43_1","first-page":"87","article-title":"AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration","volume":"6","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Wei-Ming Chen, Wei-Chen Wang, Guangxuan Xiao, Xingyu Dang, Chuang Gan, and Song Han. 2024. AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration. Proceedings of Machine Learning and Systems, Vol. 6 (2024), 87-100.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_44_1","volume-title":"Quantization Hurts Reasoning? An Empirical Study on Quantized Reasoning Models. CoRR","author":"Liu Ruikang","year":"2025","unstructured":"Ruikang Liu, Yuxuan Sun, Manyi Zhang, Haoli Bai, Xianzhi Yu, Tiezheng Yu, Chun Yuan, and Lu Hou. 2025. Quantization Hurts Reasoning? An Empirical Study on Quantized Reasoning Models. CoRR, Vol. abs\/2504.04823 (2025)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672274"},{"key":"e_1_3_2_1_46_1","volume-title":"SpinQuant-LLM quantization with learned rotations. arXiv preprint arXiv:2405.16406","author":"Liu Zechun","year":"2024","unstructured":"Zechun Liu, Changsheng Zhao, Igor Fedorov, Bilge Soran, Dhruv Choudhary, Raghuraman Krishnamoorthi, Vikas Chandra, Yuandong Tian, and Tijmen Blankevoort. 2024b. SpinQuant-LLM quantization with learned rotations. arXiv preprint arXiv:2405.16406 (2024)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS57955.2024.00064"},{"key":"e_1_3_2_1_48_1","first-page":"881","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Ma Lingxiao","year":"2020","unstructured":"Lingxiao Ma, Zhiqiang Xie, Zhi Yang, Jilong Xue, Youshan Miao, Wei Cui, Wenxiang Hu, Fan Yang, Lintao Zhang, and Lidong Zhou. 2020. Rammer: Enabling Holistic Deep Learning Compiler Optimizations with rTasks. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20). USENIX Association, 881-897. https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/ma"},{"key":"e_1_3_2_1_49_1","volume-title":"Does quantization affect models' performance on long-context tasks? arXiv preprint arXiv:2505.20276","author":"Mekala Anmol","year":"2025","unstructured":"Anmol Mekala, Anirudh Atmakuru, Yixiao Song, Marzena Karpinska, and Mohit Iyyer. 2025. Does quantization affect models' performance on long-context tasks? arXiv preprint arXiv:2505.20276 (2025)."},{"key":"e_1_3_2_1_50_1","unstructured":"NVIDIA. 2020. NVIDIA Ampere GA102 GPU Architecture Whitepaper. https:\/\/www.nvidia.com\/content\/PDF\/nvidia-ampere-ga-102-gpu-architecture-whitepaper-v2.pdf."},{"key":"e_1_3_2_1_51_1","unstructured":"NVIDIA. 2023. NVIDIA Ada GPU Architecture Whitepaper. https:\/\/images.nvidia.com\/aem-dam\/Solutions\/geforce\/ada\/nvidia-ada-gpu-architecture.pdf."},{"key":"e_1_3_2_1_52_1","unstructured":"NVIDIA. 2024. cuBLAS Docs. https:\/\/docs.nvidia.com\/cuda\/cublas\/index.html."},{"key":"e_1_3_2_1_53_1","unstructured":"NVIDIA. 2025. nvcomp: Repository for nvCOMP docs and examples. https:\/\/github.com\/NVIDIA\/nvcomp. Accessed: 2025-08-18."},{"key":"e_1_3_2_1_54_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. arXiv:2303.08774 [cs.CL]"},{"key":"e_1_3_2_1_55_1","volume-title":"Byeongwook Kim, Youngjoo Lee, and Dongsoo Lee.","author":"Park Gunho","year":"2024","unstructured":"Gunho Park, Baeseong Park, Minsub Kim, Sungjae Lee, Jeonghoon Kim, Beomseok Kwon, Se Jung Kwon, Byeongwook Kim, Youngjoo Lee, and Dongsoo Lee. 2024. LUT-GEMM: Quantized Matrix Multiplication based on LUTs for Efficient Inference in Large-Scale Generative Language Models. In ICLR. OpenReview.net."},{"key":"e_1_3_2_1_56_1","volume-title":"Byeongwook Kim, Youngjoo Lee, and Dongsoo Lee.","author":"Park Gunho","year":"2022","unstructured":"Gunho Park, Baeseong Park, Se Jung Kwon, Byeongwook Kim, Youngjoo Lee, and Dongsoo Lee. 2022. nuQmm: Quantized MatMul for Efficient Inference of Large-Scale Generative Language Models. CoRR, Vol. abs\/2206.09557 (2022)."},{"key":"e_1_3_2_1_57_1","volume-title":"QIGen: Generating Efficient Kernels for Quantized Inference on Large Language Models. CoRR","author":"Pegolotti Tommaso","year":"2023","unstructured":"Tommaso Pegolotti, Elias Frantar, Dan Alistarh, and Markus P\u00fcschel. 2023. QIGen: Generating Efficient Kernels for Quantized Inference on Large Language Models. CoRR, Vol. abs\/2307.03738 (2023)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540724"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/2370816.2370870"},{"key":"e_1_3_2_1_60_1","volume-title":"Toolformer: Language Models Can Teach Themselves to Use Tools. In NeurIPS.","author":"Schick Timo","year":"2023","unstructured":"Timo Schick, Jane Dwivedi-Yu, Roberto Dess\u00ec, Roberta Raileanu, Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom. 2023. Toolformer: Language Models Can Teach Themselves to Use Tools. In NeurIPS."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS61541.2024.00022"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"crossref","unstructured":"Jay Shah Ganesh Bikshandi Ying Zhang Vijay Thakkar Pradeep Ramani and Tri Dao. 2024. FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision. In NeurIPS.","DOI":"10.52202\/079017-2193"},{"key":"e_1_3_2_1_63_1","volume-title":"Unveiling the Mystery of Weight in Large Foundation Models: Gaussian Distribution Never Fades. CoRR","author":"Si Chongjie","year":"2025","unstructured":"Chongjie Si, Jingjing Jiang, and Wei Shen. 2025. Unveiling the Mystery of Weight in Large Foundation Models: Gaussian Distribution Never Fades. CoRR, Vol. abs\/2501.10661 (2025)."},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695964"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/3669940.3707255"},{"key":"e_1_3_2_1_66_1","volume-title":"Llumnix: Dynamic Scheduling for Large Language Model Serving","author":"Sun Biao","year":"2024","unstructured":"Biao Sun, Ziming Huang, Hanyu Zhao, Wencong Xiao, Xinyi Zhang, Yong Li, and Wei Lin. 2024a. Llumnix: Dynamic Scheduling for Large Language Model Serving. In OSDI. USENIX Association, 173-191."},{"key":"e_1_3_2_1_67_1","unstructured":"Mingjie Sun Zhuang Liu Anna Bair and J. Zico Kolter. 2024b. A Simple and Effective Pruning Approach for Large Language Models. In ICLR."},{"key":"e_1_3_2_1_68_1","volume-title":"Gemma 3 technical report. arXiv preprint arXiv:2503.19786","author":"Team Gemma","year":"2025","unstructured":"Gemma Team. 2025a. Gemma 3 technical report. arXiv preprint arXiv:2503.19786 (2025)."},{"key":"e_1_3_2_1_69_1","volume-title":"arXiv preprint arXiv:2412.15115","author":"Team Qwen","year":"2024","unstructured":"Qwen Team. 2024. Qwen2.5 technical report. arXiv preprint arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_70_1","unstructured":"Qwen Team. 2025b. Qwen3 Technical Report. arXiv preprint arXiv:2505.09388 (2025)."},{"key":"e_1_3_2_1_71_1","volume-title":"Lossless Compression for LLM Tensor Incremental Snapshots. arXiv preprint arXiv:2505.09810","author":"Waddington Daniel","year":"2025","unstructured":"Daniel Waddington and Cornel Constantinescu. 2025. Lossless Compression for LLM Tensor Incremental Snapshots. arXiv preprint arXiv:2505.09810 (2025)."},{"key":"e_1_3_2_1_72_1","first-page":"307","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Wang Lei","year":"2024","unstructured":"Lei Wang, Lingxiao Ma, Shijie Cao, Quanlu Zhang, Jilong Xue, Yining Shi, Ningxin Zheng, Ziming Miao, Fan Yang, Ting Cao, et al., 2024. Ladder: Enabling Efficient Low-Precision Deep Learning Computing through Hardware-aware Tensor Transformation. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). 307-323."},{"key":"e_1_3_2_1_73_1","first-page":"537","volume-title":"19th USENIX Symposium on Operating Systems Design and Implementation (OSDI 25)","author":"Wang Zhuang","year":"2025","unstructured":"Zhuang Wang, Zhaozhuo Xu, Jingyi Xi, Yuke Wang, Anshumali Shrivastava, and TS Eugene Ng. 2025. : Empowering Distributed Training with Sparsity-driven Data Synchronization. In 19th USENIX Symposium on Operating Systems Design and Implementation (OSDI 25). 537-556."},{"key":"e_1_3_2_1_74_1","volume-title":"Quoc V. Le, and Denny Zhou.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed H. Chi, Quoc V. Le, and Denny Zhou. 2022. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In NeurIPS."},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_2_1_76_1","volume-title":"Mirage: A Multi-Level Superoptimizer for Tensor Programs. In 19th USENIX Symposium on Operating Systems Design and Implementation (OSDI 25)","author":"Wu Mengdi","year":"2025","unstructured":"Mengdi Wu, Xinhao Cheng, Shengyu Liu, Chunan Shi, Jianan Ji, Kit Ao, Praveen Velliengiri, Xupeng Miao, Oded Padon, and Zhihao Jia. 2025. Mirage: A Multi-Level Superoptimizer for Tensor Programs. In 19th USENIX Symposium on Operating Systems Design and Implementation (OSDI 25). USENIX Association. https:\/\/www.usenix.org\/conference\/osdi25\/presentation\/wu-mengdi"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.14778\/3626292.3626303"},{"key":"e_1_3_2_1_78_1","volume-title":"International Conference on Machine Learning. PMLR, 38087-38099","author":"Xiao Guangxuan","year":"2023","unstructured":"Guangxuan Xiao, Ji Lin, Mickael Seznec, Hao Wu, Julien Demouth, and Song Han. 2023. Smoothquant: Accurate and efficient post-training quantization for large language models. In International Conference on Machine Learning. PMLR, 38087-38099."},{"key":"e_1_3_2_1_79_1","first-page":"204","article-title":"Bolt: Bridging the gap between auto-tuners and hardware-native performance","volume":"4","author":"Xing Jiarong","year":"2022","unstructured":"Jiarong Xing, Leyuan Wang, Shang Zhang, Jack Chen, Ang Chen, and Yibo Zhu. 2022. Bolt: Bridging the gap between auto-tuners and hardware-native performance. Proceedings of Machine Learning and Systems, Vol. 4 (2022), 204-216.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_80_1","volume-title":"BESA: Pruning Large Language Models with Blockwise Parameter-Efficient Sparsity Allocation. In ICLR.","author":"Xu Peng","year":"2024","unstructured":"Peng Xu, Wenqi Shao, Mengzhao Chen, Shitao Tang, Kaipeng Zhang, Peng Gao, Fengwei An, Yu Qiao, and Ping Luo. 2024. BESA: Pruning Large Language Models with Blockwise Parameter-Efficient Sparsity Allocation. In ICLR."},{"key":"e_1_3_2_1_81_1","unstructured":"Tian Ye Zicheng Xu Yuanzhi Li and Zeyuan Allen-Zhu. 2025. Physics of Language Models: Part 2.2 How to Learn From Mistakes on Grade-School Math Problems. In ICLR. OpenReview.net."},{"key":"e_1_3_2_1_82_1","first-page":"521","volume-title":"Orca: A Distributed Serving System for Transformer-Based Generative Models. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Yu Gyeong-In","year":"2022","unstructured":"Gyeong-In Yu, Joo Seong Jeong, Geon-Woo Kim, Soojeong Kim, and Byung-Gon Chun. 2022. Orca: A Distributed Serving System for Transformer-Based Generative Models. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22). USENIX Association, Carlsbad, CA, 521-538. https:\/\/www.usenix.org\/conference\/osdi22\/presentation\/yu"},{"key":"e_1_3_2_1_83_1","volume-title":"Siddharth Joshi, Chinmay Hegde, and Siddharth Garg.","author":"Yubeaton Patrick","year":"2025","unstructured":"Patrick Yubeaton, Tareq Mahmoud, Shehab Naga, Pooria Taheri, Tianhua Xia, Arun George, Yasmein Khalil, Sai Qian Zhang, Siddharth Joshi, Chinmay Hegde, and Siddharth Garg. 2025. Huff-LLM: End-to-End Lossless Compression for Efficient LLM Inference. arXiv:2502.00922 [cs.LG] https:\/\/arxiv.org\/abs\/2502.00922"},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS57875.2023.00031"},{"key":"e_1_3_2_1_85_1","volume-title":"Lossless LLM Compression for Efficient GPU Inference via Dynamic-Length Float. arXiv preprint arXiv:2504.11651","author":"Zhang Tianyi","year":"2025","unstructured":"Tianyi Zhang, Yang Sui, Shaochen Zhong, Vipin Chaudhary, Xia Hu, and Anshumali Shrivastava. 2025. 70% Size, 100% Accuracy: Lossless LLM Compression for Efficient GPU Inference via Dynamic-Length Float. arXiv preprint arXiv:2504.11651 (2025)."},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.20944\/preprints202310.1487.v2"},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1145\/2808233"},{"key":"e_1_3_2_1_88_1","first-page":"196","article-title":"Atom: Low-bit quantization for efficient and accurate llm serving","volume":"6","author":"Zhao Yilong","year":"2024","unstructured":"Yilong Zhao, Chien-Yu Lin, Kan Zhu, Zihao Ye, Lequn Chen, Size Zheng, Luis Ceze, Arvind Krishnamurthy, Tianqi Chen, and Baris Kasikci. 2024. Atom: Low-bit quantization for efficient and accurate llm serving. Proceedings of Machine Learning and Systems, Vol. 6 (2024), 196-209.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_89_1","volume-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems","author":"Zheng Lianmin","year":"2025","unstructured":"Lianmin Zheng, Liangsheng Yin, Zhiqiang Xie, Chuyue Sun, Jeff Huang, Cody Hao Yu, Shiyi Cao, Christos Kozyrakis, Ion Stoica, Joseph E. Gonzalez, Clark Barrett, and Ying Sheng. 2025. SGLang: efficient execution of structured language model programs. In Proceedings of the 38th International Conference on Neural Information Processing Systems (Vancouver, BC, Canada) (NIPS '24). Curran Associates Inc., Red Hook, NY, USA, Article 2000, 27 pages."},{"key":"e_1_3_2_1_90_1","first-page":"193","volume-title":"DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong Yinmin","year":"2024","unstructured":"Yinmin Zhong, Shengyu Liu, Junda Chen, Jianbo Hu, Yibo Zhu, Xuanzhe Liu, Xin Jin, and Hao Zhang. 2024. DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). USENIX Association, Santa Clara, CA, 193-210. https:\/\/www.usenix.org\/conference\/osdi24\/presentation\/zhong-yinmin"}],"event":{"name":"ASPLOS '26: 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems","location":"Pittsburgh PA USA","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGARCH ACM Special Interest Group on Computer Architecture","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2"],"original-title":[],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T13:56:24Z","timestamp":1773582984000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3779212.3790250"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,22]]},"references-count":90,"alternative-id":["10.1145\/3779212.3790250","10.1145\/3779212"],"URL":"https:\/\/doi.org\/10.1145\/3779212.3790250","relation":{},"subject":[],"published":{"date-parts":[[2026,3,22]]},"assertion":[{"value":"2026-03-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}