{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T13:00:07Z","timestamp":1780664407833,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":81,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,26]],"date-time":"2026-04-26T00:00:00Z","timestamp":1777161600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Key R&D Program of China","award":["2023YFB3002000"],"award-info":[{"award-number":["2023YFB3002000"]}]},{"name":"National Natural Science Foundation of China","award":["62422209"],"award-info":[{"award-number":["62422209"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,27]]},"DOI":"10.1145\/3767295.3769339","type":"proceedings-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T20:20:04Z","timestamp":1777062004000},"page":"845-861","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["LLMFolder: Revisiting Constant Folding in Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-0468-3744","authenticated-orcid":false,"given":"Gansen","family":"Hu","sequence":"first","affiliation":[{"name":"Institute of Parallel and Distributed Systems, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0220-5726","authenticated-orcid":false,"given":"Zhaoguo","family":"Wang","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-9852-4124","authenticated-orcid":false,"given":"Wei","family":"Huang","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0301-5975","authenticated-orcid":false,"given":"Jinglin","family":"Wei","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9720-0361","authenticated-orcid":false,"given":"Haibo","family":"Chen","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, School of Computer Science, Shanghai JiaoTong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,26]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"[n. d.]. Constant folding - Wikipedia. https:\/\/en.wikipedia.org\/wiki\/Constant_folding."},{"key":"e_1_3_2_1_2_1","volume-title":"d.]. CUDA C++ Programming Guide. https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/ [Online","year":"2024","unstructured":"[n. d.]. CUDA C++ Programming Guide. https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/ [Online; accessed 2024-11-01]."},{"key":"e_1_3_2_1_3_1","volume-title":"d.]. openai-community\/gpt2-xl \u00b7 Hugging Face. https:\/\/huggingface.co\/openai-community\/gpt2-xl [Online","year":"2025","unstructured":"[n. d.]. openai-community\/gpt2-xl \u00b7 Hugging Face. https:\/\/huggingface.co\/openai-community\/gpt2-xl [Online; accessed 2025-01-09]."},{"key":"e_1_3_2_1_4_1","unstructured":"[n. d.]. Optimize Options (Using the GNU Compiler Collection (GCC)). https:\/\/gcc.gnu.org\/onlinedocs\/gcc\/Optimize-Options.html. (Accessed on 11\/29\/2024)."},{"key":"e_1_3_2_1_5_1","unstructured":"[n. d.]. python\/cpython: The Python programming language. https:\/\/github.com\/python\/cpython?tab=readme-ov-file. (Accessed on 11\/29\/2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"d.]. Release PyTorch 2.5.1: bug fix release \u00b7 pytorch\/pytorch. https:\/\/github.com\/pytorch\/pytorch\/releases\/tag\/v2.5.1 [Online","year":"2024","unstructured":"[n. d.]. Release PyTorch 2.5.1: bug fix release \u00b7 pytorch\/pytorch. https:\/\/github.com\/pytorch\/pytorch\/releases\/tag\/v2.5.1 [Online; accessed 2024-11-24]."},{"key":"e_1_3_2_1_7_1","volume-title":"d.]. Release v0.6.6 \u00b7 vllm-project\/vllm. https:\/\/github.com\/vllmproject\/vllm\/releases\/tag\/v0.6.6 [Online","year":"2025","unstructured":"[n. d.]. Release v0.6.6 \u00b7 vllm-project\/vllm. https:\/\/github.com\/vllmproject\/vllm\/releases\/tag\/v0.6.6 [Online; accessed 2025-01-14]."},{"key":"e_1_3_2_1_8_1","volume-title":"d.]. Sixth Computational and Data Science school for HEP (CoDaS-HEP 2024) (22\u201326","year":"2024","unstructured":"[n. d.]. Sixth Computational and Data Science school for HEP (CoDaS-HEP 2024) (22\u201326 July 2024): Floating Point Arithmetic Is Not Real \u00b7 Indico. https:\/\/indico.cern.ch\/event\/1422680\/contributions\/5983305\/ [Online; accessed 2024-12-16]."},{"key":"e_1_3_2_1_9_1","unstructured":"[n. d.]. Spark SQL & DataFrames | Apache Spark. https:\/\/spark.apache.org\/sql\/. (Accessed on 11\/29\/2024)."},{"key":"e_1_3_2_1_10_1","unstructured":"[n. d.]. Transformers. https:\/\/huggingface.co\/docs\/transformers\/index. (Accessed on 06\/23\/2024)."},{"key":"e_1_3_2_1_11_1","unstructured":"[n. d.]. Welcome to cuML's documentation! \u2014 cuml 24.10.00 documentation. https:\/\/docs.rapids.ai\/api\/cuml\/stable\/. (Accessed on 12\/03\/2024)."},{"key":"e_1_3_2_1_12_1","volume-title":"https:\/\/huggingface.co\/datasets\/EleutherAI\/lambada_openai [Online","author":"Hugging Face Datasets","year":"2025","unstructured":"2023. EleutherAI\/lambada_openai \u00b7 Datasets at Hugging Face. https:\/\/huggingface.co\/datasets\/EleutherAI\/lambada_openai [Online; accessed 2025-01-09]."},{"key":"e_1_3_2_1_13_1","volume-title":"https:\/\/forums.developer.nvidia.com\/t\/lds-128-loads-from-shared-memory\/264465 [Online","author":"Programming CUDA","year":"2024","unstructured":"2023. LDS.128 loads from shared memory - CUDA \/ CUDA Programming and Performance - NVIDIA Developer Forums. https:\/\/forums.developer.nvidia.com\/t\/lds-128-loads-from-shared-memory\/264465 [Online; accessed 2024-11-01]."},{"key":"e_1_3_2_1_14_1","volume-title":"philschmid\/sharegpt-raw \u00b7 Datasets at Hugging Face. https:\/\/huggingface.co\/datasets\/philschmid\/sharegpt-raw [Online","year":"2025","unstructured":"2023. philschmid\/sharegpt-raw \u00b7 Datasets at Hugging Face. https:\/\/huggingface.co\/datasets\/philschmid\/sharegpt-raw [Online; accessed 2025-01-07]."},{"key":"e_1_3_2_1_15_1","volume-title":"tiiuae\/falcon-11B \u00b7 Hugging Face. https:\/\/huggingface.co\/tiiuae\/falcon-11B [Online","year":"2025","unstructured":"2023. tiiuae\/falcon-11B \u00b7 Hugging Face. https:\/\/huggingface.co\/tiiuae\/falcon-11B [Online; accessed 2025-01-13]."},{"key":"e_1_3_2_1_16_1","volume-title":"allenai\/ai2_arc \u00b7 Datasets at Hugging Face. https:\/\/huggingface.co\/datasets\/allenai\/ai2_arc [Online","year":"2025","unstructured":"2024. allenai\/ai2_arc \u00b7 Datasets at Hugging Face. https:\/\/huggingface.co\/datasets\/allenai\/ai2_arc [Online; accessed 2025-01-09]."},{"key":"e_1_3_2_1_17_1","volume-title":"facebook\/opt-6.7b \u00b7 Hugging Face. https:\/\/huggingface.co\/facebook\/opt-6.7b [Online","year":"2025","unstructured":"2024. facebook\/opt-6.7b \u00b7 Hugging Face. https:\/\/huggingface.co\/facebook\/opt-6.7b [Online; accessed 2025-01-09]."},{"key":"e_1_3_2_1_18_1","volume-title":"google\/switch-c-2048 \u00b7 Hugging Face. https:\/\/huggingface.co\/google\/switch-c-2048 [Online","year":"2025","unstructured":"2024. google\/switch-c-2048 \u00b7 Hugging Face. https:\/\/huggingface.co\/google\/switch-c-2048 [Online; accessed 2025-01-08]."},{"key":"e_1_3_2_1_19_1","unstructured":"Ebtesam Almazrouei Hamza Alobeidli Abdulaziz Alshamsi Alessandro Cappelli Ruxandra Cojocaru M\u00e9rouane Debbah \u00c9tienne Goffinet Daniel Hesslow Julien Launay Quentin Malartic Daniele Mazzotta Badreddine Noune Baptiste Pannier and Guilherme Penedo. 2023. The Falcon Series of Open Language Models. arXiv:2311.16867 [cs.CL] https:\/\/arxiv.org\/abs\/2311.16867"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"e_1_3_2_1_21_1","volume-title":"Language models are few-shot learners. arXiv preprint arXiv:2005.14165","author":"Brown Tom B","year":"2020","unstructured":"Tom B Brown. 2020. Language models are few-shot learners. arXiv preprint arXiv:2005.14165 (2020)."},{"key":"e_1_3_2_1_22_1","volume-title":"Medusa: Simple llm inference acceleration framework with multiple decoding heads. arXiv preprint arXiv:2401.10774","author":"Cai Tianle","year":"2024","unstructured":"Tianle Cai, Yuhong Li, Zhengyang Geng, Hongwu Peng, Jason D Lee, Deming Chen, and Tri Dao. 2024. Medusa: Simple llm inference acceleration framework with multiple decoding heads. arXiv preprint arXiv:2401.10774 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"Dsformer: Effective compression of text-transformers by dense-sparse weight factorization. arXiv preprint arXiv:2312.13211","author":"Chand Rahul","year":"2023","unstructured":"Rahul Chand, Yashoteja Prabhu, and Pratyush Kumar. 2023. Dsformer: Effective compression of text-transformers by dense-sparse weight factorization. arXiv preprint arXiv:2312.13211 (2023)."},{"key":"e_1_3_2_1_24_1","volume-title":"Accelerating large language model decoding with speculative sampling. arXiv preprint arXiv:2302.01318","author":"Chen Charlie","year":"2023","unstructured":"Charlie Chen, Sebastian Borgeaud, Geoffrey Irving, Jean-Baptiste Lespiau, Laurent Sifre, and John Jumper. 2023. Accelerating large language model decoding with speculative sampling. arXiv preprint arXiv:2302.01318 (2023)."},{"key":"e_1_3_2_1_25_1","volume-title":"Fast and accurate deep network learning by exponential linear units (elus). arXiv preprint arXiv:1511.07289","author":"Clevert Djork-Arn\u00e9","year":"2015","unstructured":"Djork-Arn\u00e9 Clevert. 2015. Fast and accurate deep network learning by exponential linear units (elus). arXiv preprint arXiv:1511.07289 (2015)."},{"key":"e_1_3_2_1_26_1","unstructured":"Tri Dao. [n. d.]. Stanford CRFM. https:\/\/crfm.stanford.edu\/2023\/10\/12\/flashdecoding.html [Online]."},{"key":"e_1_3_2_1_27_1","volume-title":"Flashattention-2: Faster attention with better parallelism and work partitioning. arXiv preprint arXiv:2307.08691","author":"Dao Tri","year":"2023","unstructured":"Tri Dao. 2023. Flashattention-2: Faster attention with better parallelism and work partitioning. arXiv preprint arXiv:2307.08691 (2023)."},{"key":"e_1_3_2_1_28_1","volume-title":"Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in neural information processing systems 35","author":"Dao Tri","year":"2022","unstructured":"Tri Dao, Dan Fu, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. 2022. Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in neural information processing systems 35 (2022), 16344\u201316359."},{"key":"e_1_3_2_1_29_1","volume-title":"International conference on machine learning. PMLR, 933\u2013941","author":"Dauphin Yann N","year":"2017","unstructured":"Yann N Dauphin, Angela Fan, Michael Auli, and David Grangier. 2017. Language modeling with gated convolutional networks. In International conference on machine learning. PMLR, 933\u2013941."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"Dettmers Tim","year":"2024","unstructured":"Tim Dettmers, Mike Lewis, Younes Belkada, and Luke Zettlemoyer. 2024. LLM.int8(): 8-bit matrix multiplication for transformers at scale. In Proceedings of the 36th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS '22). Curran Associates Inc., Red Hook, NY, USA, Article 2198, 15 pages."},{"key":"e_1_3_2_1_31_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_32_1","volume-title":"Sigmoid-weighted linear units for neural network function approximation in reinforcement learning. Neural networks 107","author":"Elfwing Stefan","year":"2018","unstructured":"Stefan Elfwing, Eiji Uchibe, and Kenji Doya. 2018. Sigmoid-weighted linear units for neural network function approximation in reinforcement learning. Neural networks 107 (2018), 3\u201311."},{"key":"e_1_3_2_1_33_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Frantar Elias","year":"2022","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2022. OPTQ: Accurate quantization for generative pre-trained transformers. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_34_1","volume-title":"GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers. arXiv:2210.17323 [cs.LG] https:\/\/arxiv.org\/abs\/2210.17323","author":"Frantar Elias","year":"2023","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2023. GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers. arXiv:2210.17323 [cs.LG] https:\/\/arxiv.org\/abs\/2210.17323"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the International Conference on Learning Representations.","author":"Frantar O.","unstructured":"O. Frantar and D. Alistarh. 2023. SparseGPT: A One-Shot Pruning Method for Neural Networks. In Proceedings of the International Conference on Learning Representations."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.12608602"},{"key":"e_1_3_2_1_37_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_38_1","volume-title":"Knowledge distillation of large language models. arXiv preprint arXiv:2306.08543","author":"Gu Yuxian","year":"2023","unstructured":"Yuxian Gu, Li Dong, Furu Wei, and Minlie Huang. 2023. Knowledge distillation of large language models. arXiv preprint arXiv:2306.08543 (2023)."},{"key":"e_1_3_2_1_39_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Guo Jinyang","year":"2024","unstructured":"Jinyang Guo, Jianyu Wu, Zining Wang, Jiaheng Liu, Ge Yang, Yifu Ding, Ruihao Gong, Haotong Qin, and Xianglong Liu. 2024. Compressing large language models by joint sparsification and quantization. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_40_1","volume-title":"How to Access Global Memory Efficiently in CUDA C\/C++ Kernels | NVIDIA Technical Blog. https:\/\/developer.nvidia.com\/blog\/how-access-global-memory-efficiently-cuda-c-kernels\/ [Online","author":"Harris Mark","year":"2024","unstructured":"Mark Harris. 2014. How to Access Global Memory Efficiently in CUDA C\/C++ Kernels | NVIDIA Technical Blog. https:\/\/developer.nvidia.com\/blog\/how-access-global-memory-efficiently-cuda-c-kernels\/ [Online; accessed 2024-11-01]."},{"key":"e_1_3_2_1_41_1","volume-title":"Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415","author":"Hendrycks Dan","year":"2016","unstructured":"Dan Hendrycks and Kevin Gimpel. 2016. Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415 (2016)."},{"key":"e_1_3_2_1_42_1","volume-title":"Faster large language model inference on gpus. arXiv preprint arXiv:2311.01282","author":"Hong Ke","year":"2023","unstructured":"Ke Hong, Guohao Dai, Jiaming Xu, Qiuli Mao, Xiuhong Li, Jun Liu, Kangdi Chen, Yuhan Dong, and Yu Wang. 2023. Flashdecoding++: Faster large language model inference on gpus. arXiv preprint arXiv:2311.01282 (2023)."},{"key":"e_1_3_2_1_43_1","volume-title":"Finequant: Unlocking efficiency with fine-grained weight-only quantization for llms. arXiv preprint arXiv:2308.09723","author":"Kim Young Jin","year":"2023","unstructured":"Young Jin Kim, Rawn Henry, Raffy Fahim, and Hany Hassan Awadalla. 2023. Finequant: Unlocking efficiency with fine-grained weight-only quantization for llms. arXiv preprint arXiv:2308.09723 (2023)."},{"key":"e_1_3_2_1_44_1","volume-title":"Self-normalizing neural networks. Advances in neural information processing systems 30","author":"Klambauer G\u00fcnter","year":"2017","unstructured":"G\u00fcnter Klambauer, Thomas Unterthiner, Andreas Mayr, and Sepp Hochreiter. 2017. Self-normalizing neural networks. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_46_1","volume-title":"Fran\u00e7ois Yvon, Matthias Gall\u00e9, et al.","author":"Scao Teven Le","year":"2023","unstructured":"Teven Le Scao, Angela Fan, Christopher Akiki, Ellie Pavlick, Suzana Ili\u0107, Daniel Hesslow, Roman Castagn\u00e9, Alexandra Sasha Luccioni, Fran\u00e7ois Yvon, Matthias Gall\u00e9, et al. 2023. Bloom: A 176b-parameter open-access multilingual language model. (2023)."},{"key":"e_1_3_2_1_47_1","volume-title":"International Conference on Machine Learning. PMLR","author":"Leviathan Yaniv","year":"2023","unstructured":"Yaniv Leviathan, Matan Kalman, and Yossi Matias. 2023. Fast inference from transformers via speculative decoding. In International Conference on Machine Learning. PMLR, 19274\u201319286."},{"key":"e_1_3_2_1_48_1","volume-title":"Eagle: Speculative sampling requires rethinking feature uncertainty. arXiv preprint arXiv:2401.15077","author":"Li Yuhui","year":"2024","unstructured":"Yuhui Li, Fangyun Wei, Chao Zhang, and Hongyang Zhang. 2024. Eagle: Speculative sampling requires rethinking feature uncertainty. arXiv preprint arXiv:2401.15077 (2024)."},{"key":"e_1_3_2_1_49_1","volume-title":"International Conference on Machine Learning. PMLR","author":"Li Yixiao","year":"2023","unstructured":"Yixiao Li, Yifan Yu, Qingru Zhang, Chen Liang, Pengcheng He, Weizhu Chen, and Tuo Zhao. 2023. Losparse: Structured compression of large language models based on low-rank and sparse approximation. In International Conference on Machine Learning. PMLR, 20336\u201320350."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.3390\/electronics12020267"},{"key":"e_1_3_2_1_51_1","first-page":"87","article-title":"AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration","volume":"6","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Wei-Ming Chen, Wei-Chen Wang, Guangxuan Xiao, Xingyu Dang, Chuang Gan, and Song Han. 2024. AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration. Proceedings of Machine Learning and Systems 6 (2024), 87\u2013100.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_52_1","volume-title":"International Conference on Machine Learning. PMLR, 22137\u201322176","author":"Liu Zichang","year":"2023","unstructured":"Zichang Liu, Jue Wang, Tri Dao, Tianyi Zhou, Binhang Yuan, Zhao Song, Anshumali Shrivastava, Ce Zhang, Yuandong Tian, Christopher Re, et al. 2023. Deja vu: Contextual sparsity for efficient llms at inference time. In International Conference on Machine Learning. PMLR, 22137\u201322176."},{"key":"e_1_3_2_1_53_1","volume-title":"Llm-pruner: On the structural pruning of large language models. Advances in neural information processing systems 36","author":"Ma Xinyin","year":"2023","unstructured":"Xinyin Ma, Gongfan Fang, and Xinchao Wang. 2023. Llm-pruner: On the structural pruning of large language models. Advances in neural information processing systems 36 (2023), 21702\u201321720."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.3115\/1075812.1075835"},{"key":"e_1_3_2_1_55_1","volume-title":"Pointer sentinel mixture models. arXiv preprint arXiv:1609.07843","author":"Merity Stephen","year":"2016","unstructured":"Stephen Merity, Caiming Xiong, James Bradbury, and Richard Socher. 2016. Pointer sentinel mixture models. arXiv preprint arXiv:1609.07843 (2016)."},{"key":"e_1_3_2_1_56_1","volume-title":"Oncel Tuzel, Golnoosh Samei, Mohammad Rastegari, and Mehrdad Farajtabar.","author":"Mirzadeh Iman","year":"2023","unstructured":"Iman Mirzadeh, Keivan Alizadeh, Sachin Mehta, Carlo C Del Mundo, Oncel Tuzel, Golnoosh Samei, Mohammad Rastegari, and Mehrdad Farajtabar. 2023. Relu strikes back: Exploiting activation sparsity in large language models. arXiv preprint arXiv:2310.04564 (2023)."},{"key":"e_1_3_2_1_57_1","volume-title":"ReLU Strikes Back: Exploiting Activation Sparsity in Large Language Models. In The Twelfth International Conference on Learning Representations.","author":"Mirzadeh Seyed Iman","year":"2024","unstructured":"Seyed Iman Mirzadeh, Keivan Alizadeh-Vahid, Sachin Mehta, Carlo C del Mundo, Oncel Tuzel, Golnoosh Samei, Mohammad Rastegari, and Mehrdad Farajtabar. 2024. ReLU Strikes Back: Exploiting Activation Sparsity in Large Language Models. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICM.2007.4497664"},{"key":"e_1_3_2_1_59_1","volume-title":"On estimation of a probability density function and mode. The annals of mathematical statistics 33, 3","author":"Parzen Emanuel","year":"1962","unstructured":"Emanuel Parzen. 1962. On estimation of a probability density function and mode. The annals of mathematical statistics 33, 3 (1962), 1065\u20131076."},{"key":"e_1_3_2_1_60_1","first-page":"1","article-title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J. Liu. 2020. Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer. Journal of Machine Learning Research 21, 140 (2020), 1\u201367. http:\/\/jmlr.org\/papers\/v21\/20-074.html","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_61_1","volume-title":"Matrix compression via randomized low rank and low precision factorization. Advances in Neural Information Processing Systems 36","author":"Saha Rajarshi","year":"2023","unstructured":"Rajarshi Saha, Varun Srivastava, and Mert Pilanci. 2023. Matrix compression via randomized low rank and low precision factorization. Advances in Neural Information Processing Systems 36 (2023)."},{"key":"e_1_3_2_1_62_1","unstructured":"Noam Shazeer. 2020. GLU Variants Improve Transformer. arXiv:2002.05202 [cs.LG] https:\/\/arxiv.org\/abs\/2002.05202"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695964"},{"key":"e_1_3_2_1_64_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Sun Mingjie","year":"2024","unstructured":"Mingjie Sun, Zhuang Liu, Anna Bair, and J Zico Kolter. 2024. A Simple and Effective Pruning Approach for Large Language Models. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_65_1","volume-title":"Variable kernel density estimation. The Annals of Statistics","author":"Terrell George R","year":"1992","unstructured":"George R Terrell and David W Scott. 1992. Variable kernel density estimation. The Annals of Statistics (1992), 1236\u20131265."},{"key":"e_1_3_2_1_66_1","volume-title":"Baby llama: knowledge distillation from an ensemble of teachers trained on a small dataset with no performance penalty. arXiv preprint arXiv:2308.02019","author":"Timiryasov Inar","year":"2023","unstructured":"Inar Timiryasov and Jean-Loup Tastet. 2023. Baby llama: knowledge distillation from an ensemble of teachers trained on a small dataset with no performance penalty. arXiv preprint arXiv:2308.02019 (2023)."},{"key":"e_1_3_2_1_67_1","volume-title":"https:\/\/en.wikipedia.org\/wiki\/Norm_(mathematics) [Online","author":"Wikipedia Contributors","year":"2024","unstructured":"Contributors to Wikimedia projects. 2004. Norm (mathematics) -Wikipedia. https:\/\/en.wikipedia.org\/wiki\/Norm_(mathematics) [Online; accessed 2024-11-13]."},{"key":"e_1_3_2_1_68_1","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar Aurelien Rodriguez Armand Joulin Edouard Grave and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. arXiv:2302.13971 [cs.CL] https:\/\/arxiv.org\/abs\/2302.13971"},{"key":"e_1_3_2_1_69_1","volume-title":"Model compression and efficient inference for large language models: A survey. arXiv preprint arXiv:2402.09748","author":"Wang Wenxiao","year":"2024","unstructured":"Wenxiao Wang, Wei Chen, Yicong Luo, Yongliu Long, Zhengkai Lin, Liye Zhang, Binbin Lin, Deng Cai, and Xiaofei He. 2024. Model compression and efficient inference for large language models: A survey. arXiv preprint arXiv:2402.09748 (2024)."},{"key":"e_1_3_2_1_70_1","volume-title":"Outlier suppression+: Accurate quantization of large language models by equivalent and optimal shifting and scaling. arXiv preprint arXiv:2304.09145","author":"Wei Xiuying","year":"2023","unstructured":"Xiuying Wei, Yunchen Zhang, Yuhang Li, Xiangguo Zhang, Ruihao Gong, Jinyang Guo, and Xianglong Liu. 2023. Outlier suppression+: Accurate quantization of large language models by equivalent and optimal shifting and scaling. arXiv preprint arXiv:2304.09145 (2023)."},{"key":"e_1_3_2_1_71_1","unstructured":"Mengwei Xu Wangsong Yin Dongqi Cai Rongjie Yi Daliang Xu Qipeng Wang Bingyang Wu Yihao Zhao Chen Yang Shihe Wang et al. 2024. A survey of resource-efficient llm and multimodal foundation models. arXiv preprint arXiv:2401.08092 (2024)."},{"key":"e_1_3_2_1_72_1","unstructured":"An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei et al. 2024. Qwen2. 5 Technical Report. arXiv preprint arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_73_1","volume-title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track.","author":"Yang Ge","year":"2024","unstructured":"Ge Yang, Changyi He, Jinyang Guo, Jianyu Wu, Yifu Ding, Aishan Liu, Haotong Qin, Pengliang Ji, and Xianglong Liu. 2024. LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_1_74_1","first-page":"27168","article-title":"Zeroquant: Efficient and affordable post-training quantization for large-scale transformers","volume":"35","author":"Yao Zhewei","year":"2022","unstructured":"Zhewei Yao, Reza Yazdani Aminabadi, Minjia Zhang, Xiaoxia Wu, Conglong Li, and Yuxiong He. 2022. Zeroquant: Efficient and affordable post-training quantization for large-scale transformers. Advances in Neural Information Processing Systems 35 (2022), 27168\u201327183.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_75_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"57115","author":"Yin Lu","year":"2024","unstructured":"Lu Yin, You Wu, Zhenyu Zhang, Cheng-Yu Hsieh, Yaqing Wang, Yiling Jia, Gen Li, Ajay Kumar Jaiswal, Mykola Pechenizkiy, Yi Liang, Michael Bendersky, Zhangyang Wang, and Shiwei Liu. 2024. Outlier Weighed Layerwise Sparsity (OWL): A Missing Secret Sauce for Pruning LLMs to High Sparsity. In Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 235), Ruslan Salakhutdinov, Zico Kolter, Katherine Heller, Adrian Weller, Nuria Oliver, Jonathan Scarlett, and Felix Berkenkamp (Eds.). PMLR, 57101\u201357115. https:\/\/proceedings.mlr.press\/v235\/yin24e.html"},{"key":"e_1_3_2_1_76_1","volume-title":"ShiftAddLLM: Accelerating Pretrained LLMs via Post-Training Multiplication-Less Reparameterization. arXiv preprint arXiv:2406.05981","author":"You Haoran","year":"2024","unstructured":"Haoran You, Yipin Guo, Yichao Fu, Wei Zhou, Huihong Shi, Xiaofan Zhang, Souvik Kundu, Amir Yazdanbakhsh, and Yingyan Celine Lin. 2024. ShiftAddLLM: Accelerating Pretrained LLMs via Post-Training Multiplication-Less Reparameterization. arXiv preprint arXiv:2406.05981 (2024)."},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICENCO48310.2019.9027479"},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.20944\/preprints202310.1487.v2"},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.blackboxnlp-1.29"},{"key":"e_1_3_2_1_80_1","volume-title":"QPruner: Probabilistic Decision Quantization for Structured Pruning in Large Language Models. arXiv preprint arXiv:2412.11629","author":"Zhou Changhai","year":"2024","unstructured":"Changhai Zhou, Yuhua Zhou, Shijie Han, Qian Qiao, and Hongguang Li. 2024. QPruner: Probabilistic Decision Quantization for Structured Pruning in Large Language Models. arXiv preprint arXiv:2412.11629 (2024)."},{"key":"e_1_3_2_1_81_1","unstructured":"Zixuan Zhou Xuefei Ning Ke Hong Tianyu Fu Jiaming Xu Shiyao Li Yuming Lou Luning Wang Zhihang Yuan Xiuhong Li et al. 2024. A survey on efficient inference for large language models. arXiv preprint arXiv:2404.14294 (2024)."}],"event":{"name":"EUROSYS '26: 21st European Conference on Computer Systems","location":"McEwan Hall\/The University of Edinburgh Edinburgh Scotland UK","acronym":"EUROSYS '26","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 21st European Conference on Computer Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3767295.3769339","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T12:01:56Z","timestamp":1780660916000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3767295.3769339"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,26]]},"references-count":81,"alternative-id":["10.1145\/3767295.3769339","10.1145\/3767295"],"URL":"https:\/\/doi.org\/10.1145\/3767295.3769339","relation":{},"subject":[],"published":{"date-parts":[[2026,4,26]]},"assertion":[{"value":"2026-04-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}