{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T08:01:26Z","timestamp":1780473686403,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62104128, U21B2031"],"award-info":[{"award-number":["62104128, U21B2031"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013105","name":"Shanghai Rising-Star Program","doi-asserted-by":"publisher","award":["24QB2706200"],"award-info":[{"award-number":["24QB2706200"]}],"id":[{"id":"10.13039\/501100013105","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,21]]},"DOI":"10.1145\/3695053.3730996","type":"proceedings-article","created":{"date-parts":[[2025,6,20]],"date-time":"2025-06-20T16:43:11Z","timestamp":1750437791000},"page":"467-481","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["SpecEE: Accelerating Large Language Model Inference with Speculative Early Exiting"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-7000-6537","authenticated-orcid":false,"given":"Jiaming","family":"Xu","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-0468-4592","authenticated-orcid":false,"given":"Jiayi","family":"Pan","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7732-6347","authenticated-orcid":false,"given":"Yongkang","family":"Zhou","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5931-7827","authenticated-orcid":false,"given":"Siming","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4286-6359","authenticated-orcid":false,"given":"Jinhao","family":"Li","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7858-5132","authenticated-orcid":false,"given":"Yaoxiu","family":"Lian","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-3437-2087","authenticated-orcid":false,"given":"Junyi","family":"Wu","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0849-3252","authenticated-orcid":false,"given":"Guohao","family":"Dai","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University; Infinigence-AI; SII, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"},{"key":"e_1_3_3_2_3_2","unstructured":"Tianle Cai Yuhong Li Zhengyang Geng Hongwu Peng Jason\u00a0D. Lee Deming Chen and Tri Dao. 2024. Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads. arxiv:https:\/\/arXiv.org\/abs\/2401.10774\u00a0[cs.LG]"},{"key":"e_1_3_3_2_4_2","unstructured":"CEIC. 2024. United States Monthly Earnings. [Online]. https:\/\/www.ceicdata.com\/en\/indicator\/united-states\/monthly-earnings."},{"key":"e_1_3_3_2_5_2","unstructured":"Charlie Chen Sebastian Borgeaud Geoffrey Irving Jean-Baptiste Lespiau Laurent Sifre and John Jumper. 2023. Accelerating large language model decoding with speculative sampling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.01318 (2023)."},{"key":"e_1_3_3_2_6_2","unstructured":"Mark Chen Jerry Tworek Heewoo Jun Qiming Yuan Henrique\u00a0Ponde de Oliveira\u00a0Pinto Jared Kaplan Harri Edwards Yuri Burda Nicholas Joseph Greg Brockman Alex Ray Raul Puri Gretchen Krueger Michael Petrov Heidy Khlaaf Girish Sastry Pamela Mishkin Brooke Chan Scott Gray Nick Ryder Mikhail Pavlov Alethea Power Lukasz Kaiser Mohammad Bavarian Clemens Winter Philippe Tillet Felipe\u00a0Petroski Such Dave Cummings Matthias Plappert Fotios Chantzis Elizabeth Barnes Ariel Herbert-Voss William\u00a0Hebgen Guss Alex Nichol Alex Paino Nikolas Tezak Jie Tang Igor Babuschkin Suchir Balaji Shantanu Jain William Saunders Christopher Hesse Andrew\u00a0N. Carr Jan Leike Josh Achiam Vedant Misra Evan Morikawa Alec Radford Matthew Knight Miles Brundage Mira Murati Katie Mayer Peter Welinder Bob McGrew Dario Amodei Sam McCandlish Ilya Sutskever and Wojciech Zaremba. 2021. Evaluating Large Language Models Trained on Code. arxiv:https:\/\/arXiv.org\/abs\/2107.03374\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2107.03374"},{"key":"e_1_3_3_2_7_2","unstructured":"Karl Cobbe Vineet Kosaraju Mohammad Bavarian Mark Chen Heewoo Jun Lukasz Kaiser Matthias Plappert Jerry Tworek Jacob Hilton Reiichiro Nakano Christopher Hesse and John Schulman. 2021. Training Verifiers to Solve Math Word Problems. arxiv:https:\/\/arXiv.org\/abs\/2110.14168\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2110.14168"},{"key":"e_1_3_3_2_8_2","unstructured":"Tri Dao. 2023. Flashattention-2: Faster attention with better parallelism and work partitioning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.08691 (2023)."},{"key":"e_1_3_3_2_9_2","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.03378","author":"Driess Danny","year":"2023","unstructured":"Danny Driess, Fei Xia, Mehdi S.\u00a0M. Sajjadi, Corey Lynch, Aakanksha Chowdhery, Brian Ichter, Ayzaan Wahid, Jonathan Tompson, Quan Vuong, Tianhe Yu, Wenlong Huang, Yevgen Chebotar, Pierre Sermanet, Daniel Duckworth, Sergey Levine, Vincent Vanhoucke, Karol Hausman, Marc Toussaint, Klaus Greff, Andy Zeng, Igor Mordatch, and Pete Florence. 2023. PaLM-E: An Embodied Multimodal Language Model. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.03378."},{"key":"e_1_3_3_2_10_2","unstructured":"Siqi Fan Xin Jiang Xiang Li Xuying Meng Peng Han Shuo Shang Aixin Sun Yequan Wang and Zhongyuan Wang. 2024. Not all layers of llms are necessary during inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.02181 (2024)."},{"key":"e_1_3_3_2_11_2","first-page":"10323","volume-title":"International Conference on Machine Learning","author":"Frantar Elias","year":"2023","unstructured":"Elias Frantar and Dan Alistarh. 2023. Sparsegpt: Massive language models can be accurately pruned in one-shot. In International Conference on Machine Learning. PMLR, 10323\u201310337."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2015.49"},{"key":"e_1_3_3_2_13_2","unstructured":"Yichao Fu Peter Bailis Ion Stoica and Hao Zhang. 2024. Break the Sequential Dependency of LLM Inference Using Lookahead Decoding. arxiv:https:\/\/arXiv.org\/abs\/2402.02057\u00a0[cs.LG]"},{"key":"e_1_3_3_2_14_2","unstructured":"Trevor Gale Deepak Narayanan Cliff Young and Matei Zaharia. 2022. MegaBlocks: Efficient Sparse Training with Mixture-of-Experts. arxiv:https:\/\/arXiv.org\/abs\/2211.15841\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2211.15841"},{"key":"e_1_3_3_2_15_2","unstructured":"Georgi Gerganov. 2023. LLM inference in C\/C++. [Online]. https:\/\/github.com\/ggerganov\/llama.cpp."},{"key":"e_1_3_3_2_16_2","unstructured":"Yizeng Han Gao Huang Shiji Song Le Yang Honghui Wang and Yulin Wang. 2021. Dynamic Neural Networks: A Survey. arxiv:https:\/\/arXiv.org\/abs\/2102.04906\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2102.04906"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"crossref","unstructured":"Marti\u00a0A. Hearst Susan\u00a0T Dumais Edgar Osuna John Platt and Bernhard Scholkopf. 1998. Support vector machines. IEEE Intelligent Systems and their applications 13 4 (1998) 18\u201328.","DOI":"10.1109\/5254.708428"},{"key":"e_1_3_3_2_18_2","unstructured":"Dan Hendrycks Collin Burns Steven Basart Andy Zou Mantas Mazeika Dawn Song and Jacob Steinhardt. 2021. Measuring Massive Multitask Language Understanding. arxiv:https:\/\/arXiv.org\/abs\/2009.03300\u00a0[cs.CY] https:\/\/arxiv.org\/abs\/2009.03300"},{"key":"e_1_3_3_2_19_2","unstructured":"Ke Hong Guohao Dai Jiaming Xu Qiuli Mao Xiuhong Li Jun Liu Yuhan Dong Yu Wang et\u00a0al. 2024. FlashDecoding++: Faster Large Language Model Inference with Asynchronization Flat GEMM Optimization and Heuristics. Proceedings of Machine Learning and Systems 6 (2024) 148\u2013161."},{"key":"e_1_3_3_2_20_2","unstructured":"Lianming Huang Shangyu Wu Yufei Cui Ying Xiong Xue Liu Tei-Wei Kuo Nan Guan and Chun\u00a0Jason Xue. 2024. RAEE: A Training-Free Retrieval-Augmented Early Exiting Framework for Efficient Inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.15198 (2024)."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Qingjia Huang Kai Shuang Peng Xu Jian Li Xu Liu and Sen Su. 2014. Prediction-based dynamic resource scheduling for virtualized cloud systems. Journal of Networks 9 2 (2014) 375.","DOI":"10.4304\/jnw.9.2.375-383"},{"key":"e_1_3_3_2_22_2","unstructured":"Juyong Jiang Fan Wang Jiasi Shen Sungju Kim and Sunghun Kim. 2024. A Survey on Large Language Models for Code Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.00515 (2024)."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"crossref","unstructured":"Tom Kwiatkowski Jennimaria Palomaki Olivia Redfield Michael Collins Ankur Parikh Chris Alberti Danielle Epstein Illia Polosukhin Jacob Devlin Kenton Lee et\u00a0al. 2019. Natural questions: a benchmark for question answering research. Transactions of the Association for Computational Linguistics 7 (2019) 453\u2013466.","DOI":"10.1162\/tacl_a_00276"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3469116.3470012"},{"key":"e_1_3_3_2_26_2","unstructured":"Jinhao Li Shiyao Li Jiaming Xu Shan Huang Yaoxiu Lian Jun Liu Yu Wang and Guohao Dai. 2023. Enabling Fast 2-bit LLM on GPUs: Memory Alignment Sparse Outlier and Asynchronous Dequantization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.16442 (2023)."},{"key":"e_1_3_3_2_27_2","unstructured":"Jinhao Li Jiaming Xu Shan Huang Yonghua Chen Wen Li Jun Liu Yaoxiu Lian Jiayi Pan Li Ding Hao Zhou Yu Wang and Guohao Dai. 2024. Large Language Model Inference Acceleration: A Comprehensive Hardware Perspective. arxiv:https:\/\/arXiv.org\/abs\/2410.04466\u00a0[cs.AR] https:\/\/arxiv.org\/abs\/2410.04466"},{"key":"e_1_3_3_2_28_2","volume-title":"International Conference on Machine Learning","author":"Li Yuhui","year":"2024","unstructured":"Yuhui Li, Fangyun Wei, Chao Zhang, and Hongyang Zhang. 2024. EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty. In International Conference on Machine Learning."},{"key":"e_1_3_3_2_29_2","unstructured":"Ji Lin Jiaming Tang Haotian Tang Shang Yang Wei-Ming Chen Wei-Chen Wang Guangxuan Xiao Xingyu Dang Chuang Gan and Song Han. 2024. AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration. arxiv:https:\/\/arXiv.org\/abs\/2306.00978\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2306.00978"},{"key":"e_1_3_3_2_30_2","unstructured":"Chi Ma Mincong Huang Ying Zhang Chao Wang Yujie Wang Lei Yu Chuan Liu and Wei Lin. 2024. First Activations Matter: Training-Free Methods for Dynamic Activation in Large Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.11393 (2024)."},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"crossref","unstructured":"George\u00a0A Miller and Walter\u00a0G Charles. 1991. Contextual correlates of semantic similarity. Language and cognitive processes 6 1 (1991) 1\u201328.","DOI":"10.1080\/01690969108406936"},{"key":"e_1_3_3_2_32_2","unstructured":"Iman Mirzadeh Keivan Alizadeh Sachin Mehta Carlo\u00a0C Del\u00a0Mundo Oncel Tuzel Golnoosh Samei Mohammad Rastegari and Mehrdad Farajtabar. 2023. Relu strikes back: Exploiting activation sparsity in large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.04564 (2023)."},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"crossref","unstructured":"Ramesh Nallapati Bowen Zhou Caglar Gulcehre Bing Xiang et\u00a0al. 2016. Abstractive text summarization using sequence-to-sequence rnns and beyond. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1602.06023 (2016).","DOI":"10.18653\/v1\/K16-1028"},{"key":"e_1_3_3_2_34_2","unstructured":"NVIDIA. 2017. CUTLASS: CUDA Templates for Linear Algebra Subroutines. [Online]. https:\/\/github.com\/NVIDIA\/cutlass."},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3190508.3190517"},{"key":"e_1_3_3_2_36_2","unstructured":"David Raposo Sam Ritter Blake Richards Timothy Lillicrap Peter\u00a0Conway Humphreys and Adam Santoro. 2024. Mixture-of-Depths: Dynamically allocating compute in transformer-based language models. arxiv:https:\/\/arXiv.org\/abs\/2404.02258\u00a0[cs.LG]"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"Frank Rosenblatt. 1958. The perceptron: a probabilistic model for information storage and organization in the brain. Psychological review 65 6 (1958) 386.","DOI":"10.1037\/h0042519"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1170"},{"key":"e_1_3_3_2_39_2","unstructured":"Yixin Song Zeyu Mi Haotong Xie and Haibo Chen. 2023. PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU. arxiv:https:\/\/arXiv.org\/abs\/2312.12456\u00a0[cs.LG]"},{"key":"e_1_3_3_2_40_2","unstructured":"Alon Talmor Jonathan Herzig Nicholas Lourie and Jonathan Berant. 2019. CommonsenseQA: A Question Answering Challenge Targeting Commonsense Knowledge. arxiv:https:\/\/arXiv.org\/abs\/1811.00937\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/1811.00937"},{"key":"e_1_3_3_2_41_2","unstructured":"Rohan Taori Ishaan Gulrajani Tianyi Zhang Yann Dubois Xuechen Li Carlos Guestrin Percy Liang and Tatsunori\u00a0B. Hashimoto. 2023. Stanford Alpaca: An Instruction-following LLaMA model. https:\/\/github.com\/tatsu-lab\/stanford_alpaca."},{"key":"e_1_3_3_2_42_2","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et\u00a0al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.09288 (2023)."},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_3_2_44_2","unstructured":"Qingyun Wu Gagan Bansal Jieyu Zhang Yiran Wu Shaokun Zhang Erkang Zhu Beibin Li Li Jiang Xiaoyun Zhang and Chi Wang. 2023. Autogen: Enabling next-gen llm applications via multi-agent conversation framework. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.08155 (2023)."},{"key":"e_1_3_3_2_45_2","unstructured":"xAI. 2024. Open Release of Grok-1. [Online]. https:\/\/github.com\/xai-org\/grok-1."},{"key":"e_1_3_3_2_46_2","volume-title":"The Thirty-eighth Annual Conference on Neural Information Processing Systems","author":"jiang yikun","year":"2024","unstructured":"yikun jiang, Huanyu Wang, Lei Xie, Hanbin Zhao, Chao Zhang, Hui Qian, and John\u00a0C.S. Lui. 2024. D-LLM: A Token Adaptive Computing Resource Allocation Strategy for Large Language Models. In The Thirty-eighth Annual Conference on Neural Information Processing Systems. https:\/\/openreview.net\/forum?id=UIOjGTKHQG"},{"key":"e_1_3_3_2_47_2","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric\u00a0P. Xing Hao Zhang Joseph\u00a0E. Gonzalez and Ion Stoica. 2023. Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena. arxiv:https:\/\/arXiv.org\/abs\/2306.05685\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2306.05685"},{"key":"e_1_3_3_2_48_2","unstructured":"Zixuan Zhou Xuefei Ning Ke Hong Tianyu Fu Jiaming Xu Shiyao Li Yuming Lou Luning Wang Zhihang Yuan Xiuhong Li et\u00a0al. 2024. A survey on efficient inference for large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.14294 (2024)."},{"key":"e_1_3_3_2_49_2","unstructured":"Zodhya. 2023. Optimizing Inference on Large Language Models with NVIDIA TensorRT-LLM Now Publicly Available. [Online]. https:\/\/medium.com\/@zodhyatech\/how-much-energy-does-chatgpt-consume-4cba1a7aef85."}],"event":{"name":"ISCA '25: Proceedings of the 52nd Annual International Symposium on Computer Architecture","location":"Tokyo Japan","acronym":"SIGARCH '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 52nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3695053.3730996","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T10:56:28Z","timestamp":1750503388000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3695053.3730996"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,20]]},"references-count":48,"alternative-id":["10.1145\/3695053.3730996","10.1145\/3695053"],"URL":"https:\/\/doi.org\/10.1145\/3695053.3730996","relation":{},"subject":[],"published":{"date-parts":[[2025,6,20]]},"assertion":[{"value":"2025-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}