{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:57:46Z","timestamp":1776931066622,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":67,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62322201, U23B2020"],"award-info":[{"award-number":["62322201, U23B2020"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["JKF-20240598, JKF-2025012343648"],"award-info":[{"award-number":["JKF-20240598, JKF-2025012343648"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,16]]},"DOI":"10.1145\/3712285.3759834","type":"proceedings-article","created":{"date-parts":[[2025,11,12]],"date-time":"2025-11-12T16:05:39Z","timestamp":1762963539000},"page":"973-990","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Towards Efficient LLM Inference via Collective and Adaptive Speculative Decoding"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-0761-3287","authenticated-orcid":false,"given":"Siqi","family":"Wang","sequence":"first","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1101-7927","authenticated-orcid":false,"given":"Hailong","family":"Yang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7293-545X","authenticated-orcid":false,"given":"Xuezhu","family":"Wang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2634-2788","authenticated-orcid":false,"given":"Tongxuan","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7197-2873","authenticated-orcid":false,"given":"Pengbo","family":"Wang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7787-6460","authenticated-orcid":false,"given":"Yufan","family":"Xu","sequence":"additional","affiliation":[{"name":"Independent Researcher, Cupertino, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7275-201X","authenticated-orcid":false,"given":"Xuning","family":"Liang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-0393-7627","authenticated-orcid":false,"given":"Kejie","family":"Ma","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4830-5482","authenticated-orcid":false,"given":"Tianyu","family":"Feng","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5163-4607","authenticated-orcid":false,"given":"Xin","family":"You","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0081-5395","authenticated-orcid":false,"given":"Ruihao","family":"Gong","sequence":"additional","affiliation":[{"name":"SenseTime Research, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2741-6033","authenticated-orcid":false,"given":"Rui","family":"Wang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7186-0556","authenticated-orcid":false,"given":"Zhongzhi","family":"Luan","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1829-2817","authenticated-orcid":false,"given":"Yi","family":"Liu","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5382-1473","authenticated-orcid":false,"given":"Depei","family":"Qian","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,15]]},"reference":[{"key":"e_1_3_3_3_2_2","unstructured":"Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia\u00a0Leoni Aleman Diogo Almeida Janko Altenschmidt Sam Altman Shyamal Anadkat et\u00a0al. 2023. GPT-4 Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.08774 (2023)."},{"key":"e_1_3_3_3_3_2","unstructured":"Daniel Adiwardana Minh-Thang Luong David\u00a0R So Jamie Hall Noah Fiedel Romal Thoppilan Zi Yang Apoorv Kulshreshtha Gaurav Nemade Yifeng Lu et\u00a0al. 2020. Towards a human-like open-domain chatbot. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2001.09977 (2020)."},{"key":"e_1_3_3_3_4_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1160"},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"crossref","unstructured":"Branden Butler Sixing Yu Arya Mazaheri and Ali Jannesari. 2024. PipeInfer: Accelerating LLM Inference using Asynchronous Pipelined Speculation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.11798 (2024).","DOI":"10.1109\/SC41406.2024.00046"},{"key":"e_1_3_3_3_7_2","unstructured":"Tianle Cai Yuhong Li Zhengyang Geng Hongwu Peng Jason\u00a0D Lee Deming Chen and Tri Dao. 2024. Medusa: Simple llm inference acceleration framework with multiple decoding heads. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.10774 (2024)."},{"key":"e_1_3_3_3_8_2","unstructured":"Shiyi Cao Shu Liu Tyler Griggs Peter Schafhalter Xiaoxuan Liu Ying Sheng Joseph\u00a0E Gonzalez Matei Zaharia and Ion Stoica. 2025. Moe-lightning: High-throughput moe inference on memory-constrained gpus. (2025)."},{"key":"e_1_3_3_3_9_2","unstructured":"Charlie Chen Sebastian Borgeaud Geoffrey Irving Jean-Baptiste Lespiau Laurent Sifre and John Jumper. 2023. Accelerating large language model decoding with speculative sampling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.01318 (2023)."},{"key":"e_1_3_3_3_10_2","unstructured":"Mark Chen Jerry Tworek Heewoo Jun Qiming Yuan Henrique Ponde de\u00a0Oliveira Pinto Jared Kaplan Harri Edwards Yuri Burda Nicholas Joseph Greg Brockman et\u00a0al. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2107.03374 (2021)."},{"key":"e_1_3_3_3_11_2","unstructured":"Mark Chen Jerry Tworek Heewoo Jun Qiming Yuan Henrique Ponde De\u00a0Oliveira Pinto Jared Kaplan Harri Edwards Yuri Burda Nicholas Joseph Greg Brockman et\u00a0al. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2107.03374 (2021)."},{"key":"e_1_3_3_3_12_2","unstructured":"Ziyi Chen Xiaocong Yang Jiacheng Lin Chenkai Sun Jie Huang and Kevin Chen-Chuan Chang. 2023. Cascade speculative drafting for even faster llm inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.11462 (2023)."},{"key":"e_1_3_3_3_13_2","unstructured":"Karl Cobbe Vineet Kosaraju Mohammad Bavarian Mark Chen Heewoo Jun Lukasz Kaiser Matthias Plappert Jerry Tworek Jacob Hilton Reiichiro Nakano et\u00a0al. 2021. Training verifiers to solve math word problems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2110.14168 (2021)."},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"crossref","unstructured":"Alex de Vries. 2023. The growing energy footprint of artificial intelligence. Joule 7 10 (2023) 2191\u20132194.","DOI":"10.1016\/j.joule.2023.09.004"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"crossref","unstructured":"Tim Dettmers Artidoro Pagnoni Ari Holtzman and Luke Zettlemoyer. 2023. Qlora: Efficient finetuning of quantized llms. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.14314 (2023).","DOI":"10.52202\/075280-0441"},{"key":"e_1_3_3_3_16_2","unstructured":"Cunxiao Du Jing Jiang Xu Yuanchen Jiawei Wu Sicheng Yu Yongqi Li Shenggui Li Kai Xu Liqiang Nie Zhaopeng Tu et\u00a0al. 2024. Glide with a cape: A low-hassle method to accelerate speculative decoding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.02082 (2024)."},{"key":"e_1_3_3_3_17_2","unstructured":"Zhixu Du Shiyu Li Yuhao Wu Xiangyu Jiang Jingwei Sun Qilin Zheng Yongkai Wu Ang Li Hai Li and Yiran Chen. 2024. Sida: Sparsity-inspired data-aware serving for efficient and scalable large mixture-of-experts models. Proceedings of Machine Learning and Systems 6 (2024) 224\u2013238."},{"key":"e_1_3_3_3_18_2","unstructured":"FlexFlow. 2023. Llama-160M https:\/\/huggingface.co\/JackFram\/llama-160m."},{"key":"e_1_3_3_3_19_2","first-page":"10323","volume-title":"International Conference on Machine Learning","author":"Frantar Elias","year":"2023","unstructured":"Elias Frantar and Dan Alistarh. 2023. Sparsegpt: Massive language models can be accurately pruned in one-shot. In International Conference on Machine Learning. PMLR, 10323\u201310337."},{"key":"e_1_3_3_3_20_2","volume-title":"The Eleventh International Conference on Learning Representations","author":"Frantar Elias","year":"2022","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2022. OPTQ: Accurate quantization for generative pre-trained transformers. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_3_3_21_2","unstructured":"Yichao Fu Peter Bailis Ion Stoica and Hao Zhang. 2024. Break the sequential dependency of llm inference using lookahead decoding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.02057 (2024)."},{"key":"e_1_3_3_3_22_2","unstructured":"Zhenyu He Zexuan Zhong Tianle Cai Jason\u00a0D Lee and Di He. 2023. Rest: Retrieval-based speculative decoding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.08252 (2023)."},{"key":"e_1_3_3_3_23_2","unstructured":"Coleman Hooper Sehoon Kim Hiva Mohammadzadeh Hasan Genc Kurt Keutzer Amir Gholami and Sophia Shao. 2023. Speed: Speculative pipelined execution for efficient decoding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.12072 (2023)."},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"crossref","unstructured":"Cheng-Yu Hsieh Chun-Liang Li Chih-Kuan Yeh Hootan Nakhost Yasuhisa Fujii Alexander Ratner Ranjay Krishna Chen-Yu Lee and Tomas Pfister. 2023. Distilling step-by-step! outperforming larger language models with less training data and smaller model sizes. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.02301 (2023).","DOI":"10.18653\/v1\/2023.findings-acl.507"},{"key":"e_1_3_3_3_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378530"},{"key":"e_1_3_3_3_26_2","unstructured":"HuggingFace. 2023. Chatbot https:\/\/huggingface.co\/datasets\/alespalla\/_chatbot_instruction_prompts."},{"key":"e_1_3_3_3_27_2","unstructured":"HuggingFace. 2023. Finance https:\/\/huggingface.co\/datasets\/gbharti\/finance-alpaca."},{"key":"e_1_3_3_3_28_2","unstructured":"HuggingFace. 2023. Text Generation Inference https:\/\/github.com\/huggingface\/text-generation-inference."},{"key":"e_1_3_3_3_29_2","unstructured":"HuggingFace. 2024. Summary Instruct https:\/\/huggingface.co\/datasets\/_ibivibiv\/summary_instruct."},{"key":"e_1_3_3_3_30_2","unstructured":"HuggingFace. 2024. Writing https:\/\/huggingface.co\/datasets\/Nitral-AI\/Creative_Writing-ShareGPT."},{"key":"e_1_3_3_3_31_2","unstructured":"Sehoon Kim Karttikeya Mangalam Suhong Moon Jitendra Malik Michael\u00a0W Mahoney Amir Gholami and Kurt Keutzer. 2024. Speculative decoding with big little decoder. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_3_33_2","first-page":"19274","volume-title":"International Conference on Machine Learning","author":"Leviathan Yaniv","year":"2023","unstructured":"Yaniv Leviathan, Matan Kalman, and Yossi Matias. 2023. Fast inference from transformers via speculative decoding. In International Conference on Machine Learning. PMLR, 19274\u201319286."},{"key":"e_1_3_3_3_34_2","unstructured":"Yuhui Li Fangyun Wei Chao Zhang and Hongyang Zhang. 2024. Eagle: Speculative sampling requires rethinking feature uncertainty. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.15077 (2024)."},{"key":"e_1_3_3_3_35_2","unstructured":"Ji Lin Jiaming Tang Haotian Tang Shang Yang Xingyu Dang and Song Han. 2023. AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.00978 (2023)."},{"key":"e_1_3_3_3_36_2","doi-asserted-by":"crossref","unstructured":"Nick Littlestone and Manfred\u00a0K Warmuth. 1994. The weighted majority algorithm. Information and computation 108 2 (1994) 212\u2013261.","DOI":"10.1006\/inco.1994.1009"},{"key":"e_1_3_3_3_37_2","unstructured":"Meta. 2022. OPT-125M https:\/\/huggingface.co\/facebook\/opt-125m."},{"key":"e_1_3_3_3_38_2","unstructured":"Meta. 2022. OPT-13B https:\/\/huggingface.co\/facebook\/opt-13b."},{"key":"e_1_3_3_3_39_2","unstructured":"Meta. 2023. Llama2-70B-chat https:\/\/huggingface.co\/meta-llama\/Llama-2-70b-chat-hf."},{"key":"e_1_3_3_3_40_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651335"},{"key":"e_1_3_3_3_41_2","unstructured":"Microsoft. 2023. DeepSpeed FastGen https:\/\/github.com\/microsoft\/DeepSpeed\/_tree\/master\/blogs\/deepspeed-fastgen."},{"key":"e_1_3_3_3_42_2","doi-asserted-by":"crossref","unstructured":"Ramesh Nallapati Bowen Zhou Caglar Gulcehre Bing Xiang et\u00a0al. 2016. Abstractive text summarization using sequence-to-sequence rnns and beyond. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1602.06023 (2016).","DOI":"10.18653\/v1\/K16-1028"},{"key":"e_1_3_3_3_43_2","unstructured":"NVIDIA. 2012. NVIDIA: Sharing a gpu between mpi processes: multiple-process service https:\/\/docs.nvidia.com\/deploy\/mps\/index.html."},{"key":"e_1_3_3_3_44_2","unstructured":"NVIDIA. 2023. TensorRT-LLM https:\/\/github.com\/NVIDIA\/TensorRT-LLM."},{"key":"e_1_3_3_3_45_2","unstructured":"OpenAI. 2023. OpenAI Pricing https:\/\/openai.com\/pricing."},{"key":"e_1_3_3_3_46_2","unstructured":"Romain Paulus Caiming Xiong and Richard Socher. 2017. A deep reinforced model for abstractive summarization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1705.04304 (2017)."},{"key":"e_1_3_3_3_47_2","unstructured":"Baolin Peng Chunyuan Li Pengcheng He Michel Galley and Jianfeng Gao. 2023. Instruction tuning with gpt-4. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.03277 (2023)."},{"key":"e_1_3_3_3_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378505"},{"key":"e_1_3_3_3_49_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1534"},{"key":"e_1_3_3_3_50_2","unstructured":"Stephen Roller Emily Dinan Naman Goyal Da Ju Mary Williamson Yinhan Liu Jing Xu Myle Ott Kurt Shuster Eric\u00a0M Smith et\u00a0al. 2020. Recipes for building an open-domain chatbot. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2004.13637 (2020)."},{"key":"e_1_3_3_3_51_2","unstructured":"Victor Sanh Thomas Wolf and Alexander Rush. 2020. Movement pruning: Adaptive sparsity by fine-tuning. Advances in Neural Information Processing Systems 33 (2020) 20378\u201320389."},{"key":"e_1_3_3_3_52_2","unstructured":"Abigail See Peter\u00a0J Liu and Christopher\u00a0D Manning. 2017. Get to the point: Summarization with pointer-generator networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1704.04368 (2017)."},{"key":"e_1_3_3_3_53_2","unstructured":"Ryan Smith Jason\u00a0A Fries Braden Hancock and Stephen\u00a0H Bach. 2022. Language models in the loop: Incorporating prompting into weak supervision. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.02318 (2022)."},{"key":"e_1_3_3_3_54_2","unstructured":"Mitchell Stern Noam Shazeer and Jakob Uszkoreit. 2018. Blockwise parallel decoding for deep autoregressive models. Advances in Neural Information Processing Systems 31 (2018)."},{"key":"e_1_3_3_3_55_2","unstructured":"Biao Sun Ziming Huang Hanyu Zhao Wencong Xiao Xinyi Zhang Yong Li and Wei Lin. 2024. Llumnix: Dynamic Scheduling for Large Language Model Serving. 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24) (2024)."},{"key":"e_1_3_3_3_56_2","unstructured":"Ziteng Sun Ananda\u00a0Theertha Suresh Jae\u00a0Hun Ro Ahmad Beirami Himanshu Jain and Felix Yu. 2024. Spectr: Fast speculative decoding via optimal transport. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_3_57_2","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et\u00a0al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.09288 (2023)."},{"key":"e_1_3_3_3_58_2","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_3_59_2","first-page":"2048","volume-title":"International conference on machine learning","author":"Xu Kelvin","year":"2015","unstructured":"Kelvin Xu, Jimmy Ba, Ryan Kiros, Kyunghyun Cho, Aaron Courville, Ruslan Salakhudinov, Rich Zemel, and Yoshua Bengio. 2015. Show, attend and tell: Neural image caption generation with visual attention. In International conference on machine learning. PMLR, 2048\u20132057."},{"key":"e_1_3_3_3_60_2","unstructured":"Zhilin Yang Ye Yuan Yuexin Wu William\u00a0W Cohen and Russ\u00a0R Salakhutdinov. 2016. Review networks for caption generation. Advances in neural information processing systems 29 (2016)."},{"key":"e_1_3_3_3_61_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS57955.2024.00086"},{"key":"e_1_3_3_3_62_2","doi-asserted-by":"crossref","unstructured":"Zhewei Yao Reza Yazdani\u00a0Aminabadi Minjia Zhang Xiaoxia Wu Conglong Li and Yuxiong He. 2022. Zeroquant: Efficient and affordable post-training quantization for large-scale transformers. Advances in Neural Information Processing Systems 35 (2022) 27168\u201327183.","DOI":"10.52202\/068431-1970"},{"key":"e_1_3_3_3_63_2","unstructured":"Jun Zhang Jue Wang Huan Li Lidan Shou Ke Chen Gang Chen and Sharad Mehrotra. 2023. Draft & verify: Lossless large language model acceleration via self-speculative decoding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.08168 (2023)."},{"key":"e_1_3_3_3_64_2","unstructured":"Susan Zhang Stephen Roller Naman Goyal Mikel Artetxe Moya Chen Shuohui Chen Christopher Dewan Mona Diab Xian Li Xi\u00a0Victoria Lin et\u00a0al. 2022. Opt: Open pre-trained transformer language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.01068 (2022)."},{"key":"e_1_3_3_3_65_2","unstructured":"Weilin Zhao Yuxiang Huang Xu Han Chaojun Xiao Zhiyuan Liu and Maosong Sun. 2024. Ouroboros: Speculative Decoding with Large Model Enhanced Drafting. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.13720 (2024)."},{"key":"e_1_3_3_3_66_2","unstructured":"Wayne\u00a0Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong et\u00a0al. 2023. A survey of large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.18223 (2023)."},{"key":"e_1_3_3_3_67_2","unstructured":"Yinmin Zhong Shengyu Liu Junda Chen Jianbo Hu Yibo Zhu Xuanzhe Liu Xin Jin and Hao Zhang. 2024. DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24) (2024)."},{"key":"e_1_3_3_3_68_2","unstructured":"Yongchao Zhou Kaifeng Lyu Ankit\u00a0Singh Rawat Aditya\u00a0Krishna Menon Afshin Rostamizadeh Sanjiv Kumar Jean-Fran\u00e7ois Kagy and Rishabh Agarwal. 2023. Distillspec: Improving speculative decoding via knowledge distillation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.08461 (2023)."}],"event":{"name":"SC '25: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St. Louis MO USA","acronym":"SC '25","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3712285.3759834","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T18:35:50Z","timestamp":1773254150000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3712285.3759834"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,15]]},"references-count":67,"alternative-id":["10.1145\/3712285.3759834","10.1145\/3712285"],"URL":"https:\/\/doi.org\/10.1145\/3712285.3759834","relation":{},"subject":[],"published":{"date-parts":[[2025,11,15]]},"assertion":[{"value":"2025-11-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}