{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T05:12:11Z","timestamp":1783746731094,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T00:00:00Z","timestamp":1783900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Department of Energy (DOE), Office of Advanced Scientific Computing Research (ASCR)","award":["DEAC02\u201306CH11357\\\/0F\u201360169"],"award-info":[{"award-number":["DEAC02\u201306CH11357\\\/0F\u201360169"]}]},{"name":"National Science Foundation (NSF), Office of Advanced Cyberinfrastructure","award":["CSSI-2411318\u200c Core-2313154\u200c CSSI-2104013\u200c 2411386\\\/2411387\u200c 2106635"],"award-info":[{"award-number":["CSSI-2411318\u200c Core-2313154\u200c CSSI-2104013\u200c 2411386\\\/2411387\u200c 2106635"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,13]]},"DOI":"10.1145\/3806645.3807819","type":"proceedings-article","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:21:11Z","timestamp":1783743671000},"page":"402-414","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["PKAS: Predictive KVCache-Aware Scheduling for Faster LLM and Transformer Inferences"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-3985-7896","authenticated-orcid":false,"given":"Jie","family":"Ye","sequence":"first","affiliation":[{"name":"Illinois Institute of Technology, Chicago, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8200-0148","authenticated-orcid":false,"given":"Avinash","family":"Maurya","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory, Lemont, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3027-1915","authenticated-orcid":false,"given":"Krishna Teja","family":"Chitty-Venkata","sequence":"additional","affiliation":[{"name":"Red Hat, Inc., Boston, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0661-7509","authenticated-orcid":false,"given":"Bogdan","family":"Nicolae","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory, Lemont, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3943-663X","authenticated-orcid":false,"given":"Anthony","family":"Kougkas","sequence":"additional","affiliation":[{"name":"Illinois Institute of Technology, Chicago, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1093-0792","authenticated-orcid":false,"given":"Xian-He","family":"Sun","sequence":"additional","affiliation":[{"name":"Illinois Institute of Technology, Chicago, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,13]]},"reference":[{"key":"e_1_3_3_2_2_2","first-page":"117","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Agrawal Amey","year":"2024","unstructured":"Amey Agrawal, Nitin Kedia, Ashish Panwar, Jayashree Mohan, Nipun Kwatra, Bhargav Gulavani, Alexey Tumanov, and Ramachandran Ramjee. 2024. Taming { Throughput-Latency} Tradeoff in { LLM} Inference with { Sarathi-Serve}. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). USENIX Association, Santa Clara, CA, USA, 117\u2013134."},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.776"},{"key":"e_1_3_3_2_4_2","volume-title":"ShareGPT_Vicuna_unfiltered","year":"2023","unstructured":"anon8231489123. 2023. ShareGPT_Vicuna_unfiltered. Hugging Face. https:\/\/huggingface.co\/datasets\/anon8231489123\/ShareGPT_Vicuna_unfiltered\/tree\/main"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.172"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.390"},{"key":"e_1_3_3_2_7_2","unstructured":"S\u00e9bastien Bubeck Varun Chandrasekaran Ronen Eldan Johannes Gehrke Eric Horvitz Ece Kamar Peter Lee Yin\u00a0Tat Lee Yuanzhi Li Scott Lundberg et\u00a0al. 2023. Sparks of Artificial General Intelligence: Early Experiments with GPT-4. arxiv:https:\/\/arXiv.org\/abs\/2303.12712\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2303.12712"},{"key":"e_1_3_3_2_8_2","unstructured":"Daniel Castro. 2024. Rethinking Concerns About AI\u2019s Energy Use. https:\/\/www2.datainnovation.org\/2024-ai-energy-use.pdf"},{"key":"e_1_3_3_2_9_2","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et\u00a0al. 2025. Gemini 2.5: Pushing the Frontier with Advanced Reasoning Multimodality Long Context and Next Generation Agentic Capabilities. arxiv:https:\/\/arXiv.org\/abs\/2507.06261\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2507.06261"},{"key":"e_1_3_3_2_10_2","volume-title":"Text Generation Inference","author":"Face Hugging","year":"2023","unstructured":"Hugging Face. 2023. Text Generation Inference. Hugging Face. https:\/\/github.com\/huggingface\/text-generation-inference"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"crossref","unstructured":"Yichao Fu Siqi Zhu Runlong Su Aurick Qiao Ion Stoica and Hao Zhang. 2024. Efficient llm scheduling by learning to rank. Advances in Neural Information Processing Systems 37 (2024) 59006\u201359029.","DOI":"10.52202\/079017-1882"},{"key":"e_1_3_3_2_12_2","unstructured":"Kanishk Goel Jayashree Mohan Nipun Kwatra Ravi\u00a0Shreyas Anupindi and Ramachandran Ramjee. 2025. Niyama: Breaking the Silos of LLM Inference Serving. arxiv:https:\/\/arXiv.org\/abs\/2503.22562\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2503.22562"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"crossref","unstructured":"Jianping Gou Baosheng Yu Stephen\u00a0J Maybank and Dacheng Tao. 2021. Knowledge distillation: A survey. International Journal of Computer Vision 129 6 (2021) 1789\u20131819.","DOI":"10.1007\/s11263-021-01453-z"},{"key":"e_1_3_3_2_14_2","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et\u00a0al. 2024. The Llama 3 Herd of Models. arxiv:https:\/\/arXiv.org\/abs\/2407.21783\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_3_2_15_2","unstructured":"Connor Holmes Masahiro Tanaka Michael Wyatt Ammar\u00a0Ahmad Awan Jeff Rasley Samyam Rajbhandari Reza\u00a0Yazdani Aminabadi Heyang Qin Arash Bakhtiari Lev Kurilenko et\u00a0al. 2024. DeepSpeed-FastGen: High-Throughput Text Generation for LLMs via MII and DeepSpeed-Inference. arxiv:https:\/\/arXiv.org\/abs\/2401.08671\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2401.08671"},{"key":"e_1_3_3_2_16_2","unstructured":"Cunchen Hu Heyang Huang Liangliang Xu Xusheng Chen Jiang Xu Shuang Chen Hao Feng Chenxi Wang Sa Wang Yungang Bao et\u00a0al. 2024. Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads. arxiv:https:\/\/arXiv.org\/abs\/2401.11181\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2401.11181"},{"key":"e_1_3_3_2_17_2","unstructured":"Azam Ikram Xiang Li Sameh Elnikety and Saurabh Bagchi. 2025. Ascendra: Dynamic Request Prioritization for Efficient LLM Serving. arxiv:https:\/\/arXiv.org\/abs\/2504.20828\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2504.20828"},{"key":"e_1_3_3_2_18_2","unstructured":"Albert\u00a0Q. Jiang Alexandre Sablayrolles Antoine Roux Arthur Mensch Blanche Savary Chris Bamford Devendra\u00a0Singh Chaplot Diego de\u00a0las Casas Emma\u00a0Bou Hanna Florian Bressand et\u00a0al. 2024. Mixtral of Experts. arxiv:https:\/\/arXiv.org\/abs\/2401.04088\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2401.04088"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"crossref","unstructured":"Chao Jin Zili Zhang Xuanlin Jiang Fangyue Liu Shufan Liu Xuanzhe Liu and Xin Jin. 2025. Ragcache: Efficient knowledge caching for retrieval-augmented generation. ACM Transactions on Computer Systems 44 1 (2025) 1\u201327.","DOI":"10.1145\/3768628"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Yunho Jin Chun-Feng Wu David Brooks and Gu-Yeon Wei. 2023. S3: Increasing GPU Utilization during Generative Inference for Higher Throughput. Advances in Neural Information Processing Systems 36 (2023) 18015\u201318027.","DOI":"10.52202\/075280-0791"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_22_2","unstructured":"Aixin Liu Bei Feng Bing Xue Bingxuan Wang Bochao Wu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan et\u00a0al. 2024. DeepSeek-V3 Technical Report. arxiv:https:\/\/arXiv.org\/abs\/2412.19437\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2412.19437"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.3386\/w32381"},{"key":"e_1_3_3_2_24_2","volume-title":"FastTransformer","year":"2024","unstructured":"Nvidia. 2024. FastTransformer. NVIDIA Corporation. https:\/\/github.com\/NVIDIA\/FasterTransformer"},{"key":"e_1_3_3_2_25_2","volume-title":"NVIDIA Triton Dynamic Batching","year":"2024","unstructured":"NVIDIA. 2024. NVIDIA Triton Dynamic Batching. NVIDIA Corporation. https:\/\/docs.nvidia.com\/deeplearning\/triton-inference-server\/user-guide\/docs\/user_guide\/model_configuration.html#dynamic-batcher"},{"key":"e_1_3_3_2_26_2","volume-title":"TensorRT-LLM","year":"2024","unstructured":"NVIDIA. 2024. TensorRT-LLM. NVIDIA Corporation. https:\/\/github.com\/NVIDIA\/TensorRT-LLM Accessed: 2026-04-30."},{"key":"e_1_3_3_2_27_2","unstructured":"Reiner Pope Sholto Douglas Aakanksha Chowdhery Jacob Devlin James Bradbury Jonathan Heek Kefan Xiao Shivani Agrawal and Jeff Dean. 2023. Efficiently scaling transformer inference. Proceedings of machine learning and systems 5 (2023) 606\u2013624."},{"key":"e_1_3_3_2_28_2","unstructured":"Haoran Qiu Weichao Mao Archit Patke Shengkun Cui Saurabh Jha Chen Wang Hubertus Franke Zbigniew\u00a0T. Kalbarczyk Tamer Ba\u015far and Ravishankar\u00a0K. Iyer. 2024. Efficient Interactive LLM Serving with Proxy Model-Based Sequence Length Prediction. arxiv:https:\/\/arXiv.org\/abs\/2404.08509\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2404.08509"},{"key":"e_1_3_3_2_29_2","volume-title":"The Thirteenth International Conference on Learning Representations (ICLR \u201925)","author":"Shahout Rana","year":"2025","unstructured":"Rana Shahout, Eran Malach, Chunwei Liu, Weifan Jiang, Minlan Yu, and Michael Mitzenmacher. 2025. Don\u2019t Stop Me Now: Embedding-Based Scheduling for LLMs. In The Thirteenth International Conference on Learning Representations (ICLR \u201925). OpenReview.net, Singapore. https:\/\/openreview.net\/forum?id=7JhGdZvW4T"},{"key":"e_1_3_3_2_30_2","first-page":"173","volume-title":"18th USENIX symposium on operating systems design and implementation (OSDI 24)","author":"Sun Biao","year":"2024","unstructured":"Biao Sun, Ziming Huang, Hanyu Zhao, Wencong Xiao, Xinyi Zhang, Yong Li, and Wei Lin. 2024. Llumnix: Dynamic scheduling for large language model serving. In 18th USENIX symposium on operating systems design and implementation (OSDI 24). USENIX Association, Santa Clara, CA, USA, 173\u2013191."},{"key":"e_1_3_3_2_31_2","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017) 5998\u20136008."},{"key":"e_1_3_3_2_32_2","volume-title":"A high-throughput and memory-efficient inference and serving engine for LLMs","year":"2024","unstructured":"vllm. 2024. A high-throughput and memory-efficient inference and serving engine for LLMs. vLLM Project. https:\/\/github.com\/vllm-project\/vllm\/tree\/main\/vllm"},{"key":"e_1_3_3_2_33_2","unstructured":"Yuxin Wang Yuhan Chen Zeyu Li Xueze Kang Yuchu Fang Yeju Zhou Yang Zheng Zhenheng Tang Xin He Rui Guo Xin Wang Qiang Wang Amelie\u00a0Chi Zhou and Xiaowen Chu. 2025. BurstGPT: A Real-world Workload Dataset to Optimize LLM Serving Systems. arxiv:https:\/\/arXiv.org\/abs\/2401.17644\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2401.17644"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"crossref","unstructured":"B.\u00a0P. Welford. 1962. Note on a method for calculating corrected sums of squares and products. Technometrics 4 3 (1962) 419\u2013420.","DOI":"10.1080\/00401706.1962.10490022"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695948"},{"key":"e_1_3_3_2_36_2","unstructured":"Yu Wu Tongxuan Liu Yuting Zeng Siyu Wu Jun Xiong Xianzhe Dong Hailong Yang Ke Zhang and Jing Li. 2025. Arrow: Adaptive Scheduling Mechanisms for Disaggregated LLM Inference Architecture. arxiv:https:\/\/arXiv.org\/abs\/2505.11916\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2505.11916"},{"key":"e_1_3_3_2_37_2","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chang Gao Chengen Huang Chenxu Lv et\u00a0al. 2025. Qwen3 Technical Report. arxiv:https:\/\/arXiv.org\/abs\/2505.09388\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2505.09388"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS64566.2025.00108"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"crossref","unstructured":"Zhisheng Ye Wei Gao Qinghao Hu Peng Sun Xiaolin Wang Yingwei Luo Tianwei Zhang and Yonggang Wen. 2024. Deep Learning Workload Scheduling in GPU Datacenters: A Survey. ACM Comput. Surv. 56 6 Article 146 (Jan. 2024) 38\u00a0pages.","DOI":"10.1145\/3638757"},{"key":"e_1_3_3_2_40_2","first-page":"521","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Yu Gyeong-In","year":"2022","unstructured":"Gyeong-In Yu, Joo\u00a0Seong Jeong, Geon-Woo Kim, Soojeong Kim, and Byung-Gon Chun. 2022. Orca: A distributed serving system for { Transformer-Based} generative models. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22). USENIX Association, Carlsbad, CA, USA, 521\u2013538."},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Lianmin Zheng Liangsheng Yin Zhiqiang Xie Chuyue\u00a0Livia Sun Jeff Huang Cody\u00a0Hao Yu Shiyi Cao Christos Kozyrakis Ion Stoica Joseph\u00a0E Gonzalez et\u00a0al. 2024. Sglang: Efficient execution of structured language model programs. Advances in neural information processing systems 37 (2024) 62557\u201362583.","DOI":"10.52202\/079017-2000"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"crossref","unstructured":"Zangwei Zheng Xiaozhe Ren Fuzhao Xue Yang Luo Xin Jiang and Yang You. 2023. Response length perception and sequence scheduling: An llm-empowered llm inference pipeline. Advances in Neural Information Processing Systems 36 (2023) 65517\u201365530.","DOI":"10.52202\/075280-2859"},{"key":"e_1_3_3_2_43_2","first-page":"749","volume-title":"19th USENIX Symposium on Operating Systems Design and Implementation (OSDI 25)","author":"Zhu Kan","year":"2025","unstructured":"Kan Zhu, Yufei Gao, Yilong Zhao, Liangyu Zhao, Gefei Zuo, Yile Gu, Dedong Xie, Zihao Ye, Keisuke Kamahori, Chien-Yu Lin, et\u00a0al. 2025. { NanoFlow} : Towards Optimal Large Language Model Serving Throughput. In 19th USENIX Symposium on Operating Systems Design and Implementation (OSDI 25). USENIX Association, Boston, MA, USA, 749\u2013765."}],"event":{"name":"HPDC '26: 35th International Symposium on High-Performance Parallel and Distributed Computing","location":"Cleveland USA","acronym":"HPDC '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 35th International Symposium on High-Performance Parallel and Distributed Computing"],"original-title":[],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:21:32Z","timestamp":1783743692000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3806645.3807819"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,13]]},"references-count":42,"alternative-id":["10.1145\/3806645.3807819","10.1145\/3806645"],"URL":"https:\/\/doi.org\/10.1145\/3806645.3807819","relation":{},"subject":[],"published":{"date-parts":[[2026,7,13]]},"assertion":[{"value":"2026-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}