{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T13:07:54Z","timestamp":1780664874664,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,26]],"date-time":"2026-04-26T00:00:00Z","timestamp":1777161600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,27]]},"DOI":"10.1145\/3767295.3769348","type":"proceedings-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T20:20:04Z","timestamp":1777062004000},"page":"1244-1260","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["KUNSERVE: Parameter-centric Memory Management for Efficient Memory Overloading Handling in LLM Serving"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-2380-2407","authenticated-orcid":false,"given":"Rongxin","family":"Cheng","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1089-8609","authenticated-orcid":false,"given":"Yuxin","family":"Lai","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4983-6047","authenticated-orcid":false,"given":"Xingda","family":"Wei","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6115-8130","authenticated-orcid":false,"given":"Rong","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9720-0361","authenticated-orcid":false,"given":"Haibo","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai JiaoTong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,26]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"https:\/\/docs.kernel.org\/admin-guide\/mm\/multigen_lru.html","author":"Multi","year":"2023","unstructured":"Multi-gen lru. https:\/\/docs.kernel.org\/admin-guide\/mm\/multigen_lru.html, 2023."},{"key":"e_1_3_2_1_2_1","volume-title":"fast, and cheap llm serving for everyone. https:\/\/github.com\/vllm-project\/vllm","author":"Easy","year":"2024","unstructured":"Easy, fast, and cheap llm serving for everyone. https:\/\/github.com\/vllm-project\/vllm, 2024."},{"key":"e_1_3_2_1_3_1","volume-title":"https:\/\/huggingface.co\/datasets\/shibing624\/sharegpt_gpt4","author":"Sharegpt","year":"2024","unstructured":"Sharegpt_gpt4, 2024. https:\/\/huggingface.co\/datasets\/shibing624\/sharegpt_gpt4, 2024."},{"key":"e_1_3_2_1_4_1","volume-title":"https:\/\/docs.nvidia.com\/cuda\/cuda-driver-api\/group_CUDA_VA.html","author":"Virtual","year":"2024","unstructured":"Virtual memory management. https:\/\/docs.nvidia.com\/cuda\/cuda-driver-api\/group_CUDA_VA.html, 2024."},{"key":"e_1_3_2_1_5_1","volume-title":"https:\/\/www.baseten.co\/blog\/how-multi-node-inference-works-llms-deepseek-r1\/#from-single-node-to-multi-node-infrastructure","author":"How","year":"2025","unstructured":"How multi-node inference works for massive llms like deepseek-r1. https:\/\/www.baseten.co\/blog\/how-multi-node-inference-works-llms-deepseek-r1\/#from-single-node-to-multi-node-infrastructure, 2025."},{"key":"e_1_3_2_1_6_1","volume-title":"https:\/\/www.perplexity.ai\/hub\/blog\/lower-latency-and-higher-throughput-with-multi-node-deepseek-deployment","author":"Lower","year":"2025","unstructured":"Lower latency and higher throughput with multi-node deepseek deployment. https:\/\/www.perplexity.ai\/hub\/blog\/lower-latency-and-higher-throughput-with-multi-node-deepseek-deployment, 2025."},{"key":"e_1_3_2_1_7_1","volume-title":"Forty-first International Conference on Machine Learning, ICML 2024","author":"Abhyankar R.","year":"2024","unstructured":"Abhyankar, R., He, Z., Srivatsa, V., Zhang, H., and Zhang, Y. Infercept: Efficient intercept support for augmented large language model inference. In Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21\u201327, 2024 (2024), Open-Review.net."},{"key":"e_1_3_2_1_8_1","first-page":"134","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024","author":"Agrawal A.","year":"2024","unstructured":"Agrawal, A., Kedia, N., Panwar, A., Mohan, J., Kwatra, N., Gulavani, B. S., Tumanov, A., and Ramjee, R. Taming throughput-latency tradeoff in LLM inference with sarathi-serve. In 18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024, Santa Clara, CA, USA, July 10\u201312, 2024 (2024), A. Gavrilovska and D. B. Terry, Eds., USENIX Association, pp. 117\u2013134."},{"key":"e_1_3_2_1_9_1","volume-title":"SARATHI: efficient LLM inference by piggybacking decodes with chunked prefills. CoRR abs\/2308.16369","author":"Agrawal A.","year":"2023","unstructured":"Agrawal, A., Panwar, A., Mohan, J., Kwatra, N., Gulavani, B. S., and Ramjee, R. SARATHI: efficient LLM inference by piggybacking decodes with chunked prefills. CoRR abs\/2308.16369 (2023)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"e_1_3_2_1_12_1","volume-title":"Ray serve: Scalable and programmable serving. https:\/\/docs.ray.io\/en\/latest\/serve\/index.html","author":"Anyscale","year":"2024","unstructured":"Anyscale. Ray serve: Scalable and programmable serving. https:\/\/docs.ray.io\/en\/latest\/serve\/index.html, 2024."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2600428.2609627"},{"key":"e_1_3_2_1_14_1","volume-title":"https:\/\/aws.amazon.com\/en\/bedrock\/","author":"Amazon","year":"2024","unstructured":"AWS. Amazon bedrock. https:\/\/aws.amazon.com\/en\/bedrock\/, 2024."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.172"},{"key":"e_1_3_2_1_16_1","volume-title":"Efficient and economic large language model inference with attention offloading. CoRR abs\/2405.01814","author":"Chen S.","year":"2024","unstructured":"Chen, S., Lin, Y., Zhang, M., and Wu, Y. Efficient and economic large language model inference with attention offloading. CoRR abs\/2405.01814 (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Dao T.","year":"2024","unstructured":"Dao, T. FlashAttention-2: Faster attention with better parallelism and work partitioning. In International Conference on Learning Representations (ICLR) (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1189"},{"key":"e_1_3_2_1_19_1","volume-title":"Top llm quantization methods and their impact on model quality","author":"Deepchecks","year":"2024","unstructured":"Deepchecks. Top llm quantization methods and their impact on model quality, 2024. https:\/\/www.deepchecks.com\/top-llm-quantization-methods-impact-on-model-quality\/, 2024."},{"key":"e_1_3_2_1_20_1","volume-title":"Deepseek-v3 technical report. CoRR abs\/2412.19437","author":"DeepSeek AI","year":"2024","unstructured":"DeepSeek-AI, Liu, A., Feng, B., Xue, B., Wang, B., Wu, B., Lu, C., Zhao, C., Deng, C., Zhang, C., Ruan, C., Dai, D., Guo, D., Yang, D., Chen, D., Ji, D., Li, E., Lin, F., Dai, F., Luo, F., Hao, G., Chen, G., Li, G., Zhang, H., Bao, H., Xu, H., Wang, H., Zhang, H., Ding, H., Xin, H., Gao, H., Li, H., Qu, H., Cai, J. L., Liang, J., Guo, J., Ni, J., Li, J., Wang, J., Chen, J., Chen, J., Yuan, J., Qiu, J., Li, J., Song, J., Dong, K., Hu, K., Gao, K., Guan, K., Huang, K., Yu, K., Wang, L., Zhang, L., Xu, L., Xia, L., Zhao, L., Wang, L., Zhang, L., Li, M., Wang, M., Zhang, M., Zhang, M., Tang, M., Li, M., Tian, N., Huang, P., Wang, P., Zhang, P., Wang, Q., Zhu, Q., Chen, Q., Du, Q., Chen, R. J., Jin, R. L., Ge, R., Zhang, R., Pan, R., Wang, R., Xu, R., Zhang, R., Chen, R., Li, S. S., Lu, S., Zhou, S., Chen, S., Wu, S., Ye, S., Ma, S., Wang, S., Zhou, S., Yu, S., Zhou, S., Pan, S., Wang, T., Yun, T., Pei, T., Sun, T., Xiao, W. L., and Zeng, W. Deepseek-v3 technical report. CoRR abs\/2412.19437 (2024)."},{"key":"e_1_3_2_1_21_1","volume-title":"Flashinfer: Kernel library for llm serving. https:\/\/github.com\/flashinfer-ai\/flashinfer","author":"Flashinfer","year":"2024","unstructured":"Flashinfer ai. Flashinfer: Kernel library for llm serving. https:\/\/github.com\/flashinfer-ai\/flashinfer, 2024."},{"key":"e_1_3_2_1_22_1","series-title":"Proceedings of Machine Learning Research","first-page":"10337","volume-title":"International Conference on Machine Learning, ICML","author":"Frantar E.","year":"2023","unstructured":"Frantar, E., and Alistarh, D. Sparsegpt: Massive language models can be accurately pruned in one-shot. In International Conference on Machine Learning, ICML 2023, 23\u201329 July 2023, Honolulu, Hawaii, USA (2023), A. Krause, E. Brunskill, K. Cho, B. Engelhardt, S. Sabato, and J. Scarlett, Eds., vol. 202 of Proceedings of Machine Learning Research, PMLR, pp. 10323\u201310337."},{"key":"e_1_3_2_1_23_1","first-page":"153","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024","author":"Fu Y.","year":"2024","unstructured":"Fu, Y., Xue, L., Huang, Y., Brabete, A., Ustiugov, D., Patel, Y., and Mai, L. Serverlessllm: Low-latency serverless inference for large language models. In 18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024, Santa Clara, CA, USA, July 10\u201312, 2024 (2024), A. Gavrilovska and D. B. Terry, Eds., USENIX Association, pp. 135\u2013153."},{"key":"e_1_3_2_1_24_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Furuta H.","year":"2024","unstructured":"Furuta, H., Lee, K., Nachum, O., Matsuo, Y., Faust, A., Gu, S. S., and Gur, I. Multimodal web navigation with instruction-finetuned foundation models. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7\u201311, 2024 (2024), OpenReview.net."},{"key":"e_1_3_2_1_25_1","volume-title":"Amazon found every 100ms of latency cost them 1% in sales. https:\/\/www.gigaspaces.com\/blog\/amazon-found-every-100ms-of-latency-cost-them-1-in-sales","author":"GIGASPACES.","year":"2024","unstructured":"GIGASPACES. Amazon found every 100ms of latency cost them 1% in sales. https:\/\/www.gigaspaces.com\/blog\/amazon-found-every-100ms-of-latency-cost-them-1-in-sales, 2024."},{"key":"e_1_3_2_1_26_1","volume-title":"Accelerate your development speed with copilot. https:\/\/copilot.github.com","author":"GitHub","year":"2024","unstructured":"GitHub. Accelerate your development speed with copilot. https:\/\/copilot.github.com, 2024."},{"key":"e_1_3_2_1_27_1","volume-title":"Deepspeed-fastgen: High-throughput text generation for llms via MII and deepspeed-inference. CoRR abs\/2401.08671","author":"Holmes C.","year":"2024","unstructured":"Holmes, C., Tanaka, M., Wyatt, M., Awan, A. A., Rasley, J., Rajbhandari, S., Aminabadi, R. Y., Qin, H., Bakhtiari, A., Kurilenko, L., and He, Y. Deepspeed-fastgen: High-throughput text generation for llms via MII and deepspeed-inference. CoRR abs\/2401.08671 (2024)."},{"key":"e_1_3_2_1_28_1","volume-title":"Inference without interference: Disaggregate LLM inference for mixed downstream workloads. CoRR abs\/2401.11181","author":"Hu C.","year":"2024","unstructured":"Hu, C., Huang, H., Xu, L., Chen, X., Xu, J., Chen, S., Feng, H., Wang, C., Wang, S., Bao, Y., Sun, N., and Shan, Y. Inference without interference: Disaggregate LLM inference for mixed downstream workloads. CoRR abs\/2401.11181 (2024)."},{"key":"e_1_3_2_1_29_1","first-page":"912","volume-title":"Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","volume":"2","author":"Kamath A. K.","year":"2025","unstructured":"Kamath, A. K., Prabhu, R., Mohan, J., Peter, S., Ramjee, R., and Panwar, A. Pod-attention: Unlocking full prefill-decode overlap for faster LLM inference. In Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2, ASPLOS 2025, Rotterdam, Netherlands, 30 March 2025 - 3 April 2025 (2025), L. Eeckhout, G. Smaragdakis, K. Liang, A. Sampson, M. A. Kim, and C. J. Rossbach, Eds., ACM, pp. 897\u2013912."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_31_1","volume-title":"Forty-first International Conference on Machine Learning, ICML 2024","author":"Li S.","year":"2024","unstructured":"Li, S., Ning, X., Wang, L., Liu, T., Shi, X., Yan, S., Dai, G., Yang, H., and Wang, Y. Evaluating quantized large language models. In Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21\u201327, 2024 (2024), OpenReview.net."},{"key":"e_1_3_2_1_32_1","first-page":"679","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2023","author":"Li Z.","year":"2023","unstructured":"Li, Z., Zheng, L., Zhong, Y., Liu, V., Sheng, Y., Jin, X., Huang, Y., Chen, Z., Zhang, H., Gonzalez, J. E., and Stoica, I. Al-paserve: Statistical multiplexing with model parallelism for deep learning serving. In 17th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2023, Boston, MA, USA, July 10\u201312, 2023 (2023), R. Geambasu and E. Nightingale, Eds., USENIX Association, pp. 663\u2013679."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.935"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640411"},{"key":"e_1_3_2_1_35_1","unstructured":"NVIDIA. Nvidia dgx superpod: Next generation scalable infrastructure for ai leadership. https:\/\/docs.nvidia.com\/dgx-superpod\/reference-architecture\/scalable-infrastructure-h200\/latest\/_downloads\/bbd08041e98eb913619944ead1f92373\/RA-11336-001-DSPH200-ReferenceArch.pdf#page=8.10 2024."},{"key":"e_1_3_2_1_36_1","volume-title":"https:\/\/chatgpt.com","author":"Open AI.","year":"2024","unstructured":"OpenAI. Chatgpt. https:\/\/chatgpt.com, 2024."},{"key":"e_1_3_2_1_37_1","volume-title":"Openai api. https:\/\/openai.com\/index\/openai-api\/","author":"Open AI.","year":"2024","unstructured":"OpenAI. Openai api. https:\/\/openai.com\/index\/openai-api\/, 2024."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"e_1_3_2_1_39_1","volume-title":"vattention: Dynamic memory management for serving llms without pagedattention. CoRR abs\/2405.04437","author":"Prabhu R.","year":"2024","unstructured":"Prabhu, R., Nayak, A., Mohan, J., Ramjee, R., and Panwar, A. vattention: Dynamic memory management for serving llms without pagedattention. CoRR abs\/2405.04437 (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"Mooncake: A kvcache-centric disaggregated architecture for LLM serving. CoRR abs\/2407.00079","author":"Qin R.","year":"2024","unstructured":"Qin, R., Li, Z., He, W., Zhang, M., Wu, Y., Zheng, W., and Xu, X. Mooncake: A kvcache-centric disaggregated architecture for LLM serving. CoRR abs\/2407.00079 (2024)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629581"},{"key":"e_1_3_2_1_42_1","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism. CoRR abs\/1909.08053","author":"Shoeybi M.","year":"2019","unstructured":"Shoeybi, M., Patwary, M., Puri, R., LeGresley, P., Casper, J., and Catanzaro, B. Megatron-lm: Training multi-billion parameter language models using model parallelism. CoRR abs\/1909.08053 (2019)."},{"key":"e_1_3_2_1_43_1","volume-title":"Dynamollm: Designing LLM inference clusters for performance and energy efficiency. CoRR abs\/2408.00741","author":"Stojkovic J.","year":"2024","unstructured":"Stojkovic, J., Zhang, C., Goiri, \u00cd., Torrellas, J., and Choukse, E. Dynamollm: Designing LLM inference clusters for performance and energy efficiency. CoRR abs\/2408.00741 (2024)."},{"key":"e_1_3_2_1_44_1","first-page":"191","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024","author":"Sun B.","year":"2024","unstructured":"Sun, B., Huang, Z., Zhao, H., Xiao, W., Zhang, X., Li, Y., and Lin, W. Llumnix: Dynamic scheduling for large language model serving. In 18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024, Santa Clara, CA, USA, July 10\u201312, 2024 (2024), A. Gavrilovska and D. B. Terry, Eds., USENIX Association, pp. 173\u2013191."},{"key":"e_1_3_2_1_45_1","volume-title":"September","author":"Team Q.","year":"2024","unstructured":"Team, Q. Qwen2.5: A party of foundation models, September 2024."},{"key":"e_1_3_2_1_46_1","first-page":"6008","volume-title":"Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017","author":"Vaswani A.","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, L., and Polosukhin, I. Attention is all you need. In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4\u20139, 2017, Long Beach, CA, USA (2017), I. Guyon, U. von Luxburg, S. Bengio, H. M. Wallach, R. Fergus, S. V. N. Vishwanathan, and R. Garnett, Eds., pp. 5998\u20136008."},{"key":"e_1_3_2_1_47_1","volume-title":"Llm compressor","year":"2025","unstructured":"vllm project. Llm compressor, 2025. https:\/\/github.com\/vllm-project\/llm-compressor, 2025."},{"key":"e_1_3_2_1_48_1","volume-title":"Burstgpt: A real-world workload dataset to optimize llm serving systems","author":"Wang Y.","year":"2024","unstructured":"Wang, Y., Chen, Y., Li, Z., Kang, X., Tang, Z., He, X., Guo, R., Wang, X., Wang, Q., Zhou, A. C., and Chu, X. Burstgpt: A real-world workload dataset to optimize llm serving systems, 2024."},{"key":"e_1_3_2_1_49_1","volume-title":"https:\/\/mathworld.wolfram.com\/LeastSquaresFitting.html","author":"Weisstein E. W.","year":"2025","unstructured":"Weisstein, E. W. \"least squares fitting.\" from mathworld-a wolfram resource. https:\/\/mathworld.wolfram.com\/LeastSquaresFitting.html, 2025."},{"key":"e_1_3_2_1_50_1","volume-title":"Loongserve: Efficiently serving long-context large language models with elastic sequence parallelism. CoRR abs\/2404.09526","author":"Wu B.","year":"2024","unstructured":"Wu, B., Liu, S., Zhong, Y., Sun, P., Liu, X., and Jin, X. Loongserve: Efficiently serving long-context large language models with elastic sequence parallelism. CoRR abs\/2404.09526 (2024)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695948"},{"key":"e_1_3_2_1_52_1","volume-title":"Fast distributed inference serving for large language models. CoRR abs\/2305.05920","author":"Wu B.","year":"2023","unstructured":"Wu, B., Zhong, Y., Zhang, Z., Huang, G., Liu, X., and Jin, X. Fast distributed inference serving for large language models. CoRR abs\/2305.05920 (2023)."},{"key":"e_1_3_2_1_53_1","first-page":"293","volume-title":"19th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2025","author":"Zhang D.","year":"2025","unstructured":"Zhang, D., Wang, H., Liu, Y., Wei, X., Shan, Y., Chen, R., and Chen, H. Blitzscale: Fast and live large model autoscaling with O(1) host caching. In 19th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2025, Boston, MA, USA, July 7\u20139, 2025 (2025), L. Zhou and Y. Zhou, Eds., USENIX Association, pp. 275\u2013293."},{"key":"e_1_3_2_1_54_1","first-page":"578","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2022","author":"Zheng L.","year":"2022","unstructured":"Zheng, L., Li, Z., Zhang, H., Zhuang, Y., Chen, Z., Huang, Y., Wang, Y., Xu, Y., Zhuo, D., Xing, E. P., Gonzalez, J. E., and Stoica, I. Alpa: Automating inter- and intra-operator parallelism for distributed deep learning. In 16th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2022, Carlsbad, CA, USA, July 11\u201313, 2022 (2022), M. K. Aguilera and H. Weatherspoon, Eds., USENIX Association, pp. 559\u2013578."},{"key":"e_1_3_2_1_55_1","first-page":"210","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024","author":"Zhong Y.","year":"2024","unstructured":"Zhong, Y., Liu, S., Chen, J., Hu, J., Zhu, Y., Liu, X., Jin, X., and Zhang, H. Distserve: Disaggregating prefill and decoding for goodput-optimized large language model serving. In 18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024, Santa Clara, CA, USA, July 10\u201312, 2024 (2024), A. Gavrilovska and D. B. Terry, Eds., USENIX Association, pp. 193\u2013210."},{"key":"e_1_3_2_1_56_1","volume-title":"Nanoflow: Towards optimal large language model serving throughput. CoRR abs\/2408.12757","author":"Zhu K.","year":"2024","unstructured":"Zhu, K., Zhao, Y., Zhao, L., Zuo, G., Gu, Y., Xie, D., Gao, Y., Xu, Q., Tang, T., Ye, Z., Kamahori, K., Lin, C., Wang, S., Krishnamurthy, A., and Kasikci, B. Nanoflow: Towards optimal large language model serving throughput. CoRR abs\/2408.12757 (2024)."}],"event":{"name":"EUROSYS '26: 21st European Conference on Computer Systems","location":"McEwan Hall\/The University of Edinburgh Edinburgh Scotland UK","acronym":"EUROSYS '26","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 21st European Conference on Computer Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3767295.3769348","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T12:21:59Z","timestamp":1780662119000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3767295.3769348"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,26]]},"references-count":56,"alternative-id":["10.1145\/3767295.3769348","10.1145\/3767295"],"URL":"https:\/\/doi.org\/10.1145\/3767295.3769348","relation":{},"subject":[],"published":{"date-parts":[[2026,4,26]]},"assertion":[{"value":"2026-04-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}