{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T09:02:57Z","timestamp":1784624577817,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":31,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T00:00:00Z","timestamp":1785888000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"NSFC","award":["62572304"],"award-info":[{"award-number":["62572304"]}]},{"name":"NSFC","award":["62402407"],"award-info":[{"award-number":["62402407"]}]},{"name":"National Key Research and Development Plan of China","award":["2024YFB2906600"],"award-info":[{"award-number":["2024YFB2906600"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,6]]},"DOI":"10.1145\/3820441.3820479","type":"proceedings-article","created":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T08:11:31Z","timestamp":1784621491000},"page":"260-266","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards Efficient Serving of Network-intensive LLM Inferences"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-5548-9458","authenticated-orcid":false,"given":"Weiye","family":"Wang","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9480-5632","authenticated-orcid":false,"given":"Chen","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9859-2045","authenticated-orcid":false,"given":"Junxue","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8086-6036","authenticated-orcid":false,"given":"Zhusheng","family":"Wang","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5352-1277","authenticated-orcid":false,"given":"Hui","family":"Yuan","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-3780-1203","authenticated-orcid":false,"given":"Zixuan","family":"Guan","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9549-186X","authenticated-orcid":false,"given":"Xiaolong","family":"Zheng","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9195-6443","authenticated-orcid":false,"given":"Qizhen","family":"Weng","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence (TeleAI), China Telecom, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4930-1028","authenticated-orcid":false,"given":"Yin","family":"Chen","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence (TeleAI), China Telecom, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0034-2302","authenticated-orcid":false,"given":"Minyi","family":"Guo","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,8,5]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","unstructured":"Shubham Agarwal Sai Sundaresan Subrata Mitra Debabrata Mahapatra Archit Gupta Rounak Sharma Nirmal\u00a0Joshua Kapu Tong Yu and Shiv Saini. 2025. Cache-Craft: Managing Chunk-Caches for Efficient Retrieval-Augmented Generation. arXiv. 10.48550\/arXiv.2502.15734","DOI":"10.48550\/arXiv.2502.15734"},{"key":"e_1_3_3_2_3_2","unstructured":"Meta AI. [n. d.]. Introducing Llama 3.1: Our Most Capable Models to Date. https:\/\/ai.meta.com\/blog\/meta-llama-3-1\/."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","unstructured":"Egor Bogomolov Aleksandra Eliseeva Timur Galimzyanov Evgeniy Glukhov Anton Shapkin Maria Tigina Yaroslav Golubev Alexander Kovrigin Arie van Deursen Maliheh Izadi and Timofey Bryksin. 2024. Long Code Arena: A Set of Benchmarks for Long-Context Code Models. arxiv:https:\/\/arXiv.org\/abs\/2406.11612\u00a0[cs] 10.48550\/arXiv.2406.11612","DOI":"10.48550\/arXiv.2406.11612"},{"key":"e_1_3_3_2_5_2","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared\u00a0D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et\u00a0al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877\u20131901."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","unstructured":"L.\u00a0L. Cheng. 1985. A Proof of the Optimality of the Shortest-Job-First Policy for Monotone Processors. Operations Research 33 5 (1985) 1035\u20131040. 10.1287\/opre.33.5.1035","DOI":"10.1287\/opre.33.5.1035"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/3731569.3764834"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"crossref","unstructured":"Luciano Floridi and Massimo Chiriatti. 2020. GPT-3: Its nature scope limits and consequences. Minds and Machines 30 (2020) 681\u2013694.","DOI":"10.1007\/s11023-020-09548-1"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575721"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3611643.3617850"},{"key":"e_1_3_3_2_11_2","unstructured":"iMatix Corporation. [n. d.]. ZeroMQ. https:\/\/zeromq.org."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1663"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.859"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3604237.3626869"},{"key":"e_1_3_3_2_16_2","volume-title":"USENIX OSDI","author":"Lin Chaofan","year":"2024","unstructured":"Chaofan Lin, Zhenhua Han, Chengruidong Zhang, Yuqing Yang, Fan Yang, Chen Chen, and Lili Qiu. 2024. Parrot: Efficient Serving of LLM-based Applications with Semantic Variable. In USENIX OSDI."},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","unstructured":"C.\u00a0L. Liu and J.\u00a0W. Layland. 1973. Scheduling Algorithms for Multiprogramming in a Hard-Real-Time Environment. J. ACM 20 1 (1973) 46\u201361. 10.1145\/321738.321743","DOI":"10.1145\/321738.321743"},{"key":"e_1_3_3_2_18_2","unstructured":"Jiachen Liu Zhiyu Wu Jae-Won Chung Fan Lai Myungjin Lee and Mosharaf Chowdhury. 2024. Andes: Defining and Enhancing Quality-of-Experience in LLM-Based Text Streaming Services."},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672274"},{"key":"e_1_3_3_2_20_2","unstructured":"LMCache. [n. d.]. LMCache. https:\/\/lmcache.ai\/."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Christos Makridis. 2025. The Impact of Generative Artificial Intelligence on Artists. Available at SSRN (2025).","DOI":"10.2139\/ssrn.5179390"},{"key":"e_1_3_3_2_22_2","volume-title":"ASPLOS 23","author":"Narayanan Deepak","year":"2023","unstructured":"Deepak Narayanan, Keshav Santhanam, and Amar Phanishayee. 2023. Heterogeneity-Aware Cluster Scheduling Policies for Deep Learning Workloads. In ASPLOS 23."},{"key":"e_1_3_3_2_23_2","first-page":"155","volume-title":"23rd USENIX Conference on File and Storage Technologies (FAST 25)","author":"Qin Ruoyu","year":"2025","unstructured":"Ruoyu Qin, Zheming Li, Weiran He, Jialei Cui, Feng Ren, Mingxing Zhang, Yongwei Wu, Weimin Zheng, and Xinran Xu. 2025. Mooncake: Trading More Storage for Less Computation \u2014 A KVCache-centric Architecture for Serving LLM Chatbot. In 23rd USENIX Conference on File and Storage Technologies (FAST 25). USENIX Association, Santa Clara, CA, 155\u2013170."},{"key":"e_1_3_3_2_24_2","unstructured":"Shuo Ren Pu Jian Zhenjiang Ren Chunlin Leng Can Xie and Jiajun Zhang. 2025. Towards Scientific Intelligence: A Survey of LLM-based Scientific Agents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.24047 (2025)."},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","unstructured":"John\u00a0Paul Shen and Mikko\u00a0H. Lipasti. 1995. The Microarchitecture of Superscalar Processors. Proc. IEEE 83 12 (1995) 1603\u20131623. 10.1109\/5.476077","DOI":"10.1109\/5.476077"},{"key":"e_1_3_3_2_26_2","unstructured":"Qwen Team. 2025. Qwen2.5-1M: Deploy Your Own Qwen with Context Length up to 1M Tokens. https:\/\/qwenlm.github.io\/blog\/qwen2.5-1m\/."},{"key":"e_1_3_3_2_27_2","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar et\u00a0al. 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.13971 (2023)."},{"key":"e_1_3_3_2_28_2","unstructured":"Junchao Wu Shu Yang Runzhe Zhan Yulin Yuan Lidia\u00a0Sam Chao and Derek\u00a0Fai Wong. 2025. A survey on LLM-generated text detection: Necessity methods and future directions. Computational Linguistics (2025) 1\u201366."},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2000"},{"key":"e_1_3_3_2_30_2","unstructured":"Zhiqiang Xie. [n. d.]. SGLang HiCache: Fast Hierarchical KV Caching with Your Favorite Storage Backends | LMSYS Org. https:\/\/lmsys.org\/blog\/2025-09-10-sglang-hicache."},{"key":"e_1_3_3_2_31_2","first-page":"193","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong Yinmin","year":"2024","unstructured":"Yinmin Zhong, Shengyu Liu, Junda Chen, Jianbo Hu, Yibo Zhu, Xuanzhe Liu, Xin Jin, and Hao Zhang. 2024. DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). USENIX Association, Santa Clara, CA, 193\u2013210."},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1245"}],"event":{"name":"APNet 2026: The 10th Asia-Pacific Workshop on Networking","location":"Singapore Singapore","acronym":"APNet '26"},"container-title":["Proceedings of the 10th Asia-Pacific Workshop on Networking"],"original-title":[],"deposited":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T08:16:27Z","timestamp":1784621787000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3820441.3820479"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,5]]},"references-count":31,"alternative-id":["10.1145\/3820441.3820479","10.1145\/3820441"],"URL":"https:\/\/doi.org\/10.1145\/3820441.3820479","relation":{},"subject":[],"published":{"date-parts":[[2026,8,5]]},"assertion":[{"value":"2026-08-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}