{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T04:35:25Z","timestamp":1782448525292,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":80,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,13]]},"DOI":"10.1145\/3731569.3764810","type":"proceedings-article","created":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T12:43:24Z","timestamp":1759322604000},"page":"431-445","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["DiffKV: Differentiated Memory Management for Large Language Models with Parallel KV Compaction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0130-1502","authenticated-orcid":false,"given":"Yanqi","family":"Zhang","sequence":"first","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6246-4253","authenticated-orcid":false,"given":"Yuwei","family":"Hu","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1305-901X","authenticated-orcid":false,"given":"Runyuan","family":"Zhao","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7466-0384","authenticated-orcid":false,"given":"John C. S.","family":"Lui","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9720-0361","authenticated-orcid":false,"given":"Haibo","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai JiaoTong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,12]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"FasterTransformer. https:\/\/github.com\/NVIDIA\/FasterTransformer."},{"key":"e_1_3_2_1_2_1","unstructured":"FlashInfer: Kernel Library for LLM Serving. https:\/\/github.com\/flashinfer-ai\/flashinfer."},{"key":"e_1_3_2_1_3_1","volume-title":"https:\/\/huggingface.co\/datasets\/AI-MO\/aimo-validation-aime","author":"Aime","year":"2024","unstructured":"Aime 2024. https:\/\/huggingface.co\/datasets\/AI-MO\/aimo-validation-aime, 2024."},{"key":"e_1_3_2_1_4_1","volume-title":"Gpt-4 technical report. arXiv preprint arXiv:2303.08774","author":"Achiam J.","year":"2023","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F. L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., et al. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_5_1","first-page":"114","article-title":"Keyformer: Kv cache reduction through key tokens selection for efficient generative inference","volume":"6","author":"Adnan M.","year":"2024","unstructured":"Adnan, M., Arunkumar, A., Jain, G., Nair, P. J., Soloveychik, I., and Kamath, P. Keyformer: Kv cache reduction through key tokens selection for efficient generative inference. Proceedings of Machine Learning and Systems 6 (2024), 114\u2013127.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_6_1","volume-title":"Gqa: Training generalized multi-query transformer models from multi-head checkpoints. arXiv preprint arXiv:2305.13245","author":"Ainslie J.","year":"2023","unstructured":"Ainslie, J., Lee-Thorp, J., de Jong, M., Zemlyanskiy, Y., Lebr\u00f3n, F., and Sanghai, S. Gqa: Training generalized multi-query transformer models from multi-head checkpoints. arXiv preprint arXiv:2305.13245 (2023)."},{"key":"e_1_3_2_1_7_1","first-page":"15","volume-title":"SC22: International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Aminabadi R. Y.","year":"2022","unstructured":"Aminabadi, R. Y., Rajbhandari, S., Awan, A. A., Li, C., Li, D., Zheng, E., Ruwase, O., Smith, S., Zhang, M., Rasley, J., et al. Deepspeed-inference: enabling efficient inference of transformer models at unprecedented scale. In SC22: International Conference for High Performance Computing, Networking, Storage and Analysis (2022), IEEE, pp. 1\u201315."},{"key":"e_1_3_2_1_8_1","volume-title":"Program synthesis with large language models. arXiv preprint arXiv:2108.07732","author":"Austin J.","year":"2021","unstructured":"Austin, J., Odena, A., Nye, M., Bosma, M., Michalewski, H., Dohan, D., Jiang, E., Cai, C., Terry, M., Le, Q., et al. Program synthesis with large language models. arXiv preprint arXiv:2108.07732 (2021)."},{"key":"e_1_3_2_1_9_1","volume-title":"Longbench: A bilingual, multitask benchmark for long context understanding. arXiv preprint arXiv:2308.14508","author":"Bai Y.","year":"2023","unstructured":"Bai, Y., Lv, X., Zhang, J., Lyu, H., Tang, J., Huang, Z., Du, Z., Liu, X., Zeng, A., Hou, L., et al. Longbench: A bilingual, multitask benchmark for long context understanding. arXiv preprint arXiv:2308.14508 (2023)."},{"key":"e_1_3_2_1_10_1","volume-title":"Language models are few-shot learners. arXiv preprint arXiv:2005.14165","author":"Brown T. B.","year":"2020","unstructured":"Brown, T. B. Language models are few-shot learners. arXiv preprint arXiv:2005.14165 (2020)."},{"key":"e_1_3_2_1_11_1","volume-title":"Pyramidkv: Dynamic kv cache compression based on pyramidal information funneling. arXiv preprint arXiv:2406.02069","author":"Cai Z.","year":"2024","unstructured":"Cai, Z., Zhang, Y., Gao, B., Liu, T., Lu, K., Xiong, W., Dong, Y., Chang, B., Hu, J., and Xiao, W. Pyramidkv: Dynamic kv cache compression based on pyramidal information funneling. arXiv preprint arXiv:2406.02069 (2024)."},{"key":"e_1_3_2_1_12_1","volume-title":"Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374","author":"Chen M.","year":"2021","unstructured":"Chen, M., Tworek, J., Jun, H., Yuan, Q., Pinto, H. P. D. O., Kaplan, J., Edwards, H., Burda, Y., Joseph, N., Brockman, G., et al. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374 (2021)."},{"key":"e_1_3_2_1_13_1","volume-title":"March","author":"Chiang W.-L.","year":"2023","unstructured":"Chiang, W.-L., Li, Z., Lin, Z., Sheng, Y., Wu, Z., Zhang, H., Zheng, L., Zhuang, S., Zhuang, Y., Gonzalez, J. E., Stoica, I., and Xing, E. P. Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality, March 2023."},{"key":"e_1_3_2_1_14_1","volume-title":"Chatbot arena: An open platform for evaluating llms by human preference. arXiv preprint arXiv:2403.04132","author":"Chiang W.-L.","year":"2024","unstructured":"Chiang, W.-L., Zheng, L., Sheng, Y., Angelopoulos, A. N., Li, T., Li, D., Zhang, H., Zhu, B., Jordan, M., Gonzalez, J. E., et al. Chatbot arena: An open platform for evaluating llms by human preference. arXiv preprint arXiv:2403.04132 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Training verifiers to solve math word problems. arXiv preprint arXiv:2110.14168","author":"Cobbe K.","year":"2021","unstructured":"Cobbe, K., Kosaraju, V., Bavarian, M., Chen, M., Jun, H., Kaiser, L., Plappert, M., Tworek, J., Hilton, J., Nakano, R., et al. Training verifiers to solve math word problems. arXiv preprint arXiv:2110.14168 (2021)."},{"key":"e_1_3_2_1_16_1","volume-title":"Flashattention-2: Faster attention with better parallelism and work partitioning. arXiv preprint arXiv:2307.08691","author":"Dao T.","year":"2023","unstructured":"Dao, T. Flashattention-2: Faster attention with better parallelism and work partitioning. arXiv preprint arXiv:2307.08691 (2023)."},{"key":"e_1_3_2_1_17_1","first-page":"16344","article-title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","volume":"35","author":"Dao T.","year":"2022","unstructured":"Dao, T., Fu, D., Ermon, S., Rudra, A., and R\u00e9, C. Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in Neural Information Processing Systems 35 (2022), 16344\u201316359.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_18_1","first-page":"30318","article-title":"Gpt3. int8 (): 8-bit matrix multiplication for transformers at scale","volume":"35","author":"Dettmers T.","year":"2022","unstructured":"Dettmers, T., Lewis, M., Belkada, Y., and Zettlemoyer, L. Gpt3. int8 (): 8-bit matrix multiplication for transformers at scale. Advances in Neural Information Processing Systems 35 (2022), 30318\u201330332.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_19_1","volume-title":"Qaq: Quality adaptive quantization for llm kv cache. arXiv preprint arXiv:2403.04643","author":"Dong S.","year":"2024","unstructured":"Dong, S., Cheng, W., Qin, J., and Wang, W. Qaq: Quality adaptive quantization for llm kv cache. arXiv preprint arXiv:2403.04643 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"The llama 3 herd of models. arXiv preprint arXiv:2407.21783","author":"Dubey A.","year":"2024","unstructured":"Dubey, A., Jauhri, A., Pandey, A., Kadian, A., Al-Dahle, A., Letman, A., Mathur, A., Schelten, A., Yang, A., Fan, A., et al. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_21_1","first-page":"402","volume-title":"Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","author":"Fang J.","year":"2021","unstructured":"Fang, J., Yu, Y., Zhao, C., and Zhou, J. Turbotransformers: an efficient gpu serving system for transformer models. In Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (2021), pp. 389\u2013402."},{"key":"e_1_3_2_1_22_1","first-page":"15","volume-title":"Proceedings of the Thirteenth EuroSys Conference","author":"Gao P.","year":"2018","unstructured":"Gao, P., Yu, L., Wu, Y., and Li, J. Low latency rnn inference with cellular batching. In Proceedings of the Thirteenth EuroSys Conference (2018), pp. 1\u201315."},{"key":"e_1_3_2_1_23_1","volume-title":"Model tells you what to discard: Adaptive kv cache compression for llms. arXiv preprint arXiv:2310.01801","author":"Ge S.","year":"2023","unstructured":"Ge, S., Zhang, Y., Liu, L., Zhang, M., Han, J., and Gao, J. Model tells you what to discard: Adaptive kv cache compression for llms. arXiv preprint arXiv:2310.01801 (2023)."},{"key":"e_1_3_2_1_24_1","volume-title":"Github copilot. https:\/\/github.com\/features\/copilot","author":"Github","year":"2023","unstructured":"Github. Github copilot. https:\/\/github.com\/features\/copilot, 2023."},{"key":"e_1_3_2_1_25_1","first-page":"15","volume-title":"Proceedings of the 50th Annual International Symposium on Computer Architecture","author":"Guo C.","year":"2023","unstructured":"Guo, C., Tang, J., Hu, W., Leng, J., Zhang, C., Yang, F., Liu, Y., Guo, M., and Zhu, Y. Olive: Accelerating large language models via hardware-friendly outlier-victim pair quantization. In Proceedings of the 50th Annual International Symposium on Computer Architecture (2023), pp. 1\u201315."},{"key":"e_1_3_2_1_26_1","volume-title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:2501.12948","author":"Guo D.","year":"2025","unstructured":"Guo, D., Yang, D., Zhang, H., Song, J., Zhang, R., Xu, R., Zhu, Q., Ma, S., Wang, P., Bi, X., et al. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:2501.12948 (2025)."},{"key":"e_1_3_2_1_27_1","volume-title":"Deepseek-coder: When the large language model meets programming-the rise of code intelligence. arXiv preprint arXiv:2401.14196","author":"Guo D.","year":"2024","unstructured":"Guo, D., Zhu, Q., Yang, D., Xie, Z., Dong, K., Zhang, W., Chen, G., Bi, X., Wu, Y., Li, Y., et al. Deepseek-coder: When the large language model meets programming-the rise of code intelligence. arXiv preprint arXiv:2401.14196 (2024)."},{"key":"e_1_3_2_1_28_1","first-page":"558","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Han M.","year":"2022","unstructured":"Han, M., Zhang, H., Chen, R., and Chen, H. Microsecond-scale preemption for concurrent {GPU-accelerated} {DNN} inferences. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22) (2022), pp. 539\u2013558."},{"key":"e_1_3_2_1_29_1","volume-title":"Parallel prefix sum (scan) with cuda. GPU gems 3, 39","author":"Harris M.","year":"2007","unstructured":"Harris, M., Sengupta, S., and Owens, J. D. Parallel prefix sum (scan) with cuda. GPU gems 3, 39 (2007), 851\u2013876."},{"key":"e_1_3_2_1_30_1","first-page":"68287","article-title":"Zipcache: Accurate and efficient kv cache quantization with salient token identification","volume":"37","author":"He Y.","year":"2024","unstructured":"He, Y., Zhang, L., Wu, W., Liu, J., Zhou, H., and Zhuang, B. Zipcache: Accurate and efficient kv cache quantization with salient token identification. Advances in Neural Information Processing Systems 37 (2024), 68287\u201368307.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_31_1","volume-title":"Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300","author":"Hendrycks D.","year":"2020","unstructured":"Hendrycks, D., Burns, C., Basart, S., Zou, A., Mazeika, M., Song, D., and Steinhardt, J. Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300 (2020)."},{"key":"e_1_3_2_1_32_1","volume-title":"Kvquant: Towards 10 million context length llm inference with kv cache quantization. arXiv preprint arXiv:2401.18079","author":"Hooper C.","year":"2024","unstructured":"Hooper, C., Kim, S., Mohammadzadeh, H., Mahoney, M. W., Shao, Y. S., Keutzer, K., and Gholami, A. Kvquant: Towards 10 million context length llm inference with kv cache quantization. arXiv preprint arXiv:2401.18079 (2024)."},{"key":"e_1_3_2_1_33_1","volume-title":"Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems 32","author":"Huang Y.","year":"2019","unstructured":"Huang, Y., Cheng, Y., Bapna, A., Firat, O., Chen, D., Chen, M., Lee, H., Ngiam, J., Le, Q. V., Wu, Y., et al. Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_2_1_34_1","volume-title":"5-coder technical report. arXiv preprint arXiv:2409.12186","author":"Hui B.","year":"2024","unstructured":"Hui, B., Yang, J., Cui, Z., Yang, J., Liu, D., Zhang, L., Liu, T., Zhang, J., Yu, B., Lu, K., et al. Qwen2. 5-coder technical report. arXiv preprint arXiv:2409.12186 (2024)."},{"key":"e_1_3_2_1_35_1","volume-title":"Openai o1 system card. arXiv preprint arXiv:2412.16720","author":"Jaech A.","year":"2024","unstructured":"Jaech, A., Kalai, A., Lerer, A., Richardson, A., El-Kishky, A., Low, A., Helyar, A., Madry, A., Beutel, A., Carney, A., et al. Openai o1 system card. arXiv preprint arXiv:2412.16720 (2024)."},{"key":"e_1_3_2_1_36_1","volume-title":"B., Bamford, C., Chaplot, D. S., Casas, D. d. l., Hanna, E. B., Bressand, F., et al. Mixtral of experts. arXiv preprint arXiv:2401.04088","author":"Jiang A. Q.","year":"2024","unstructured":"Jiang, A. Q., Sablayrolles, A., Roux, A., Mensch, A., Sava ry, B., Bamford, C., Chaplot, D. S., Casas, D. d. l., Hanna, E. B., Bressand, F., et al. Mixtral of experts. arXiv preprint arXiv:2401.04088 (2024)."},{"key":"e_1_3_2_1_37_1","first-page":"12","volume-title":"Proceedings of the 44th annual international symposium on computer architecture","author":"Jouppi N. P.","year":"2017","unstructured":"Jouppi, N. P., Young, C., Patil, N., Patterson, D., Agrawal, G., Bajwa, R., Bates, S., Bhatia, S., Boden, N., Borchers, A., et al. In-datacenter performance analysis of a tensor processing unit. In Proceedings of the 44th annual international symposium on computer architecture (2017), pp. 1\u201312."},{"key":"e_1_3_2_1_38_1","first-page":"310","volume-title":"Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","volume":"2","author":"Kao S.-C.","year":"2023","unstructured":"Kao, S.-C., Subramanian, S., Agrawal, G., Yazdanbakhsh, A., and Krishna, T. Flat: An optimized dataflow for mitigating attention bottlenecks. In Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2 (2023), pp. 295\u2013310."},{"key":"e_1_3_2_1_39_1","first-page":"626","volume-title":"Proceedings of the 29th Symposium on Operating Systems Principles","author":"Kwon W.","year":"2023","unstructured":"Kwon, W., Li, Z., Zhuang, S., Sheng, Y., Zheng, L., Yu, C. H., Gonzalez, J., Zhang, H., and Stoica, I. Efficient memory management for large language model serving with pagedattention. In Proceedings of the 29th Symposium on Operating Systems Principles (2023), pp. 611\u2013626."},{"key":"e_1_3_2_1_40_1","first-page":"172","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Lee W.","year":"2024","unstructured":"Lee, W., Lee, J., Seo, J., and Sim, J. {InfiniGen}: Efficient generative inference of large language models with dynamic {KV} cache management. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24) (2024), pp. 155\u2013172."},{"key":"e_1_3_2_1_41_1","first-page":"3843","article-title":"Solving quantitative reasoning problems with language models","volume":"35","author":"Lewkowycz A.","year":"2022","unstructured":"Lewkowycz, A., Andreassen, A., Dohan, D., Dyer, E., Michalewski, H., Ramasesh, V., Slone, A., Anil, C., Schlag, I., Gutman-Solo, T., et al. Solving quantitative reasoning problems with language models. Advances in Neural Information Processing Systems 35 (2022), 3843\u20133857.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_42_1","volume-title":"Snapkv: Llm knows what you are looking for before generation. arXiv preprint arXiv:2404.14469","author":"Li Y.","year":"2024","unstructured":"Li, Y., Huang, Y., Yang, B., Venkitesh, B., Locatelli, A., Ye, H., Cai, T., Lewis, P., and Chen, D. Snapkv: Llm knows what you are looking for before generation. arXiv preprint arXiv:2404.14469 (2024)."},{"key":"e_1_3_2_1_43_1","first-page":"679","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Li Z.","year":"2023","unstructured":"Li, Z., Zheng, L., Zhong, Y., Liu, V., Sheng, Y., Jin, X., Huang, Y., Chen, Z., Zhang, H., Gonzalez, J. E., et al. {AlpaServe}: Statistical multiplexing with model parallelism for deep learning serving. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23) (2023), pp. 663\u2013679."},{"key":"e_1_3_2_1_44_1","first-page":"44","volume-title":"2019 IEEE Hot Chips 31 Symposium (HCS)","author":"Liao H.","year":"2019","unstructured":"Liao, H., Tu, J., Xia, J., and Zhou, X. Davinci: A scalable architecture for neural network computing. In 2019 IEEE Hot Chips 31 Symposium (HCS) (2019), IEEE Computer Society, pp. 1\u201344."},{"key":"e_1_3_2_1_45_1","volume-title":"Qserve: W4a8kv4 quantization and system co-design for efficient llm serving. arXiv preprint arXiv:2405.04532","author":"Lin Y.","year":"2024","unstructured":"Lin, Y., Tang, H., Yang, S., Zhang, Z., Xiao, G., Gan, C., and Han, S. Qserve: W4a8kv4 quantization and system co-design for efficient llm serving. arXiv preprint arXiv:2405.04532 (2024)."},{"key":"e_1_3_2_1_46_1","volume-title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model. arXiv preprint arXiv:2405.04434","author":"Liu A.","year":"2024","unstructured":"Liu, A., Feng, B., Wang, B., Wang, B., Liu, B., Zhao, C., Dengr, C., Ruan, C., Dai, D., Guo, D., et al. Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model. arXiv preprint arXiv:2405.04434 (2024)."},{"key":"e_1_3_2_1_47_1","volume-title":"Thirty-seventh Conference on Neural Information Processing Systems","author":"Liu J.","year":"2023","unstructured":"Liu, J., Xia, C. S., Wang, Y., and Zhang, L. Is your code generated by chatGPT really correct? rigorous evaluation of large language models for code generation. In Thirty-seventh Conference on Neural Information Processing Systems (2023)."},{"key":"e_1_3_2_1_48_1","first-page":"36","article-title":"Scissorhands: Exploiting the persistence of importance hypothesis for llm kv cache compression at test time","author":"Liu Z.","year":"2024","unstructured":"Liu, Z., Desai, A., Liao, F., Wang, W., Xie, V., Xu, Z., Kyrillidis, A., and Shrivastava, A. Scissorhands: Exploiting the persistence of importance hypothesis for llm kv cache compression at test time. Advances in Neural Information Processing Systems 36 (2024).","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_49_1","volume-title":"Kivi: A tuning-free asymmetric 2bit quantization for kv cache. arXiv preprint arXiv:2402.02750","author":"Liu Z.","year":"2024","unstructured":"Liu, Z., Yuan, J., Jin, H., Zhong, S., Xu, Z., Braverman, V., Chen, B., and Hu, X. Kivi: A tuning-free asymmetric 2bit quantization for kv cache. arXiv preprint arXiv:2402.02750 (2024)."},{"key":"e_1_3_2_1_50_1","volume-title":"Pointer sentinel mixture models. arXiv preprint arXiv:1609.07843","author":"Merity S.","year":"2016","unstructured":"Merity, S., Xiong, C., Bradbury, J., and Socher, R. Pointer sentinel mixture models. arXiv preprint arXiv:1609.07843 (2016)."},{"key":"e_1_3_2_1_51_1","volume-title":"A white paper on neural network quantization. arXiv preprint arXiv:2106.08295","author":"Nagel M.","year":"2021","unstructured":"Nagel, M., Fournarakis, M., Amjad, R. A., Bondarenko, Y., Van Baalen, M., and Blankevoort, T. A white paper on neural network quantization. arXiv preprint arXiv:2106.08295 (2021)."},{"key":"e_1_3_2_1_52_1","first-page":"113","volume-title":"International Conference on Algorithms and Architectures for Parallel Processing","author":"Nakano K.","year":"2012","unstructured":"Nakano, K. An optimal parallel prefix-sums algorithm on the memory machine models for gpus. In International Conference on Algorithms and Architectures for Parallel Processing (2012), Springer, pp. 99\u2013113."},{"key":"e_1_3_2_1_53_1","first-page":"15","volume-title":"Proceedings of the 27th ACM symposium on operating systems principles","author":"Narayanan D.","year":"2019","unstructured":"Narayanan, D., Harlap, A., Phanishayee, A., Seshadri, V., Devanur, N. R., Ganger, G. R., Gibbons, P. B., and Zaharia, M. Pipedream: Generalized pipeline parallelism for dnn training. In Proceedings of the 27th ACM symposium on operating systems principles (2019), pp. 1\u201315."},{"key":"e_1_3_2_1_54_1","first-page":"15","volume-title":"Proceedings of the international conference for high performance computing, networking, storage and analysis","author":"Narayanan D.","year":"2021","unstructured":"Narayanan, D., Shoeybi, M., Casper, J., LeGresley, P., Patwary, M., Korthikanti, V., Vainbrand, D., Kashinkunti, P., Bernauer, J., Catanzaro, B., et al. Efficient large-scale language model training on gpu clusters using megatron-lm. In Proceedings of the international conference for high performance computing, networking, storage and analysis (2021), pp. 1\u201315."},{"key":"e_1_3_2_1_55_1","volume-title":"Nvidia L40 datasheet. https:\/\/www.nvidia.com\/content\/dam\/en-zz\/Solutions\/design-visualization\/support-guide\/NVIDIA-L40-Datasheet-January-2023.pdf","author":"Nvidia","year":"2023","unstructured":"Nvidia. Nvidia L40 datasheet. https:\/\/www.nvidia.com\/content\/dam\/en-zz\/Solutions\/design-visualization\/support-guide\/NVIDIA-L40-Datasheet-January-2023.pdf, 2023."},{"key":"e_1_3_2_1_56_1","volume-title":"Training language models to follow instructions with human feedback. Advances in neural information processing systems 35","author":"Ouyang L.","year":"2022","unstructured":"Ouyang, L., Wu, J., Jiang, X., Almeida, D., Wainwright, C., Mishkin, P., Zhang, C., Agarwal, S., Slama, K., Ray, A., et al. Training language models to follow instructions with human feedback. Advances in neural information processing systems 35 (2022), 27730\u201327744."},{"key":"e_1_3_2_1_57_1","first-page":"606","article-title":"Efficiently scaling transformer inference","volume":"5","author":"Pope R.","year":"2023","unstructured":"Pope, R., Douglas, S., Chowdhery, A., Devlin, J., Bradbury, J., Heek, J., Xiao, K., Agrawal, S., and Dean, J. Efficiently scaling transformer inference. Proceedings of Machine Learning and Systems 5 (2023), 606\u2013624.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_58_1","volume-title":"First Conference on Language Modeling","author":"Rein D.","year":"2024","unstructured":"Rein, D., Hou, B. L., Stickland, A. C., Petty, J., Pang, R. Y., Dirani, J., Michael, J., and Bowman, S. R. Gpqa: A graduate-level google-proof q&a benchmark. In First Conference on Language Modeling (2024)."},{"key":"e_1_3_2_1_59_1","volume-title":"Capabilities of gemini models in medicine. arXiv preprint arXiv:2404.18416","author":"Saab K.","year":"2024","unstructured":"Saab, K., Tu, T., Weng, W.-H., Tanno, R., Stutz, D., Wulczyn, E., Zhang, F., Strother, T., Park, C., Vedadi, E., et al. Capabilities of gemini models in medicine. arXiv preprint arXiv:2404.18416 (2024)."},{"key":"e_1_3_2_1_60_1","volume-title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300","author":"Shao Z.","year":"2024","unstructured":"Shao, Z., Wang, P., Zhu, Q., Xu, R., Song, J., Zhang, M., Li, Y., Wu, Y., and Guo, D. Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300 (2024)."},{"key":"e_1_3_2_1_61_1","volume-title":"Fast transformer decoding: One write-head is all you need. arXiv preprint arXiv:1911.02150","author":"Shazeer N.","year":"2019","unstructured":"Shazeer, N. Fast transformer decoding: One write-head is all you need. arXiv preprint arXiv:1911.02150 (2019)."},{"key":"e_1_3_2_1_62_1","first-page":"31116","volume-title":"International Conference on Machine Learning","author":"Sheng Y.","year":"2023","unstructured":"Sheng, Y., Zheng, L., Yuan, B., Li, Z., Ryabinin, M., Chen, B., Liang, P., R\u00e9, C., Stoica, I., and Zhang, C. Flexgen: High-throughput generative inference of large language models with a single gpu. In International Conference on Machine Learning (2023), PMLR, pp. 31094\u201331116."},{"key":"e_1_3_2_1_63_1","first-page":"718","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Shi Y.","year":"2023","unstructured":"Shi, Y., Yang, Z., Xue, J., Ma, L., Xia, Y., Miao, Z., Guo, Y., Yang, F., and Zhou, L. Welder: Scheduling deep learning memory access via tile-graph. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23) (2023), pp. 701\u2013718."},{"key":"e_1_3_2_1_64_1","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053","author":"Shoeybi M.","year":"2019","unstructured":"Shoeybi, M., Patwary, M., Puri, R., LeGresley, P., Casper, J., and Catanzaro, B. Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053 (2019)."},{"key":"e_1_3_2_1_65_1","volume-title":"Mmlu-pro+: Evaluating higher-order reasoning and shortcut learning in llms. arXiv preprint arXiv:2409.02257","author":"Taghanaki S. A.","year":"2024","unstructured":"Taghanaki, S. A., Khani, A., and Khasahmadi, A. Mmlu-pro+: Evaluating higher-order reasoning and shortcut learning in llms. arXiv preprint arXiv:2409.02257 (2024)."},{"key":"e_1_3_2_1_66_1","volume-title":"Quest: Query-aware sparsity for efficient long-context llm inference. arXiv preprint arXiv:2406.10774","author":"Tang J.","year":"2024","unstructured":"Tang, J., Zhao, Y., Zhu, K., Xiao, G., Kasikci, B., and Han, S. Quest: Query-aware sparsity for efficient long-context llm inference. arXiv preprint arXiv:2406.10774 (2024)."},{"key":"e_1_3_2_1_67_1","volume-title":"Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805","author":"Team G.","year":"2023","unstructured":"Team, G., Anil, R., Borgeaud, S., Wu, Y., Alayrac, J.-B., Yu, J., Soricut, R., Schalkwyk, J., Dai, A. M., Hauth, A., et al. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_68_1","volume-title":"March","author":"The Qwen Team","year":"2025","unstructured":"The Qwen Team. Qwq-32b: Embracing the power of reinforcement learning, March 2025."},{"key":"e_1_3_2_1_69_1","volume-title":"Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288","author":"Touvron H.","year":"2023","unstructured":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y., Bashlykov, N., Batra, S., Bhargava, P., Bhosale, S., et al. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_70_1","volume-title":"Attention is all you need. Advances in Neural Information Processing Systems","author":"Vaswani A.","year":"2017","unstructured":"Vaswani, A. Attention is all you need. Advances in Neural Information Processing Systems (2017)."},{"key":"e_1_3_2_1_71_1","volume-title":"Duoattention: Efficient long-context llm inference with retrieval and streaming heads. arXiv preprint arXiv:2410.10819","author":"Xiao G.","year":"2024","unstructured":"Xiao, G., Tang, J., Zuo, J., Guo, J., Yang, S., Tang, H., Fu, Y., and Han, S. Duoattention: Efficient long-context llm inference with retrieval and streaming heads. arXiv preprint arXiv:2410.10819 (2024)."},{"key":"e_1_3_2_1_72_1","volume-title":"Efficient streaming language models with attention sinks. arXiv preprint arXiv:2309.17453","author":"Xiao G.","year":"2023","unstructured":"Xiao, G., Tian, Y., Chen, B., Han, S., and Lewis, M. Efficient streaming language models with attention sinks. arXiv preprint arXiv:2309.17453 (2023)."},{"key":"e_1_3_2_1_73_1","volume-title":"A preliminary study of o1 in medicine: Are we closer to an ai doctor? arXiv preprint arXiv:2409.15277","author":"Xie Y.","year":"2024","unstructured":"Xie, Y., Wu, J., Tu, H., Yang, S., Zhao, B., Zong, Y., Jin, Q., Xie, C., and Zhou, Y. A preliminary study of o1 in medicine: Are we closer to an ai doctor? arXiv preprint arXiv:2409.15277 (2024)."},{"key":"e_1_3_2_1_74_1","first-page":"781","volume-title":"Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","author":"Yang Y.","year":"2022","unstructured":"Yang, Y., Zhao, L., Li, Y., Zhang, H., Li, J., Zhao, M., Chen, X., and Li, K. Infiless: a native serverless system for low-latency, high-throughput inference. In Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems (2022), pp. 768\u2013781."},{"key":"e_1_3_2_1_75_1","first-page":"538","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Yu G.-I.","year":"2022","unstructured":"Yu, G.-I., Jeong, J. S., Kim, G.-W., Kim, S., and Chun, B.-G. Orca: A distributed serving system for {Transformer-Based} generative models. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22) (2022), pp. 521\u2013538."},{"key":"e_1_3_2_1_76_1","first-page":"808","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Zhang H.","year":"2023","unstructured":"Zhang, H., Tang, Y., Khandelwal, A., and Stoica, I. {SHEPHERD}: Serving {DNNs} in the wild. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23) (2023), pp. 787\u2013808."},{"key":"e_1_3_2_1_77_1","first-page":"36","article-title":"H2o: Heavy-hitter oracle for efficient generative inference of large language models","author":"Zhang Z.","year":"2024","unstructured":"Zhang, Z., Sheng, Y., Zhou, T., Chen, T., Zheng, L., Cai, R., Song, Z., Tian, Y., R\u00e9, C., Barrett, C., et al. H2o: Heavy-hitter oracle for efficient generative inference of large language models. Advances in Neural Information Processing Systems 36 (2024).","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_78_1","first-page":"196","article-title":"Atom: Low-bit quantization for efficient and accurate llm serving","volume":"6","author":"Zhao Y.","year":"2024","unstructured":"Zhao, Y., Lin, C.-Y., Zhu, K., Ye, Z., Chen, L., Zheng, S., Ceze, L., Krishnamurthy, A., Chen, T., and Kasikci, B. Atom: Low-bit quantization for efficient and accurate llm serving. Proceedings of Machine Learning and Systems 6 (2024), 196\u2013209.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_79_1","volume-title":"Evaluation of openai o1: Opportunities and challenges of agi. arXiv preprint arXiv:2409.18486","author":"Zhong T.","year":"2024","unstructured":"Zhong, T., Liu, Z., Pan, Y., Zhang, Y., Zhou, Y., Liang, S., Wu, Z., Lyu, Y., Shu, P., Yu, X., et al. Evaluation of openai o1: Opportunities and challenges of agi. arXiv preprint arXiv:2409.18486 (2024)."},{"key":"e_1_3_2_1_80_1","volume-title":"Serving large language models on huawei cloudmatrix384. arXiv preprint arXiv:2506.12708","author":"Zuo P.","year":"2025","unstructured":"Zuo, P., Lin, H., Deng, J., Zou, N., Yang, X., Diao, Y., Gao, W., Xu, K., Chen, Z., Lu, S., et al. Serving large language models on huawei cloudmatrix384. arXiv preprint arXiv:2506.12708 (2025)."}],"event":{"name":"SOSP '25: ACM SIGOPS 31st Symposium on Operating Systems Principles","location":"Lotte Hotel World Seoul Republic of Korea","acronym":"SOSP '25","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","USENIX"]},"container-title":["Proceedings of the ACM SIGOPS 31st Symposium on Operating Systems Principles"],"original-title":[],"deposited":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T12:50:46Z","timestamp":1759323046000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731569.3764810"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,12]]},"references-count":80,"alternative-id":["10.1145\/3731569.3764810","10.1145\/3731569"],"URL":"https:\/\/doi.org\/10.1145\/3731569.3764810","relation":{},"subject":[],"published":{"date-parts":[[2025,10,12]]},"assertion":[{"value":"2025-10-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}