{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T16:56:50Z","timestamp":1782925010882,"version":"3.54.5"},"reference-count":51,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,25]],"date-time":"2026-05-25T00:00:00Z","timestamp":1779667200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,25]],"date-time":"2026-05-25T00:00:00Z","timestamp":1779667200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,25]]},"DOI":"10.1109\/ipdps65963.2026.00058","type":"proceedings-article","created":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T20:55:16Z","timestamp":1782852916000},"page":"612-625","source":"Crossref","is-referenced-by-count":0,"title":["Near-Zero Cost KV Cache Compression for Large Language Model Inference"],"prefix":"10.1109","author":[{"given":"Boyuan","family":"Zhang","sequence":"first","affiliation":[{"name":"Indiana University,Bloomington,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ding","family":"Zhou","sequence":"additional","affiliation":[{"name":"ByteDance,San Jose,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yafan","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Iowa,Iowa City,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shihui","family":"Song","sequence":"additional","affiliation":[{"name":"University of Iowa,Iowa City,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Feng","sequence":"additional","affiliation":[{"name":"Indiana University,Bloomington,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinda","family":"Jia","sequence":"additional","affiliation":[{"name":"Indiana University,Bloomington,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chengming","family":"Zhang","sequence":"additional","affiliation":[{"name":"Independent Researcher,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhi","family":"Zhang","sequence":"additional","affiliation":[{"name":"ByteDance,San Jose,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Beyond the limits: A survey of techniques to extend the context length in large language models","author":"Wang","year":"2024"},{"key":"ref2","article-title":"Scaling instruction-tuned llms to million-token contexts via hierarchical synthetic data generation","author":"He","year":"2025"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00638"},{"key":"ref4","article-title":"Blenderbot 3: a deployed conversational agent that continually learns to responsibly engage","author":"Shuster","year":"2022"},{"key":"ref5","first-page":"9459","article-title":"Retrieval-augmented generation for knowledge-intensive nlp tasks","volume-title":"Advances in neural information processing systems","volume":"33","author":"Lewis"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1506"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.183"},{"key":"ref8","article-title":"Kivi: A tuning-free asymmetric 2bit quantization for kv cache","author":"Liu","year":"2024"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0040"},{"key":"ref10","article-title":"Commvq: Commutative vector quantization for kv cache compression","author":"Li","year":"2025"},{"key":"ref11","first-page":"193","article-title":"{DistServe}: Disaggregating prefill and decoding for goodput-optimized large language model serving","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong"},{"key":"ref12","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume-title":"Advances in neural information processing systems","volume":"32","author":"Paszke"},{"key":"ref13","article-title":"Triton: An open-source language and compiler for custom deep learning operations","author":"Tillet","year":"2021"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2014.2346458"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3588195.3592994"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607048"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3721145.3725743"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00021"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/BigData50022.2020.9378449"},{"key":"ref20","volume-title":"Atmosphere Model","year":"2019"},{"key":"ref22","volume-title":"QMCPACK: many-body ab initio Quantum Monte Carlo code","year":"2019"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1978.1055934"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1977.1055714"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3577193.3593706"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/JRPROC.1952.273898"},{"key":"ref28","article-title":"Asymmetric numeral systems: entropy coding combining speed of huffman coding with compression rate of arithmetic coding","author":"Duda","year":"2013"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/B978-0-12-385963-1.00026-5"},{"key":"ref30","article-title":"CUB \u2014 CUDA Core Compute Libraries"},{"key":"ref31","article-title":"The llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"ref32","article-title":"Efficient streaming language models with attention sinks","author":"Xiao","year":"2023"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.460"},{"key":"ref34","article-title":"Meta llama 3.1 8b instruct","year":"2024"},{"key":"ref35","first-page":"3119","article-title":"LongBench: A bilingual, multitask benchmark for long context understanding","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics","volume":"1","author":"Bai"},{"key":"ref36","first-page":"15 262","article-title":"Bench: Extending long context evaluation beyond 100K tokens","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics","volume":"1","author":"Zhang"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.859"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.818"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1986"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3381"},{"key":"ref41","article-title":"kvpress","author":"Jegou","year":"2024"},{"key":"ref42","article-title":"Mistral-nemo-instruct-2407","year":"2025"},{"key":"ref43","article-title":"Qwen3 technical report","author":"Yang","year":"2025"},{"key":"ref44","article-title":"Qwen3-30b-a3b-instruct-2507","year":"2025"},{"key":"ref45","volume-title":"nvCOMP: A library for fast lossless compression\/decompression on the GPU"},{"key":"ref46","article-title":"Duoattention: Efficient long-context llm inference with retrieval and streaming heads","author":"Xiao","year":"2024"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2279"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0109"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1145\/3733104"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2024.05.022"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00095"},{"key":"ref52","article-title":"Transformers \u2014 hugging face documentation","year":"2025"}],"event":{"name":"2026 IEEE International Parallel and Distributed Processing Symposium (IPDPS)","location":"New Orleans, LA, USA","start":{"date-parts":[[2026,5,25]]},"end":{"date-parts":[[2026,5,29]]}},"container-title":["2026 IEEE International Parallel and Distributed Processing Symposium (IPDPS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11575315\/11575316\/11575358.pdf?arnumber=11575358","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T16:30:12Z","timestamp":1782923412000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11575358\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,25]]},"references-count":51,"URL":"https:\/\/doi.org\/10.1109\/ipdps65963.2026.00058","relation":{},"subject":[],"published":{"date-parts":[[2026,5,25]]}}}