{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T05:52:56Z","timestamp":1781848376492,"version":"3.54.5"},"reference-count":27,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,24]],"date-time":"2026-05-24T00:00:00Z","timestamp":1779580800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,24]],"date-time":"2026-05-24T00:00:00Z","timestamp":1779580800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,24]]},"DOI":"10.1109\/iscas66217.2026.11562764","type":"proceedings-article","created":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T20:06:41Z","timestamp":1781813201000},"page":"4252-4256","source":"Crossref","is-referenced-by-count":0,"title":["AdaCGen: Heterogeneity-Aware Layer Management for Efficient KV Cache Offloading in LLMs"],"prefix":"10.1109","author":[{"given":"Yujiao","family":"Wang","sequence":"first","affiliation":[{"name":"Chongqing University,School of Computer Science,Chongqing,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiayi","family":"Guo","sequence":"additional","affiliation":[{"name":"Chongqing University,School of Computer Science,Chongqing,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zongjie","family":"Wang","sequence":"additional","affiliation":[{"name":"Chongqing University,School of Computer Science,Chongqing,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yujuan","family":"Tan","sequence":"additional","affiliation":[{"name":"Chongqing University,School of Computer Science,Chongqing,China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1177\/2053951720943234"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.ausmj.2020.03.005"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.482"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref6","first-page":"31094","article-title":"Flexgen: High-throughput generative inference of large language models with a single gpu","volume-title":"International Conference on Machine Learning","author":"Sheng"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"ref8","first-page":"155","article-title":"{InfiniGen}: Efficient generative inference of large language models with dynamic {KV} cache management","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Lee"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/dac63849.2025.11132479"},{"key":"ref10","article-title":"Shadowkv: Kv cache in shadows for high-throughput long-context llm inference","author":"Sun","year":"2024"},{"key":"ref11","article-title":"Infllm: Training-free long-context extrapolation for llms with an efficient context memory","author":"Xiao","year":"2024"},{"key":"ref12","article-title":"QUEST: query-aware sparsity for efficient long-context LLM inference","volume-title":"Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21-27, 2024","author":"Tang"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1506"},{"key":"ref14","article-title":"Scissorhands: Exploiting the persistence of importance hypothesis for LLM KV cache compression at test time","volume-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023","author":"Liu"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0722"},{"key":"ref16","article-title":"Pyramidkv: Dynamic kv cache compression based on pyramidal information funneling","author":"Cai","year":"2024"},{"key":"ref17","doi-asserted-by":"crossref","first-page":"3258","DOI":"10.18653\/v1\/2024.findings-acl.195","article-title":"PyramidInfer: Pyramid KV cache compression for high-throughput LLM inference","volume-title":"Findings of the Association for Computational Linguistics: ACL 2024","author":"Yang","year":"2024"},{"key":"ref18","article-title":"Windowkv: Task-adaptive group-wise KV cache window selection for efficient LLM inference","volume-title":"CoRR","author":"Zuo","year":"2025"},{"key":"ref19","first-page":"5998","article-title":"Attention is all you need","volume-title":"Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4-9, 2017, Long Beach, CA, USA","author":"Vaswani"},{"key":"ref20","article-title":"Opt: Open pre-trained transformer language models","author":"Zhang","year":"2022"},{"key":"ref21","article-title":"Llama 2: Open foundation and fine-tuned chat models","volume-title":"CoRR","author":"Touvron","year":"2023"},{"key":"ref22","article-title":"Compressive transformers for long-range sequence modelling","volume-title":"8th International Conference on Learning Representations, ICLR 2020, Addis Ababa, Ethiopia, April 26-30, 2020","author":"Rae"},{"key":"ref23","article-title":"A framework for few-shot language model evaluation","author":"Gao","year":"2021"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d19-1630"},{"key":"ref25","first-page":"1694","article-title":"Identifying similar words and contexts in natural language with senseclusters","volume-title":"Proceedings, The Twentieth National Conference on Artificial Intelligence and the Seventeenth Innovative Applications of Artificial Intelligence Conference, July 9-13, 2005, Pittsburgh, Pennsylvania, USA","author":"Pedersen"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-1260"}],"event":{"name":"2026 IEEE International Symposium on Circuits and Systems (ISCAS)","location":"Shanghai, China","start":{"date-parts":[[2026,5,24]]},"end":{"date-parts":[[2026,5,28]]}},"container-title":["2026 IEEE International Symposium on Circuits and Systems (ISCAS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11561899\/11561804\/11562764.pdf?arnumber=11562764","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T05:28:38Z","timestamp":1781846918000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11562764\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,24]]},"references-count":27,"URL":"https:\/\/doi.org\/10.1109\/iscas66217.2026.11562764","relation":{},"subject":[],"published":{"date-parts":[[2026,5,24]]}}}