{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T08:03:47Z","timestamp":1784189027631,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":23,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234165","type":"print"},{"value":"9789819234172","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T00:00:00Z","timestamp":1784246400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T00:00:00Z","timestamp":1784246400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3417-2_11","type":"book-chapter","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T07:10:14Z","timestamp":1784185814000},"page":"119-130","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["When Sophisticated Eviction Fails: An Empirical Study of KV Cache Optimization for Hybrid Attention Models"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-6208-4222","authenticated-orcid":false,"given":"Zihan","family":"Tian","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,17]]},"reference":[{"key":"11_CR1","volume-title":"NeurIPS","author":"Z Zhang","year":"2023","unstructured":"Zhang, Z., et al.: H2O: heavy-hitter oracle for efficient generative inference of large language models. In: NeurIPS (2023)"},{"key":"11_CR2","unstructured":"Xiao, G., et al.: Efficient streaming language models with attention sinks. arXiv:2309.17453. (2023)"},{"key":"11_CR3","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: SnapKV: LLM knows what you are looking for before generation. arXiv:2404.14469. (2024)","DOI":"10.52202\/079017-0722"},{"key":"11_CR4","unstructured":"Cai, Z., et al.: Pyramid KV: dynamic KV cache compression based on pyramidal information funneling. arXiv:2406.02069. (2024)"},{"key":"11_CR5","doi-asserted-by":"crossref","unstructured":"Zhou, X., et al.: DynamicKV: task-aware adaptive KV cache compression. arXiv:2412.14838. (2024)","DOI":"10.18653\/v1\/2025.findings-emnlp.426"},{"key":"11_CR6","unstructured":"Zeng, H., et al.: Lethe: layer and time-adaptive KV cache pruning. arXiv:2511.06029. (2025)"},{"key":"11_CR7","unstructured":"Tang, H., et al.: RazorAttention: efficient KV cache compression through retrieval heads. arXiv:2407.15891. (2024)"},{"key":"11_CR8","unstructured":"Tang, J., et al.: Quest: query-aware sparsity for efficient long-context LLM inference. arXiv:2406.10774. (2024)"},{"key":"11_CR9","unstructured":"Kim, J.H., et al.: KVzip: query-agnostic KV cache compression. arXiv:2505.23416. (2025)"},{"key":"11_CR10","unstructured":"Yang, A., et al.: Qwen2.5: a party of foundation models. arXiv:2412.15115. (2024)"},{"key":"11_CR11","unstructured":"Lieber, O., et al.: Jamba: a hybrid transformer-mamba language model. arXiv:2403.19887. (2024)"},{"key":"11_CR12","unstructured":"De, S., et al.: Griffin: mixing gated linear recurrences with local attention. arXiv:2402.19427. (2024)"},{"key":"11_CR13","volume-title":"SOSP","author":"W Kwon","year":"2023","unstructured":"Kwon, W., et al.: Efficient memory management for large language model serving with PagedAttention. In: SOSP (2023)"},{"key":"11_CR14","doi-asserted-by":"crossref","unstructured":"Dao, T., et al.: FlashAttention: fast and memory-efficient exact attention. arXiv:2205.14135. (2022)","DOI":"10.52202\/068431-1189"},{"key":"11_CR15","unstructured":"Gu, A., Dao, T.: Mamba: linear-time sequence modeling with selective state spaces. arXiv:2312.00752. (2023)"},{"key":"11_CR16","doi-asserted-by":"crossref","unstructured":"Peng, B., et al.: RWKV: reinventing RNNs for the transformer era. arXiv:2305.13048. (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.936"},{"key":"11_CR17","unstructured":"Sun, Y., et al.: Retentive network: a successor to transformer. arXiv:2307.08621. (2023)"},{"key":"11_CR18","unstructured":"Yang, S., et al.: Gated linear attention transformers with hardware-efficient training. arXiv:2312.06635. (2023)"},{"key":"11_CR19","volume-title":"Needle in a Haystack-Pressure Testing LLMs","author":"G Kamradt","year":"2023","unstructured":"Kamradt, G.: Needle in a Haystack-Pressure Testing LLMs. GitHub Repository (2023)"},{"key":"11_CR20","unstructured":"Hsieh, C.P., et al.: RULER: what\u2019s the real context size of your long-context LLMs? arXiv:2404.06654. (2024)"},{"key":"11_CR21","unstructured":"Bai, Y., et al.: LongBench: a bilingual, multitask benchmark for long context understanding. arXiv:2308.14508. (2023)"},{"key":"11_CR22","doi-asserted-by":"crossref","unstructured":"Liu, N.F., et al.: Lost in the middle: how language models use long contexts. TACL. (2024)","DOI":"10.1162\/tacl_a_00638"},{"issue":"6","key":"11_CR23","doi-asserted-by":"publisher","first-page":"657","DOI":"10.1007\/BF01068419","volume":"15","author":"DJW Schuirmann","year":"1987","unstructured":"Schuirmann, D.J.W.: A comparison of the two one-sided tests procedure. J. Pharmacokinet. Biopharm. 15(6), 657\u2013680 (1987)","journal-title":"J. Pharmacokinet. Biopharm."}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3417-2_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T07:10:25Z","timestamp":1784185825000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3417-2_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,17]]},"ISBN":["9789819234165","9789819234172"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3417-2_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,17]]},"assertion":[{"value":"17 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}