{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T07:11:18Z","timestamp":1784099478875,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":36,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819227587","type":"print"},{"value":"9789819227594","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T00:00:00Z","timestamp":1784160000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T00:00:00Z","timestamp":1784160000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-2759-4_32","type":"book-chapter","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T06:51:06Z","timestamp":1784098266000},"page":"430-441","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["EMS-GL: Adaptive Evict-then-Merge Strategy for\u00a0KV Cache Compression Based on\u00a0Global-Local Importance"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-2924-869X","authenticated-orcid":false,"given":"Yingxin","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9690-2119","authenticated-orcid":false,"given":"Ye","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7450-9438","authenticated-orcid":false,"given":"Yuan","family":"Meng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0504-0186","authenticated-orcid":false,"given":"Xinzhu","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2690-3374","authenticated-orcid":false,"given":"Zihan","family":"Geng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8639-982X","authenticated-orcid":false,"given":"Shutao","family":"Xia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5462-6178","authenticated-orcid":false,"given":"Zhi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,16]]},"reference":[{"key":"32_CR1","unstructured":"Devlin, J.: Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"32_CR2","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, et al.: In: Larochelle, H., Ranzato, M., Hadsell, R., Balcan, M. F., Lin, H. (eds.) Advances in Neural Information Processing Systems, vol. 33, pp. 1877\u20131901. Curran Associates, Inc. (2020)"},{"key":"32_CR3","unstructured":"Anil, R., et al.: Palm 2 technical report. arXiv preprint arXiv:2305.10403 (2023)"},{"key":"32_CR4","unstructured":"Dubey, A., et al.: The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)"},{"key":"32_CR5","unstructured":"Jiang, A.Q., et al: Mistral 7B. arXiv preprint arXiv:2310.06825 (2023)"},{"key":"32_CR6","doi-asserted-by":"crossref","unstructured":"Ram, O., et al.: In-context retrieval-augmented language models. Trans. Assoc. Comput. Linguist. 1316\u20131331 (2023)","DOI":"10.1162\/tacl_a_00605"},{"key":"32_CR7","doi-asserted-by":"crossref","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models. In: Advances in Neural Information Processing Systems, pp. 24824\u201324837 (2022)","DOI":"10.52202\/068431-1800"},{"key":"32_CR8","unstructured":"Roziere, B., et al.: Code llama: open foundation models for code. arXiv preprint arXiv:2308.12950 (2023)"},{"key":"32_CR9","unstructured":"Jin, H., et al.: Llm maybe longlm: self-extend llm context window without tuning. arXiv preprint arXiv:2401.01325 (2024)"},{"key":"32_CR10","unstructured":"Chen, Y., et al.: LongLoRA: efficient fine-tuning of long-context large language models. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"32_CR11","unstructured":"Chen, S., Wong, S., Chen, L., Tian, Y.: Extending context window of large language models via positional interpolation. arXiv preprint arXiv:2306.15595 (2023)"},{"key":"32_CR12","unstructured":"Lazaridou, A., Gribovskaya, E., Stokowiec, W., Grigorev, N.: Internet-augmented language models through few-shot prompting for open-domain question answering. arXiv preprint arXiv:2203.05115 (2022)"},{"key":"32_CR13","unstructured":"Yuan, Z., et al.: Llm inference unveiled: survey and roofline model insights. arXiv preprint arXiv:2402.16363 (2024)"},{"key":"32_CR14","unstructured":"Zhou, Z., et al.: A survey on efficient inference for large language models. arXiv preprint arXiv:2404.14294 (2024)"},{"key":"32_CR15","unstructured":"Dao, T.: FlashAttention-2: faster attention with better parallelism and work partitioning. In: International Conference on Learning Representations (ICLR) (2024)"},{"key":"32_CR16","unstructured":"Xiao, G., Tian, Y., Chen, B., Han, S., Lewis, M.: Efficient streaming language models with attention sinks. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"32_CR17","unstructured":"Yu, Z., Wang, Z., Fu, Y., Shi, H., Shaikh, K., Lin, Y.C.: Unveiling and harnessing hidden attention sinks: enhancing large language models without training through attention calibration. In: Forty-first International Conference on Machine Learning (2024)"},{"key":"32_CR18","doi-asserted-by":"crossref","unstructured":"Zhang, Z., et al.: H2O: heavy-hitter oracle for efficient generative inference of large language models. In: Thirty-Seventh Conference on Neural Information Processing Systems (2023)","DOI":"10.52202\/075280-1506"},{"key":"32_CR19","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Snapkv: Llm knows what you are looking for before generation. arXiv preprint arXiv:2404.14469 (2024)","DOI":"10.52202\/079017-0722"},{"key":"32_CR20","first-page":"114","volume":"6","author":"M Adnan","year":"2024","unstructured":"Adnan, M., Arunkumar, A., Jain, G., Nair, P., Soloveychik, I., Kamath, P.: Keyformer: Kv cache reduction through key tokens selection for efficient generative inference. Proc. Mach. Learn. Syst. 6, 114\u2013127 (2024)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"32_CR21","doi-asserted-by":"crossref","unstructured":"Wang, H., Zhang, Z.K., Han, S.: Spatten: efficient sparse attention architecture with cascade token and head pruning. In: 2021 IEEE International Symposium on High-Performance Computer Architecture (HPCA), pp. 97\u2013110 (2021)","DOI":"10.1109\/HPCA51647.2021.00018"},{"key":"32_CR22","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Scissorhands: exploiting the persistence of importance hypothesis for LLM KV cache compression at test time. In: Thirty-seventh Conference on Neural Information Processing Systems (2023)","DOI":"10.52202\/075280-2279"},{"key":"32_CR23","doi-asserted-by":"crossref","unstructured":"Anagnostidis, S., Pavllo, D., Biggio, L., Noci, L., Lucchi, A., Hofmann, T.: Dynamic context pruning for efficient and interpretable autoregressive transformers. In: Thirty-seventh Conference on Neural Information Processing Systems (2023)","DOI":"10.52202\/075280-2845"},{"key":"32_CR24","unstructured":"Ge, S., Zhang, Y., Liu, L., Zhang, M., Han, J., Gao, J.: Model tells you what to discard: adaptive KV cache compression for LLMs. In: Workshop on Advancing Neural Network Training: Computational Efficiency, Scalability, and Resource Optimization (WANT@NeurIPS 2023) (2023)"},{"key":"32_CR25","unstructured":"Zhang, Y., et al.: CaM: cache merging for memory-efficient LLMs inference. In: International Conference on Machine Learning (2024)"},{"key":"32_CR26","unstructured":"Nawrot, P., \u0141a\u0144cucki, A., Chochowski, M., Tarjan, D., Ponti, E.: Dynamic memory compression: retrofitting LLMs for accelerated inference. In: Forty-first International Conference on Machine Learning (2024)"},{"key":"32_CR27","unstructured":"Dong, H., Yang, X., Zhang, Z., Wang, Z., Chi, Y., Chen, B.: Get More with LESS: synthesizing recurrence with KV cache compression for efficient LLM inference. In: Forty-first International Conference on Machine Learning (2024)"},{"issue":"8","key":"32_CR28","doi-asserted-by":"publisher","first-page":"2441","DOI":"10.1007\/s11431-022-2216-9","volume":"66","author":"Y Li","year":"2023","unstructured":"Li, Y., Liu, Z.X., Lan, G., Sader, M., Chen, Z.Q.: A DDPG-based solution for optimal consensus of continuous-time linear multi-agent systems. SCIENCE CHINA Technol. Sci. 66(8), 2441\u20132453 (2023)","journal-title":"SCIENCE CHINA Technol. Sci."},{"key":"32_CR29","doi-asserted-by":"publisher","first-page":"111430","DOI":"10.1016\/j.knosys.2024.111430","volume":"287","author":"ZX Liu","year":"2024","unstructured":"Liu, Z.X., Li, Y., Lan, G., Chen, Z.Q.: A novel data-driven model-free synchronization protocol for discrete-time multi-agent systems via TD3 based algorithm. Knowl.-Based Syst. 287, 111430 (2024)","journal-title":"Knowl.-Based Syst."},{"key":"32_CR30","unstructured":"Li, D., et al.: How long can context length of open-source LLMs truly promise?. In: NeurIPS 2023 Workshop on Instruction Tuning and Instruction Following (2023)"},{"key":"32_CR31","unstructured":"Touvron, H., et al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"32_CR32","unstructured":"Bai, Y., et al.: Longbench: a bilingual, multitask benchmark for long context understanding. arXiv preprint arXiv:2308.14508 (2023)"},{"key":"32_CR33","doi-asserted-by":"publisher","unstructured":"Huang, L., Cao, S., Parulian, N., Ji, H., Wang, L.: Efficient attentions for long document summarization. In: Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 1419\u20131436. ACL (2021). https:\/\/doi.org\/10.18653\/v1\/2021.naacl-main.112","DOI":"10.18653\/v1\/2021.naacl-main.112"},{"key":"32_CR34","unstructured":"Kamradt, G.: Needle in a Haystack-pressure testing LLMs (2023)"},{"key":"32_CR35","unstructured":"Shi, F., et al.: Large language models can be easily distracted by irrelevant context. In: International Conference on Machine Learning, pp. 31210\u201331227. PMLR (2023)"},{"key":"32_CR36","doi-asserted-by":"crossref","unstructured":"Yang, Z., et al.: HotpotQA: a dataset for diverse, explainable multi-hop question answering. arXiv preprint arXiv:1809.09600 (2018)","DOI":"10.18653\/v1\/D18-1259"}],"container-title":["Lecture Notes in Computer Science","Knowledge Science, Engineering and Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-2759-4_32","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T06:51:11Z","timestamp":1784098271000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-2759-4_32"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,16]]},"ISBN":["9789819227587","9789819227594"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-2759-4_32","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,16]]},"assertion":[{"value":"16 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"KSEM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Knowledge Science, Engineering and Management","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Beijing","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ksem2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ksem2026.rosc.org.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}