{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:30:55Z","timestamp":1787495455062,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":30,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_6","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:47:50Z","timestamp":1787492870000},"page":"82-96","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["LocalKV: Leveraging Sparse Attention Locality for Efficient Long-Context LLM Inference"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-6652-1188","authenticated-orcid":false,"given":"Chengwei","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6108-688X","authenticated-orcid":false,"given":"Guangda","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8211-2812","authenticated-orcid":false,"given":"Jieru","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5832-0347","authenticated-orcid":false,"given":"Quan","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0034-2302","authenticated-orcid":false,"given":"Minyi","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"6_CR1","doi-asserted-by":"publisher","unstructured":"Ainslie, J., Lee-Thorp, J., de\u00a0Jong, M., Zemlyanskiy, Y., Lebron, F., Sanghai, S.: GQA: training generalized multi-query transformer models from multi-head checkpoints. In: Bouamor, H., Pino, J., Bali, K. (eds.) Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 4895\u20134901. Association for Computational Linguistics, Singapore (2023). https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.298. https:\/\/aclanthology.org\/2023.emnlp-main.298","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"6_CR2","unstructured":"Bai, Y., et al.: Longbench: a bilingual, multitask benchmark for long context understanding. arXiv preprint arXiv:2308.14508 (2023)"},{"key":"6_CR3","doi-asserted-by":"crossref","unstructured":"Chen, R., et al.: Arkvale: efficient generative LLM inference with recallable key-value eviction. In: Globerson, A., et al. (eds.) Advances in Neural Information Processing Systems, vol. 37, pp. 113134\u2013113155. Curran Associates, Inc. (2024). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2024\/file\/cd4b49379efac6e84186a3ffce108c37-Paper-Conference.pdf","DOI":"10.52202\/079017-3595"},{"key":"6_CR4","unstructured":"Chew, R., Bollenbacher, J., Wenger, M., Speer, J., Kim, A.: LLM-assisted content analysis: using large language models to support deductive coding (2023). https:\/\/arxiv.org\/abs\/2306.14924"},{"key":"6_CR5","unstructured":"DeepMind, G.: Gemini 2.5: our most intelligent AI model (2025)"},{"key":"6_CR6","unstructured":"Guo, D., et al.: Deepseek-r1: incentivizing reasoning capability in LLMs via reinforcement learning (2025). https:\/\/arxiv.org\/abs\/2501.12948"},{"key":"6_CR7","unstructured":"Liu, A., et al.: Deepseek-v2: a strong, economical, and efficient mixture-of-experts language model (2024). https:\/\/arxiv.org\/abs\/2405.04434"},{"key":"6_CR8","unstructured":"Hendrycks, D., et al.: Measuring mathematical problem solving with the math dataset. In: Vanschoren, J., Yeung, S. (eds.) Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks, vol. 1 (2021). https:\/\/datasets-benchmarks-proceedings.neurips.cc\/paper_files\/paper\/2021\/file\/be83ab3ecd0db773eb2dc1b0a17836a1-Paper-round2.pdf"},{"key":"6_CR9","unstructured":"Hochlehnert, A., Bhatnagar, H., Udandarao, V., Albanie, S., Prabhu, A., Bethge, M.: A sober look at progress in language model reasoning: pitfalls and paths to reproducibility (2025). https:\/\/arxiv.org\/abs\/2504.07086"},{"key":"6_CR10","unstructured":"Hu, J., et al.: Efficient long-decoding inference with reasoning-aware attention sparsity (2025). https:\/\/arxiv.org\/abs\/2502.11147"},{"key":"6_CR11","unstructured":"Huggingface: Maxwell-jia\/aime_2024 (2025). https:\/\/huggingface.co\/datasets\/Maxwell-Jia\/AIME_2024"},{"key":"6_CR12","doi-asserted-by":"crossref","unstructured":"Jiang, H., et al.: Minference 1.0: accelerating pre-filling for long-context LLMs via dynamic sparse attention (2024). https:\/\/arxiv.org\/abs\/2407.02490","DOI":"10.52202\/079017-1663"},{"key":"6_CR13","unstructured":"Jiang, J., Wang, F., Shen, J., Kim, S., Kim, S.: A survey on large language models for code generation (2024). https:\/\/arxiv.org\/abs\/2406.00515"},{"key":"6_CR14","unstructured":"Kamradt, G.: Needle in a haystack - pressure testing LLMs (2023). https:\/\/github.com\/gkamradt\/LLMTest_NeedleInAHaystack"},{"key":"6_CR15","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Snapkv: LLM knows what you are looking for before generation (2024). https:\/\/arxiv.org\/abs\/2404.14469","DOI":"10.52202\/079017-0722"},{"key":"6_CR16","unstructured":"Liu, G., Li, C., Ning, Z., Guo, M., Zhao, J.: Freekv: boosting kv cache retrieval for efficient LLM inference (2025). https:\/\/arxiv.org\/abs\/2505.13109"},{"key":"6_CR17","unstructured":"Liu, G., Li, C., Zhao, J., Zhang, C., Guo, M.: Clusterkv: manipulating LLM kv cache in semantic space for recallable compression (2024). https:\/\/arxiv.org\/abs\/2412.03213"},{"key":"6_CR18","unstructured":"Meta: The llama 3 herd of models (2024). https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"6_CR19","unstructured":"Qwen: Qwen3: Think deeper, act faster (2025)"},{"key":"6_CR20","unstructured":"Rein, D., et al.: GPQA: a graduate-level google-proof Q&A benchmark (2023). https:\/\/arxiv.org\/abs\/2311.12022"},{"key":"6_CR21","doi-asserted-by":"crossref","unstructured":"Su, J., Lu, Y., Pan, S., Murtadha, A., Wen, B., Liu, Y.: Roformer: enhanced transformer with rotary position embedding (2023). https:\/\/arxiv.org\/abs\/2104.09864","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"6_CR22","unstructured":"Tang, H., et al.: Razorattention: efficient kv cache compression through retrieval heads (2024). https:\/\/arxiv.org\/abs\/2407.15891"},{"key":"6_CR23","unstructured":"Tang, J., Zhao, Y., Zhu, K., Xiao, G., Kasikci, B., Han, S.: Quest: query-aware sparsity for efficient long-context LLM inference. In: ICML (2024)"},{"key":"6_CR24","unstructured":"xAI: Grok 3 beta \u2014 the age of reasoning agents (2025). https:\/\/x.ai\/news\/grok-3"},{"key":"6_CR25","unstructured":"Xiao, G., et al.: Duoattention: efficient long-context LLM inference with retrieval and streaming heads (2024). https:\/\/arxiv.org\/abs\/2410.10819"},{"key":"6_CR26","unstructured":"Xiao, G., Tian, Y., Chen, B., Han, S., Lewis, M.: Efficient streaming language models with attention sinks. In: ICLR (2024)"},{"key":"6_CR27","doi-asserted-by":"publisher","unstructured":"Xie, X., Yu, Z., Liao, Y., Wang, T., Toh, K.C., Yan, S.: Slow-Fast Inference: Training-Free Inference Acceleration via Within-Sentence Support Stability (2026). https:\/\/doi.org\/10.48550\/arXiv.2603.12038","DOI":"10.48550\/arXiv.2603.12038"},{"key":"6_CR28","doi-asserted-by":"publisher","unstructured":"Xu, R., Xiao, G., Huang, H., Guo, J., Han, S.: XAttention: Block Sparse Attention with Antidiagonal Scoring (2025). https:\/\/doi.org\/10.48550\/arXiv.2503.16428","DOI":"10.48550\/arXiv.2503.16428"},{"key":"6_CR29","unstructured":"Yuan, J., et al.: Native sparse attention: hardware-aligned and natively trainable sparse attention (2025). https:\/\/arxiv.org\/abs\/2502.11089"},{"key":"6_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, Z., et al.: H2o: heavy-hitter oracle for efficient generative inference of large language models. In: Oh, A., Naumann, T., Globerson, A., Saenko, K., Hardt, M., Levine, S. (eds.) Advances in Neural Information Processing Systems, vol.\u00a036, pp. 34661\u201334710. Curran Associates, Inc. (2023)","DOI":"10.52202\/075280-1506"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:47:54Z","timestamp":1787492874000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.\n                      Artificial intelligence tools, if used, were used only for language organization and figure\/table polishing, and did not participate in the generation of the core ideas or technical content.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}