{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:30:55Z","timestamp":1787495455024,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":39,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_32","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:47:47Z","timestamp":1787492867000},"page":"483-497","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["InputSnatch: Stealing Input in\u00a0LLM Services via\u00a0Cache-Sharing Timing Side-Channel Attacks"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-7727-440X","authenticated-orcid":false,"given":"Xinyao","family":"Zheng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0873-3471","authenticated-orcid":false,"given":"Husheng","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3443-0474","authenticated-orcid":false,"given":"Shangyi","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-3645-5708","authenticated-orcid":false,"given":"Qiyan","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7603-4210","authenticated-orcid":false,"given":"Zidong","family":"Du","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9979-0561","authenticated-orcid":false,"given":"Xing","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2530-5874","authenticated-orcid":false,"given":"Qi","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"32_CR1","unstructured":"Anthropic: Prompt caching (beta) (2024). https:\/\/docs.anthropic.com\/en\/docs\/build-with-claude\/prompt-caching"},{"key":"32_CR2","unstructured":"Azure, M.: Tutorial: use azure cache for Redis as a semantic cache (2024). https:\/\/learn.microsoft.com\/en-us\/azure\/azure-cache-for-redis\/cache-tutorial-semantic-cache"},{"key":"32_CR3","doi-asserted-by":"crossref","unstructured":"Bang, F.: GPTCache: an open-source semantic cache for LLM applications enabling faster answers and cost savings. In: Proceedings of the 3rd Workshop for Natural Language Processing Open Source Software (NLP-OSS 2023), pp. 212\u2013218 (2023)","DOI":"10.18653\/v1\/2023.nlposs-1.24"},{"key":"32_CR4","unstructured":"Carlini, N., Nasr, M.: Remote timing attacks on efficient language model inference. arXiv preprint arXiv:2410.17175 (2024)"},{"key":"32_CR5","doi-asserted-by":"crossref","unstructured":"Chen, Y., Lent, H., Bjerva, J.: Text embedding inversion security for multilingual language models. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 7808\u20137827 (2024)","DOI":"10.18653\/v1\/2024.acl-long.422"},{"key":"32_CR6","unstructured":"DeepSeek: DeepSeek API introduces context caching on disk, cutting prices by an order of magnitude (2024). https:\/\/api-docs.deepseek.com\/news\/news0802\/"},{"key":"32_CR7","unstructured":"Ding, Y., et al.: LongRoPE: extending LLM context window beyond 2 million tokens. arXiv preprint arXiv:2402.13753 (2024)"},{"key":"32_CR8","unstructured":"Gill, W., Elidrisi, M., Kalapatapu, P., Anwar, A., Gulzar, M.A.: Privacy-aware semantic cache for large language models. arXiv:2403.02694 arXiv preprint (2024)"},{"key":"32_CR9","unstructured":"Google: Gemini API: Google AI for Developers (2024). https:\/\/ai.google.dev\/gemini-api\/docs\/caching?lang=python"},{"key":"32_CR10","unstructured":"Hugging Face: Text Generation Inference: Large language model text generation inference (2022). https:\/\/github.com\/huggingface\/text-generation-inference"},{"key":"32_CR11","unstructured":"Juravsky, J., Brown, B., Ehrlich, R., Fu, D.Y., R\u00e9, C., Mirhoseini, A.: Hydragen: high-throughput LLM inference with shared prefixes. arXiv preprint arXiv:2402.05099 (2024)"},{"key":"32_CR12","unstructured":"Kanter, S.F.: Improve speed and reduce cost for generative AI workloads with a persistent semantic cache in Amazon MemoryDB (2024). https:\/\/aws.amazon.com\/cn\/blogs\/database\/improve-speed-and-reduce-cost-for-generative-ai-workloads-with-a-persistent-semantic-cache-in-amazon-memorydb\/"},{"key":"32_CR13","doi-asserted-by":"crossref","unstructured":"Kwon, W., et al.: Efficient memory management for large language model serving with pagedattention. In: Proceedings of the 29th Symposium on Operating Systems Principles, pp. 611\u2013626 (2023)","DOI":"10.1145\/3600006.3613165"},{"key":"32_CR14","doi-asserted-by":"crossref","unstructured":"Li, J., Xu, C., Wang, F., von Riedemann, I.M., Zhang, C., Liu, J.: SCALM: towards semantic caching for automated chat services with large language models. arXiv preprint arXiv:2406.00025 (2024)","DOI":"10.1109\/IWQoS61813.2024.10682957"},{"key":"32_CR15","doi-asserted-by":"crossref","unstructured":"Li, Y., Li, Z., Zhang, K., Dan, R., Jiang, S., Zhang, Y.: ChatDoctor: a medical chat model fine-tuned on a large language model meta-AI (LLaMA) using medical domain knowledge. Cureus 15(6) (2023)","DOI":"10.7759\/cureus.40895"},{"key":"32_CR16","unstructured":"Liang, Z., Hu, H., Ye, Q., Xiao, Y., Li, H.: Why are my prompts leaked? Unraveling prompt extraction threats in customized large language models. arXiv preprint arXiv:2408.02416 (2024)"},{"key":"32_CR17","unstructured":"Liu, H.: CrimeKgAssitant: crime assistant including crime type prediction and crime consult service based on NLP methods and crime KG (2018). https:\/\/github.com\/liuhuanyong\/CrimeKgAssitant"},{"key":"32_CR18","doi-asserted-by":"publisher","unstructured":"Mahajan, S., Rahman, T., Yi, K.M., Sigal, L.: Prompting hard or hardly prompting: prompt inversion for text-to-image diffusion models. In: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6808\u20136817 (2024). https:\/\/doi.org\/10.1109\/CVPR52733.2024.00650","DOI":"10.1109\/CVPR52733.2024.00650"},{"key":"32_CR19","doi-asserted-by":"crossref","unstructured":"Mohandoss, R.: Context-based semantic caching for LLM applications. In: 2024 IEEE Conference on Artificial Intelligence (CAI), pp. 371\u2013376. IEEE (2024)","DOI":"10.1109\/CAI59869.2024.00075"},{"key":"32_CR20","unstructured":"Morris, J.X., Zhao, W., Chiu, J.T., Shmatikov, V., Rush, A.M.: Language model inversion. arXiv preprint arXiv:2311.13647 (2023)"},{"key":"32_CR21","doi-asserted-by":"crossref","unstructured":"Morris, J.X., Kuleshov, V., Shmatikov, V., Rush, A.M.: Text embeddings reveal (almost) as much as text. In: The 2023 Conference on Empirical Methods in Natural Language Processing (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.765"},{"key":"32_CR22","unstructured":"NVIDIA Corporation: NVIDIA TensorRT-LLM (2026). https:\/\/docs.nvidia.com\/tensorrt-llm\/index.html"},{"key":"32_CR23","unstructured":"OpenAI: Introducing the GPT Store We\u2019re launching the GPT Store to help you find (2024). https:\/\/chat.openai.com\/gpts"},{"key":"32_CR24","unstructured":"OpenAI: Prompt caching: Reduce latency and cost with prompt caching (2024). https:\/\/platform.openai.com\/docs\/guides\/prompt-caching"},{"key":"32_CR25","unstructured":"OpenAI: OpenAI API: Rate limits (2026). https:\/\/developers.openai.com\/api\/docs\/guides\/rate-limits"},{"key":"32_CR26","unstructured":"Perez, F., Ribeiro, I.: Ignore previous prompt: attack techniques for language models. arXiv preprint arXiv:2211.09527 (2022)"},{"key":"32_CR27","unstructured":"Shen, X., Qu, Y., Backes, M., Zhang, Y.: Prompt stealing attacks against $$\\{$$Text-to-Image$$\\}$$ generation models. In: 33rd USENIX Security Symposium (USENIX Security 24), pp. 5823\u20135840 (2024)"},{"key":"32_CR28","doi-asserted-by":"crossref","unstructured":"Song, C., Raghunathan, A.: Information leakage in embedding models. In: Proceedings of the 2020 ACM SIGSAC Conference on Computer and Communications Security, pp. 377\u2013390 (2020)","DOI":"10.1145\/3372297.3417270"},{"key":"32_CR29","doi-asserted-by":"publisher","unstructured":"Song, L., et al.: The early bird catches the leak: unveiling timing side channels in LLM serving systems. IEEE Trans. Inf. Forensics Secur. 20, 11431\u201311446 (2025). https:\/\/doi.org\/10.1109\/TIFS.2025.3622954","DOI":"10.1109\/TIFS.2025.3622954"},{"key":"32_CR30","unstructured":"Talarian: OpenAI GPT prompt generator (2024). https:\/\/gptforwork.com\/tools\/prompt-generator"},{"key":"32_CR31","doi-asserted-by":"crossref","unstructured":"Tao, C., et al.: Scaling laws with vocabulary: larger models deserve larger vocabularies. arXiv preprint arXiv:2407.13623 (2024)","DOI":"10.52202\/079017-3626"},{"key":"32_CR32","unstructured":"Technology, Z.: Zuoshouyisheng open platforms (2024). https:\/\/open.zuoshouyisheng.com\/"},{"key":"32_CR33","unstructured":"Wei, J., Abdulrazzag, A., Zhang, T., Muursepp, A., Saileshwar, G.: When speculation spills secrets: side channels via speculative decoding in LLMs (2025). https:\/\/arxiv.org\/abs\/2411.01076"},{"key":"32_CR34","unstructured":"Weiss, R., Ayzenshteyn, D., Amit, G., Mirsky, Y.: What was your prompt? A remote keylogging attack on ai assistants. arXiv preprint arXiv:2403.09751 (2024)"},{"key":"32_CR35","doi-asserted-by":"crossref","unstructured":"Wu, G., et al.: I know what you asked: prompt leakage via kv-cache sharing in multi-tenant LLM serving. In: Proceedings of the 2025 Network and Distributed System Security (NDSS) Symposium, San Diego, CA, USA (2025)","DOI":"10.14722\/ndss.2025.241772"},{"key":"32_CR36","unstructured":"Yang, Y., et al.: PRSA: prompt reverse stealing attacks against large language models. arXiv preprint arXiv:2402.19200 (2024)"},{"key":"32_CR37","doi-asserted-by":"crossref","unstructured":"Ye, L., Tao, Z., Huang, Y., Li, Y.: ChunkAttention: efficient self-attention with prefix-aware kv cache and two-phase partition. arXiv preprint arXiv:2402.15220 (2024)","DOI":"10.18653\/v1\/2024.acl-long.623"},{"key":"32_CR38","unstructured":"Zhang, Y., Carlini, N., Ippolito, D.: Effective prompt extraction from language models. In: First Conference on Language Modeling (2024)"},{"key":"32_CR39","unstructured":"Zheng, L., et al.: Efficiently programming large language models using SGLang. arXiv preprint arXiv:2312.07104 (2023)"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_32","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:47:50Z","timestamp":1787492870000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_32"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":39,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_32","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The technical solution and experimental design presented in this paper were independently completed by the authors. AI tools were used solely for language polishing and formatting optimization, and did not participate in the development of the research ideas or core content.","order":1,"name":"Ethics","label":"AI Disclosure","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}