{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T22:24:49Z","timestamp":1775082289087,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":47,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819584048","type":"print"},{"value":"9789819584055","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-8405-5_9","type":"book-chapter","created":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T20:15:01Z","timestamp":1775074501000},"page":"151-171","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["EACC: Efficient Agent Context Cache Sharing for\u00a0Multi-Agent Systems"],"prefix":"10.1007","author":[{"given":"Sihao","family":"Cheng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunfei","family":"Gu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chentao","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Meng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,2]]},"reference":[{"key":"9_CR1","unstructured":"Agarwal, M., Qureshi, A., Sardana, N., Li, L., Quevedo, J., Khudia, D.: LLM inference performance engineering: Best practices. https:\/\/www.databricks.com\/blog\/llm-inference-performance-engineering-best-practices (October 2023). Accessed 30 July 2025"},{"key":"9_CR2","unstructured":"AI, M.: Meta\u2011llama\u20113\u20118b\u2011instruct. https:\/\/huggingface.co\/meta-llama\/Meta-Llama-3-8B-Instruct (2024), released: April 18, 2024; Accessed3 Aug 2025"},{"key":"9_CR3","unstructured":"Ashish, V.: Attention is all you need. Advances in neural information processing systems 30, I (2017)"},{"key":"9_CR4","doi-asserted-by":"publisher","unstructured":"Bai, Y., L et al.: LongBench: A bilingual, multitask benchmark for long context understanding. In: Ku, L.W., Martins, A., Srikumar, V. (eds.) Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 3119\u20133137. Association for Computational Linguistics, Bangkok, Thailand (Aug 2024). https:\/\/doi.org\/10.18653\/v1\/2024.acl-long.172, https:\/\/aclanthology.org\/2024.acl-long.172\/","DOI":"10.18653\/v1\/2024.acl-long.172"},{"key":"9_CR5","unstructured":"Banerjee, D., Singh, P., Avadhanam, A., Srivastava, S.: Benchmarking LLM powered chatbots: Methods and metrics (2023). https:\/\/arxiv.org\/abs\/2308.04624"},{"key":"9_CR6","unstructured":"Briggs, J.: the\u00a0vllm\u2011project contributors: agent\u2011conversations\u2011retrieval\u2011tool dataset. https:\/\/huggingface.co\/datasets\/jamescalam\/agent-conversations-retrieval-tool (2023), updated: August 27, 2023; Accessed 18 July 2025"},{"key":"9_CR7","unstructured":"Chen, W., et al.: Agentverse: Facilitating multi-agent collaboration and exploring emergent behaviors. In: The Twelfth International Conference on Learning Representations (2024). https:\/\/openreview.net\/forum?id=EHg5GDnyq1"},{"key":"9_CR8","unstructured":"Cheng, Y., et al.: Exploring large language model based intelligent agents: Definitions, methods, and prospects (2024). https:\/\/arxiv.org\/abs\/2401.03428"},{"key":"9_CR9","unstructured":"vLLM Contributors: vLLM: Fast and easy LLM inference. https:\/\/github.com\/vllm-project\/vllm\/tree\/v0.4.1 (2025). Accessed 16 July 2025"},{"key":"9_CR10","unstructured":"Contributors, R.P.: Ray: A framework for scalable distributed computing. https:\/\/github.com\/ray-project\/ray (2025). Accessed 18 July 2025"},{"key":"9_CR11","doi-asserted-by":"publisher","unstructured":"Dong, X., Zhang, X., Bu, W., Zhang, D., Cao, F.: A survey of llm-based agents: Theories, technologies, applications and suggestions. In: 2024 3rd International Conference on Artificial Intelligence, Internet of Things and Cloud Computing Technology (AIoTC), pp. 407\u2013413 (2024). https:\/\/doi.org\/10.1109\/AIoTC63215.2024.10748304","DOI":"10.1109\/AIoTC63215.2024.10748304"},{"key":"9_CR12","unstructured":"Du, Y., Li, S., Torralba, A., Tenenbaum, J.B., Mordatch, I.: Improving factuality and reasoning in language models through multiagent debate. In: Forty-first International Conference on Machine Learning (2023)"},{"key":"9_CR13","unstructured":"Foundation, P.: Pytorch: an open source machine learning framework. https:\/\/pytorch.org (2025). Accessed 30 July 2025"},{"key":"9_CR14","unstructured":"Gim, I., Chen, G., Lee, S.S., Sarda, N., Khandelwal, A., Zhong, L.: Prompt cache: Modular attention reuse for low-latency inference. In: MLSys (2024). https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2024\/hash\/ a66caa1703fe34705a4368c3014c1966-Abstract-Conference.html"},{"key":"9_CR15","doi-asserted-by":"publisher","unstructured":"Ho, X., Duong\u00a0Nguyen, A.K., Sugawara, S., Aizawa, A.: Constructing a multi-hop QA dataset for comprehensive evaluation of reasoning steps. In: Scott, D., Bel, N., Zong, C. (eds.) Proceedings of the 28th International Conference on Computational Linguistics, pp. 6609\u20136625. International Committee on Computational Linguistics, Barcelona, Spain (Online) (Dec 2020). https:\/\/doi.org\/10.18653\/v1\/2020.coling-main.580, https:\/\/aclanthology.org\/2020.coling-main.580\/","DOI":"10.18653\/v1\/2020.coling-main.580"},{"key":"9_CR16","unstructured":"Hu, C., et\u00a0al.: Memserve: Context caching for disaggregated LLM serving with elastic memory pool. arXiv preprint arXiv:2406.17565 (2024)"},{"key":"9_CR17","unstructured":"Hu, J., et al.: EPIC: Efficient position-independent caching for serving large language models. In: Forty-second International Conference on Machine Learning (2025). https:\/\/openreview.net\/forum?id=qjd3ZUiHRT"},{"key":"9_CR18","doi-asserted-by":"publisher","unstructured":"Hu, J., Wang, C., Huang, H., Luo, H., Jin, Y., Deng, Y., Xie, T.: Predicting compilation resources for adaptive build in an industrial setting. In: 2023 38th IEEE\/ACM International Conference on Automated Software Engineering (ASE), pp. 1808\u20131813 (2023). https:\/\/doi.org\/10.1109\/ASE56229.2023.00128","DOI":"10.1109\/ASE56229.2023.00128"},{"key":"9_CR19","doi-asserted-by":"publisher","unstructured":"Ji, Z., Yu, T., Xu, Y., Lee, N., Ishii, E., Fung, P.: Towards mitigating LLM hallucination via self reflection. In: Bouamor, H., Pino, J., Bali, K. (eds.) Findings of the Association for Computational Linguistics: EMNLP 2023, pp. 1827\u20131843. Association for Computational Linguistics, Singapore (Dec 2023). https:\/\/doi.org\/10.18653\/v1\/2023.findings-emnlp.123, https:\/\/aclanthology.org\/2023.findings-emnlp.123\/","DOI":"10.18653\/v1\/2023.findings-emnlp.123"},{"key":"9_CR20","unstructured":"Ke, Z., Shao, Y., Lin, H., Konishi, T., Kim, G., Liu, B.: Continual pre-training of language models. In: The Eleventh International Conference on Learning Representations (2023). https:\/\/openreview.net\/forum?id=m_GDIItaI3o"},{"key":"9_CR21","doi-asserted-by":"crossref","unstructured":"Kwon, W., et al.: Efficient memory management for large language model serving with pagedattention. In: Proceedings of the 29th Symposium on Operating Systems Principles, pp. 611\u2013626 (2023)","DOI":"10.1145\/3600006.3613165"},{"key":"9_CR22","unstructured":"LangChain: Langchain \u2013 the platform for reliable agents. https:\/\/www.langchain.com (2025). Accessed 16 July 2025"},{"key":"9_CR23","unstructured":"Li, G., Hammoud, H.A.A.K., Itani, H., Khizbullin, D., Ghanem, B.: CAMEL: Communicative agents for \u201dmind\u201d exploration of large language model society. In: Thirty-seventh Conference on Neural Information Processing Systems (2023). https:\/\/openreview.net\/forum?id=3IyL2XWDkG"},{"issue":"1","key":"9_CR24","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1007\/s44336-024-00009-2","volume":"1","author":"X Li","year":"2024","unstructured":"Li, X., Wang, S., Zeng, S., Wu, Y., Yang, Y.: A survey on LLM-based multi-agent systems: workflow, infrastructure, and challenges. Vicinagearth 1(1), 9 (2024)","journal-title":"Vicinagearth"},{"key":"9_CR25","unstructured":"Li, X.: A review of prominent paradigms for LLM-based agents: Tool use (including rag), planning, and feedback learning (2024). https:\/\/arxiv.org\/abs\/2406.05804"},{"key":"9_CR26","unstructured":"Liu, Y., et\u00a0al.: Droidspeak: Kv cache sharing for cross-LLM communication and multi-LLM serving. arXiv preprint arXiv:2411.02820 (2024)"},{"key":"9_CR27","doi-asserted-by":"crossref","unstructured":"Maharana, A., Lee, D.H., Tulyakov, S., Bansal, M., Barbieri, F., Fang, Y.: Evaluating very long-term conversational memory of LLM agents (2024). https:\/\/arxiv.org\/abs\/2402.17753","DOI":"10.18653\/v1\/2024.acl-long.747"},{"key":"9_CR28","doi-asserted-by":"crossref","unstructured":"Maldonado, D., Cruz, E., Torres, J.A., Cruz, P.J., Benitez, S.d.P.G.: Multi-agent systems: a survey about its components, framework and workflow. IEEE Access 12, 80950\u201380975 (2024)","DOI":"10.1109\/ACCESS.2024.3409051"},{"key":"9_CR29","unstructured":"NVIDIA: Connectx infiniband. https:\/\/www.nvidia.cn\/networking\/infiniband-adapters\/ (2025). Accessed 18 July 2025"},{"key":"9_CR30","unstructured":"OpenAI: Function calling. https:\/\/platform.openai.com\/docs\/guides\/function-calling?api-mode=responses (2024). Accessed 16 July 2025"},{"key":"9_CR31","doi-asserted-by":"crossref","unstructured":"Patel, P., Choukse, E., Zhang, C., Shah, A., Goiri, \u00cd., Maleki, S., Bianchini, R.: Splitwise: Efficient generative LLM inference using phase splitting. In: 2024 ACM\/IEEE 51st Annual International Symposium on Computer Architecture (ISCA), pp. 118\u2013132. IEEE (2024)","DOI":"10.1109\/ISCA59077.2024.00019"},{"issue":"1","key":"9_CR32","doi-asserted-by":"publisher","first-page":"69","DOI":"10.3233\/MGS-230089","volume":"20","author":"A Qasim","year":"2024","unstructured":"Qasim, A., Ghouri, A., Munawar, A.: An effective approach for reducing data redundancy in multi-agent system communication. Multiagent Grid Syst. 20(1), 69\u201388 (2024)","journal-title":"Multiagent Grid Syst."},{"key":"9_CR33","unstructured":"Qin, R., et al.: Mooncake: Trading more storage for less computation\u2014a $$\\{$$KVCache-centric$$\\}$$ architecture for serving $$\\{$$LLM$$\\}$$ chatbot. In: 23rd USENIX Conference on File and Storage Technologies (FAST 25), pp. 155\u2013170 (2025)"},{"key":"9_CR34","unstructured":"Rajpurkar, P., Zhang, J., Lopyrev, K., Liang, P.: Stanford question answering dataset (squad). https:\/\/huggingface.co\/datasets\/rajpurkar\/squad (2016), based on original dataset (arXiv:1606.05250). Accessed 18 July 2025"},{"key":"9_CR35","unstructured":"Ram\u00edrez, S., the FastAPI\u00a0community: Fastapi: high performance, easy to learn, fast to code, ready for production. https:\/\/fastapi.tiangolo.com (2025). Accessed 17 July 2025"},{"key":"9_CR36","unstructured":"Research, S.A., contributors: xlam\u2011function\u2011calling\u201160k dataset. https:\/\/huggingface.co\/datasets\/Salesforce\/xlam-function-calling-60k (2024). arXiv:2406.18518, released: July 2024; Accessed 18 July 2025"},{"key":"9_CR37","doi-asserted-by":"publisher","unstructured":"Tan, X., Jiang, Y., Yang, Y., Xu, H.: Towards end-to-end optimization of LLM-based applications with ayo. In: Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2, pp. 1302\u20131316. ASPLOS \u201925, Association for Computing Machinery, New York, NY, USA (2025). https:\/\/doi.org\/10.1145\/3676641.3716278","DOI":"10.1145\/3676641.3716278"},{"issue":"8","key":"9_CR38","doi-asserted-by":"publisher","first-page":"274","DOI":"10.3390\/fi16080274","volume":"16","author":"A Tara","year":"2024","unstructured":"Tara, A., Turesson, H.K., Natea, N.: Dynamic storage optimization for communication between ai agents. Future Internet 16(8), 274 (2024)","journal-title":"Future Internet"},{"key":"9_CR39","unstructured":"Team, T.M.A.: Mistral\u20117b\u2011instruct\u2011v0.2. https:\/\/huggingface.co\/mistralai\/Mistral-7B-Instruct-v0.2 Accessed 18 July 2025"},{"key":"9_CR40","unstructured":"Wang, Q., Tang, Z., JIANG, Z., Chen, N., Wang, T., He, B.: Agenttaxo: Dissecting and benchmarking token distribution of LLM multi-agent systems. In: ICLR 2025 Workshop on Foundation Models in the Wild (2025). https:\/\/openreview.net\/forum?id=0iLbiYYIpC"},{"key":"9_CR41","unstructured":"Wu, Q., et al.: Autogen: Enabling next-gen LLM applications via multi-agent conversations. In: First Conference on Language Modeling (2024). https:\/\/openreview.net\/forum?id=BAakY1hNKS"},{"key":"9_CR42","unstructured":"Xu, W., Mei, K., Gao, H., Tan, J., Liang, Z., Zhang, Y.: A-mem: Agentic memory for LLM agents (2025). https:\/\/arxiv.org\/abs\/2502.12110"},{"key":"9_CR43","unstructured":"Xue, F., Fu, Y., Zhou, W., Zheng, Z., You, Y.: To repeat or not to repeat: insights from scaling LLM under token-crisis. In: Thirty-seventh Conference on Neural Information Processing Systems (2023). https:\/\/openreview.net\/forum?id=Af5GvIj3T5"},{"key":"9_CR44","doi-asserted-by":"crossref","unstructured":"Yao, J., et al.: Cacheblend: fast large language model serving for rag with cached knowledge fusion. In: Proceedings of the Twentieth European Conference on Computer Systems, pp. 94\u2013109 (2025)","DOI":"10.1145\/3689031.3696098"},{"key":"9_CR45","unstructured":"Zhao, W.X., et al.: A survey of large language models (2025). https:\/\/arxiv.org\/abs\/2303.18223"},{"key":"9_CR46","first-page":"62557","volume":"37","author":"L Zheng","year":"2024","unstructured":"Zheng, L., et al.: Sglang: efficient execution of structured language model programs. Adv. Neural. Inf. Process. Syst. 37, 62557\u201362583 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"9_CR47","unstructured":"Zhong, Y., et al.: DistServe: Disaggregating prefill and decoding for goodput-optimized large language model serving. In: 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24), pp. 193\u2013210. USENIX Association, Santa Clara, CA (Jul 2024). https:\/\/www.usenix.org\/conference\/osdi24\/presentation\/zhong-yinmin"}],"container-title":["Lecture Notes in Computer Science","Algorithms and Architectures for Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-8405-5_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T20:15:08Z","timestamp":1775074508000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-8405-5_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819584048","9789819584055"],"references-count":47,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-8405-5_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 April 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICA3PP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Algorithms and Architectures for Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Zhengzhou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 October 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 November 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ica3pp2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ieee-cybermatics.org\/2025\/ica3pp\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}