{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T16:15:09Z","timestamp":1784736909789,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","funder":[{"name":"Bundesministerium f\u00fcr Forschung, Technologie und Raumfahrt (BMFTR, German Federal Ministry of Research, Technology and Space)","award":["16KISK183"],"award-info":[{"award-number":["16KISK183"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,27]]},"DOI":"10.1145\/3805621.3807656","type":"proceedings-article","created":{"date-parts":[[2026,4,28]],"date-time":"2026-04-28T13:08:45Z","timestamp":1777381725000},"page":"296-303","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["DisCEdge: Distributed Context Management for Large Language Models at the Edge"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-0627-3600","authenticated-orcid":false,"given":"Mohammadreza","family":"Malekabbasi","sequence":"first","affiliation":[{"name":"Scalable Software Systems, Technische Universit\u00e4t Berlin, Berlin, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3780-5828","authenticated-orcid":false,"given":"Minghe","family":"Wang","sequence":"additional","affiliation":[{"name":"Scalable Software Systems, Technische Universit\u00e4t Berlin, Berlin, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7524-3256","authenticated-orcid":false,"given":"David","family":"Bermbach","sequence":"additional","affiliation":[{"name":"Scalable Software Systems, Technische Universit\u00e4t Berlin, Berlin, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/IC2E.2013.32"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Kevin Black Noah Brown Danny Driess Adnan Esmail Michael Equi Chelsea Finn Niccolo Fusai Lachy Groom Karol Hausman Brian Ichter et al. 2024. \u03c00: A Vision-Language-Action Flow Model for General Robot Control. arXiv preprint arXiv:2410.24164 (2024).","DOI":"10.15607\/RSS.2025.XXI.010"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/343477.343502"},{"key":"e_1_3_2_1_4_1","volume-title":"FlexQuant: Elastic Quantization Framework for Locally Hosted LLM on Edge Devices. arXiv preprint arXiv:2501.07139","author":"Chai Yuji","year":"2025","unstructured":"Yuji Chai, Mujin Kwen, David Brooks, and Gu-Yeon Wei. 2025. FlexQuant: Elastic Quantization Framework for Locally Hosted LLM on Edge Devices. arXiv preprint arXiv:2501.07139 (2025)."},{"key":"e_1_3_2_1_5_1","volume-title":"ChatFly: Low-Latency Drone Planning with Large Language Models","author":"Chen Guojun","year":"2025","unstructured":"Guojun Chen, Xiaojing Yu, Neiwen Ling, and Lin Zhong. 2025. ChatFly: Low-Latency Drone Planning with Large Language Models. IEEE Transactions on Mobile Computing (2025)."},{"key":"e_1_3_2_1_6_1","volume-title":"FogROS2-PLR: Probabilistic Latency-Reliability For Cloud Robotics. arXiv preprint arXiv:2410.05562","author":"Chen Kaiyuan","year":"2024","unstructured":"Kaiyuan Chen, Nan Tian, Christian Juette, Tianshuang Qiu, Liu Ren, John Kubiatowicz, and Ken Goldberg. 2024. FogROS2-PLR: Probabilistic Latency-Reliability For Cloud Robotics. arXiv preprint arXiv:2410.05562 (2024)."},{"key":"e_1_3_2_1_7_1","unstructured":"Cloudflare Workers AI. [n.d.]. Cloudflare Workers AI. https:\/\/developers.cloudflare.com\/workers-ai. Accessed: 2025-04-23."},{"key":"e_1_3_2_1_8_1","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:2507.06261 (2025)."},{"key":"e_1_3_2_1_9_1","volume-title":"Chengruidong Zhang, Yuanyuan Xu, Ning Shang, Jiahang Xu, Fan Yang, and Mao Yang.","author":"Ding Yiran","year":"2024","unstructured":"Yiran Ding, Li Lyna Zhang, Chengruidong Zhang, Yuanyuan Xu, Ning Shang, Jiahang Xu, Fan Yang, and Mao Yang. 2024. Longrope: Extending llm context window beyond 2 million tokens. arXiv preprint arXiv:2402.13753 (2024)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3638550.3641126"},{"key":"e_1_3_2_1_11_1","volume-title":"Extending context window of large language models via semantic compression. arXiv preprint arXiv:2312.09571","author":"Fei Weizhi","year":"2023","unstructured":"Weizhi Fei, Xueyan Niu, Pingyi Zhou, Lu Hou, Bo Bai, Lei Deng, and Wei Han. 2023. Extending context window of large language models via semantic compression. arXiv preprint arXiv:2312.09571 (2023)."},{"key":"e_1_3_2_1_12_1","volume-title":"Rethinking Key-Value Cache Compression Techniques for Large Language Model Serving. arXiv preprint arXiv:2503.24000","author":"Gao Wei","year":"2025","unstructured":"Wei Gao, Xinyu Zhou, Peng Sun, Tianwei Zhang, and Yonggang Wen. 2025. Rethinking Key-Value Cache Compression Techniques for Large Language Model Serving. arXiv preprint arXiv:2503.24000 (2025)."},{"key":"e_1_3_2_1_13_1","volume-title":"Retrieval-augmented generation for large language models: A survey. arXiv preprint arXiv:2312.10997 2, 1","author":"Gao Yunfan","year":"2023","unstructured":"Yunfan Gao, Yun Xiong, Xinyu Gao, Kangxiang Jia, Jinliu Pan, Yuxi Bi, Yixin Dai, Jiawei Sun, and Haofen Wang. 2023. Retrieval-augmented generation for large language models: A survey. arXiv preprint arXiv:2312.10997 2, 1 (2023)."},{"key":"e_1_3_2_1_14_1","unstructured":"Georgi Gerganov et al. 2023. llama.cpp: LLM inference in C\/C++. GitHub repository. https:\/\/github.com\/ggml-org\/llama.cpp Commit a33e6a (Feb 26 2024); accessed 2025-04-01."},{"key":"e_1_3_2_1_15_1","unstructured":"Google Cloud. [n.d.]. Vertex AI. https:\/\/cloud.google.com\/vertex-ai. Accessed: 2025-04-23."},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the 12th ACM International Conference on Distributed and Event-based Systems. 148\u2013159","author":"Gupta Harshit","year":"2018","unstructured":"Harshit Gupta and Umakishore Ramachandran. 2018. Fogstore: A geo-distributed key-value store guaranteeing low latency for strongly consistent access. In Proceedings of the 12th ACM International Conference on Distributed and Event-based Systems. 148\u2013159."},{"key":"e_1_3_2_1_17_1","unstructured":"Hugging Face. [n. d.]. Hugging Face Inference Endpoints. https:\/\/endpoints.huggingface.co. Accessed: 2025-04-23."},{"key":"e_1_3_2_1_18_1","volume-title":"2023 IEEE international conference on robotics and automation (ICRA). IEEE, 5493\u20135500","author":"Ichnowski Jeffrey","year":"2023","unstructured":"Jeffrey Ichnowski, Kaiyuan Chen, Karthik Dharmarajan, Simeon Adebola, Michael Danielczuk, V\u00edctor Mayoral-Vilches, Nikhil Jha, Hugo Zhan, Edith LLontop, Derek Xu, et al. 2023. Fogros2: An adaptive platform for cloud and fog robotics using ros 2. In 2023 IEEE international conference on robotics and automation (ICRA). IEEE, 5493\u20135500."},{"key":"e_1_3_2_1_19_1","volume-title":"Small models, big tasks: An exploratory empirical study on small language models for function calling. arXiv preprint arXiv:2504.19277","author":"Kavathekar Ishan","year":"2025","unstructured":"Ishan Kavathekar, Raghav Donakanti, Ponnurangam Kumaraguru, and Karthik Vaidhyanathan. 2025. Small models, big tasks: An exploratory empirical study on small language models for function calling. arXiv preprint arXiv:2504.19277 (2025)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_21_1","volume-title":"Adaptive Contextual Caching for Mobile Edge Large Language Model Service. arXiv preprint arXiv:2501.09383","author":"Liu Guangyuan","year":"2025","unstructured":"Guangyuan Liu, Yinqiu Liu, Jiacheng Wang, Hongyang Du, Dusit Niyato, Jiawen Kang, and Zehui Xiong. 2025. Adaptive Contextual Caching for Mobile Edge Large Language Model Service. arXiv preprint arXiv:2501.09383 (2025)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/IC2E61754.2024.00014"},{"key":"e_1_3_2_1_23_1","unstructured":"Kim Martineau. 2024. What's an LLM context window and why is it getting larger? https:\/\/research.ibm.com\/blog\/larger-context-window"},{"key":"e_1_3_2_1_24_1","volume-title":"2024 IEEE International Conference on Pervasive Computing and Communications (PerCom). IEEE, 47\u201356","author":"Mendula Matteo","year":"2024","unstructured":"Matteo Mendula, Paolo Bellavista, Marco Levorato, and Sharon Ladron de Guevara Contreras. 2024. Furcifer: a Context Adaptive Middleware for Real-world Object Detection Exploiting Local, Edge, and Split Computing in the Cloud Continuum. In 2024 IEEE International Conference on Pervasive Computing and Communications (PerCom). IEEE, 47\u201356."},{"key":"e_1_3_2_1_25_1","volume-title":"2021 international conference on advanced computer science and information systems (icacsis). IEEE, 1\u20135.","author":"Mutasodirin Mirza Alim","year":"2021","unstructured":"Mirza Alim Mutasodirin and Radityo Eko Prasojo. 2021. Investigating text shortening strategy in bert: Truncation vs summarization. In 2021 international conference on advanced computer science and information systems (icacsis). IEEE, 1\u20135."},{"key":"e_1_3_2_1_26_1","volume-title":"Minions: Cost-efficient collaboration between on-device and cloud language models. arXiv preprint arXiv:2502.15964","author":"Narayan Avanika","year":"2025","unstructured":"Avanika Narayan, Dan Biderman, Sabri Eyuboglu, Avner May, Scott Linderman, James Zou, and Christopher Re. 2025. Minions: Cost-efficient collaboration between on-device and cloud language models. arXiv preprint arXiv:2502.15964 (2025)."},{"key":"e_1_3_2_1_27_1","unstructured":"Artur Niederfahrenhorst and Kourosh Hakhamaneshi. 2024. Fine-tuning LLMs for longer context and better RAG systems. https:\/\/www.anyscale.com\/blog\/fine-tuning-llms-for- longer-context-and-better-rag-systems"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1002\/spe.3237"},{"key":"e_1_3_2_1_29_1","unstructured":"SearchWing Project. [n.d.]. SearchWing: Drones for Sea Rescue. https:\/\/tha.de\/searchwing\/. Accessed: 2025-04-18."},{"key":"e_1_3_2_1_30_1","volume-title":"Mobile edge intelligence for large language models: A contemporary survey","author":"Qu Guanqiao","year":"2025","unstructured":"Guanqiao Qu, Qiyuan Chen, Wei Wei, Zheng Lin, Xianhao Chen, and Kaibin Huang. 2025. Mobile edge intelligence for large language models: A contemporary survey. IEEE Communications Surveys & Tutorials (2025)."},{"key":"e_1_3_2_1_31_1","unstructured":"Replicate. [n.d.]. Replicate. https:\/\/replicate.com\/home. Accessed: 2025-04-23."},{"key":"e_1_3_2_1_32_1","volume-title":"2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS). IEEE, 5559\u20135566","author":"Schafhalter Peter","year":"2023","unstructured":"Peter Schafhalter, Sukrit Kalra, Le Xu, Joseph E Gonzalez, and Ion Stoica. 2023. Leveraging cloud computing to make autonomous vehicles safer. In 2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS). IEEE, 5559\u20135566."},{"key":"e_1_3_2_1_33_1","volume-title":"Bandwidth Allocation for Cloud-Augmented Autonomous Driving. arXiv preprint arXiv:2503.20127","author":"Schafhalter Peter","year":"2025","unstructured":"Peter Schafhalter, Alexander Krentsel, Joseph E Gonzalez, Sylvia Ratnasamy, Scott Shenker, and Ion Stoica. 2025. Bandwidth Allocation for Cloud-Augmented Autonomous Driving. arXiv preprint arXiv:2503.20127 (2025)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01202"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2025.130193"},{"key":"e_1_3_2_1_36_1","volume-title":"Token Pruning in Multimodal Large Language Models: Are We Solving the Right Problem? arXiv preprint arXiv:2502.11501","author":"Wen Zichen","year":"2025","unstructured":"Zichen Wen, Yifeng Gao, Weijia Li, Conghui He, and Linfeng Zhang. 2025. Token Pruning in Multimodal Large Language Models: Are We Solving the Right Problem? arXiv preprint arXiv:2502.11501 (2025)."},{"key":"e_1_3_2_1_37_1","volume-title":"Reducing Distraction in Long-Context Language Models by Focused Learning. arXiv preprint arXiv:2411.05928","author":"Wu Zijun","year":"2024","unstructured":"Zijun Wu, Bingyuan Liu, Ran Yan, Lei Chen, and Thomas Delteil. 2024. Reducing Distraction in Long-Context Language Models by Focused Learning. arXiv preprint arXiv:2411.05928 (2024)."},{"key":"e_1_3_2_1_38_1","volume-title":"EdgeLLM: Fast On-device LLM Inference with Speculative Decoding","author":"Xu Daliang","year":"2024","unstructured":"Daliang Xu, Wangsong Yin, Hao Zhang, Xin Jin, Ying Zhang, Shiyun Wei, Mengwei Xu, and Xuanzhe Liu. 2024. EdgeLLM: Fast On-device LLM Inference with Speculative Decoding. IEEE Transactions on Mobile Computing (2024)."},{"key":"e_1_3_2_1_39_1","volume-title":"Jupiter: Fast and resource-efficient collaborative inference of generative llms on edge devices. arXiv preprint arXiv:2504.08242","author":"Ye Shengyuan","year":"2025","unstructured":"Shengyuan Ye, Bei Ouyang, Liekang Zeng, Tianyi Qian, Xiaowen Chu, Jian Tang, and Xu Chen. 2025. Jupiter: Fast and resource-efficient collaborative inference of generative llms on edge devices. arXiv preprint arXiv:2504.08242 (2025)."},{"key":"e_1_3_2_1_40_1","volume-title":"Llm as a system service on mobile devices. arXiv preprint arXiv:2403.11805","author":"Yin Wangsong","year":"2024","unstructured":"Wangsong Yin, Mengwei Xu, Yuanchun Li, and Xuanzhe Liu. 2024. Llm as a system service on mobile devices. arXiv preprint arXiv:2403.11805 (2024)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689031.3696086"},{"key":"e_1_3_2_1_42_1","volume-title":"Edgeshard: Efficient llm inference via collaborative edge computing","author":"Zhang Mingjin","year":"2024","unstructured":"Mingjin Zhang, Xiaoming Shen, Jiannong Cao, Zeyang Cui, and Shan Jiang. 2024. Edgeshard: Efficient llm inference via collaborative edge computing. IEEE Internet of Things Journal (2024)."},{"key":"e_1_3_2_1_43_1","volume-title":"Tinyllama: An open-source small language model. arXiv preprint arXiv:2401.02385","author":"Zhang Peiyuan","year":"2024","unstructured":"Peiyuan Zhang, Guangtao Zeng, Tianduo Wang, and Wei Lu. 2024. Tinyllama: An open-source small language model. arXiv preprint arXiv:2401.02385 (2024)."},{"key":"e_1_3_2_1_44_1","volume-title":"Sglang: Efficient execution of structured language model programs. Advances in neural information processing systems 37","author":"Zheng Lianmin","year":"2024","unstructured":"Lianmin Zheng, Liangsheng Yin, Zhiqiang Xie, Chuyue Sun, Jeff Huang, Cody H Yu, Shiyi Cao, Christos Kozyrakis, Ion Stoica, Joseph E Gonzalez, et al. 2024. Sglang: Efficient execution of structured language model programs. Advances in neural information processing systems 37 (2024), 62557\u201362583."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3719664"}],"event":{"name":"EuroSys '26: 21st European Conference on Computer Systems","location":"Edinburgh Scotland Uk","acronym":"EuroMLSys '26","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the Sixth European Workshop on Machine Learning and Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3805621.3807656","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,28]],"date-time":"2026-04-28T13:14:23Z","timestamp":1777382063000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805621.3807656"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,27]]},"references-count":45,"alternative-id":["10.1145\/3805621.3807656","10.1145\/3805621"],"URL":"https:\/\/doi.org\/10.1145\/3805621.3807656","relation":{},"subject":[],"published":{"date-parts":[[2026,4,27]]},"assertion":[{"value":"2026-04-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}