{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T16:29:04Z","timestamp":1784737744923,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","funder":[{"name":"Office of Naval Research (ONR)","award":["N00014-23-C-1016"],"award-info":[{"award-number":["N00014-23-C-1016"]}]},{"name":"National Science Foundation (NSF)","award":["CPS-2313109"],"award-info":[{"award-number":["CPS-2313109"]}]},{"name":"Air Force Office of Scientific Research (AFOSR)","award":["FA9550-24-1-0083"],"award-info":[{"award-number":["FA9550-24-1-0083"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3704413.3764429","type":"proceedings-article","created":{"date-parts":[[2025,10,23]],"date-time":"2025-10-23T17:08:23Z","timestamp":1761239303000},"page":"201-210","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Local-Cloud Inference Offloading for LLMs in Multi-Modal, Multi-Task, Multi-Dialogue Settings"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9994-6773","authenticated-orcid":false,"given":"Liangqi","family":"Yuan","sequence":"first","affiliation":[{"name":"Purdue University, West Lafayette, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2970-7336","authenticated-orcid":false,"given":"Dong-Jun","family":"Han","sequence":"additional","affiliation":[{"name":"Yonsei University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2090-5512","authenticated-orcid":false,"given":"Shiqiang","family":"Wang","sequence":"additional","affiliation":[{"name":"IBM Research, Yorktown Heights, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2771-3521","authenticated-orcid":false,"given":"Christopher","family":"Brinton","sequence":"additional","affiliation":[{"name":"Purdue University, West Lafayette, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,23]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641512.3686392"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM52122.2024.10621325"},{"key":"e_1_3_2_1_3_1","volume-title":"An analysis of deep neural network models for practical applications. arXiv preprint arXiv:1605.07678","author":"Canziani Alfredo","year":"2016","unstructured":"Alfredo Canziani, Adam Paszke, and Eugenio Culurciello. 2016. An analysis of deep neural network models for practical applications. arXiv preprint arXiv:1605.07678 (2016)."},{"key":"e_1_3_2_1_4_1","volume-title":"Frugalgpt: How to use large language models while reducing cost and improving performance. arXiv preprint arXiv:2305.05176","author":"Chen Lingjiao","year":"2023","unstructured":"Lingjiao Chen, Matei Zaharia, and James Zou. 2023. Frugalgpt: How to use large language models while reducing cost and improving performance. arXiv preprint arXiv:2305.05176 (2023)."},{"key":"e_1_3_2_1_5_1","volume-title":"TileSR: Accelerate On-Device Super-Resolution with Parallel Offloading in Tile Granularity. In IEEE INFOCOM 2024-IEEE Conference on Computer Communications. IEEE, 2538\u20132547","author":"Chen Ning","year":"2024","unstructured":"Ning Chen, Sheng Zhang, Yu Liang, Jie Wu, Yu Chen, Yuting Yan, Zhuzhong Qian, and Sanglu Lu. 2024. TileSR: Accelerate On-Device Super-Resolution with Parallel Offloading in Tile Granularity. In IEEE INFOCOM 2024-IEEE Conference on Computer Communications. IEEE, 2538\u20132547."},{"key":"e_1_3_2_1_6_1","volume-title":"Mobilevlm: A fast, reproducible and strong vision language assistant for mobile devices. arXiv preprint arXiv:2312.16886","author":"Chu Xiangxiang","year":"2023","unstructured":"Xiangxiang Chu, Limeng Qiao, Xinyang Lin, Shuang Xu, Yang Yang, Yiming Hu, Fei Wei, Xinyu Zhang, Bo Zhang, Xiaolin Wei, et al. 2023. Mobilevlm: A fast, reproducible and strong vision language assistant for mobile devices. arXiv preprint arXiv:2312.16886 (2023)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3638550.3641126"},{"key":"e_1_3_2_1_8_1","volume-title":"Large language models (llms) inference offloading and resource allocation in cloud-edge computing: An active inference approach","author":"He Ying","year":"2024","unstructured":"Ying He, Jingcheng Fang, F Richard Yu, and Victor C Leung. 2024. Large language models (llms) inference offloading and resource allocation in cloud-edge computing: An active inference approach. IEEE Transactions on Mobile Computing (2024)."},{"key":"e_1_3_2_1_9_1","volume-title":"Uncertainty in natural language processing: Sources, quantification, and applications. arXiv preprint arXiv:2306.04459","author":"Hu Mengting","year":"2023","unstructured":"Mengting Hu, Zhen Zhang, Shiwan Zhao, Minlie Huang, and Bingzhe Wu. 2023. Uncertainty in natural language processing: Sources, quantification, and applications. arXiv preprint arXiv:2306.04459 (2023)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642230"},{"key":"e_1_3_2_1_11_1","first-page":"87","article-title":"AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration","volume":"6","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Wei-Ming Chen, Wei-Chen Wang, Guangxuan Xiao, Xingyu Dang, Chuang Gan, and Song Han. 2024. AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration. Proceedings of Machine Learning and Systems 6 (2024), 87\u2013100.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641512.3686358"},{"key":"e_1_3_2_1_13_1","volume-title":"Visual instruction tuning. Advances in neural information processing systems 36","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2024. Visual instruction tuning. Advances in neural information processing systems 36 (2024)."},{"key":"e_1_3_2_1_14_1","volume-title":"Agentbench: Evaluating llms as agents. arXiv preprint arXiv:2308.03688","author":"Liu Xiao","year":"2023","unstructured":"Xiao Liu, Hao Yu, Hanchen Zhang, Yifan Xu, Xuanyu Lei, Hanyu Lai, Yu Gu, Hangliang Ding, Kaiwen Men, Kejuan Yang, et al. 2023. Agentbench: Evaluating llms as agents. arXiv preprint arXiv:2308.03688 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"International conference on machine learning. PMLR","author":"Mnih Volodymyr","year":"2016","unstructured":"Volodymyr Mnih, Adria Puigdomenech Badia, Mehdi Mirza, Alex Graves, Timothy Lillicrap, Tim Harley, David Silver, and Koray Kavukcuoglu. 2016. Asynchronous methods for deep reinforcement learning. In International conference on machine learning. PMLR, 1928\u20131937."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Volodymyr Mnih Koray Kavukcuoglu David Silver Andrei A Rusu Joel Veness Marc G Bellemare Alex Graves Martin Riedmiller Andreas K Fidjeland Georg Ostrovski et al. 2015. Human-level control through deep reinforcement learning. nature 518 7540 (2015) 529\u2013533.","DOI":"10.1038\/nature14236"},{"key":"e_1_3_2_1_17_1","unstructured":"Long Ouyang Jeffrey Wu Xu Jiang Diogo Almeida Carroll Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray et al. 2022. Training language models to follow instructions with human feedback. Advances in neural information processing systems 35 (2022) 27730\u201327744."},{"key":"e_1_3_2_1_18_1","volume-title":"International conference on machine learning. PMLR, 8748\u20138763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_2_1_19_1","volume-title":"International Conference on Machine Learning. PMLR, 28701\u201328717","author":"Ran Yuhang","year":"2023","unstructured":"Yuhang Ran, Yi-Chen Li, Fuxiang Zhang, Zongzhang Zhang, and Yang Yu. 2023. Policy regularization with dataset constraint for offline reinforcement learning. In International Conference on Machine Learning. PMLR, 28701\u201328717."},{"key":"e_1_3_2_1_20_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.001.2300550"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM52122.2024.10621218"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM52122.2024.10621164"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641512.3686373"},{"key":"e_1_3_2_1_25_1","unstructured":"Zhiheng Xi Wenxiang Chen Xin Guo Wei He Yiwen Ding Boyang Hong Ming Zhang Junzhe Wang Senjie Jin Enyu Zhou et al. 2023. The rise and potential of large language model based agents: A survey. arXiv preprint arXiv:2309.07864 (2023)."},{"key":"e_1_3_2_1_26_1","volume-title":"PerLLM: Personalized Inference Scheduling with Edge-Cloud Collaboration for Diverse LLM Services. arXiv preprint arXiv:2405.14636","author":"Yang Zheming","year":"2024","unstructured":"Zheming Yang, Yuanhao Yang, Chang Zhao, Qi Guo, Wenkai He, and Wen Ji. 2024. PerLLM: Personalized Inference Scheduling with Edge-Cloud Collaboration for Diverse LLM Services. arXiv preprint arXiv:2405.14636 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"Carlee Joe-Wong, Samet Oymak, and Jiasi Chen.","author":"Zhang Xuechen","year":"2024","unstructured":"Xuechen Zhang, Zijian Huang, Ege Onur Taga, Carlee Joe-Wong, Samet Oymak, and Jiasi Chen. 2024. TREACLE: Thrifty Reasoning via Context-Aware LLM and Prompt Selection. arXiv preprint arXiv:2404.13082 (2024)."}],"event":{"name":"MobiHoc '25: Twenty-sixth International Symposium on Theory, Algorithmic Foundations, and Protocol Design for Mobile Networks and Mobile Computing","location":"Rice University Houston TX USA","acronym":"MobiHoc '25","sponsor":["SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing"]},"container-title":["Proceedings of the Twenty-sixth International Symposium on Theory, Algorithmic Foundations, and Protocol Design for Mobile Networks and Mobile Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3704413.3764429","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,23]],"date-time":"2025-10-23T17:10:00Z","timestamp":1761239400000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3704413.3764429"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,23]]},"references-count":27,"alternative-id":["10.1145\/3704413.3764429","10.1145\/3704413"],"URL":"https:\/\/doi.org\/10.1145\/3704413.3764429","relation":{},"subject":[],"published":{"date-parts":[[2025,10,23]]},"assertion":[{"value":"2025-10-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}