{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,2]],"date-time":"2025-08-02T16:33:06Z","timestamp":1754152386445,"version":"3.41.2"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,7]]},"DOI":"10.1145\/3735358.3735368","type":"proceedings-article","created":{"date-parts":[[2025,7,17]],"date-time":"2025-07-17T23:08:27Z","timestamp":1752793707000},"page":"172-177","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["FlexSpark: Robust and Efficient Multi-Device Collaborative Inference over Wireless Network"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-1373-3830","authenticated-orcid":false,"given":"Yiyang","family":"Shao","sequence":"first","affiliation":[{"name":"Huawei, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4856-0719","authenticated-orcid":false,"given":"Hongyi","family":"Li","sequence":"additional","affiliation":[{"name":"Huawei, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3620-4843","authenticated-orcid":false,"given":"Shuihai","family":"Hu","sequence":"additional","affiliation":[{"name":"Huawei, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8918-9580","authenticated-orcid":false,"given":"Xinle","family":"Du","sequence":"additional","affiliation":[{"name":"Huawei, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7704-5116","authenticated-orcid":false,"given":"Hao","family":"Wu","sequence":"additional","affiliation":[{"name":"Huawei, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4781-1876","authenticated-orcid":false,"given":"Jingbin","family":"Zhou","sequence":"additional","affiliation":[{"name":"Huawei, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5876-9096","authenticated-orcid":false,"given":"Kun","family":"Tan","sequence":"additional","affiliation":[{"name":"Huawei, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,8,6]]},"reference":[{"key":"e_1_3_3_1_2_2","first-page":"117","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Agrawal Amey","year":"2024","unstructured":"Amey Agrawal, Nitin Kedia, Ashish Panwar, Jayashree Mohan, Nipun Kwatra, Bhargav Gulavani, Alexey Tumanov, and Ramachandran Ramjee. 2024. Taming { Throughput-Latency} tradeoff in { LLM} inference with { Sarathi-Serve}. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). 117\u2013134."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"},{"volume-title":"Apple Intelligence","key":"e_1_3_3_1_4_2","unstructured":"Apple. [n. d.]. Apple Intelligence. https:\/\/www.apple.com\/apple-intelligence\/ 2024."},{"key":"e_1_3_3_1_5_2","unstructured":"Jerry Chee Yaohui Cai Volodymyr Kuleshov and Christopher\u00a0M De\u00a0Sa. 2023. Quip: 2-bit quantization of large language models with guarantees. Advances in Neural Information Processing Systems 36 (2023) 4396\u20134429."},{"key":"e_1_3_3_1_6_2","unstructured":"Alex Cheema. 2024. exo. https:\/\/github.com\/exo-explore\/exo."},{"key":"e_1_3_3_1_7_2","unstructured":"NVIDIA Corporation. 2019. Fastertransformer. https:\/\/github.com\/NVIDIA\/FasterTransformer."},{"key":"e_1_3_3_1_8_2","unstructured":"NVIDIA Corporation. 2023. Tensorrt-llm. https:\/\/github.com\/NVIDIA\/TensorRT-LLM."},{"volume-title":"Get started with Gemini Nano on Android (ondevice)","key":"e_1_3_3_1_9_2","unstructured":"Google. [n. d.]. Get started with Gemini Nano on Android (ondevice). https:\/\/ai.google.dev\/gemini-api\/docs\/getstarted\/android_aicore 2024."},{"key":"e_1_3_3_1_10_2","first-page":"353","volume-title":"17th USENIX Symposium on Networked Systems Design and Implementation (NSDI 20)","author":"Goyal Prateesh","year":"2020","unstructured":"Prateesh Goyal, Anup Agarwal, Ravi Netravali, Mohammad Alizadeh, and Hari Balakrishnan. 2020. { ABC} : A simple explicit congestion controller for wireless networks. In 17th USENIX Symposium on Networked Systems Design and Implementation (NSDI 20). 353\u2013372."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575698"},{"key":"e_1_3_3_1_12_2","unstructured":"Ajay Jaiswal Shiwei Liu Tianlong Chen Zhangyang Wang et\u00a0al. 2023. The emergence of essential sparsity in large pre-trained models: The weights that matter. Advances in Neural Information Processing Systems 36 (2023) 38887\u201338901."},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538948"},{"key":"e_1_3_3_1_14_2","unstructured":"Jaehun Jung Peter West Liwei Jiang Faeze Brahman Ximing Lu Jillian Fisher Taylor Sorensen and Yejin Choi. 2023. Impossible distillation: from low-quality model to high-quality dataset & model for summarization and paraphrasing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.16635 (2023)."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_1_16_2","unstructured":"Ji Lin Jiaming Tang Haotian Tang Shang Yang Wei-Ming Chen Wei-Chen Wang Guangxuan Xiao Xingyu Dang Chuang Gan and Song Han. 2024. Awq: Activation-aware weight quantization for on-device llm compression and acceleration. Proceedings of Machine Learning and Systems 6 (2024) 87\u2013100."},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672274"},{"key":"e_1_3_3_1_18_2","unstructured":"Xinyin Ma Gongfan Fang and Xinchao Wang. 2023. Llm-pruner: On the structural pruning of large language models. Advances in neural information processing systems 36 (2023) 21702\u201321720."},{"key":"e_1_3_3_1_19_2","unstructured":"Simone Margaritelli. 2024. Cake. https:\/\/github.com\/evilsocket\/cake."},{"key":"e_1_3_3_1_20_2","volume-title":"MLC-LLM","author":"team MLC","year":"2023","unstructured":"MLC team. 2023-2025. MLC-LLM. https:\/\/github.com\/mlc-ai\/mlc-llm"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"e_1_3_3_1_22_2","first-page":"155","volume-title":"23rd USENIX Conference on File and Storage Technologies (FAST 25)","author":"Qin Ruoyu","year":"2025","unstructured":"Ruoyu Qin, Zheming Li, Weiran He, Jialei Cui, Feng Ren, Mingxing Zhang, Yongwei Wu, Weimin Zheng, and Xinran Xu. 2025. Mooncake: Trading More Storage for Less Computation\u2014A { KVCache-centric} Architecture for Serving { LLM} Chatbot. In 23rd USENIX Conference on File and Storage Technologies (FAST 25). 155\u2013170."},{"volume-title":"MCS TABLE (UPDATED WITH 802.11AX DATA RATES)","key":"e_1_3_3_1_23_2","unstructured":"semfionetworks. [n. d.]. MCS TABLE (UPDATED WITH 802.11AX DATA RATES). https:\/\/semfionetworks.com\/blog\/mcs-table-updated-with-80211ax-data-rates\/ 2019."},{"key":"e_1_3_3_1_24_2","first-page":"31094","volume-title":"International Conference on Machine Learning","author":"Sheng Ying","year":"2023","unstructured":"Ying Sheng, Lianmin Zheng, Binhang Yuan, Zhuohan Li, Max Ryabinin, Beidi Chen, Percy Liang, Christopher R\u00e9, Ion Stoica, and Ce Zhang. 2023. Flexgen: High-throughput generative inference of large language models with a single gpu. In International Conference on Machine Learning. PMLR, 31094\u201331116."},{"key":"e_1_3_3_1_25_2","unstructured":"Mohammad Shoeybi Mostofa Patwary Raul Puri Patrick LeGresley Jared Casper and Bryan Catanzaro. 2019. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. CoRR abs\/1909.08053 (2019). arXiv:https:\/\/arXiv.org\/abs\/1909.08053http:\/\/arxiv.org\/abs\/1909.08053"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695964"},{"key":"e_1_3_3_1_27_2","unstructured":"Bart Tadych. 2024. Distributed Llama. https:\/\/github.com\/b4rtaz\/distributed-llama."},{"key":"e_1_3_3_1_28_2","first-page":"459","volume-title":"10th USENIX Symposium on Networked Systems Design and Implementation (NSDI 13)","author":"Winstein Keith","year":"2013","unstructured":"Keith Winstein, Anirudh Sivaraman, and Hari Balakrishnan. 2013. Stochastic forecasts achieve high throughput and low delay over cellular networks. In 10th USENIX Symposium on Networked Systems Design and Implementation (NSDI 13). 459\u2013471."},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695948"},{"key":"e_1_3_3_1_30_2","unstructured":"Daliang Xu Hao Zhang Liming Yang Ruiqi Liu Gang Huang Mengwei Xu and Xuanzhe Liu. 2024. Empowering 1000 tokens\/second on-device llm prefilling with mllm-npu. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.05858 (2024)."},{"key":"e_1_3_3_1_31_2","unstructured":"Mengwei Xu Wangsong Yin Dongqi Cai Rongjie Yi Daliang Xu Qipeng Wang Bingyang Wu Yihao Zhao Chen Yang Shihe Wang et\u00a0al. 2024. A survey of resource-efficient llm and multimodal foundation models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.08092 (2024)."},{"key":"e_1_3_3_1_32_2","unstructured":"Zhenliang Xue Yixin Song Zeyu Mi Xinrui Zheng Yubin Xia and Haibo Chen. 2024. Powerinfer-2: Fast large language model inference on a smartphone. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.06282 (2024)."},{"key":"e_1_3_3_1_33_2","first-page":"521","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Yu Gyeong-In","year":"2022","unstructured":"Gyeong-In Yu, Joo\u00a0Seong Jeong, Geon-Woo Kim, Soojeong Kim, and Byung-Gon Chun. 2022. Orca: A distributed serving system for { Transformer-Based} generative models. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22). 521\u2013538."},{"key":"e_1_3_3_1_34_2","first-page":"193","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong Yinmin","year":"2024","unstructured":"Yinmin Zhong, Shengyu Liu, Junda Chen, Jianbo Hu, Yibo Zhu, Xuanzhe Liu, Xin Jin, and Hao Zhang. 2024. { DistServe} : Disaggregating prefill and decoding for goodput-optimized large language model serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). 193\u2013210."}],"event":{"name":"APNET 2025: The 9th Asia-Pacific Workshop on Networking","acronym":"APNET 2025","location":"Shang Hai China"},"container-title":["Proceedings of the 9th Asia-Pacific Workshop on Networking"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3735358.3735368","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,22]],"date-time":"2025-07-22T05:09:29Z","timestamp":1753160969000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3735358.3735368"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,6]]},"references-count":33,"alternative-id":["10.1145\/3735358.3735368","10.1145\/3735358"],"URL":"https:\/\/doi.org\/10.1145\/3735358.3735368","relation":{},"subject":[],"published":{"date-parts":[[2025,8,6]]},"assertion":[{"value":"2025-08-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}