{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T16:29:17Z","timestamp":1783528157809,"version":"3.55.0"},"reference-count":67,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2023YFB2703700"],"award-info":[{"award-number":["2023YFB2703700"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62472410"],"award-info":[{"award-number":["62472410"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Institute of Computing Technology, Chinese Academy of Sciences - China Mobile Research Institute Joint Innovation Platform"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. on Mobile Comput."],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1109\/tmc.2025.3642580","type":"journal-article","created":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T18:34:42Z","timestamp":1765391682000},"page":"7508-7523","source":"Crossref","is-referenced-by-count":2,"title":["Recursive Offloading for LLM Serving in Multi-Tier Networks"],"prefix":"10.1109","volume":"25","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8925-4896","authenticated-orcid":false,"given":"Zhiyuan","family":"Wu","sequence":"first","affiliation":[{"name":"State Key Lab of Processers, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0260-2692","authenticated-orcid":false,"given":"Sheng","family":"Sun","sequence":"additional","affiliation":[{"name":"State Key Lab of Processers, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3228-7371","authenticated-orcid":false,"given":"Yuwei","family":"Wang","sequence":"additional","affiliation":[{"name":"State Key Lab of Processers, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2824-9601","authenticated-orcid":false,"given":"Min","family":"Liu","sequence":"additional","affiliation":[{"name":"State Key Lab of Processers, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4377-2970","authenticated-orcid":false,"given":"Bo","family":"Gao","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology and Engineering Research Center of Network Management Technology for High-Speed Railway of Ministry of Education, Beijing Jiaotong University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7322-9942","authenticated-orcid":false,"given":"Jinda","family":"Lu","sequence":"additional","affiliation":[{"name":"School of Cyber Science and Technology, University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6982-6994","authenticated-orcid":false,"given":"Tingting","family":"Wu","sequence":"additional","affiliation":[{"name":"China Mobile Research Institute, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9957-5792","authenticated-orcid":false,"given":"Zheming","family":"Yang","sequence":"additional","affiliation":[{"name":"State Key Lab of Processers, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tian","family":"Wen","sequence":"additional","affiliation":[{"name":"State Key Lab of Processers, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/3708528"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2024.3522130"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2025.3560560"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2025.3565130"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3362031"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2022.3218527"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2024.3393230"},{"issue":"8","key":"ref8","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3744746"},{"key":"ref11","article-title":"Gpt-3.5 turbo: Legacy GPT model for cheaper chat and non-chat tasks","year":"2022"},{"key":"ref12","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Brown","year":"2020"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-025-09422-z"},{"key":"ref14","article-title":"MiniCPM4: Ultra-efficient llms on end devices","author":"MiniCPM","year":"2025"},{"key":"ref15","first-page":"32431","article-title":"MobileLLM: Optimizing sub-billion parameter language models for on-device use cases","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","author":"Liu","year":"2024"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.001.2400764"},{"key":"ref17","first-page":"135","article-title":"$\\lbrace${ServerlessLLM $\\rbrace$}:$\\lbrace${ Low-Latency$\\rbrace$} serverless inference for large language models","volume-title":"Proc. 18th USENIX Symp. Operating Syst. Des. Implementation","volume":"24","author":"Fu","year":"2024"},{"key":"ref18","first-page":"193","article-title":"$\\lbrace${DistServe$\\rbrace$}: Disaggregating prefill and decoding for goodput-optimized large language model serving","volume-title":"Proc. 18th USENIX Symp. Operating Syst. Des. Implementation","volume":"24","author":"Zhong","year":"2024"},{"key":"ref19","article-title":"PerLLM: Personalized inference scheduling with edge-cloud collaboration for diverse LLM services","author":"Yang","year":"2024"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.734"},{"key":"ref21","article-title":"Efficient inference with model cascades","volume-title":"Trans. Mach. Learn. Res.","author":"Lebovitz","year":"2023"},{"key":"ref22","article-title":"Qwen2. 5 Technical Report","author":"Yang","year":"2024"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref24","article-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"key":"ref25","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"issue":"2","key":"ref26","article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Proc. Int. Conf. Learn. Representations","volume":"1","author":"Hu","year":"2022"},{"key":"ref27","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Houlsby","year":"2019"},{"key":"ref28","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Ouyang","year":"2022"},{"issue":"140","key":"ref29","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2022.3218527"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2023.3294688"},{"key":"ref32","article-title":"Beyond model scale limits: End-edge-cloud federated learning with self-rectified knowledge agglomeration","author":"Wu","year":"2025"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2016.2579198"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM52122.2024.10621254"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysarc.2021.102225"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1016\/j.comnet.2021.108177"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1016\/j.comnet.2024.110791"},{"key":"ref38","article-title":"Understanding softmax confidence and uncertainty","author":"Pearce","year":"2021"},{"key":"ref39","article-title":"State-of-the-art machine learning for pytorch, tensorflow, and jax","year":"2025"},{"key":"ref40","first-page":"142","article-title":"Learning word vectors for sentiment analysis","volume-title":"Proc. 49th Annu. Meeting Assoc. Comput. Linguistics, Hum. Lang. Technol.","author":"Maas","year":"2011"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1170"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.3115\/1219840.1219855"},{"key":"ref43","first-page":"649","article-title":"Character-level convolutional networks for text classification","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Zhang","year":"2015"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W16-2301"},{"key":"ref45","first-page":"1","article-title":"Findings of the 2019 conference on machine translation","volume-title":"Proc. ACL","author":"Barrault","year":"2019"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.63317\/59wvhpiwbv7y"},{"key":"ref47","article-title":"DistilBERT, a distilled version of BERT: Smaller, faster, cheaper and lighter","author":"Sanh","year":"2019"},{"key":"ref48","article-title":"azizbarank\/distilroberta-base-sst2-distilled","year":"2022"},{"key":"ref49","article-title":"textattack\/roberta-base-sst-2","year":"2023"},{"key":"ref50","article-title":"howey\/roberta-large-sst2","year":"2021"},{"key":"ref51","article-title":"Sebis\/legal_t5_small_trans_de_en_small_finetuned","year":"2021"},{"key":"ref52","first-page":"479","article-title":"OPUS-MT\u2013Building open translation services for the World","volume-title":"Proc. Annu. Conf. Eur. Assoc. Mach. Transl.","author":"Tiedemann","year":"2020"},{"key":"ref53","article-title":"Pontifexmaximus\/opus-mt-de-en-finetuned-de-to-en","year":"2022"},{"key":"ref54","article-title":"Deep encoder, shallow decoder: Reevaluating non-autoregressive machine translation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kasai","year":"2021"},{"key":"ref55","article-title":"allenai\/wmt19-de-en-6-6-big","year":"2023"},{"key":"ref56","article-title":"Minicpm: Unveiling the potential of small language models with scalable training strategies","author":"Hu","year":"2024"},{"key":"ref57","first-page":"973","article-title":"Gemel: Model merging for $\\lbrace${Memory-Efficient $\\rbrace,\\lbrace$},{ Real-Time$\\rbrace$} video analytics at the edge","volume-title":"Proc. 20th USENIX Symp. Networked Syst. Des. Implementation","volume":"23","author":"Padmanabhan","year":"2023"},{"key":"ref58","article-title":"Deberta: Decoding-enhanced BERT with disentangled attention","author":"He","year":"2020"},{"key":"ref59","article-title":"Tomor0720\/deberta-large-finetuned-sst2","year":"2023"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672274"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1145\/3603269.3604830"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2024.3524255"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1145\/3527155"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM41043.2020.9155237"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1145\/3495243.3560545"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/hpca61900.2025.00113"}],"container-title":["IEEE Transactions on Mobile Computing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7755\/11511860\/11293838.pdf?arnumber=11293838","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T19:42:59Z","timestamp":1778269379000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11293838\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":67,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/tmc.2025.3642580","relation":{},"ISSN":["1536-1233","1558-0660","2161-9875"],"issn-type":[{"value":"1536-1233","type":"print"},{"value":"1558-0660","type":"electronic"},{"value":"2161-9875","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]}}}