{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,28]],"date-time":"2026-08-28T16:51:56Z","timestamp":1787935916238,"version":"build-2784847793"},"reference-count":47,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Guangdong S&#x0026;T Programme","award":["2024B0101040007"],"award-info":[{"award-number":["2024B0101040007"]}]},{"DOI":"10.13039\/501100021171","name":"Basic and Applied Basic Research Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2023B1515120058"],"award-info":[{"award-number":["2023B1515120058"]}],"id":[{"id":"10.13039\/501100021171","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Guangzhou Basic and Applied Basic Research Program","award":["2024A04J6367"],"award-info":[{"award-number":["2024A04J6367"]}]},{"name":"Program for Guangdong Introducing Innovative and Entrepreneurial Teams","award":["2017ZT07X355"],"award-info":[{"award-number":["2017ZT07X355"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. on Mobile Comput."],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1109\/tmc.2025.3590969","type":"journal-article","created":{"date-parts":[[2025,7,21]],"date-time":"2025-07-21T18:10:33Z","timestamp":1753121433000},"page":"13648-13662","source":"Crossref","is-referenced-by-count":13,"title":["Quality-of-Service Aware LLM Routing for Edge Computing With Multiple Experts"],"prefix":"10.1109","volume":"24","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-2788-3368","authenticated-orcid":false,"given":"Jin","family":"Yang","sequence":"first","affiliation":[{"name":"School of Computer Science and Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2156-4433","authenticated-orcid":false,"given":"Qiong","family":"Wu","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, Hong Kong University of Science and Technology, Clear Water Bay, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5156-340X","authenticated-orcid":false,"given":"Zhiying","family":"Feng","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0987-9344","authenticated-orcid":false,"given":"Zhi","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4894-5540","authenticated-orcid":false,"given":"Deke","family":"Guo","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9943-6020","authenticated-orcid":false,"given":"Xu","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2024.3471904"},{"key":"ref2","first-page":"7627","article-title":"Large language models for human-AI co-creation of robotic dance performances","volume-title":"Proc. 33rd Int. Joint Conf. Artif. Intell.","author":"De Filippo"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3625687.3625793"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/GLOBECOM52923.2024.10901084"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICMC60390.2024.00008"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2019.2918951"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/mwc.004.2300015"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/mwc.003.2400046"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.792"},{"key":"ref10","article-title":"Intelligent router for LLM workloads: Improving performance through workload-aware scheduling","author":"Jain","year":"2024"},{"key":"ref11","first-page":"521","article-title":"ORCA: A distributed serving system for transformer-based generative models","volume-title":"Proc. 16th USENIX Symp. Operating Syst. Des. Implementation","author":"Yu"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref13","article-title":"PolyRouter: A multi-LLM querying system","author":"Stripelis","year":"2024"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.109"},{"key":"ref15","article-title":"Hybrid LLM: Cost-efficient and quality-aware query routing","author":"Ding","year":"2024"},{"key":"ref16","article-title":"RouteLLM: Learning to route LLMs with preference data","author":"Ong","year":"2024"},{"key":"ref17","article-title":"Towards efficient and reliable LLM serving: A real-world workload study","author":"Wang","year":"2024"},{"key":"ref18","article-title":"Efficient interactive LLM serving with proxy model-based sequence length prediction","author":"Qiu","year":"2024"},{"key":"ref19","first-page":"18015","article-title":"S3: Increasing GPU utilization during generative inference for higher throughput","volume-title":"Proc. 37th Int. Conf. Neural Inf. Process. Syst.","author":"Jin"},{"key":"ref20","first-page":"31094","article-title":"FlexGen: High-throughput generative inference of large language models with a single GPU","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Sheng"},{"key":"ref21","first-page":"16344","article-title":"FlashAttention: Fast and memory-efficient exact attention with IO-awareness","volume-title":"Proc. 36th Int. Conf. Neural Inf. Process. Syst.","author":"Dao"},{"key":"ref22","first-page":"162","article-title":"HeteGen: Efficient heterogeneous parallel inference for large language models on resource-constrained devices","volume-title":"Proc. Mach. Learn. Syst.","author":"Xuanlei"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640383"},{"key":"ref24","article-title":"AttentionStore: Cost-effective attention reuse across multi-turn conversations in large language model serving","author":"Gao","year":"2024"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"ref26","article-title":"DistServe: Disaggregating prefill and decoding for goodput-optimized large language model serving","author":"Zhong","year":"2024"},{"key":"ref27","article-title":"Merge, ensemble, and cooperate! A survey on collaborative strategies in the era of large language models","author":"Lu","year":"2024"},{"key":"ref28","article-title":"Large language model routing with benchmark datasets","author":"Shnitzer","year":"2023"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.245"},{"key":"ref30","article-title":"GraphRouter: A graph-based router for LLM selections","author":"Feng","year":"2024"},{"key":"ref31","article-title":"Eagle: Efficient training-free router for multi-LLM inference","author":"Zhao","year":"2024"},{"key":"ref32","article-title":"RouterBench: A benchmark for multi-LLM routing system","author":"Hu","year":"2024"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/s42979-020-00326-5"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TNET.2023.3267168"},{"key":"ref35","first-page":"613","article-title":"Clipper: A low-latency online prediction serving system","volume-title":"Proc. 14th USENIX Symp. Netw. Syst. Des. Implementation","author":"Crankshaw"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM53939.2023.10229031"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1137\/S0097539701398375"},{"issue":"6","key":"ref38","first-page":"7","article-title":"Alpaca: A strong, replicable instruction-following model","volume-title":"Stanford Center Res. Found. Models","volume":"3","author":"Taori","year":"2023"},{"key":"ref39","article-title":"ChatGLM: A family of large language models from GLM-130B to GLM-4 all rools","author":"GLM","year":"2024"},{"key":"ref40","article-title":"Introducing MPT-7B: A new standard for open-source, commercially usable LLMs","year":"2023"},{"key":"ref41","first-page":"5333","article-title":"BERTScore: Evaluating text generation with BERT","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Zhang"},{"key":"ref42","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a Stochastic actor","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref43","article-title":"DistilBERT, a distilled version of BERT: Smaller, faster, cheaper and lighter","author":"Sanh","year":"2019"},{"key":"ref44","first-page":"8024","article-title":"PyTorch: An imperative style, high-performance deep learning library","volume-title":"Proc. 33rd Int. Conf. Neural Inf. Process. Syst.","author":"Paszke"},{"key":"ref45","article-title":"TorchRL: A data-driven decision-making library for pytorch","author":"Bou","year":"2023"},{"key":"ref46","article-title":"Fast graph representation learning with PyTorch geometric","author":"Fey","year":"2019"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"}],"container-title":["IEEE Transactions on Mobile Computing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7755\/11229982\/11086391.pdf?arnumber=11086391","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,6]],"date-time":"2025-11-06T18:55:08Z","timestamp":1762455308000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11086391\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12]]},"references-count":47,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tmc.2025.3590969","relation":{},"ISSN":["1536-1233","1558-0660","2161-9875"],"issn-type":[{"value":"1536-1233","type":"print"},{"value":"1558-0660","type":"electronic"},{"value":"2161-9875","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12]]}}}