{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,13]],"date-time":"2026-01-13T12:53:21Z","timestamp":1768308801203,"version":"3.49.0"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,1,4]],"date-time":"2026-01-04T00:00:00Z","timestamp":1767484800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,4]],"date-time":"2026-01-04T00:00:00Z","timestamp":1767484800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s11432-024-4487-8","type":"journal-article","created":{"date-parts":[[2026,1,13]],"date-time":"2026-01-13T05:40:10Z","timestamp":1768282810000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Mico: efficient query scheduling for multi-cloud deployed LLM inference service"],"prefix":"10.1007","volume":"69","author":[{"given":"Peizhuang","family":"Cong","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tong","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuchao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wendong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ke","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,4]]},"reference":[{"key":"4487_CR1","doi-asserted-by":"publisher","first-page":"208","DOI":"10.1016\/j.aiopen.2023.08.012","volume":"5","author":"X Liu","year":"2024","unstructured":"Liu X, Zheng Y, Du Z, et al. GPT understands, too. AI Open, 2024, 5: 208\u2013215","journal-title":"AI Open"},{"key":"4487_CR2","first-page":"7283","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle","author":"X Li","year":"2024","unstructured":"Li X, Feng X, Hu S, et al. DTLLM-VLT: diverse text generation for visual language tracking based on LLM. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, 2024. 7283\u20137292"},{"key":"4487_CR3","first-page":"2811","volume-title":"Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, Washington DC","author":"E Kamalloo","year":"2024","unstructured":"Kamalloo E, Upadhyay S, Lin J. Towards robust QA evaluation via open LLMs. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, Washington DC, 2024. 2811\u20132816"},{"key":"4487_CR4","doi-asserted-by":"publisher","first-page":"9662","DOI":"10.18653\/v1\/2023.emnlp-main.600","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, Singapore","author":"P Laban","year":"2023","unstructured":"Laban P, Kry\u015bci\u0144ski W, Agarwal D, et al. SUMMEDITS: measuring LLM ability at factual reasoning through the lens of summarization. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, Singapore, 2023. 9662\u20139676"},{"key":"4487_CR5","unstructured":"Google Cloud. Deployments and endpoints Generative AI on Vertex AI. Google Cloud Vertex AI Documentation. Retrieved May 7, 2025. https:\/\/cloud.google.com\/vertex-ai\/generative-ai\/docs\/learn\/locations"},{"key":"4487_CR6","doi-asserted-by":"publisher","first-page":"1122","DOI":"10.1109\/JAS.2023.123618","volume":"10","author":"T Wu","year":"2023","unstructured":"Wu T, He S, Liu J, et al. A BRIEF OVERview of ChatGPT: the history, status Quo and potential future development. IEEE CAA J Autom Sin, 2023, 10: 1122\u20131136","journal-title":"IEEE CAA J Autom Sin"},{"key":"4487_CR7","doi-asserted-by":"publisher","first-page":"2352","DOI":"10.1162\/neco_a_00990","volume":"29","author":"W Rawat","year":"2017","unstructured":"Rawat W, Wang Z. Deep convolutional neural networks for image classification: a comprehensive review. Neural Comput, 2017, 29: 2352\u20132449","journal-title":"Neural Comput"},{"key":"4487_CR8","doi-asserted-by":"publisher","first-page":"38","DOI":"10.18653\/v1\/2020.emnlp-demos.6","volume-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations","author":"T Wolf","year":"2020","unstructured":"Wolf T, Debut L, Sanh V, et al. Transformers: state-of-the-art natural language processing. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, 2020. 38\u201345"},{"key":"4487_CR9","doi-asserted-by":"publisher","first-page":"611","DOI":"10.1145\/3600006.3613165","volume-title":"Proceedings of the 29th Symposium on Operating Systems Principles, Koblenz","author":"W Kwon","year":"2023","unstructured":"Kwon W, Li Z, Zhuang S, et al. Efficient memory management for large language model serving with paged attention. In: Proceedings of the 29th Symposium on Operating Systems Principles, Koblenz, 2023. 611\u2013626"},{"key":"4487_CR10","first-page":"3505","volume-title":"Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, San Diego","author":"J Rasley","year":"2020","unstructured":"Rasley J, Rajbhandari S, Ruwase O, et al. Deepspeed: system optimizations enable training deep learning models with over 100 billion parameters. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, San Diego, 2020. 3505\u20133506"},{"key":"4487_CR11","unstructured":"Chelba C, Chen M, Bapna A, et al. Faster transformer decoding: N-gram masked self-attention. 2020. ArXiv:2001.04589"},{"key":"4487_CR12","first-page":"521","volume-title":"Proceedings of the 16th USENIX Symposium on Operating Systems Design and Implementation, Carlsbad","author":"G I Yu","year":"2022","unstructured":"Yu G I, Jeong J S, Kim G W, et al. Orca: a distributed serving system for transformer-based generative models. In: Proceedings of the 16th USENIX Symposium on Operating Systems Design and Implementation, Carlsbad, 2022. 521\u2013538"},{"key":"4487_CR13","first-page":"18015","volume-title":"Proceedings of the Advances in Neural Information Processing Systems, New Orleans","author":"Y Jin","year":"2023","unstructured":"Jin Y, Wu C F, Brooks D, et al. S3: increasing GPU utilization during generative inference for higher throughput. In: Proceedings of the Advances in Neural Information Processing Systems, New Orleans, 2023. 18015\u201318027"},{"key":"4487_CR14","unstructured":"Wu B, Zhong Y, Zhang Z, et al. Fast distributed inference serving for large language models. 2023. ArXiv:2305.05920"},{"key":"4487_CR15","first-page":"65517","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems, New Orleans","author":"Z Zheng","year":"2023","unstructured":"Zheng Z, Ren X, Xue F, et al. Response length perception and sequence scheduling: an LLM-empowered LLM inference pipeline. In: Proceedings of the 37th International Conference on Neural Information Processing Systems, New Orleans, 2023. 65517\u201365530"},{"key":"4487_CR16","unstructured":"Achiam J, Adler S, Agarwal S, et al. GPT-4 technical report. 2023. ArXiv:2303.08774"},{"key":"4487_CR17","unstructured":"Touvron H, Martin L, Stone K, et al. Llama 2: open foundation and fine-tuned chat models. 2023. ArXiv:2307.09288"},{"key":"4487_CR18","first-page":"5998","volume-title":"Proceedings of the Advances in Neural Information Processing Systems, Long Beach","author":"A Vaswani","year":"2017","unstructured":"Vaswani A, Shazeer N, Parmar N, et al. Attention is all you need. In: Proceedings of the Advances in Neural Information Processing Systems, Long Beach, 2017. 5998\u20136008"},{"key":"4487_CR19","volume-title":"Proceedings of the 41st International Conference on Machine Learning, Vienna","author":"F Strati","year":"2024","unstructured":"Strati F, McAllister S, Phanishayee A, et al. D\u00e9j\u00e0Vu: KV-cache streaming for fast, fault-tolerant generative LLM serving. In: Proceedings of the 41st International Conference on Machine Learning, Vienna, 2024"},{"key":"4487_CR20","first-page":"193","volume-title":"Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation, Santa Clara","author":"Y Zhong","year":"2024","unstructured":"Zhong Y, Liu S, Chen J, et al. DistServe: disaggregating prefill and decoding for goodput-optimized large language model serving. In: Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation, Santa Clara, 2024. 193\u2013210"},{"key":"4487_CR21","first-page":"2309","volume-title":"Proceedings of the ACM on Web Conference, Sydney","author":"P Cong","year":"2025","unstructured":"Cong P, Chen Q, Zhao H, et al. BATON: enhancing batch-wise inference efficiency for large language models via dynamic re-batching. In: Proceedings of the ACM on Web Conference, Sydney, 2025. 2309\u20132318"},{"key":"4487_CR22","volume-title":"Proceedings of the Advances in Neural Information Processing Systems, Vancouver","author":"V Sanh","year":"2020","unstructured":"Sanh V, Debut L, Chaumond J, et al. DistilBERT: a distilled version of BERT: smaller, faster, cheaper and lighter. In: Proceedings of the Advances in Neural Information Processing Systems, Vancouver, 2020"},{"key":"4487_CR23","volume-title":"Proceedings of the 41st International Conference on Machine Learning, Vienna","author":"Z Wan","year":"2024","unstructured":"Wan Z, Feng X, Wen M, et al. Alphazero-like tree-search can guide large language model decoding and training. In: Proceedings of the 41st International Conference on Machine Learning, Vienna, 2024"},{"key":"4487_CR24","first-page":"1877","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"T Brown","year":"2020","unstructured":"Brown T, Mann B, Ryder N, et al. Language models are few-shot learners. In: Proceedings of the Advances in Neural Information Processing Systems, 2020. 1877\u20131901"},{"key":"4487_CR25","first-page":"1","volume":"24","author":"A Chowdhery","year":"2023","unstructured":"Chowdhery A, Narang S, Devlin J, et al. Palm: scaling language modeling with pathways. J Mach Learn Res, 2023, 24: 1\u2013113","journal-title":"J Mach Learn Res"},{"key":"4487_CR26","first-page":"30016","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems, New Orleans","author":"J Hoffmann","year":"2022","unstructured":"Hoffmann J, Borgeaud S, Mensch A, et al. Training compute-optimal large language models. In: Proceedings of the 36th International Conference on Neural Information Processing Systems, New Orleans, 2022. 30016\u201330030"},{"key":"4487_CR27","first-page":"27730","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems, New Orleans","author":"L Ouyang","year":"2022","unstructured":"Ouyang L, Wu J, Jiang X, et al. Training language models to follow instructions with human feedback. In: Proceedings of the 36th International Conference on Neural Information Processing Systems, New Orleans, 2022. 27730\u201327744"},{"key":"4487_CR28","first-page":"1","volume-title":"Proceedings of the Architecture and System Support for Transformer Models, Oralndo","author":"S Kim","year":"2023","unstructured":"Kim S, Hooper C, Wattanawong T, et al. Full stack optimization of transformer inference. In: Proceedings of the Architecture and System Support for Transformer Models, Oralndo, 2023. 1\u20136"},{"key":"4487_CR29","doi-asserted-by":"publisher","first-page":"102990","DOI":"10.1016\/j.sysarc.2023.102990","volume":"144","author":"K T Chitty-Venkata","year":"2023","unstructured":"Chitty-Venkata K T, Mittal S, Emani M, et al. A survey of techniques for optimizing transformer inference. J Syst Architecture, 2023, 144: 102990","journal-title":"J Syst Architecture"},{"key":"4487_CR30","first-page":"92","volume-title":"Proceedings of the 2022 IEEE International Symposium on Workload Characterization, Austin","author":"J Choi","year":"2022","unstructured":"Choi J, Li H, Kim B, et al. Accelerating transformer networks through recomposing softmax layers. In: Proceedings of the 2022 IEEE International Symposium on Workload Characterization, Austin, 2022. 92\u2013103"},{"key":"4487_CR31","doi-asserted-by":"publisher","first-page":"169","DOI":"10.1109\/LCA.2023.3323482","volume":"22","author":"H Li","year":"2023","unstructured":"Li H, Choi J, Kwon Y, et al. A hardware-friendly tiled singular-value decomposition-based matrix multiplication for transformer-based models. IEEE Comput Arch Lett, 2023, 22: 169\u2013172","journal-title":"IEEE Comput Arch Lett"},{"key":"4487_CR32","first-page":"16344","volume-title":"Proceedings of the Advances in Neural Information Processing Systems, New Orleans","author":"T Dao","year":"2022","unstructured":"Dao T, Fu D, Ermon S, et al. Flashattention: fast and memory-efficient exact attention with IO-awareness. In: Proceedings of the Advances in Neural Information Processing Systems, New Orleans, 2022. 16344\u201316359"},{"key":"4487_CR33","unstructured":"Shoeybi M, Patwary M, Puri R, et al. Megatron-LM: training multi-billion parameter language models using model parallelism. 2019. ArXiv:1909.08053"},{"key":"4487_CR34","first-page":"1","volume-title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Mexico City","author":"R Ma","year":"2024","unstructured":"Ma R, Yang X, Wang J, et al. HPipe: large language model pipeline parallelism for long context on heterogeneous cost-effective devices. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Mexico City, 2024. 1\u20139"},{"key":"4487_CR35","volume-title":"Proceedings of the 41st International Conference on Machine Learning, Vienna","author":"Y Chen","year":"2024","unstructured":"Chen Y, Pan X, Li Y, et al. EE-LLM: large-scale training and inference of early-exit large language models with 3D parallelism. In: Proceedings of the 41st International Conference on Machine Learning, Vienna, 2024"},{"key":"4487_CR36","first-page":"663","volume-title":"Proceedings of the 17th USENIX Symposium on Operating Systems Design and Implementation, Boston","author":"Z Li","year":"2023","unstructured":"Li Z, Zheng L, Zhong Y, et al. AlpaServe: statistical multiplexing with model parallelism for deep learning serving. In: Proceedings of the 17th USENIX Symposium on Operating Systems Design and Implementation, Boston, 2023. 663\u2013679"},{"key":"4487_CR37","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1145\/3688351.3689164","volume-title":"Proceedings of the 17th ACM International Systems and Storage Conference, Haifa","author":"C Yu","year":"2024","unstructured":"Yu C, Wang T, Shao Z, et al. TwinPilots: a new computing paradigm for GPU-CPU parallel LLM inference. In: Proceedings of the 17th ACM International Systems and Storage Conference, Haifa, 2024. 91\u2013103"},{"key":"4487_CR38","first-page":"37524","volume-title":"Proceedings of the International Conference on Machine Learning, Honolulu","author":"X Wu","year":"2023","unstructured":"Wu X, Li C, Aminabadi R Y, et al. Understanding Int4 quantization for language models: latency speedup, composability, and failure cases. In: Proceedings of the International Conference on Machine Learning, Honolulu, 2023. 37524\u201337539"},{"key":"4487_CR39","first-page":"1","volume-title":"Proceedings of the 11th International Conference on Learning Representations, Kigali","author":"E Frantar","year":"2023","unstructured":"Frantar E, Ashkboos S, Hoefler T, et al. GPTQ: accurate post-training quantization for generative pre-trained transformers. In: Proceedings of the 11th International Conference on Learning Representations, Kigali, 2023. 1\u201316"},{"key":"4487_CR40","first-page":"792","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, Singapore","author":"Z Cheng","year":"2023","unstructured":"Cheng Z, Kasai J, Yu T. Batch prompting: efficient inference with large language model APIs. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, Singapore, 2023. 792\u2013810"},{"key":"4487_CR41","first-page":"31094","volume-title":"Proceedings of the International Conference on Machine Learning, Honolulu","author":"Y Sheng","year":"2023","unstructured":"Sheng Y, Zheng L, Yuan B, et al. Flexgen: high-throughput generative inference of large language models single GPU. In: Proceedings of the International Conference on Machine Learning, Honolulu, 2023. 31094\u201331116"},{"key":"4487_CR42","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1145\/3437801.3441578","volume-title":"Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","author":"J Fang","year":"2021","unstructured":"Fang J, Yu Y, Zhao C, et al. Turbotransformers: an efficient GPU serving system for transformer models. In: Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, 2021. 389\u2013402"},{"key":"4487_CR43","doi-asserted-by":"publisher","first-page":"1126","DOI":"10.1145\/3603269.3610856","volume-title":"Proceedings of the ACM SIGCOMM 2023 Conference, New York City","author":"R Ma","year":"2023","unstructured":"Ma R, Wang J, Qi Q, et al. Poster: PipeLLM: pipeline LLM inference on heterogeneous devices with sequence slicing. In: Proceedings of the ACM SIGCOMM 2023 Conference, New York City, 2023. 1126\u20131128"},{"key":"4487_CR44","doi-asserted-by":"publisher","first-page":"13119","DOI":"10.1109\/JIOT.2024.3524255","volume":"12","author":"M Zhang","year":"2025","unstructured":"Zhang M, Shen X, Cao J, et al. EdgeShard: efficient LLM inference via collaborative edge computing. IEEE Internet Things J, 2025, 12: 13119\u201313131","journal-title":"IEEE Internet Things J"},{"key":"4487_CR45","first-page":"12312","volume":"36","author":"A Borzunov","year":"2023","unstructured":"Borzunov A, Ryabinin M, Chumachenko A, et al. Distributed inference and fine-tuning of large language models over the internet. In: Proceedings of the 37th International Conference on Neural Information Processing Systems, New Orleans, 2023. 36: 12312\u201312331","journal-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems, New Orleans"},{"key":"4487_CR46","first-page":"118","volume-title":"Proceedings of the ACM\/IEEE 51st Annual International Symposium on Computer Architecture, Buenos Aires","author":"P Patel","year":"2024","unstructured":"Patel P, Choukse E, Zhang C, et al. Splitwise: efficient generative LLM inference using phase splitting. In: Proceedings of the ACM\/IEEE 51st Annual International Symposium on Computer Architecture, Buenos Aires, 2024. 118\u2013132"},{"key":"4487_CR47","doi-asserted-by":"publisher","first-page":"149","DOI":"10.18653\/v1\/2020.acl-main.15","volume-title":"Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics","author":"Y Ren","year":"2020","unstructured":"Ren Y, Liu J, Tan X, et al. A study of non-autoregressive model for sequence generation. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, 2020. 149\u2013159"},{"key":"4487_CR48","first-page":"3011","volume-title":"Proceedings of the Advances in Neural Information Processing Systems, Vancouver","author":"Z Sun","year":"2019","unstructured":"Sun Z, Li Z, Wang H, et al. Fast structured decoding for sequence models. In: Proceedings of the Advances in Neural Information Processing Systems, Vancouver, 2019. 3011\u20133020"},{"key":"4487_CR49","first-page":"6112","volume-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing, Hong Kong","author":"M Ghazvininejad","year":"2019","unstructured":"Ghazvininejad M, Levy O, Liu Y, et al. Mask-predict: parallel decoding of conditional masked language models. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing, Hong Kong, 2019. 6112\u20136121"},{"key":"4487_CR50","first-page":"4171","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Minneapolis","author":"J Devlin","year":"2019","unstructured":"Devlin J, Chang M W, Lee K, et al. BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Minneapolis, 2019. 4171\u20134186"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4487-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-024-4487-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4487-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,13]],"date-time":"2026-01-13T07:02:35Z","timestamp":1768287755000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-024-4487-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,4]]},"references-count":50,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["4487"],"URL":"https:\/\/doi.org\/10.1007\/s11432-024-4487-8","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,4]]},"assertion":[{"value":"8 November 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 March 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 June 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 January 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"132102"}}