{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:43:50Z","timestamp":1782834230839,"version":"3.54.5"},"reference-count":97,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"name":"National Key Research and Development Plan","award":["2022ZD0115301"],"award-info":[{"award-number":["2022ZD0115301"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62172453"],"award-info":[{"award-number":["62172453"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Natural Science Foundation of Guangdong","award":["2022A1515010154"],"award-info":[{"award-number":["2022A1515010154"]}]},{"name":"Major Key Project of PCL","award":["PCL2023AS7-1"],"award-info":[{"award-number":["PCL2023AS7-1"]}]},{"name":"CAAI-MindSpore Open Fund"},{"DOI":"10.13039\/100016691","name":"Guangdong Provincial Pearl River Talents Program","doi-asserted-by":"publisher","award":["2019QN01X130"],"award-info":[{"award-number":["2019QN01X130"]}],"id":[{"id":"10.13039\/100016691","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Open J. Comput. Soc."],"published-print":{"date-parts":[[2024]]},"DOI":"10.1109\/ojcs.2024.3380828","type":"journal-article","created":{"date-parts":[[2024,3,22]],"date-time":"2024-03-22T18:13:44Z","timestamp":1711131224000},"page":"107-119","source":"Crossref","is-referenced-by-count":19,"title":["Training and Serving System of Foundation Models: A Comprehensive Survey"],"prefix":"10.1109","volume":"5","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-9041-5874","authenticated-orcid":false,"given":"Jiahang","family":"Zhou","sequence":"first","affiliation":[{"name":"School of Systems Science and Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8932-7352","authenticated-orcid":false,"given":"Yanyu","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Informatics, Xiamen University, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5689-382X","authenticated-orcid":false,"given":"Zicong","family":"Hong","sequence":"additional","affiliation":[{"name":"Department of Computing, The Hong Kong Polytechnic University, Hong Kong SAR, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4430-7904","authenticated-orcid":false,"given":"Wuhui","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Software Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9865-2212","authenticated-orcid":false,"given":"Yue","family":"Yu","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9161-3210","authenticated-orcid":false,"given":"Tao","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Systems Science and Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0499-727X","authenticated-orcid":false,"given":"Hui","family":"Wang","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1525-5830","authenticated-orcid":false,"given":"Chuanfu","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Systems Science and Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7872-7718","authenticated-orcid":false,"given":"Zibin","family":"Zheng","sequence":"additional","affiliation":[{"name":"School of Software Engineering, Sun Yat-sen University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brown","year":"2020"},{"key":"ref2","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref3","article-title":"PanGu-$\\Sigma$: Towards trillion parameter language model with sparse heterogeneous computing","author":"Ren","year":"2023"},{"key":"ref4","article-title":"PengCheng Mind","volume-title":"Peng Cheng Laboratory"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3641289"},{"key":"ref6","article-title":"Large language models: A comprehensive survey of its applications, challenges, limitations, and future prospects","author":"Hadi","year":"2023","journal-title":"Authorea Preprints"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3639372"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-022-1410-8"},{"key":"ref9","article-title":"A survey on multimodal large language models","author":"Yin","year":"2023"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00090"},{"key":"ref11","article-title":"A comprehensive survey on pretrained foundation models: A history from bert to chatgpt","author":"Zhou","year":"2023"},{"key":"ref12","article-title":"A survey of large language models","author":"Zhao","year":"2023"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.14778\/3415478.3415530"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.14778\/3611540.3611569"},{"key":"ref16","article-title":"Automatic cross-replica sharding of weight update in data-parallel training","author":"Xu","year":"2020"},{"key":"ref17","first-page":"1","article-title":"Efficient large-scale language model training on GPU clusters using megatron-Lm","volume-title":"Proc. Int. Conf. High Perform. Comput., Netw., Storage Anal.","author":"Narayanan","year":"2021"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS54959.2023.00031"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1002\/(SICI)1096-9128(199704)9:4<255::AID-CPE250>3.0.CO;2-2"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/3545008.3545087"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-23397-5_10"},{"key":"ref22","article-title":"Maximizing parallelism in distributed training for huge neural networks","author":"Bian","year":"2021"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.48550\/arxiv.1811.06965"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3519584"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441593"},{"key":"ref27","first-page":"7937","article-title":"Memory-efficient pipeline-parallel DNN training","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Narayanan","year":"2021"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476145"},{"key":"ref29","first-page":"1","article-title":"Hanayo: Harnessing wave-like pipeline parallelism for enhanced large model training efficiency","volume-title":"Proc. Int. Conf. High Perform. Comput., Netw., Storage Anal.","author":"Liu","year":"2023"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/DAC56929.2023.10247730"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3572848.3577484"},{"key":"ref32","doi-asserted-by":"crossref","DOI":"10.1145\/3627703.3629585","article-title":"DynaPipe: Optimizing multi-task training through dynamic pipelines","volume-title":"Proc. EuroSys","author":"Jiang","year":"2024"},{"key":"ref33","first-page":"16639","article-title":"BPIPE: Memory-balanced pipeline parallelism for training large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kim","year":"2023"},{"key":"ref34","first-page":"497","article-title":"Bamboo: Making preemptible instances resilient for affordable training of large DNNs","volume-title":"Proc. 20th USENIX Symp. Netw. Syst. Des. Implementation","author":"Thorpe","year":"2023"},{"key":"ref35","first-page":"381","article-title":"Fine-tuning giant neural networks on commodity hardware with automatic pipeline model parallelism","volume-title":"Proc. USENIX Annu. Tech. Conf.","author":"Eliad","year":"2021"},{"key":"ref36","article-title":"Gshard: Scaling giant models with conditional computation and automatic sharding","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lepikhin","year":"2021"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1991.3.1.79"},{"key":"ref38","article-title":"Fastmoe: A fast mixture-of-expert training system","author":"He","year":"2021"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.5555\/3454287.3455008"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508418"},{"key":"ref41","first-page":"18332","article-title":"Deepspeed-Moe: Advancing mixture-of-experts inference and training to power next-generation AI scale","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Rajbhandari","year":"2022"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3577193.3593704"},{"key":"ref43","first-page":"961","article-title":"SmartMoE: Efficiently training sparsely-activated models through combining offline and online parallelization","volume-title":"Proc. USENIX Annu. Tech. Conf.","author":"Zhai","year":"2023"},{"key":"ref44","first-page":"945","article-title":"Accelerating distributed MoE training and inference with Lina","volume-title":"Proc. USENIX Annu. Tech. Conf.","author":"Li","year":"2023"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1145\/3603269.3604869"},{"key":"ref46","article-title":"Using deepspeed and megatron to train megatron-turing NLG 530b, a large-scale generative language model","author":"Smith","year":"2022"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"ref48","first-page":"559","article-title":"Alpa: Automating inter-and intra-operator parallelism for distributed deep learning","volume-title":"Proc. 16th USENIX Symp. Operating Syst. Des. Implementation","author":"Zheng","year":"2022"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.14778\/3570690.3570697"},{"key":"ref50","article-title":"Training deep nets with sublinear memory cost","author":"Chen","year":"2016"},{"key":"ref51","first-page":"497","article-title":"Checkmate: Breaking the memory wall with optimal tensor rematerialization","volume-title":"Proc. Mach. Learn. Syst.","volume":"2","author":"Jain","year":"2022"},{"key":"ref52","article-title":"Mixed precision training","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Micikevicius","year":"2018"},{"key":"ref53","article-title":"Highly scalable deep learning training system with mixed-precision: Training imagenet in four minutes","author":"Jia","year":"2018"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1145\/3225058.3225069"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2016.7783721"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378530"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378465"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00076"},{"key":"ref59","first-page":"387","article-title":"FlashNeuron:SSD-enabled large-batch training of very deep neural networks","volume-title":"Proc. 19th USENIX Conf. File Storage Technol.","author":"Bae","year":"2021"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575736"},{"key":"ref61","doi-asserted-by":"crossref","first-page":"395","DOI":"10.1145\/3613424.3614309","article-title":"G10: Enabling an efficient unified GPU memory and storage architecture with smart tensor migrations","volume-title":"Proc. 56th Annu. IEEE\/ACM Int. Symp. Microarchitecture","author":"Zhang","year":"2023"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3219819"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"ref64","first-page":"551","article-title":"ZeRO-Offload: Democratizing billion-scale model training","volume-title":"Proc. USENIX Annu. Techn. Conf.","author":"Ren","year":"2021"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.14778\/3503585.3503590"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1145\/3567955.3567959"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575703"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3519563"},{"key":"ref70","article-title":"ZeRO : Extremely efficient collective communication for giant model training","author":"Wang","year":"2023"},{"key":"ref71","first-page":"36058","article-title":"CocktailSGD: Fine-tuning foundation models over 500 Mbps networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wang","year":"2023"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575712"},{"key":"ref73","first-page":"183","article-title":"DVABatch: Diversity-aware multi-entry multi-exit batching for efficient processing of DNN services on GPUs","volume-title":"Proc. USENIX Annu. Techn. Conf.","author":"Cui","year":"2022"},{"key":"ref74","first-page":"521","article-title":"Orca: A distributed serving system for transformer-based generative models","volume-title":"Proc. 16th USENIX Symp. Operating Syst. Des. Implementation","author":"Yu","year":"2022"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/DAC56929.2023.10247993"},{"key":"ref76","first-page":"22137","article-title":"Deja Vu: Contextual sparsity for efficient llms at inference time","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Liu","year":"2023"},{"key":"ref77","article-title":"H2O: Heavy-hitter oracle for efficient generative inference of large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhang","year":"2023"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575698"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589038"},{"key":"ref80","first-page":"443","article-title":"Serving DNNs like clockwork: Performance predictability from the bottom up","volume-title":"Proc. 14th USENIX Symp. Operating Syst. Des. Implementation","author":"Gujarati","year":"2020"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"},{"key":"ref82","first-page":"539","article-title":"Microsecond-scale preemption for concurrent GPU-accelerated DNN inferences","volume-title":"Proc. 16th USENIX Symp. Operating Syst. Des. Implementation","author":"Han","year":"2022"},{"key":"ref83","first-page":"663","article-title":"AlpaServe: Statistical multiplexing with model parallelism for deep learning serving","volume-title":"Proc. 17th USENIX Symp. Operating Syst. Des. Implementation","author":"Li","year":"2023"},{"key":"ref84","article-title":"Fast distributed inference serving for large language models","author":"Wu","year":"2023"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421285"},{"key":"ref86","first-page":"397","article-title":"INFaaS: Automated model-less inference serving","volume-title":"Proc. USENIX Annu. Techn. Conf.","author":"Romero","year":"2021"},{"key":"ref87","first-page":"787","article-title":"SHEPHERD: Serving DNNs in the wild","volume-title":"Proc. 20th USENIX Symp. Netw. Syst. Des. Implementation","author":"Zhang","year":"2023"},{"key":"ref88","first-page":"153","article-title":"Intelligent resource scheduling for Co-located latency-critical services: A multi-model collaborative learning approach","volume-title":"Proc. 21st USENIX Conf. File Storage Technol.","author":"Liu","year":"2023"},{"key":"ref89","first-page":"199","article-title":"Serving heterogeneous machine learning models on multi-GPU servers with spatio-temporal sharing","volume-title":"Proc. USENIX Annu. Techn. Conf.","author":"Choi","year":"2022"},{"key":"ref90","first-page":"31094","article-title":"FlexGen: High-throughput generative inference of large language models with a single GPU","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sheng","year":"2023"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3567508"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref93","first-page":"489","article-title":"PetS: A unified framework for parameter-efficient transformers serving","volume-title":"Proc. USENIX Annu. Techn. Conf.","author":"Zhou","year":"2022"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i18.18018"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3587438"},{"key":"ref96","first-page":"19274","article-title":"Fast inference from transformers via speculative decoding","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Leviathan","year":"2023"},{"key":"ref97","article-title":"LLMCad: Fast and scalable on-device large language model inference","author":"Xu","year":"2023"}],"container-title":["IEEE Open Journal of the Computer Society"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8782664\/10375894\/10478189.pdf?arnumber=10478189","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,14]],"date-time":"2024-11-14T22:23:31Z","timestamp":1731623011000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10478189\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":97,"URL":"https:\/\/doi.org\/10.1109\/ojcs.2024.3380828","relation":{},"ISSN":["2644-1268"],"issn-type":[{"value":"2644-1268","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]}}}