{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,14]],"date-time":"2026-08-14T16:36:07Z","timestamp":1786725367828,"version":"build-2736575974"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032352507","type":"print"},{"value":"9783032352514","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,15]],"date-time":"2026-08-15T00:00:00Z","timestamp":1786752000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,15]],"date-time":"2026-08-15T00:00:00Z","timestamp":1786752000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-35251-4_7","type":"book-chapter","created":{"date-parts":[[2026,8,14]],"date-time":"2026-08-14T16:03:06Z","timestamp":1786723386000},"page":"91-105","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Frenzy:\u00a0A\u00a0 Memory-Aware\u00a0Serverless\u00a0LLM\u00a0Training\u00a0System\u00a0for\u00a0Heterogeneous\u00a0GPU\u00a0Clusters"],"prefix":"10.1007","author":[{"given":"Zihan","family":"Chang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sheng","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuibing","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuechen","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siling","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhenxin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhe","family":"Pan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weijian","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,15]]},"reference":[{"key":"7_CR1","unstructured":"Zheng, L., et al.: Alpa: automating inter- and intra-operator parallelism for distributed deep learning. In: 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI), 2022"},{"key":"7_CR2","unstructured":"Qiao, A., et al.: Pollux: co-adaptive cluster scheduling for goodput-optimized deep learning. In: 15th USENIX Symposium on Operating Systems Design and Implementation (OSDI), 2021"},{"key":"7_CR3","unstructured":"Xiao, W., et al.: Gandiva: introspective cluster scheduling for deep learning. In: 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI), 2018"},{"key":"7_CR4","doi-asserted-by":"crossref","unstructured":"Weng, Q., et al.: MLaaS in the wild: workload analysis and scheduling in large-scale heterogeneous GPU clusters. In: 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI), 2022","DOI":"10.21203\/rs.3.rs-2266264\/v1"},{"key":"7_CR5","unstructured":"Jeon, M., et al.: Analysis of large-scale multi-tenant GPU Clusters for DNN training workloads. In: USENIX Annual Technical Conference (USENIX ATC), 2019"},{"key":"7_CR6","unstructured":"Narayanan, D., et al.: Heterogeneity-aware cluster scheduling policies for deep learning workloads. In: 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI), 2020"},{"key":"7_CR7","unstructured":"Um, T., et al.: Metis: fast automatic distributed training on heterogeneous GPUs. In: USENIX Annual Technical Conference (USENIX ATC), 2024"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Jayaram Subramanya, S., et al.: Sia: heterogeneity-aware, goodput-optimized ML-cluster scheduling. In: Proceedings of the 29th Symposium on Operating Systems Principles, 2023","DOI":"10.1145\/3600006.3613175"},{"key":"7_CR9","doi-asserted-by":"crossref","unstructured":"Gu, D., et al.: ElasticFlow: an elastic serverless training platform for distributed deep learning. In: Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, vol. 2, 2023","DOI":"10.1145\/3575693.3575721"},{"key":"7_CR10","unstructured":"Achiam, J., et al.: GPT-4 Technical Report. arXiv preprint arXiv:2303.08774, 2023"},{"key":"7_CR11","unstructured":"Barrault, L., et al.: SeamlessM4T: Massively Multilingual and Multimodal Machine Translation. arXiv preprint arXiv:2308.11596, 2023"},{"key":"7_CR12","unstructured":"NVIDIA. \u201cNVIDIA A100 Tensor Core GPU.\u201d https:\/\/www.nvidia.com\/en-us\/data-center\/a100"},{"key":"7_CR13","unstructured":"NVIDIA. \u201cNVIDIA GeForce RTX 3090.\u201d https:\/\/www.nvidia.com\/en-us\/geforce\/graphics-cards\/30-series\/rtx-3090-3090ti"},{"key":"7_CR14","doi-asserted-by":"crossref","unstructured":"Rajbhandari, S., et al.: ZeRO: memory optimizations toward training trillion-parameter models. In: SC20: International Conference for High Performance Computing, Networking, Storage and Analysis. IEEE, 2020","DOI":"10.1109\/SC41405.2020.00024"},{"key":"7_CR15","unstructured":"Shoeybi, M., et al.: Megatron-LM: Training Multi-Billion-Parameter Language Models Using Model Parallelism. arXiv preprint arXiv:1909.08053, 2019"},{"key":"7_CR16","unstructured":"Micikevicius, P., et al.: Mixed Precision Training. arXiv preprint arXiv:1710.03740, 2017"},{"key":"7_CR17","unstructured":"Kingma, D.P.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980, 2014"},{"key":"7_CR18","unstructured":"Bottou, L.: Large-scale machine learning with stochastic gradient descent. In: Proceedings of COMPSTAT\u20192010: 19th International Conference on Computational Statistics, Paris, France, August 22\u201327, 2010: Keynote, Invited and Contributed Papers. Physica-Verlag HD, 2010"},{"key":"7_CR19","unstructured":"Korthikanti, V.A., et al.: Reducing activation recomputation in large transformer models. Proc. Mach. Learn. Syst. 5, 341\u2013353 (2023)"},{"key":"7_CR20","doi-asserted-by":"crossref","unstructured":"Hu, Q., et al.: Characterization and prediction of deep learning workloads in large-scale GPU datacenters. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, 2021","DOI":"10.1145\/3458817.3476223"},{"key":"7_CR21","unstructured":"Radford, A., et al.: Language models are unsupervised multitask learners. OpenAI Blog 1(8), 9 (2019)"},{"key":"7_CR22","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of NAACL-HLT, vol. 1 (2019)"},{"key":"7_CR23","doi-asserted-by":"crossref","unstructured":"Li, J., Xu, H., Zhu, Y., et al.: Lyra: elastic scheduling for deep learning clusters. In: Proceedings of the Eighteenth European Conference on Computer Systems, pp. 835\u2013850 (2023)","DOI":"10.1145\/3552326.3587445"},{"key":"7_CR24","unstructured":"Smith, S., Patwary, M., Norick, B., et al.: Using DeepSpeed and Megatron to Train Megatron-Turing NLG 530B, a Large-Scale Generative Language Model. arXiv preprint arXiv:2201.11990, 2022"},{"key":"7_CR25","unstructured":"Alibaba. \u201cAlibaba Cluster Data.\u201d https:\/\/github.com\/alibaba\/clusterdata"}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2026: Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-35251-4_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,14]],"date-time":"2026-08-14T16:03:12Z","timestamp":1786723392000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-35251-4_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,15]]},"ISBN":["9783032352507","9783032352514"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-35251-4_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,15]]},"assertion":[{"value":"15 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors\u00a0have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"Euro-Par","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pisa","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"32","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"europar2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2026.euro-par.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}