{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T18:57:02Z","timestamp":1772823422154,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":19,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,6,3]],"date-time":"2024-06-03T00:00:00Z","timestamp":1717372800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000015","name":"U.S. Department of Energy","doi-asserted-by":"publisher","award":["DEAC02-06CH11357"],"award-info":[{"award-number":["DEAC02-06CH11357"]}],"id":[{"id":"10.13039\/100000015","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000015","name":"U.S. Department of Energy","doi-asserted-by":"publisher","award":["0F-60169"],"award-info":[{"award-number":["0F-60169"]}],"id":[{"id":"10.13039\/100000015","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2106634"],"award-info":[{"award-number":["2106634"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2106635"],"award-info":[{"award-number":["2106635"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,6,3]]},"DOI":"10.1145\/3659995.3660038","type":"proceedings-article","created":{"date-parts":[[2024,9,11]],"date-time":"2024-09-11T16:04:13Z","timestamp":1726070653000},"page":"9-16","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Breaking the Memory Wall: A Study of I\/O Patterns and GPU Memory Utilization for Hybrid CPU-GPU Offloaded Optimizers"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8200-0148","authenticated-orcid":false,"given":"Avinash","family":"Maurya","sequence":"first","affiliation":[{"name":"Department of Computer Science, Rochester Institute of Technology, Rochester, New York, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3985-7896","authenticated-orcid":false,"given":"Jie","family":"Ye","sequence":"additional","affiliation":[{"name":"Illinois Institute of Technology, Chicago, Illinois, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5034-2880","authenticated-orcid":false,"given":"M. Mustafa","family":"Rafique","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Rochester Institute of Technology, Rochester, New York, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7890-3934","authenticated-orcid":false,"given":"Franck","family":"Cappello","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory, Lemont, Illinois, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0661-7509","authenticated-orcid":false,"given":"Bogdan","family":"Nicolae","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory, Lemont, Illinois, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,9,11]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"A survey of large language models,\" arXiv preprint arXiv:2303.18223","author":"Zhao W. X.","year":"2023","unstructured":"W. X. Zhao, K. Zhou, J. Li, T. Tang, X. Wang, Y. Hou, Y. Min, B. Zhang, J. Zhang, Z.Dong et al., \"A survey of large language models,\" arXiv preprint arXiv:2303.18223, 2023."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3275156"},{"key":"e_1_3_2_1_3_1","volume-title":"Chen et al., \"DeepSpeed4Science Initiative: Enabling Large-Scale Scientific Discovery through Sophisticated AI System Technologies","author":"Song S. L.","year":"2023","unstructured":"S. L. Song, B. Kruft, M. Zhang, C. Li, S. Chen et al., \"DeepSpeed4Science Initiative: Enabling Large-Scale Scientific Discovery through Sophisticated AI System Technologies,\" 2023."},{"key":"e_1_3_2_1_4_1","volume-title":"Ili\u0106 et al., \"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","author":"Workshop B.","year":"2023","unstructured":"B. Workshop, T. L. Scao, A. Fan, C. Akiki, E. Pavlick, S. Ili\u0106 et al., \"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model,\" 2023."},{"key":"e_1_3_2_1_5_1","volume-title":"Almahairi et al., \"Llama 2: Open Foundation and Fine-Tuned Chat Models","author":"Touvron H.","year":"2023","unstructured":"H. Touvron, L. Martin, K. Stone, P. Albert, A. Almahairi et al., \"Llama 2: Open Foundation and Fine-Tuned Chat Models,\" 2023."},{"issue":"1","key":"e_1_3_2_1_6_1","first-page":"2022","article-title":"Switch transformers: scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus W.","unstructured":"W. Fedus, B. Zoph, and N. Shazeer, \"Switch transformers: scaling to trillion parameter models with simple and efficient sparsity,\" J. Mach. Learn. Res., vol. 23, no. 1, jan 2022.","journal-title":"J. Mach. Learn. Res."},{"key":"e_1_3_2_1_7_1","volume-title":"Xia et al., \"Glm-130b: An open bilingual pre-trained model,\" arXiv preprint arXiv:2210.02414","author":"Zeng A.","year":"2022","unstructured":"A. Zeng, X. Liu, Z. Du, Z. Wang, H. Lai, M. Ding, Z. Yang, Y. Xu, W. Zheng, X. Xia et al., \"Glm-130b: An open bilingual pre-trained model,\" arXiv preprint arXiv:2210.02414, 2022."},{"key":"e_1_3_2_1_8_1","volume-title":"Lin et al., \"M6-10t: A sharing-delinking paradigm for efficient multi-trillion parameter pretraining,\" arXiv preprint arXiv:2110.03888","author":"Lin J.","year":"2021","unstructured":"J. Lin, A. Yang, J. Bai, C. Zhou, L. Jiang, X. Jia, A. Wang, J. Zhang, Y. Li, W. Lin et al., \"M6-10t: A sharing-delinking paradigm for efficient multi-trillion parameter pretraining,\" arXiv preprint arXiv:2110.03888, 2021."},{"key":"e_1_3_2_1_9_1","volume-title":"Training 175b parameter language models at 1000 gpu scale with alpa and ray,\" https:\/\/www.anyscale.com\/blog\/training-175b-parameter-language-models-at-1000-gpu-scale-with-alpa-and-ray","author":"Dong J.","year":"2023","unstructured":"J. Dong, H. Zhang, L. Zheng, J. Gong, J. S. Damji, and P. Nguyen, \"Training 175b parameter language models at 1000 gpu scale with alpa and ray,\" https:\/\/www.anyscale.com\/blog\/training-175b-parameter-language-models-at-1000-gpu-scale-with-alpa-and-ray, 2023."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_11_1","volume-title":"USA","author":"Rajbhandari S.","year":"2021","unstructured":"S. Rajbhandari, O. Ruwase, J. Rasley, S. Smith, and Y. He, \"Zero-infinity: breaking the gpu memory wall for extreme scale deep learning,\" in SC'21: The 2021 International Conference for High Performance Computing, Networking, Storage and Analysis, St. Louis, USA, 2021."},{"key":"e_1_3_2_1_12_1","first-page":"551","volume-title":"{Zero-offload}: Democratizing {billion-scale} model training,\" in 2021 USENIX Annual Technical Conference (USENIX ATC 21)","author":"Ren J.","year":"2021","unstructured":"J. Ren, S. Rajbhandari, R. Y. Aminabadi, O. Ruwase, S. Yang, M. Zhang, D. Li, and Y. He, \"{Zero-offload}: Democratizing {billion-scale} model training,\" in 2021 USENIX Annual Technical Conference (USENIX ATC 21), 2021, pp. 551--564."},{"key":"e_1_3_2_1_13_1","volume-title":"Deepspeed zero-offload++: 6x higher training throughput via collaborative cpu\/gpu twin-flow\" https:\/\/github.com\/microsoft\/DeepSpeed\/tree\/offloadpp-news\/blogs\/deepspeed-offloadpp","author":"Wang G.","year":"2023","unstructured":"G. Wang, M. Tanaka, X. Wu, L. C. Koppaka, S. Rajbhandari, O. Ruwase, and Y. He, \"Deepspeed zero-offload++: 6x higher training throughput via collaborative cpu\/gpu twin-flow\" https:\/\/github.com\/microsoft\/DeepSpeed\/tree\/offloadpp-news\/blogs\/deepspeed-offloadpp, 2023."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.14778\/3415478.3415530"},{"key":"e_1_3_2_1_15_1","first-page":"1","volume-title":"Storage and Analysis. IEEE","author":"Rajbhandari S.","year":"2020","unstructured":"S. Rajbhandari, J. Rasley, O. Ruwase, and Y. He, \"Zero: Memory optimizations toward training trillion parameter models,\" in SC20: International Conference for High Performance Computing, Networking, Storage and Analysis. IEEE, 2020, pp. 1--16."},{"key":"e_1_3_2_1_16_1","volume-title":"Reducing activation recomputation in large transformer models,\" Proceedings of Machine Learning and Systems","author":"Korthikanti V. A.","year":"2023","unstructured":"V. A. Korthikanti, J. Casper, S. Lym, L. McAfee, M. Andersch, M. Shoeybi, and B. Catanzaro, \"Reducing activation recomputation in large transformer models,\" Proceedings of Machine Learning and Systems, vol. 5, 2023."},{"key":"e_1_3_2_1_17_1","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism,\" arXiv preprint arXiv:1909.08053","author":"Shoeybi M.","year":"2019","unstructured":"M. Shoeybi, M. Patwary, R. Puri, P. LeGresley, J. Casper, and B. Catanzaro, \"Megatron-lm: Training multi-billion parameter language models using model parallelism,\" arXiv preprint arXiv:1909.08053, 2019."},{"key":"e_1_3_2_1_18_1","volume-title":"including: Bert & gpt-2,\" https:\/\/github.com\/microsoft\/Megatron-DeepSpeed","author":"Research M.","year":"2023","unstructured":"M. Research, \"Ongoing research training transformer language models at scale, including: Bert & gpt-2,\" https:\/\/github.com\/microsoft\/Megatron-DeepSpeed, 2023."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","first-page":"92","DOI":"10.1145\/3605573.3605647","volume-title":"Cotrain: Efficient scheduling for large-model training upon gpu and cpu in parallel,\" in Proceedings of the 52nd International Conference on Parallel Processing","author":"Li Z.","year":"2023","unstructured":"Z. Li, Q. Cao, Y. Chen, and W. Yan, \"Cotrain: Efficient scheduling for large-model training upon gpu and cpu in parallel,\" in Proceedings of the 52nd International Conference on Parallel Processing, 2023, pp. 92--101."}],"event":{"name":"FlexScience'24: 14th Workshop on AI and Scientific Computing at Scale using Flexible Computing Infrastructures","location":"Pisa Italy","acronym":"FlexScience'24","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the 14th Workshop on AI and Scientific Computing at Scale using Flexible Computing Infrastructures"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3659995.3660038","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3659995.3660038","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:58:30Z","timestamp":1750294710000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3659995.3660038"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,3]]},"references-count":19,"alternative-id":["10.1145\/3659995.3660038","10.1145\/3659995"],"URL":"https:\/\/doi.org\/10.1145\/3659995.3660038","relation":{},"subject":[],"published":{"date-parts":[[2024,6,3]]},"assertion":[{"value":"2024-09-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}