{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T16:21:27Z","timestamp":1781713287915,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000015","name":"U.S. Department of Energy","doi-asserted-by":"publisher","award":["DEAC02-06CH11357"],"award-info":[{"award-number":["DEAC02-06CH11357"]}],"id":[{"id":"10.13039\/100000015","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000015","name":"U.S. Department of Energy","doi-asserted-by":"publisher","award":["0F-60169"],"award-info":[{"award-number":["0F-60169"]}],"id":[{"id":"10.13039\/100000015","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2411386"],"award-info":[{"award-number":["2411386"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2411387"],"award-info":[{"award-number":["2411387"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2106635"],"award-info":[{"award-number":["2106635"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,2]]},"DOI":"10.1145\/3652892.3700781","type":"proceedings-article","created":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T19:36:13Z","timestamp":1732736173000},"page":"404-416","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Deep Optimizer States: Towards Scalable Training of Transformer Models using Interleaved Offloading"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8200-0148","authenticated-orcid":false,"given":"Avinash","family":"Maurya","sequence":"first","affiliation":[{"name":"Rochester Institute of Technology, Rochester, New York, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3985-7896","authenticated-orcid":false,"given":"Jie","family":"Ye","sequence":"additional","affiliation":[{"name":"Illinois Institute of Technology, Chicago, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5034-2880","authenticated-orcid":false,"given":"M. Mustafa","family":"Rafique","sequence":"additional","affiliation":[{"name":"Rochester Institute of Technology, Rochester, New York, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7890-3934","authenticated-orcid":false,"given":"Franck","family":"Cappello","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory, Lemont, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0661-7509","authenticated-orcid":false,"given":"Bogdan","family":"Nicolae","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory, Lemont, Illinois, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,12,2]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00046"},{"key":"e_1_3_2_1_2_1","volume-title":"Gpt-neox-20b: An open-source autoregressive language model. arXiv preprint arXiv:2204.06745","author":"Black Sid","year":"2022","unstructured":"Sid Black, Stella Biderman, Eric Hallahan, Quentin Anthony, Leo Gao, Laurence Golding, Horace He, Connor Leahy, Kyle McDonell, Jason Phang, et al. Gpt-neox-20b: An open-source autoregressive language model. arXiv preprint arXiv:2204.06745, 2022."},{"key":"e_1_3_2_1_3_1","first-page":"13","volume-title":"Automation & Test in Europe Conference & Exhibition (DATE)","author":"Chen Xiaoming","year":"2018","unstructured":"Xiaoming Chen, Danny Z Chen, and Xiaobo Sharon Hu. modnn: Memory optimal dnn training on gpus. In 2018 Design, Automation & Test in Europe Conference & Exhibition (DATE), pages 13--18. IEEE, 2018."},{"key":"e_1_3_2_1_4_1","volume-title":"Training 175b parameter language models at 1000 gpu scale with alpa and ray. https:\/\/www.anyscale.com\/blog\/training-175b-parameter-language-models-at-1000-gpu-scale-with-alpa-and-ray","author":"Dong Jiao","year":"2023","unstructured":"Jiao Dong, Hao Zhang, Lianmin Zheng, Jun Gong, Jules S. Damji, and Phi Nguyen. Training 175b parameter language models at 1000 gpu scale with alpa and ray. https:\/\/www.anyscale.com\/blog\/training-175b-parameter-language-models-at-1000-gpu-scale-with-alpa-and-ray, 2023."},{"key":"e_1_3_2_1_5_1","volume-title":"Adaptive subgradient methods for online learning and stochastic optimization. Journal of machine learning research, 12(7)","author":"Duchi John","year":"2011","unstructured":"John Duchi, Elad Hazan, and Yoram Singer. Adaptive subgradient methods for online learning and stochastic optimization. Journal of machine learning research, 12(7), 2011."},{"key":"e_1_3_2_1_6_1","first-page":"431","volume-title":"Proc. of the SIGPLAN Symposium on Principles and Practice of Parallel Programming","author":"Fan Shiqing","year":"2021","unstructured":"Shiqing Fan, Yi Rong, Chen Meng, Zongyan Cao, Siyu Wang, Zhen Zheng, Chuan Wu, Guoping Long, Jun Yang, Lixue Xia, et al. Dapple: A pipelined data parallel approach for training large models. In Proc. of the SIGPLAN Symposium on Principles and Practice of Parallel Programming, pages 431--445, 2021."},{"key":"e_1_3_2_1_7_1","volume-title":"Switch transformers: scaling to trillion parameter models with simple and efficient sparsity. Journal of Machine Learning Research, 23(1), jan","author":"Fedus William","year":"2022","unstructured":"William Fedus, Barret Zoph, and Noam Shazeer. Switch transformers: scaling to trillion parameter models with simple and efficient sparsity. Journal of Machine Learning Research, 23(1), jan 2022."},{"key":"e_1_3_2_1_8_1","volume-title":"Generating sequences with recurrent neural networks","author":"Graves Alex","year":"2014","unstructured":"Alex Graves. Generating sequences with recurrent neural networks, 2014."},{"key":"e_1_3_2_1_9_1","volume-title":"Xpipe: Efficient pipeline model parallelism for multi-gpu dnn training","author":"Guan Lei","year":"2020","unstructured":"Lei Guan, Wotao Yin, Dongsheng Li, and Xicheng Lu. Xpipe: Efficient pipeline model parallelism for multi-gpu dnn training, 2020."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378465"},{"key":"e_1_3_2_1_11_1","volume-title":"et al. Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems, 32","author":"Huang Yanping","year":"2019","unstructured":"Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Dehao Chen, Mia Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc V Le, Yonghui Wu, et al. Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems, 32, 2019."},{"key":"e_1_3_2_1_12_1","unstructured":"HuggingFace. Nanotron: Minimalistic large language model 3d-parallelism training. https:\/\/github.com\/huggingface\/nanotron."},{"key":"e_1_3_2_1_13_1","first-page":"673","volume-title":"2022 USENIX Annual Technical Conference (USENIX ATC 22)","author":"Jia Xianyan","year":"2022","unstructured":"Xianyan Jia, Le Jiang, Ang Wang, Wencong Xiao, Ziji Shi, Jie Zhang, Xinyuan Li, Langshi Chen, Yong Li, Zhen Zheng, et al. Whale: Efficient giant model training over heterogeneous {GPUs}. In 2022 USENIX Annual Technical Conference (USENIX ATC 22), pages 673--688, 2022."},{"key":"e_1_3_2_1_14_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980, 2014."},{"key":"e_1_3_2_1_15_1","volume-title":"Pytorch distributed: Experiences on accelerating data parallel training. arXiv preprint arXiv:2006.15704","author":"Li Shen","year":"2020","unstructured":"Shen Li, Yanli Zhao, Rohan Varma, Omkar Salpekar, Pieter Noordhuis, Teng Li, Adam Paszke, Jeff Smith, Brian Vaughan, Pritam Damania, et al. Pytorch distributed: Experiences on accelerating data parallel training. arXiv preprint arXiv:2006.15704, 2020."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3605573.3605647"},{"key":"e_1_3_2_1_17_1","volume-title":"M6-10t: A sharing-delinking paradigm for efficient multi-trillion parameter pretraining","author":"Lin Junyang","year":"2021","unstructured":"Junyang Lin, An Yang, Jinze Bai, Chang Zhou, Le Jiang, Xianyan Jia, Ang Wang, Jie Zhang, Yong Li, Wei Lin, et al. M6-10t: A sharing-delinking paradigm for efficient multi-trillion parameter pretraining, 2021."},{"key":"e_1_3_2_1_18_1","volume-title":"Scheduling for Collaborative Multi-Level Checkpointing. In 2021 29th International Symposium on Modeling, Analysis, and Simulation of Computer and Telecommunication Systems (MASCOTS)","author":"Maurya Avinash","year":"2021","unstructured":"Avinash Maurya, Bogdan Nicolae, M. Mustafa Rafique, Thierry Tonellot, and Franck Cappello. Towards Efficient I\/O Scheduling for Collaborative Multi-Level Checkpointing. In 2021 29th International Symposium on Modeling, Analysis, and Simulation of Computer and Telecommunication Systems (MASCOTS), 2021."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588195.3592987"},{"key":"e_1_3_2_1_20_1","volume-title":"Bogdan Nicolae. DataStates-LLM: Lazy Asynchronous Checkpointing for Large Language Models. In Proc. of the International Symposium on High-Performance Parallel and Distributed Computing, HPDC'24","author":"Maurya Avinash","year":"2024","unstructured":"Avinash Maurya, Robert Underwood, M. Mustafa Rafique, Franck Cappello, and Bogdan Nicolae. DataStates-LLM: Lazy Asynchronous Checkpointing for Large Language Models. In Proc. of the International Symposium on High-Performance Parallel and Distributed Computing, HPDC'24, 2024."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3659995.3660038"},{"key":"e_1_3_2_1_22_1","volume-title":"Mixed precision training","author":"Micikevicius Paulius","year":"2018","unstructured":"Paulius Micikevicius, Sharan Narang, Jonah Alben, Gregory Diamos, Erich Elsen, David Garcia, Boris Ginsburg, Michael Houston, Oleksii Kuchaiev, Ganesh Venkatesh, and Hao Wu. Mixed precision training, 2018."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid49817.2020.00-76"},{"key":"e_1_3_2_1_25_1","unstructured":"Nvidia. NVIDIA Management Library (NVML). https:\/\/developer.nvidia.com\/management-library-nvml."},{"key":"e_1_3_2_1_26_1","first-page":"891","volume-title":"The 25th International Conference on Architectural Support for Programming Languages and Operating Systems","author":"Peng Xuan","year":"2020","unstructured":"Xuan Peng, Xuanhua Shi, Hulin Dai, Hai Jin, Weiliang Ma, Qian Xiong, Fan Yang, and Xuehai Qian. Capuchin: Tensor-based gpu memory management for deep learning. In The 25th International Conference on Architectural Support for Programming Languages and Operating Systems, pages 891--905, 2020."},{"key":"e_1_3_2_1_27_1","volume-title":"Gradients accumulation-pytorch. https:\/\/gist.github.com\/thomwolf\/ac7a7da6b1888c2eeac8ac8b9b05d3d3","year":"2019","unstructured":"Pytorch. Gradients accumulation-pytorch. https:\/\/gist.github.com\/thomwolf\/ac7a7da6b1888c2eeac8ac8b9b05d3d3, 2019."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"e_1_3_2_1_31_1","first-page":"551","volume-title":"2021 USENIX Annual Technical Conference (USENIX ATC 21)","author":"Ren Jie","year":"2021","unstructured":"Jie Ren, Samyam Rajbhandari, Reza Yazdani Aminabadi, Olatunji Ruwase, Shuangyan Yang, Minjia Zhang, Dong Li, and Yuxiong He. {Zero-offload}: Democratizing {billion-scale} model training. In 2021 USENIX Annual Technical Conference (USENIX ATC 21), pages 551--564, 2021."},{"key":"e_1_3_2_1_32_1","volume-title":"Horovod: fast and easy distributed deep learning in tensorflow. arXiv preprint arXiv:1802.05799","author":"Sergeev Alexander","year":"2018","unstructured":"Alexander Sergeev and Mike Del Balso. Horovod: fast and easy distributed deep learning in tensorflow. arXiv preprint arXiv:1802.05799, 2018."},{"key":"e_1_3_2_1_33_1","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053, 2019."},{"key":"e_1_3_2_1_34_1","unstructured":"Shaden Smith Mostofa Patwary Brandon Norick Patrick LeGresley Samyam Rajbhandari Jared Casper Zhun Liu Shrimai Prabhumoye George Zerveas Vijay Korthikanti Elton Zhang Rewon Child Reza Yazdani Aminabadi Julie Bernauer Xia Song Mohammad Shoeybi Yuxiong He Michael Houston Saurabh Tiwary and Bryan Catanzaro. Using deepspeed and megatron to train megatron-turing nlg 530b a large-scale generative language model 2022."},{"key":"e_1_3_2_1_35_1","volume-title":"DeepSpeed4Science Initiative: Enabling Large-Scale Scientific Discovery through Sophisticated AI System Technologies","author":"Song Shuaiwen Leon","year":"2023","unstructured":"Shuaiwen Leon Song, Bonnie Kruft, Minjia Zhang, Conglong Li, Shiyang Chen, et al. DeepSpeed4Science Initiative: Enabling Large-Scale Scientific Discovery through Sophisticated AI System Technologies, 2023."},{"key":"e_1_3_2_1_36_1","volume-title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Louis Martin, Kevin Stone, Peter Albert, Amjad Almahairi, et al. Llama 2: Open Foundation and Fine-Tuned Chat Models, 2023."},{"key":"e_1_3_2_1_37_1","volume-title":"Samyam Rajbhandari, Olatunji Ruwase, and Yuxiong He. Deepspeed zero-offload++: 6x higher training throughput via collaborative cpu\/gpu twin-flow","author":"Wang Guanhua","year":"2023","unstructured":"Guanhua Wang, Masahiro Tanaka, Xiaoxia Wu, Lok Chand Koppaka, Samyam Rajbhandari, Olatunji Ruwase, and Yuxiong He. Deepspeed zero-offload++: 6x higher training throughput via collaborative cpu\/gpu twin-flow, 2023."},{"key":"e_1_3_2_1_38_1","volume-title":"Proc. of the ACM SIGPLAN symposium on principles and practice of parallel programming","author":"Wang Linnan","year":"2018","unstructured":"Linnan Wang, Jinmian Ye, Yiyang Zhao, Wei Wu, Ang Li, Shuaiwen Leon Song, Zenglin Xu, and Tim Kraska. Superneurons: Dynamic gpu memory management for training deep neural networks. In Proc. of the ACM SIGPLAN symposium on principles and practice of parallel programming, 2018."},{"key":"e_1_3_2_1_39_1","volume-title":"Training of 1-trillion parameter ai begins. https:\/\/www.hpcwire.com\/2023\/11\/13\/training-of-1-trillion-parameter-scientific-ai-begins\/","author":"Wire HPC","year":"2023","unstructured":"HPC Wire. Training of 1-trillion parameter ai begins. https:\/\/www.hpcwire.com\/2023\/11\/13\/training-of-1-trillion-parameter-scientific-ai-begins\/, 2023."},{"key":"e_1_3_2_1_40_1","volume-title":"Workshop. BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","year":"2023","unstructured":"BigScience Workshop. BLOOM: A 176B-Parameter Open-Access Multilingual Language Model, 2023."},{"key":"e_1_3_2_1_41_1","volume-title":"Understanding the performance and estimating the cost of llm fine-tuning. arXiv preprint arXiv:2408.04693","author":"Xia Yuchen","year":"2024","unstructured":"Yuchen Xia, Jiho Kim, Yuhan Chen, Haojie Ye, Souvik Kundu, Nishil Talati, et al. Understanding the performance and estimating the cost of llm fine-tuning. arXiv preprint arXiv:2408.04693, 2024."},{"issue":"10","key":"e_1_3_2_1_42_1","volume":"45","author":"Xu P.","year":"2023","unstructured":"P. Xu, X. Zhu, and D. A. Clifton. Multimodal learning with transformers: A survey. IEEE Transactions on Pattern Analysis & Machine Intelligence, 45(10), 2023.","journal-title":"Multimodal learning with transformers: A survey. IEEE Transactions on Pattern Analysis & Machine Intelligence"},{"key":"e_1_3_2_1_43_1","volume-title":"Large batch optimization for deep learning: Training bert in 76 minutes. arXiv preprint arXiv:1904.00962","author":"You Yang","year":"2019","unstructured":"Yang You, Jing Li, Sashank Reddi, Jonathan Hseu, Sanjiv Kumar, Srinadh Bhojanapalli, Xiaodan Song, James Demmel, Kurt Keutzer, and Cho-Jui Hsieh. Large batch optimization for deep learning: Training bert in 76 minutes. arXiv preprint arXiv:1904.00962, 2019."},{"key":"e_1_3_2_1_44_1","volume-title":"Glm-130b: An open bilingual pre-trained model. arXiv preprint arXiv:2210.02414","author":"Zeng Aohan","year":"2022","unstructured":"Aohan Zeng, Xiao Liu, Zhengxiao Du, Zihan Wang, Hanyu Lai, Ming Ding, Zhuoyi Yang, Yifan Xu, Wendi Zheng, Xiao Xia, et al. Glm-130b: An open bilingual pre-trained model. arXiv preprint arXiv:2210.02414, 2022."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPADS60453.2023.00126"},{"key":"e_1_3_2_1_46_1","volume-title":"Xi Victoria Lin, et al. Opt: Open pre-trained transformer language models. arXiv preprint arXiv:2205.01068","author":"Zhang Susan","year":"2022","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, Shuohui Chen, Christopher Dewan, Mona Diab, Xian Li, Xi Victoria Lin, et al. Opt: Open pre-trained transformer language models. arXiv preprint arXiv:2205.01068, 2022."},{"key":"e_1_3_2_1_47_1","volume-title":"A survey of large language models. arXiv preprint arXiv:2303.18223","author":"Zhao Wayne Xin","year":"2023","unstructured":"Wayne Xin Zhao, Kun Zhou, Junyi Li, Tianyi Tang, Xiaolei Wang, Yupeng Hou, Yingqian Min, Beichen Zhang, Junjie Zhang, Zican Dong, et al. A survey of large language models. arXiv preprint arXiv:2303.18223, 2023."},{"key":"e_1_3_2_1_48_1","volume-title":"Jun Yang, and Michael Andersch. Scalefold: Reducing alphafold initial training time to 10 hours","author":"Zhu Feiwen","year":"2024","unstructured":"Feiwen Zhu, Arkadiusz Nowaczynski, Rundong Li, Jie Xin, Yifei Song, Michal Marcinkiewicz, Sukru Burc Eryilmaz, Jun Yang, and Michael Andersch. Scalefold: Reducing alphafold initial training time to 10 hours, 2024."}],"event":{"name":"Middleware '24: 25th International Middleware Conference","location":"Hong Kong Hong Kong","acronym":"Middleware '24","sponsor":["IFIP","Usenix"]},"container-title":["Proceedings of the 25th International Middleware Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3652892.3700781","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3652892.3700781","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3652892.3700781","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T22:53:57Z","timestamp":1750287237000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3652892.3700781"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"references-count":48,"alternative-id":["10.1145\/3652892.3700781","10.1145\/3652892"],"URL":"https:\/\/doi.org\/10.1145\/3652892.3700781","relation":{},"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2024-12-02","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}