{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:44:40Z","timestamp":1782834280675,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,4]],"date-time":"2024-08-04T00:00:00Z","timestamp":1722729600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-sa\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["RS-2024-00340099"],"award-info":[{"award-number":["RS-2024-00340099"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Institute of Information & Communications Technology Planning & Evaluation (IITP)","award":["RS-2024-00418784"],"award-info":[{"award-number":["RS-2024-00418784"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,4]]},"DOI":"10.1145\/3651890.3672228","type":"proceedings-article","created":{"date-parts":[[2024,7,31]],"date-time":"2024-07-31T13:11:43Z","timestamp":1722431503000},"page":"707-720","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":17,"title":["Accelerating Model Training in Multi-cluster Environments with Consumer-grade GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9872-6234","authenticated-orcid":false,"given":"Hwijoon","family":"Lim","sequence":"first","affiliation":[{"name":"KAIST, Daejeon, Korea, South ? Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7673-1279","authenticated-orcid":false,"given":"Juncheol","family":"Ye","sequence":"additional","affiliation":[{"name":"KAIST, Daejeon, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0503-4478","authenticated-orcid":false,"given":"Sangeetha","family":"Abdu Jyothi","sequence":"additional","affiliation":[{"name":"UC Irvine, VMware Research, Irvine, California, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6922-7244","authenticated-orcid":false,"given":"Dongsu","family":"Han","sequence":"additional","affiliation":[{"name":"KAIST, Daejeon, Korea, South ? Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,8,4]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM42981.2021.9488810"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of Machine Learning and Systems 4","author":"Agarwal Saurabh","year":"2022","unstructured":"Saurabh Agarwal, Hongyi Wang, Shivaram Venkataraman, and Dimitris Papailiopoulos. 2022. On the utility of gradient compression in distributed training systems. Proceedings of Machine Learning and Systems 4 (2022)."},{"key":"e_1_3_2_1_3_1","volume-title":"Sparse communication for distributed gradient descent. arXiv preprint arXiv:1704.05021","author":"Aji Alham Fikri","year":"2017","unstructured":"Alham Fikri Aji and Kenneth Heafield. 2017. Sparse communication for distributed gradient descent. arXiv preprint arXiv:1704.05021 (2017)."},{"key":"e_1_3_2_1_4_1","volume-title":"The convergence of sparsified gradient methods. Advances in Neural Information Processing Systems 31","author":"Alistarh Dan","year":"2018","unstructured":"Dan Alistarh, Torsten Hoefler, Mikael Johansson, Nikola Konstantinov, Sarit Khirirat, and C\u00e9dric Renggli. 2018. The convergence of sparsified gradient methods. Advances in Neural Information Processing Systems 31 (2018)."},{"key":"e_1_3_2_1_5_1","unstructured":"The ZeroMQ authors. 2023. ZeroMQ. https:\/\/zeromq.org\/."},{"key":"e_1_3_2_1_6_1","unstructured":"BIZON. 2023. GPU Deep Learning Benchmarks 2023--2024. https:\/\/bizon-tech.com\/gpu-benchmarks\/NVIDIA-RTX-3090-vs-NVIDIA-A100-40-GB-(PCIe)\/579vs592. [Accessed 01-02-2024]."},{"key":"e_1_3_2_1_7_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020)."},{"key":"e_1_3_2_1_8_1","unstructured":"Yangrui Chen Cong Xie Meng Ma Juncheng Gu Yanghua Peng Haibin Lin Chuan Wu and Yibo Zhu. 2022. SAPipe: Staleness-Aware Pipeline for Data Parallel DNN Training. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_9_1","unstructured":"Jeffrey Dean Greg Corrado Rajat Monga Kai Chen Matthieu Devin Mark Mao Marc'aurelio Ranzato Andrew Senior Paul Tucker Ke Yang et al. 2012. Large scale distributed deep networks. Advances in neural information processing systems 25 (2012)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_11_1","unstructured":"Tim Dettmers. 2023. The Best GPUs for Deep Learning in 2023 --- An In-depth Analysis. https:\/\/timdettmers.com\/2023\/01\/30\/which-gpu-for-deep-learning\/. [Accessed 27-Jun-2023]."},{"key":"e_1_3_2_1_12_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the USENIX Annual Technical Conference (ATC). https:\/\/www.flux.utah.edu\/paper\/duplyakin-atc19","author":"Duplyakin Dmitry","year":"2019","unstructured":"Dmitry Duplyakin, Robert Ricci, Aleksander Maricq, Gary Wong, Jonathon Duerig, Eric Eide, Leigh Stoller, Mike Hibler, David Johnson, Kirk Webb, Aditya Akella, Kuangching Wang, Glenn Ricart, Larry Landweber, Chip Elliott, Michael Zink, Emmanuel Cecchet, Snigdhaswin Kar, and Prabodh Mishra. 2019. The Design and Operation of CloudLab. In Proceedings of the USENIX Annual Technical Conference (ATC). https:\/\/www.flux.utah.edu\/paper\/duplyakin-atc19"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5793"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3452296.3472904"},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of Machine Learning and Systems 1","author":"Hashemi Sayed Hadi","year":"2019","unstructured":"Sayed Hadi Hashemi, Sangeetha Abdu Jyothi, and Roy Campbell. 2019. Tictac: Accelerating distributed deep learning with communication scheduling. Proceedings of Machine Learning and Systems 1 (2019)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_18_1","volume-title":"Phillip B Gibbons, Garth A Gibson, Greg Ganger, and Eric P Xing.","author":"Ho Qirong","year":"2013","unstructured":"Qirong Ho, James Cipar, Henggang Cui, Seunghak Lee, Jin Kyu Kim, Phillip B Gibbons, Garth A Gibson, Greg Ganger, and Eric P Xing. 2013. More effective distributed ML via a stale synchronous parallel parameter server. Advances in neural information processing systems 26 (2013)."},{"key":"e_1_3_2_1_19_1","volume-title":"Marco Canini, and Peter Richt\u00e1rik.","author":"Horv\u00f3th Samuel","year":"2022","unstructured":"Samuel Horv\u00f3th, Chen-Yu Ho, Ludovit Horvath, Atal Narayan Sahu, Marco Canini, and Peter Richt\u00e1rik. 2022. Natural compression for distributed deep learning. In Mathematical and Scientific Machine Learning. PMLR."},{"key":"e_1_3_2_1_20_1","volume-title":"Gaia: Geo-distributed machine learning approaching LAN speeds.. In NSDI.","author":"Hsieh Kevin","year":"2017","unstructured":"Kevin Hsieh, Aaron Harlap, Nandita Vijaykumar, Dimitris Konomis, Gregory R Ganger, Phillip B Gibbons, and Onur Mutlu. 2017. Gaia: Geo-distributed machine learning approaching LAN speeds.. In NSDI."},{"key":"e_1_3_2_1_21_1","volume-title":"13th Generation Intel\u00ae Core\u2122 Processors Datasheet","unstructured":"Intel. 2023. 13th Generation Intel\u00ae Core\u2122 Processors Datasheet."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of Machine Learning and Systems 1","author":"Jayarajan Anand","year":"2019","unstructured":"Anand Jayarajan, Jinliang Wei, Garth Gibson, Alexandra Fedorova, and Gennady Pekhimenko. 2019. Priority-based parameter propagation for distributed DNN training. Proceedings of Machine Learning and Systems 1 (2019)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.5555\/3488766.3488792"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2640087.2644155"},{"key":"e_1_3_2_1_25_1","volume-title":"Communication efficient distributed machine learning with the parameter server. Advances in Neural Information Processing Systems 27","author":"Li Mu","year":"2014","unstructured":"Mu Li, David G Andersen, Alexander J Smola, and Kai Yu. 2014. Communication efficient distributed machine learning with the parameter server. Advances in Neural Information Processing Systems 27 (2014)."},{"key":"e_1_3_2_1_26_1","unstructured":"Shen Li Yanli Zhao Rohan Varma Omkar Salpekar Pieter Noordhuis Teng Li Adam Paszke Jeff Smith Brian Vaughan Pritam Damania et al. 2020. Pytorch distributed: Experiences on accelerating data parallel training. arXiv preprint arXiv:2006.15704 (2020)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_28_1","volume-title":"Pointer Sentinel Mixture Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Byj72udxe","author":"Merity Stephen","year":"2017","unstructured":"Stephen Merity, Caiming Xiong, James Bradbury, and Richard Socher. 2017. Pointer Sentinel Mixture Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Byj72udxe"},{"key":"e_1_3_2_1_29_1","unstructured":"NVIDIA. 2022. NVIDIA DGX A100 : The Universal System for AI Infrastructure --- nvidia.com. https:\/\/www.nvidia.com\/en-us\/data-center\/dgx-a100\/. [Accessed 22-09-2023]."},{"key":"e_1_3_2_1_30_1","unstructured":"NVIDIA. 2023. NVIDIA H100 Tensor Core GPU. https:\/\/nvidia.com\/en-us\/data-center\/h100. [Accessed 27-Jun-2023]."},{"key":"e_1_3_2_1_31_1","unstructured":"NVIDIA. 2023. NVLink & NVSwitch for Advanced Multi-GPU Communication. https:\/\/nvidia.com\/en-us\/data-center\/nvlink. [Accessed 27-Jun-2023]."},{"key":"e_1_3_2_1_32_1","unstructured":"NVIDIA. 2024. GPUDirect. https:\/\/developer.nvidia.com\/gpudirect."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359642"},{"key":"e_1_3_2_1_34_1","volume-title":"A lock-free approach to parallelizing stochastic gradient descent. Advances in neural information processing systems 24","author":"Recht Benjamin","year":"2011","unstructured":"Benjamin Recht, Christopher Re, Stephen Wright, and Feng Niu. 2011. Hogwild!: A lock-free approach to parallelizing stochastic gradient descent. Advances in neural information processing systems 24 (2011)."},{"key":"e_1_3_2_1_35_1","volume-title":"2021 USENIX Annual Technical Conference (USENIX ATC 21)","author":"Ren Jie","year":"2021","unstructured":"Jie Ren, Samyam Rajbhandari, Reza Yazdani Aminabadi, Olatunji Ruwase, Shuangyan Yang, Minjia Zhang, Dong Li, and Yuxiong He. 2021. {ZeRO-Offload}: Democratizing {Billion-Scale} model training. In 2021 USENIX Annual Technical Conference (USENIX ATC 21)."},{"key":"e_1_3_2_1_36_1","volume-title":"A stochastic approximation method. The annals of mathematical statistics","author":"Robbins Herbert","year":"1951","unstructured":"Herbert Robbins and Sutton Monro. 1951. A stochastic approximation method. The annals of mathematical statistics (1951)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3387514.3405899"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Frank Seide Hao Fu Jasha Droppo Gang Li and Dong Yu. 2014. 1-bit stochastic gradient descent and its application to data-parallel distributed training of speech dnns. In Fifteenth annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2014-274"},{"key":"e_1_3_2_1_39_1","unstructured":"Amazon Web Services. 2023. Compute - Amazon EC2 Instance Types - AWS --- aws.amazon.com. https:\/\/aws.amazon.com\/en\/ec2\/instance-types\/. [Accessed 21-09-2023]."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2019.00220"},{"key":"e_1_3_2_1_41_1","volume-title":"Proceedings of Machine Learning and Systems 3","author":"Shi Shaohuai","year":"2021","unstructured":"Shaohuai Shi, Xianhao Zhou, Shutao Song, Xingyao Wang, Zilin Zhu, Xue Huang, Xinan Jiang, Feihu Zhou, Zhenyu Guo, Liqiang Xie, et al. 2021. Towards scalable distributed training of deep learning on public cloud clusters. Proceedings of Machine Learning and Systems 3 (2021)."},{"key":"e_1_3_2_1_42_1","volume-title":"Don't decay the learning rate, increase the batch size. arXiv preprint arXiv:1711.00489","author":"Smith Samuel L","year":"2017","unstructured":"Samuel L Smith, Pieter-Jan Kindermans, Chris Ying, and Quoc V Le. 2017. Don't decay the learning rate, increase the batch size. arXiv preprint arXiv:1711.00489 (2017)."},{"key":"e_1_3_2_1_43_1","volume-title":"Sparsified SGD with memory. Advances in Neural Information Processing Systems 31","author":"Stich Sebastian U","year":"2018","unstructured":"Sebastian U Stich, Jean-Baptiste Cordonnier, and Martin Jaggi. 2018. Sparsified SGD with memory. Advances in Neural Information Processing Systems 31 (2018)."},{"key":"e_1_3_2_1_44_1","volume-title":"Interspeech","author":"Str\u00f6m Nikko","year":"2015","unstructured":"Nikko Str\u00f6m. 2015. Scalable Distributed DNN Training Using Commodity GPU Cloud Computing. In Interspeech 2015. https:\/\/www.amazon.science\/publications\/scalable-distributed-dnn-training-using-commodity-gpu-cloud-computing"},{"key":"e_1_3_2_1_45_1","volume-title":"Connor Holmes, Samyam Rajbhandari, Olatunji Ruwase, Feng Yan, Lei Yang, and Yuxiong He.","author":"Wang Guanhua","year":"2023","unstructured":"Guanhua Wang, Heyang Qin, Sam Ade Jacobs, Connor Holmes, Samyam Rajbhandari, Olatunji Ruwase, Feng Yan, Lei Yang, and Yuxiong He. 2023. ZeRO++: Extremely Efficient Collective Communication for Giant Model Training. arXiv preprint arXiv:2306.10209 (2023)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3567505"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3630048.3630184"},{"key":"e_1_3_2_1_48_1","volume-title":"Konstantinos Karatsenidis, Marco Canini, and Panos Kalnis.","author":"Xu Hang","year":"2020","unstructured":"Hang Xu, Chen-Yu Ho, Ahmed M Abdelmoniem, Aritra Dutta, El Houcine Bergou, Konstantinos Karatsenidis, Marco Canini, and Panos Kalnis. 2020. Compressed communication for distributed deep learning: Survey and quantitative evaluation. Technical Report. KAUST."},{"key":"e_1_3_2_1_49_1","volume-title":"Poseidon: A system architecture for efficient gpu-based deep learning on multiple machines. arXiv preprint arXiv:1512.06216","author":"Zhang Hao","year":"2015","unstructured":"Hao Zhang, Zhiting Hu, Jinliang Wei, Pengtao Xie, Gunhee Kim, Qirong Ho, and Eric Xing. 2015. Poseidon: A system architecture for efficient gpu-based deep learning on multiple machines. arXiv preprint arXiv:1512.06216 (2015)."},{"key":"e_1_3_2_1_50_1","volume-title":"International conference on machine learning. PMLR.","author":"Zhang Ruiliang","year":"2014","unstructured":"Ruiliang Zhang and James Kwok. 2014. Asynchronous distributed ADMM for consensus optimization. In International conference on machine learning. PMLR."},{"key":"e_1_3_2_1_51_1","volume-title":"Xi Victoria Lin, et al","author":"Zhang Susan","year":"2022","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, Shuohui Chen, Christopher Dewan, Mona Diab, Xian Li, Xi Victoria Lin, et al. 2022. OPT: Open Pre-trained Transformer Language Models. arXiv preprint arXiv:2205.01068 (2022)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638950"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3472456.3472467"},{"key":"e_1_3_2_1_54_1","volume-title":"International Conference on Machine Learning. PMLR.","author":"Zheng Shuxin","year":"2017","unstructured":"Shuxin Zheng, Qi Meng, Taifeng Wang, Wei Chen, Nenghai Yu, Zhi-Ming Ma, and Tie-Yan Liu. 2017. Asynchronous stochastic gradient descent with delay compensation. In International Conference on Machine Learning. PMLR."},{"key":"e_1_3_2_1_55_1","volume-title":"Compressed communication for distributed training: Adaptive methods and system. arXiv preprint arXiv:2105.07829","author":"Zhong Yuchen","year":"2021","unstructured":"Yuchen Zhong, Cong Xie, Shuai Zheng, and Haibin Lin. 2021. Compressed communication for distributed training: Adaptive methods and system. arXiv preprint arXiv:2105.07829 (2021)."},{"key":"e_1_3_2_1_56_1","volume-title":"Slow learners are fast. Advances in neural information processing systems 22","author":"Zinkevich Martin","year":"2009","unstructured":"Martin Zinkevich, John Langford, and Alex Smola. 2009. Slow learners are fast. Advances in neural information processing systems 22 (2009)."}],"event":{"name":"ACM SIGCOMM '24: ACM SIGCOMM 2024 Conference","location":"Sydney NSW Australia","acronym":"ACM SIGCOMM '24","sponsor":["SIGCOMM ACM Special Interest Group on Data Communication"]},"container-title":["Proceedings of the ACM SIGCOMM 2024 Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3651890.3672228","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3651890.3672228","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T17:49:12Z","timestamp":1750268952000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3651890.3672228"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,4]]},"references-count":56,"alternative-id":["10.1145\/3651890.3672228","10.1145\/3651890"],"URL":"https:\/\/doi.org\/10.1145\/3651890.3672228","relation":{},"subject":[],"published":{"date-parts":[[2024,8,4]]},"assertion":[{"value":"2024-08-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}