{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,4]],"date-time":"2026-02-04T18:00:43Z","timestamp":1770228043224,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":39,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T00:00:00Z","timestamp":1723420800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,12]]},"DOI":"10.1145\/3673038.3673140","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T18:29:01Z","timestamp":1723141741000},"page":"148-157","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Sparse Gradient Communication with AlltoAll for Accelerating Distributed Deep Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7221-1216","authenticated-orcid":false,"given":"Jing","family":"Peng","sequence":"first","affiliation":[{"name":"Chengdu University of Information Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1817-9059","authenticated-orcid":false,"given":"Zihan","family":"Li","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1418-5160","authenticated-orcid":false,"given":"Shaohuai","family":"Shi","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2955-750X","authenticated-orcid":false,"given":"Bo","family":"Li","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,8,12]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/181014.181756"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the 25th International Conference on Neural Information Processing Systems (NIPS). 1223\u20131231","author":"Dean Jeffrey","year":"2012","unstructured":"Jeffrey Dean, Greg\u00a0S. Corrado, Rajat Monga, Kai Chen, Matthieu Devin, Quoc\u00a0V. Le, Mark\u00a0Z. Mao, Marc\u2019Aurelio Ranzato, Andrew Senior, Paul Tucker, Ke Yang, and Andrew\u00a0Y. Ng. 2012. Large scale distributed deep networks. In Proceedings of the 25th International Conference on Neural Information Processing Systems (NIPS). 1223\u20131231."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_4_1","first-page":"5496","article-title":"Scalable global optimization via local Bayesian optimization. In Proceedings of the 25th International Conference on Neural Information Processing Systems (NIPS)","volume":"493","author":"Eriksson David","year":"2019","unstructured":"David Eriksson, Michael Pearce, Jacob Gardner, Ryan\u00a0D Turner, and Matthias Poloczek. 2019. Scalable global optimization via local Bayesian optimization. In Proceedings of the 25th International Conference on Neural Information Processing Systems (NIPS). Article 493, 5496\u20135507\u00a0pages.","journal-title":"Article"},{"key":"e_1_3_2_1_5_1","volume-title":"large minibatch sgd: Training imagenet in 1 hour. arXiv:1706.02677","author":"Goyal Priya","year":"2017","unstructured":"Priya Goyal, Piotr Doll\u00e1r, Ross Girshick, Pieter Noordhuis, Lukasz Wesolowski, Aapo Kyrola, Andrew Tulloch, Yangqing Jia, and Kaiming He. 2017. Accurate, large minibatch sgd: Training imagenet in 1 hour. arXiv:1706.02677 (2017)."},{"key":"e_1_3_2_1_6_1","volume-title":"The communication challenge for MPP: Intel Paragon and Meiko CS-2. Parallel computing 20, 3","author":"Hockney W","year":"1994","unstructured":"Roger\u00a0W Hockney. 1994. The communication challenge for MPP: Intel Paragon and Meiko CS-2. Parallel computing 20, 3 (1994), 389\u2013398."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-15646-5_3"},{"key":"e_1_3_2_1_8_1","unstructured":"Andrej Karpathy. 2015. char-rnn. github.com\/karpathy\/char-rnn."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the Annual Conference of the North American Chapter of the Association for Computational Linguistics (NAACL-HLT), Vol.\u00a01. 4171\u20134186","author":"Ming-Wei\u00a0Chang Jacob Devlin","year":"2019","unstructured":"Jacob Devlin Ming-Wei\u00a0Chang Kenton and Lee\u00a0Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the Annual Conference of the North American Chapter of the Association for Computational Linguistics (NAACL-HLT), Vol.\u00a01. 4171\u20134186."},{"key":"e_1_3_2_1_10_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma P","year":"2014","unstructured":"Diederik\u00a0P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508399"},{"key":"e_1_3_2_1_12_1","volume-title":"Pytorch distributed: Experiences on accelerating data parallel training. arXiv preprint arXiv:2006.15704","author":"Li Shen","year":"2020","unstructured":"Shen Li, Yanli Zhao, Rohan Varma, Omkar Salpekar, Pieter Noordhuis, Teng Li, Adam Paszke, Jeff Smith, Brian Vaughan, Pritam Damania, 2020. Pytorch distributed: Experiences on accelerating data parallel training. arXiv preprint arXiv:2006.15704 (2020)."},{"key":"e_1_3_2_1_13_1","volume-title":"International Conference on Learning Representations","author":"Lin Yujun","year":"2018","unstructured":"Yujun Lin, Song Han, Huizi Mao, Yu Wang, and William\u00a0J Dally. 2018. Deep gradient compression: Reducing the communication bandwidth for distributed training. International Conference on Learning Representations (2018)."},{"key":"e_1_3_2_1_14_1","first-page":"297","article-title":"An efficient statistical-based gradient compression technique for distributed training systems","volume":"3","author":"M\u00a0Abdelmoniem Ahmed","year":"2021","unstructured":"Ahmed M\u00a0Abdelmoniem, Ahmed Elzanaty, Mohamed-Slim Alouini, and Marco Canini. 2021. An efficient statistical-based gradient compression technique for distributed training systems. Proceedings of Machine Learning and Systems 3 (2021), 297\u2013322.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359642"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-24685-5_1"},{"key":"e_1_3_2_1_17_1","volume-title":"Language models are unsupervised multitask learners. OpenAI blog 1, 8","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, Ilya Sutskever, 2019. Language models are unsupervised multitask learners. OpenAI blog 1, 8 (2019), 9."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356222"},{"key":"e_1_3_2_1_19_1","volume-title":"19th USENIX Symposium on Networked Systems Design and Implementation (NSDI). 1027\u20131040","author":"Romero Joshua","year":"2022","unstructured":"Joshua Romero, Junqi Yin, Nouamane Laanait, Bing Xie, M\u00a0Todd Young, Sean Treichler, Vitalii Starchenko, Albina Borisevich, Alex Sergeev, and Michael Matheson. 2022. Accelerating collective communication in data parallel training across deep learning frameworks. In 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI). 1027\u20131040."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2014-274"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737367"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM42981.2021.9488803"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM41043.2020.9155269"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2019.00220"},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the 4th Machine Learning and Systems Conference (MLSys), Vol.\u00a03. 401\u2013412","author":"Shi Shaohuai","year":"2021","unstructured":"Shaohuai Shi, Xianhao Zhou, Shutao Song, Xingyao Wang, Zilin Zhu, Xue Huang, Xinan Jiang, Feihu Zhou, Zhenyu Guo, Liqiang Xie, 2021. Towards scalable distributed training of deep learning on public cloud clusters. In Proceedings of the 4th Machine Learning and Systems Conference (MLSys), Vol.\u00a03. 401\u2013412."},{"key":"e_1_3_2_1_26_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 25th International Conference on Neural Information Processing Systems. 2951\u20132959","author":"Snoek Jasper","year":"2012","unstructured":"Jasper Snoek, Hugo Larochelle, and Ryan\u00a0P Adams. 2012. Practical bayesian optimization of machine learning algorithms. In Proceedings of the 25th International Conference on Neural Information Processing Systems. 2951\u20132959."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342005051521"},{"key":"e_1_3_2_1_29_1","volume-title":"2017 USENIX Annual Technical Conference (ATC 17)","author":"Vogels Thijs","year":"2017","unstructured":"Thijs Vogels, Sai\u00a0Praneeth Karimireddy, and Martin Jaggi. 2017. Poseidon: An efficient communication architecture for distributed deep learning on GPU clusters. In 2017 USENIX Annual Technical Conference (ATC 17). 181\u2013193."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 25th International Conference on Neural Information Processing Systems (NIPS). 14269\u201314278","author":"Vogels Thijs","year":"2019","unstructured":"Thijs Vogels, Sai\u00a0Praneeth Karimireddy, and Martin Jaggi. 2019. PowerSGD: Practical Low-Rank Gradient Compression for Distributed Optimization. In Proceedings of the 25th International Conference on Neural Information Processing Systems (NIPS). 14269\u201314278."},{"key":"e_1_3_2_1_31_1","unstructured":"U von Luxburg S Bengio HM Wallach R Fergus SVN Vishwanathan and R Garnett. [n. d.]. https:\/\/github. com\/baidu-research\/baidu-allreduce. ([n. d.])."},{"key":"e_1_3_2_1_32_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Wang Guanhua","year":"2024","unstructured":"Guanhua Wang, Heyang Qin, Sam\u00a0Ade Jacobs, Connor Holmes, Samyam Rajbhandari, Olatunji Ruwase, Feng Yan, Lei Yang, and Yuxiong He. 2024. ZeRO++: Extremely Efficient Collective Communication for Giant Model Training. The Twelfth International Conference on Learning Representations (2024)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3577193.3593711"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3567505"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of Machine Learning and Systems 5","author":"Wang Zhuang","year":"2023","unstructured":"Zhuang Wang, Xinyu Wu, Zhaozhuo Xu, and TS Ng. 2023. Cupcake: A Compression Scheduler for Scalable Communication-Efficient Distributed Training. Proceedings of Machine Learning and Systems 5 (2023)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS51616.2021.00060"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2023.3323282"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS57875.2023.00054"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS57875.2023.00031"}],"event":{"name":"ICPP '24: the 53rd International Conference on Parallel Processing","location":"Gotland Sweden","acronym":"ICPP '24"},"container-title":["Proceedings of the 53rd International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673140","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3673038.3673140","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T17:27:36Z","timestamp":1758648456000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673140"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,12]]},"references-count":39,"alternative-id":["10.1145\/3673038.3673140","10.1145\/3673038"],"URL":"https:\/\/doi.org\/10.1145\/3673038.3673140","relation":{},"subject":[],"published":{"date-parts":[[2024,8,12]]},"assertion":[{"value":"2024-08-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}