{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T07:06:56Z","timestamp":1772867216570,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":62,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,2,19]],"date-time":"2020-02-19T00:00:00Z","timestamp":1582070400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"European Research Council (ERC) under the European Union?s Horizon 2020 programme, grant agreement DAPP,","award":["No. 678880"],"award-info":[{"award-number":["No. 678880"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,2,19]]},"DOI":"10.1145\/3332466.3374528","type":"proceedings-article","created":{"date-parts":[[2020,2,19]],"date-time":"2020-02-19T19:13:53Z","timestamp":1582139633000},"page":"45-61","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":44,"title":["Taming unbalanced training workloads in deep learning with partial collective operations"],"prefix":"10.1145","author":[{"given":"Shigang","family":"Li","sequence":"first","affiliation":[{"name":"ETH Zurich"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tal","family":"Ben-Nun","sequence":"additional","affiliation":[{"name":"ETH Zurich"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Salvatore Di","family":"Girolamo","sequence":"additional","affiliation":[{"name":"ETH Zurich"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dan","family":"Alistarh","sequence":"additional","affiliation":[{"name":"IST Austria"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Torsten","family":"Hoefler","sequence":"additional","affiliation":[{"name":"ETH Zurich"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,2,19]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Mart\u00edn Abadi Ashish Agarwal Paul Barham Eugene Brevdo Zhifeng Chen Craig Citro Greg S. Corrado Andy Davis Jeffrey Dean Matthieu Devin Sanjay Ghemawat Ian Goodfellow Andrew Harp Geoffrey Irving Michael Isard Yangqing Jia Rafal Jozefowicz Lukasz Kaiser Manjunath Kudlur Josh Levenberg Dandelion Man\u00e9 Rajat Monga Sherry Moore Derek Murray Chris Olah Mike Schuster Jonathon Sillens Benoit Steiner Ilya Sutskever Kunal Talwar Paul Tucker Vincent Vanhoucke Vijay Vasudevan Fernanda Vi\u00e9gas Oriol Vinyals Pete Warden Martin Wattenberg Martin Wicke Yuan Yu and Xiaoqiang Zheng. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems https:\/\/www.tensorflow.org\/ Software available from tensorflow.org.  Mart\u00edn Abadi Ashish Agarwal Paul Barham Eugene Brevdo Zhifeng Chen Craig Citro Greg S. Corrado Andy Davis Jeffrey Dean Matthieu Devin Sanjay Ghemawat Ian Goodfellow Andrew Harp Geoffrey Irving Michael Isard Yangqing Jia Rafal Jozefowicz Lukasz Kaiser Manjunath Kudlur Josh Levenberg Dandelion Man\u00e9 Rajat Monga Sherry Moore Derek Murray Chris Olah Mike Schuster Jonathon Sillens Benoit Steiner Ilya Sutskever Kunal Talwar Paul Tucker Vincent Vanhoucke Vijay Vasudevan Fernanda Vi\u00e9gas Oriol Vinyals Pete Warden Martin Wattenberg Martin Wicke Yuan Yu and Xiaoqiang Zheng. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems https:\/\/www.tensorflow.org\/ Software available from tensorflow.org."},{"key":"e_1_3_2_1_2_1","volume-title":"Advances in Neural Information Processing Systems 31. Curran Associates","author":"Alistarh Dan"},{"key":"e_1_3_2_1_3_1","unstructured":"Dario Amodei and Danny Hernandez. 2018. AI and Compute. https:\/\/openai.com\/blog\/ai-and-compute\/.  Dario Amodei and Danny Hernandez. 2018. AI and Compute. https:\/\/openai.com\/blog\/ai-and-compute\/."},{"key":"e_1_3_2_1_4_1","volume-title":"Stochastic gradient push for distributed deep learning. arXiv preprint arXiv:1811.10792","author":"Assran Mahmoud","year":"2018"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"A. Awan K. Hamidouche J. Hashmi and D. Panda. 2017. S-Caffe: Co-designing MPI Runtimes and Caffe for Scalable Deep Learning on Modern GPU Clusters.  A. Awan K. Hamidouche J. Hashmi and D. Panda. 2017. S-Caffe: Co-designing MPI Runtimes and Caffe for Scalable Deep Learning on Modern GPU Clusters.","DOI":"10.1145\/3018743.3018769"},{"key":"e_1_3_2_1_6_1","unstructured":"Jimmy Ba and Brendan Frey. 2013. Adaptive dropout for training deep neural networks. In Advances in Neural Information Processing Systems. 3084--3092.  Jimmy Ba and Brendan Frey. 2013. Adaptive dropout for training deep neural networks. In Advances in Neural Information Processing Systems. 3084--3092."},{"key":"e_1_3_2_1_7_1","volume-title":"Delving Deeper into Convolutional Networks for Learning Video Representations. arXiv e-prints","author":"Ballas Nicolas","year":"2015"},{"key":"e_1_3_2_1_8_1","volume-title":"Sandia National Laboratories","author":"Barrett Brian W","year":"2018"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2019.00018"},{"key":"e_1_3_2_1_10_1","unstructured":"T. Ben-Nun and T. Hoefler. 2018. Demystifying Parallel and Distributed Deep Learning: An In-Depth Concurrency Analysis. CoRR abs\/1802.09941 (Feb. 2018).  T. Ben-Nun and T. Hoefler. 2018. Demystifying Parallel and Distributed Deep Learning: An In-Depth Concurrency Analysis. CoRR abs\/1802.09941 (Feb. 2018)."},{"key":"e_1_3_2_1_11_1","volume-title":"11th USENIX Symposium on Operating Systems Design and Implementation (OSDI 14)","author":"Chilimbi Trishul","year":"2014"},{"key":"e_1_3_2_1_12_1","volume-title":"GossipGraD: Scalable Deep Learning using Gossip Communication based Asynchronous Gradient Descent. CoRR abs\/1803.05880","author":"Daily Jeff","year":"2018"},{"key":"e_1_3_2_1_13_1","volume-title":"Ng","author":"Dean Jeffrey","year":"2012"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_15_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. CoRR abs\/1810.04805","author":"Devlin Jacob","year":"2018"},{"key":"e_1_3_2_1_16_1","volume-title":"Taylor","author":"Devries Terrance","year":"2017"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/HOTI.2015.21"},{"key":"e_1_3_2_1_18_1","volume-title":"Sergio Guadarrama, Marcus Rohrbach, Subhashini Venugopalan, Kate Saenko, and Trevor Darrell.","author":"Donahue Jeff","year":"2014"},{"key":"e_1_3_2_1_19_1","volume-title":"Bringing HPC techniques to deep learning. (2017). URL http:\/\/research.baidu.com\/bringing-hpc-techniques-deep-learning","author":"Gibiansky Andrew","year":"2017"},{"key":"e_1_3_2_1_20_1","volume-title":"Model Accuracy and Runtime Tradeoff in Distributed Deep Learning: A Systematic Study. arXiv e-prints (Sep","author":"Gupta Suyog","year":"2015"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the 26th International Conference on Neural Information Processing Systems -","volume":"1","author":"Ho Qirong"},{"key":"e_1_3_2_1_23_1","volume-title":"Long short-term memory. Neural computation 9, 8","author":"Hochreiter Sepp","year":"1997"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126970"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/1362622.1362692"},{"key":"e_1_3_2_1_26_1","first-page":"2","article-title":"Energy, Memory, and Runtime Tradeoffs for Implementing Collective Communication Operations","volume":"1","author":"Hoefler T.","year":"2014","journal-title":"Journal of Supercomputing Frontiers and Innovations"},{"key":"e_1_3_2_1_27_1","first-page":"4","article-title":"The Effect of Network Noise on Large-Scale Collective Communications","volume":"19","author":"Hoefler T.","year":"2009","journal-title":"Parallel Processing Letters (PPL)"},{"key":"e_1_3_2_1_28_1","volume-title":"International Conference for High Performance Computing, Networking, Storage and Analysis (SC'10)","author":"Hoefler T."},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of the 14th USENIX Conference on Networked Systems Design and Implementation (NSDI'17)","author":"Hsieh Kevin","year":"2017"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid.2011.22"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CloudCom.2010.69"},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the 2nd SysML Conference.","author":"Jayarajan Anand","year":"2019"},{"key":"e_1_3_2_1_34_1","volume-title":"How to scale distributed deep learning? CoRR abs\/1611.04581","author":"Jin Peter H.","year":"2016"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00054"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Y. LeCun Y. Bengio and G. Hinton. 2015. Deep learning. Nature 521 7553 (2015) 436--444.  Y. LeCun Y. Bengio and G. Hinton. 2015. Deep learning. Nature 521 7553 (2015) 436--444.","DOI":"10.1038\/nature14539"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.5555\/2685048.2685095"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295285"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the 35th International Conference on Machine Learning (Proceedings of Machine Learning Research), Jennifer Dy and Andreas Krause (Eds.)","volume":"80","author":"Lian Xiangru","year":"2018"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00068"},{"key":"e_1_3_2_1_42_1","volume-title":"MPI: A Message-Passing Interface Standard Version 3.1.","author":"Interface Forum Message Passing","year":"2015"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.09.002"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-24685-5_1"},{"key":"e_1_3_2_1_45_1","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei and Ilya Sutskever. 2018. Language Models are Unsupervised Multitask Learners. (2018). https:\/\/d4mucfpksywv.cloudfront.net\/better-language-models\/language-models.pdf  Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei and Ilya Sutskever. 2018. Language Models are Unsupervised Multitask Learners. (2018). https:\/\/d4mucfpksywv.cloudfront.net\/better-language-models\/language-models.pdf"},{"key":"e_1_3_2_1_46_1","volume-title":"Hogwild: A Lock-Free Approach to Parallelizing Stochastic Gradient Descent. In Advances in Neural Information Processing Systems 24. 693--701.","author":"Recht B.","year":"2011"},{"key":"e_1_3_2_1_47_1","volume-title":"SparCML: High-Performance Sparse Communication for Machine Learning. CoRR abs\/1802.08021","author":"Renggli C\u00e9dric","year":"2018"},{"key":"e_1_3_2_1_48_1","volume-title":"A Stochastic Approximation Method. The Annals of Mathematical Statistics","author":"Robbins Herbert","year":"1951"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.14778\/1920841.1920902"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2014-274"},{"key":"e_1_3_2_1_51_1","volume-title":"Horovod: fast and easy distributed deep learning in TensorFlow. arXiv preprint arXiv:1802.05799","author":"Sergeev Alexander","year":"2018"},{"key":"e_1_3_2_1_52_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv.1409.1556","author":"Simonyan Karen","year":"2014"},{"key":"e_1_3_2_1_53_1","volume-title":"Amir Roshan Zamir, and Mubarak Shah","author":"Soomro Khurram","year":"2012"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-354"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342005051521"},{"key":"e_1_3_2_1_57_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukliin. 2017. Attention is all you need. In Advances in Neural Information Processing Systems. 5998--6008.  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukliin. 2017. Attention is all you need. In Advances in Neural Information Processing Systems. 5998--6008."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.5555\/3020948.3021030"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3225058.3225069"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299101"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299101"},{"key":"e_1_3_2_1_62_1","volume-title":"Staleness-aware async-sgd for distributed deep learning. arXiv preprint arXiv:1511.05950","author":"Zhang Wei","year":"2015"}],"event":{"name":"PPoPP '20: 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","location":"San Diego California","acronym":"PPoPP '20","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3332466.3374528","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3332466.3374528","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:54:37Z","timestamp":1750204477000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3332466.3374528"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,2,19]]},"references-count":62,"alternative-id":["10.1145\/3332466.3374528","10.1145\/3332466"],"URL":"https:\/\/doi.org\/10.1145\/3332466.3374528","relation":{},"subject":[],"published":{"date-parts":[[2020,2,19]]},"assertion":[{"value":"2020-02-19","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}