{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,24]],"date-time":"2025-08-24T01:53:39Z","timestamp":1756000419019},"reference-count":75,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"European Research Council ERC under the European Unions Horizon 2020 programme","award":["EPiGRAM-HS, No. 801039","ERC Starting Grant ScaleML, No. 805223","grant agreement DAPP, No. 678880"],"award-info":[{"award-number":["EPiGRAM-HS, No. 801039","ERC Starting Grant ScaleML, No. 805223","grant agreement DAPP, No. 678880"]}]},{"DOI":"10.13039\/501100001711","name":"Swiss National Science Foundation","doi-asserted-by":"crossref","award":["Ambizione Project No. 185778"],"award-info":[{"award-number":["Ambizione Project No. 185778"]}],"id":[{"id":"10.13039\/501100001711","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Parallel Distrib. Syst."],"published-print":{"date-parts":[[2020]]},"DOI":"10.1109\/tpds.2020.3040606","type":"journal-article","created":{"date-parts":[[2020,11,25]],"date-time":"2020-11-25T20:58:14Z","timestamp":1606337894000},"page":"1-1","source":"Crossref","is-referenced-by-count":3,"title":["Breaking (Global) Barriers in Parallel Stochastic Optimization with Wait-Avoiding Group Averaging"],"prefix":"10.1109","author":[{"given":"Shigang","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tal","family":"Ben-Nun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giorgi","family":"Nadiradze","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Salvatore","family":"Digirolamo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nikoli","family":"Dryden","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dan","family":"Alistarh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Torsten","family":"Hoefler","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00945"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref71","first-page":"8026","article-title":"Pytorch: An imperative style, high-performance deep learning library","author":"paszke","year":"2019","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2020.2974843"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2017.00081"},{"key":"ref39","first-page":"1273","article-title":"Communication-efficient learning of deep networks from decentralized data","author":"mcmahan","year":"2017","journal-title":"Proc 20th Int Conf Artif Intell Statist"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2017.2768413"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126916"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2016.0028"},{"key":"ref32","doi-asserted-by":"crossref","DOI":"10.1109\/ICDCS47774.2020.00153","article-title":"Communication-efficient decentralized learning with sparsification and adaptive peer selection","author":"tang","year":"2020"},{"key":"ref31","article-title":"Asynchronous distributed learning with sparse communications and identification","author":"grishchenko","year":"2018"},{"key":"ref30","article-title":"Priority-based parameter propagation for distributed DNN training","author":"jayarajan","year":"2019","journal-title":"Proc 2nd SysML Conf"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/JSAIT.2020.2985917"},{"key":"ref36","first-page":"803","article-title":"Slow and stale gradients can win the race: Error-runtime trade-offs in distributed sgd","author":"dutta","year":"2018","journal-title":"Proc Int Conf Artif Intell Statist"},{"key":"ref35","first-page":"629","article-title":"Gaia: Geo-distributed machine learning approaching lan speeds","author":"hsieh","year":"2017","journal-title":"Proc 14th USENIX Conf Netw Syst Des Implementation"},{"key":"ref34","first-page":"2350","article-title":"Staleness-aware async-SGD for distributed deep learning","author":"zhang","year":"2016","journal-title":"Proc 25th Int Joint Conf Artif Intell"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/HOTI.2015.21"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1137\/120880811"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126970"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref28","first-page":"1223","article-title":"Large scale distributed deep networks","author":"dean","year":"2012","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1137\/16M1080173"},{"key":"ref65","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00943"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/2640087.2644155"},{"key":"ref67","first-page":"1","article-title":"Measuring the effects of data parallelism on neural network training","volume":"20","author":"shallue","year":"0"},{"key":"ref68","first-page":"265","article-title":"TensorFlow: Large-scale machine learning on heterogeneous systems","author":"abadi","year":"2016","journal-title":"Proc 12th USENIX Conf Operating Syst Des Implementation"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.14778\/1920841.1920902"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1923-7"},{"key":"ref1","article-title":"End to end learning for self-driving cars","author":"bojarski","year":"2016"},{"key":"ref20","first-page":"3043","article-title":"Asynchronous decentralized parallel stochastic gradient descent","author":"lian","year":"2018","journal-title":"Proc 35th Int Conf Mach Learn"},{"key":"ref22","first-page":"1223","article-title":"More effective distributed ML via a stale synchronous parallel parameter server","author":"ho","year":"2013","journal-title":"Proc 26th Int Conf Neural Inf Process Syst"},{"key":"ref21","article-title":"GossipGraD: Scalable deep learning using gossip communication based asynchronous gradient descent","author":"daily","year":"2018"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472805"},{"key":"ref23","first-page":"685","article-title":"Deep learning with elastic averaging SGD","author":"zhang","year":"2015","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref26","article-title":"SwarmSGD: Scalable decentralized SGD with local updates","author":"nadiradze","year":"2019"},{"key":"ref25","article-title":"Local SGD converges fast and communicates little","author":"stich","year":"2019","journal-title":"Proc 7th Int Conf Learn Representations"},{"key":"ref50","first-page":"5904","article-title":"Collaborative deep learning in fixed topology networks","author":"jiang","year":"2017","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref51","article-title":"Don&#x2019;t use large mini-batches, use local SGD","author":"lin","year":"2018"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.09.002"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1145\/1362622.1362692"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356196"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1007\/s10586-014-0361-4"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-24685-5_1"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1145\/2493123.2462903"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid.2011.22"},{"key":"ref52","article-title":"Adaptive communication strategies to achieve the best error-runtime trade-off in local-update SGD","author":"wang","year":"2019","journal-title":"Proc 2nd SysML Conf"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356222"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014780"},{"key":"ref40","article-title":"Federated learning: Strategies for improving communication efficiency","author":"kone?n?","year":"2016","journal-title":"Proc NIPS Workshop Private Multi-Party Mach Learn"},{"key":"ref12","article-title":"MPI: A message-passing interface standard version 3.1","year":"2015"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3332466.3374528"},{"key":"ref14","article-title":"Dd-ppo: Learning near-perfect pointgoal navigators from 2.5 billion frames","author":"wijmans","year":"2020"},{"key":"ref15","article-title":"Communication-efficient distributed deep learning: A comprehensive survey","author":"tang","year":"2020"},{"key":"ref16","first-page":"5336","article-title":"Can decentralized algorithms outperform centralized algorithms? a case study for decentralized parallel stochastic gradient descent","author":"lian","year":"2017","journal-title":"Proc 31st Int Conf Neural Inf Process Syst"},{"key":"ref17","first-page":"344","article-title":"Stochastic gradient push for distributed deep learning","author":"assran","year":"2019","journal-title":"Proc 36th Int Conf Mach Learn"},{"key":"ref18","first-page":"693","article-title":"Hogwild: A lock-free approach to parallelizing stochastic gradient descent","author":"recht","year":"2011","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref19","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","author":"mnih","year":"2016","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/3320060"},{"key":"ref3","article-title":"AI and compute","author":"amodei","year":"2018"},{"key":"ref6","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"radford","year":"0"},{"key":"ref5","article-title":"Scaling laws for neural language models","author":"kaplan","year":"2020"},{"key":"ref8","article-title":"An empirical model of large-batch training","author":"mccandlish","year":"2018"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref49","article-title":"How to scale distributed deep learning?","author":"jin","year":"2016","journal-title":"Proc Workshop Mach Learn Syst"},{"key":"ref9","article-title":"Darts: Differentiable architecture search","author":"liu","year":"2018"},{"key":"ref46","first-page":"2595","article-title":"Parallelized stochastic gradient descent","author":"zinkevich","year":"2010","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2020.06.011"},{"key":"ref48","article-title":"Decentralized stochastic optimization and gossip algorithms with compressed communication","author":"koloskova","year":"2019"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00054"},{"key":"ref42","article-title":"Horovod: Fast and easy distributed deep learning in TensorFlow","author":"sergeev","year":"2018"},{"key":"ref41","article-title":"Bringing HPC techniques to deep learning","author":"gibiansky","year":"2017"},{"key":"ref44","first-page":"1231","article-title":"Efficient large-scale distributed training of conditional maximum entropy models","author":"mcdonald","year":"2009","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2019.00018"}],"container-title":["IEEE Transactions on Parallel and Distributed Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/71\/4359390\/09271898.pdf?arnumber=9271898","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,29]],"date-time":"2022-11-29T21:22:05Z","timestamp":1669756925000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9271898\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"references-count":75,"URL":"https:\/\/doi.org\/10.1109\/tpds.2020.3040606","relation":{},"ISSN":["1045-9219","1558-2183","2161-9883"],"issn-type":[{"value":"1045-9219","type":"print"},{"value":"1558-2183","type":"electronic"},{"value":"2161-9883","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]}}}