{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T08:33:24Z","timestamp":1777106004785,"version":"3.51.4"},"reference-count":39,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,5,10]],"date-time":"2021-05-10T00:00:00Z","timestamp":1620604800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,5,10]],"date-time":"2021-05-10T00:00:00Z","timestamp":1620604800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,5,10]],"date-time":"2021-05-10T00:00:00Z","timestamp":1620604800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,5,10]]},"DOI":"10.1109\/infocom42981.2021.9488815","type":"proceedings-article","created":{"date-parts":[[2021,7,27]],"date-time":"2021-07-27T00:07:32Z","timestamp":1627344452000},"page":"1-10","source":"Crossref","is-referenced-by-count":25,"title":["Live Gradient Compensation for Evading Stragglers in Distributed Learning"],"prefix":"10.1109","author":[{"given":"Jian","family":"Xu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shao-Lun","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Linqi","family":"Song","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tian","family":"Lan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","article-title":"Very deep convolutional networks for large-scale image recognition","author":"simonyan","year":"2015","journal-title":"International Conference on Learning Representations (ICLR)"},{"key":"ref38","article-title":"Learning multiple layers of features from tiny images","author":"krizhevsky","year":"2009"},{"key":"ref33","article-title":"Accurate, large minibatch SGD: training imagenet in 1 hour","author":"goyal","year":"2017"},{"key":"ref32","article-title":"On the convergence of fedavg on non-iid data","author":"li","year":"2020","journal-title":"International Conference on Learning Representations (ICLR)"},{"key":"ref31","article-title":"Federated learning with non-iid data","author":"zhao","year":"2018"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2016.2525015"},{"key":"ref37","article-title":"Serverless straggler mitigation using local error-correcting codes","author":"gupta","year":"2020"},{"key":"ref36","article-title":"Adaptive communication strategies to achieve the best error-runtime trade-off in local-update SGD","author":"wang","year":"2019","journal-title":"Proceedings of Machine Learning and Systems (MLSys)"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1137\/16M1080173"},{"key":"ref34","first-page":"7184","article-title":"On the linear speedup analysis of communication efficient momentum SGD for distributed non-convex optimization","volume":"97","author":"yu","year":"2019","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ref10","first-page":"5434","article-title":"Straggler mitigation in distributed optimization through data encoding","author":"karakus","year":"2017","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2018.00137"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2017.2736066"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ISIT.2018.8437669"},{"key":"ref14","article-title":"Polynomially coded regression: Optimal straggler mitigation via data encoding","author":"li","year":"2018"},{"key":"ref15","article-title":"Anytime minibatch: Exploiting stragglers in online distributed optimization","author":"ferdinand","year":"2019","journal-title":"International Conference on Learning Representations (ICLR)"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ISIT.2019.8849684"},{"key":"ref17","article-title":"Revisiting distributed synchronous sgd","author":"chen","year":"2016"},{"key":"ref18","article-title":"Adaptive distributed stochastic gradient descent for minimizing delay in the presence of stragglers","author":"kas hanna","year":"2020","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref19","first-page":"5606","article-title":"Communication-computation efficient gradient coding","author":"ye","year":"2018","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638950"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref27","first-page":"693","article-title":"Hogwild! a lock-free approach to parallelizing stochastic gradient descent","author":"niu","year":"2011","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref3","article-title":"Imagenet classification with deep convolutional neural networks","author":"krizhevsky","year":"2012","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/2408776.2408794"},{"key":"ref29","first-page":"2737","article-title":"Asynchronous parallel stochastic gradient for nonconvex optimization","author":"lian","year":"2015","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref5","first-page":"19","article-title":"Communication efficient distributed machine learning with the parameter server","volume":"1","author":"li","year":"2014","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref8","article-title":"Slow and stale gradients can win the race: Error-runtime trade-offs in distributed sgd","author":"dutta","year":"2018","journal-title":"AISTATS"},{"key":"ref7","article-title":"Gradient coding: Avoiding stragglers in distributed learning","author":"tandon","year":"2017","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ref2","first-page":"1223","article-title":"Large scale distributed deep networks","author":"dean","year":"2012","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref9","first-page":"1","article-title":"Redundancy techniques for straggler mitigation in distributed optimization and learning","volume":"20","author":"karakus","year":"2019","journal-title":"Journal of Machine Learning Research"},{"key":"ref20","article-title":"Erasurehead: Distributed gradient descent without delays using approximate gradient coding","author":"wang","year":"2019"},{"key":"ref22","article-title":"Sparsified SGD with memory","author":"stich","year":"2018","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/JSAIT.2020.2991361"},{"key":"ref24","article-title":"Communication-efficient distributed blockwise momentum sgd with error-feedback","author":"zheng","year":"2019","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref23","first-page":"1","article-title":"Robust and communication-efficient federated learning from non-iid data","author":"sattler","year":"2019","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"ref26","first-page":"1","article-title":"The error-feedback framework: Sgd with delayed gradients","volume":"21","author":"stich","year":"2020","journal-title":"Journal of Machine Learning Research"},{"key":"ref25","first-page":"6155","article-title":"DoubleSqueeze: Parallel stochastic gradient descent with double-pass error-compensated compression","author":"tang","year":"2019","journal-title":"International Conference on Machine Learning (ICML)"}],"event":{"name":"IEEE INFOCOM 2021 - IEEE Conference on Computer Communications","location":"Vancouver, BC, Canada","start":{"date-parts":[[2021,5,10]]},"end":{"date-parts":[[2021,5,13]]}},"container-title":["IEEE INFOCOM 2021 - IEEE Conference on Computer Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9488422\/9488423\/09488815.pdf?arnumber=9488815","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T15:43:36Z","timestamp":1652197416000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9488815\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,5,10]]},"references-count":39,"URL":"https:\/\/doi.org\/10.1109\/infocom42981.2021.9488815","relation":{},"subject":[],"published":{"date-parts":[[2021,5,10]]}}}