{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T20:39:06Z","timestamp":1773261546739,"version":"3.50.1"},"reference-count":44,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2021,9,1]],"date-time":"2021-09-01T00:00:00Z","timestamp":1630454400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,9,1]],"date-time":"2021-09-01T00:00:00Z","timestamp":1630454400000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,9,1]],"date-time":"2021-09-01T00:00:00Z","timestamp":1630454400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,9,1]],"date-time":"2021-09-01T00:00:00Z","timestamp":1630454400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["CCF-1527767"],"award-info":[{"award-number":["CCF-1527767"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["CCF 1642658"],"award-info":[{"award-number":["CCF 1642658"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["1618512"],"award-info":[{"award-number":["1618512"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CCF-1748585"],"award-info":[{"award-number":["CCF-1748585"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CNS-1748692"],"award-info":[{"award-number":["CNS-1748692"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Sel. Areas Inf. Theory"],"published-print":{"date-parts":[[2021,9]]},"DOI":"10.1109\/jsait.2021.3105076","type":"journal-article","created":{"date-parts":[[2021,8,16]],"date-time":"2021-08-16T20:27:36Z","timestamp":1629145656000},"page":"942-953","source":"Crossref","is-referenced-by-count":18,"title":["Communication-Efficient and Byzantine-Robust Distributed Learning With Error Feedback"],"prefix":"10.1109","volume":"2","author":[{"given":"Avishek","family":"Ghosh","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4759-696X","authenticated-orcid":false,"given":"Raj Kumar","family":"Maity","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2006-1030","authenticated-orcid":false,"given":"Swanand","family":"Kadhe","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4605-7996","authenticated-orcid":false,"given":"Arya","family":"Mazumdar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kannan","family":"Ramchandran","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ISIT44484.2020.9174075"},{"key":"ref38","author":"karimireddy","year":"2019","journal-title":"SCAFFOLD Stochastic controlled averaging for federated learning"},{"key":"ref33","year":"2019","journal-title":"AWS News Blog"},{"key":"ref32","first-page":"19","article-title":"AGGREGATHOR: Byzantine machine learning via robust gradient aggregation","author":"damaskinos","year":"2019","journal-title":"Proc Conf Syst Mach Learn (SysML)"},{"key":"ref31","author":"mhamdi","year":"2018","journal-title":"The hidden vulnerability of distributed learning in byzantium"},{"key":"ref30","author":"horv\u00e1th","year":"2019","journal-title":"Stochastic distributed learning with gradient quantization and variance reduction"},{"key":"ref37","author":"haykin","year":"1994","journal-title":"An Introduction to Analog and Digital Communication"},{"key":"ref36","article-title":"Lecture notes for ece598yw: Information-theoretic methods for high-dimensional statistics","author":"wu","year":"2016"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/2785956.2787492"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ISIT.2017.8006962"},{"key":"ref10","first-page":"1509","article-title":"TernGrad: Ternary gradients to reduce communication in distributed deep learning","author":"wen","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1561\/9781601988614"},{"key":"ref11","first-page":"1709","article-title":"QSGD: Communication-efficient SGD via gradient quantization and encoding","author":"alistarh","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref12","author":"gandikota","year":"2019","journal-title":"vqSGD Vector quantized stochastic gradient descent"},{"key":"ref13","article-title":"Communication-efficient stochastic gradient descent, with applications to neural networks","author":"alistarh","year":"2017","journal-title":"Advances in Neural IInformation Processing Systems"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/2933057.2933105"},{"key":"ref15","author":"feng","year":"2014","journal-title":"Robust Distributed Learning"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3154503"},{"key":"ref17","first-page":"7074","article-title":"Defending against saddle point attack in Byzantine-robust distributed learning","volume":"97","author":"yin","year":"2019","journal-title":"Proc 36th Int Conf Mach Learn"},{"key":"ref18","author":"blanchard","year":"2017","journal-title":"Byzantine-tolerant machine learning"},{"key":"ref19","author":"ghosh","year":"2019","journal-title":"Robust federated learning in a heterogeneous environment"},{"key":"ref28","first-page":"1058","article-title":"1-bit stochastic gradient descent and its application to data-parallel distributed training of speech DNNs","author":"seide","year":"2014","journal-title":"Proc 15th Annu Conf Int Speech Commun Assoc"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/357172.357176"},{"key":"ref27","first-page":"1","article-title":"Scalable distributed DNN training using commodity GPU cloud computing","author":"strom","year":"2015","journal-title":"Proc 16th Annu Conf Int Speech Commun Assoc"},{"key":"ref3","author":"kone?n?","year":"2016","journal-title":"Federated learning Strategies for improving communication efficiency"},{"key":"ref6","first-page":"4447","article-title":"Sparsified SGD with memory","author":"stich","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref29","first-page":"2328","article-title":"Information-theoretic lower bounds for distributed statistical estimation with communication constraints","author":"zhang","year":"2013","journal-title":"Advances in neural information processing systems"},{"key":"ref5","first-page":"5973","article-title":"The convergence of sparsified gradient methods","author":"alistarh","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref8","first-page":"3329","article-title":"Distributed mean estimation with limited communication","author":"suresh","year":"2017","journal-title":"Proc 34th Int Conf Mach Learn Vol 70"},{"key":"ref7","author":"ivkin","year":"2019","journal-title":"Communication-efficient distributed SGD with sketching"},{"key":"ref2","first-page":"3252","article-title":"Error feedback fixes SignSGD and other gradient compression schemes","author":"karimireddy","year":"2019","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref9","first-page":"9850","article-title":"ATOMO: Communication-efficient learning via atomic sparsification","author":"wang","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref1","first-page":"5650","article-title":"Byzantine-robust distributed learning: Towards optimal statistical rates","author":"yin","year":"2018","journal-title":"Proc 35th Int Conf Mach Learn"},{"key":"ref20","author":"bernstein","year":"2018","journal-title":"signsgd with majority vote is communication efficient and fault tolerant"},{"key":"ref22","author":"zhu","year":"2021","journal-title":"BROADCAST Reducing both stochastic and compression noise to robustify communication-efficient federated learning"},{"key":"ref21","author":"bernstein","year":"2018","journal-title":"signsgd Compressed optimisation for non-convex problems"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref24","first-page":"1233","article-title":"No spurious local minima in nonconvex low rank problems: A unified geometric analysis","volume":"70","author":"ge","year":"2017","journal-title":"Proc 34th Int Conf Mach Learn"},{"key":"ref41","author":"hardt","year":"2021","journal-title":"Patterns predictions and actions A story about machine learning"},{"key":"ref23","author":"soudry","year":"2016","journal-title":"No bad local minima Data independent training error guarantees for multilayer neural networks"},{"key":"ref44","doi-asserted-by":"crossref","DOI":"10.1017\/9781108627771","volume":"48","author":"wainwright","year":"2019","journal-title":"High-dimensional statistics A non-asymptotic viewpoint"},{"key":"ref26","first-page":"13773","article-title":"Understanding Gradient Clipping in Private SGD: A Geometric Perspective","volume":"33","author":"chen","year":"0","journal-title":"Advances in neural information processing systems"},{"key":"ref43","author":"vershynin","year":"2010","journal-title":"Introduction to the Non-Asymptotic Analysis of Random Matrices"},{"key":"ref25","first-page":"1504","article-title":"Understanding gradient clipping in incremental gradient methods","author":"qian","year":"2021","journal-title":"Proc Int Conf Artif Intell Stat"}],"container-title":["IEEE Journal on Selected Areas in Information Theory"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/8700143\/9540917\/9514574-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8700143\/9540917\/09514574.pdf?arnumber=9514574","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,4,8]],"date-time":"2022-04-08T18:55:11Z","timestamp":1649444111000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9514574\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,9]]},"references-count":44,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/jsait.2021.3105076","relation":{},"ISSN":["2641-8770"],"issn-type":[{"value":"2641-8770","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,9]]}}}