{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:18:06Z","timestamp":1784135886465,"version":"3.55.0"},"reference-count":51,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2020,11,1]],"date-time":"2020-11-01T00:00:00Z","timestamp":1604188800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,11,1]],"date-time":"2020-11-01T00:00:00Z","timestamp":1604188800000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,11,1]],"date-time":"2020-11-01T00:00:00Z","timestamp":1604188800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,11,1]],"date-time":"2020-11-01T00:00:00Z","timestamp":1604188800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["CCF-1704624"],"award-info":[{"award-number":["CCF-1704624"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006785","name":"Google Faculty Research Award","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006785","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Sel. Areas Inf. Theory"],"published-print":{"date-parts":[[2020,11]]},"DOI":"10.1109\/jsait.2020.3042094","type":"journal-article","created":{"date-parts":[[2020,12,2]],"date-time":"2020-12-02T22:02:53Z","timestamp":1606946573000},"page":"897-907","source":"Crossref","is-referenced-by-count":42,"title":["rTop-<i>k<\/i>: A Statistical Estimation Approach to Distributed SGD"],"prefix":"10.1109","volume":"1","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9472-2770","authenticated-orcid":false,"given":"Leighton Pate","family":"Barnes","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7465-7214","authenticated-orcid":false,"given":"Huseyin A.","family":"Inan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4926-5443","authenticated-orcid":false,"given":"Berivan","family":"Isik","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ayfer","family":"Ozgur","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1137\/070704277"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-010-0420-4"},{"key":"ref33","first-page":"6394","article-title":"Communication-efficient distributed learning of discrete probability distributions","author":"diakonikolas","year":"2017","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref32","first-page":"1","article-title":"Geometric lower bounds for distributed parameter estimation under communication constraints","volume":"75","author":"han","year":"2018","journal-title":"Mach Learn Res"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2016.2646342"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/2897518.2897582"},{"key":"ref37","author":"stich","year":"2019","journal-title":"The error-feedback framework Better rates for SGD with delayed gradients and compressed communication"},{"key":"ref36","author":"barnes","year":"2019","journal-title":"Lower bounds for learning distributions under communication constraints via Fisher information"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ALLERTON.2018.8635899"},{"key":"ref34","first-page":"1","article-title":"Inference under information constraints: Lower bounds from chi-square contraction","volume":"99","author":"acharya","year":"2019","journal-title":"Mach Learn Res"},{"key":"ref28","first-page":"2328","article-title":"Information-theoretic lower bounds for distributed statistical estimation with communication constraints","author":"zhang","year":"2013","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref27","first-page":"4447","article-title":"Sparsified SGD with memory","author":"stich","year":"2018","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref29","first-page":"2726","article-title":"On communication cost of distributed statistical estimation and dimensionality","author":"garg","year":"2014","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref2","author":"kairouz","year":"2019","journal-title":"Advances and Open Problems in Federated Learning"},{"key":"ref1","first-page":"1","article-title":"Deep gradient compression: Reducing the communication bandwidth for distributed training","author":"lin","year":"2018","journal-title":"Proc 6th Int Conf Learn Represent (ICLR)"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICC47138.2019.9123209"},{"key":"ref22","first-page":"2634","article-title":"Efficient distributed hessian free algorithm for large-scale empirical risk minimization via accumulating sample strategy","author":"jahani","year":"2020","journal-title":"Proc Int Conf Artif Intell Statist"},{"key":"ref21","first-page":"ii-1000","article-title":"Communication-efficient distributed optimization using an approximate Newton-type method","volume":"32","author":"shamir","year":"2014","journal-title":"Proc 31st Int Conf Int Conf Mach Learn"},{"key":"ref24","first-page":"9498","article-title":"DINGO: Distributed Newton-type method for gradient-norm optimization","author":"crane","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref23","first-page":"1","article-title":"Distributed second-order optimization using kronecker-factored approximations","author":"ba","year":"2016","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref26","first-page":"14668","article-title":"Qsparse-local-SGD: Distributed SGD with quantization, sparsification, and local computations","author":"basu","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref25","author":"jahani","year":"2019","journal-title":"Scaling up Quasi-Newton algorithms Communication efficient distributed SR1"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ISIT.2019.8849821"},{"key":"ref51","author":"vershynin","year":"2010","journal-title":"Introduction to the Non-Asymptotic Analysis of Random Matrices"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1045"},{"key":"ref11","first-page":"1509","article-title":"Terngrad: Ternary gradients to reduce communication in distributed deep learning","author":"wen","year":"2017","journal-title":"Proc NIPS"},{"key":"ref40","first-page":"693","article-title":"Hogwild: A lock-free approach to parallelizing stochastic gradient","volume":"24","author":"niu","year":"2011","journal-title":"Proc NIPS"},{"key":"ref12","first-page":"559","article-title":"Signsgd: Compressed optimisation for non-convex problems","author":"bernstein","year":"2018","journal-title":"Proc ICML"},{"key":"ref13","first-page":"9872","article-title":"ATOMO: Communication-efficient learning via atomic sparsification","author":"wang","year":"2018","journal-title":"Proc NeurIPS"},{"key":"ref14","first-page":"1299","article-title":"Gradient sparsification for communication-efficient distributed optimization","author":"wangni","year":"2018","journal-title":"Advances in Neural IInformation Processing Systems"},{"key":"ref15","author":"gandikota","year":"2019","journal-title":"vqSGD Vector quantized stochastic gradient descent"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"ref17","first-page":"693","article-title":"Hogwild!: A lock-free approach to parallelizing stochastic gradient descent","volume":"24","author":"recht","year":"2011","journal-title":"Advances in neural information processing systems"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.1986.1104412"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054491"},{"key":"ref4","first-page":"1058","article-title":"1-bit stochastic gradient descent and application to data-parallel distributed training of speech DNNs","author":"seide","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref3","first-page":"19","article-title":"Communication efficient distributed machine learning with the parameter server","volume":"1","author":"li","year":"2014","journal-title":"Proc 27th Int Conf Neural Inf Process Syst"},{"key":"ref6","first-page":"1709","article-title":"QSGD: Communication-efficient SGD via gradient quantization and encoding","author":"alistarh","year":"2017","journal-title":"Proc Adv Neural Inf Process Syst 30"},{"key":"ref5","first-page":"1488","article-title":"Scalable distributed dnn training using commodity gpu cloud computing","author":"strom","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref8","author":"konecn\u00fd","year":"2016","journal-title":"Federated optimization distributed machine learning for on-device intelligence"},{"key":"ref7","first-page":"1273","article-title":"Communication-efficient learning of deep networks from decentralized data","author":"mcmahan","year":"2017","journal-title":"Proc Int Conf Artif Intell Statist (AISTATS)"},{"key":"ref49","author":"zaremba","year":"2014","journal-title":"Recurrent Neural Network Regularization"},{"key":"ref9","first-page":"1","article-title":"Federated learning: Strategies for improving communication efficiency","author":"konecn\u00fd","year":"2016","journal-title":"Proc NIPS Workshop Private Multi Party Mach Learn"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/E17-2025"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.21236\/ADA273556"},{"key":"ref48","doi-asserted-by":"crossref","first-page":"1045","DOI":"10.21437\/Interspeech.2010-343","article-title":"Recurrent neural network based language model","author":"mikolov","year":"2010","journal-title":"Proc INTERSPEECH"},{"key":"ref47","author":"inan","year":"2016","journal-title":"Tying word vectors and word classifiers A loss framework for language modeling"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref41","first-page":"5977","article-title":"The convergence of sparsified gradient methods","author":"alistarh","year":"2018","journal-title":"Proc NeurIPS"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref43","author":"krizhevsky","year":"2009","journal-title":"Learning multiple layers of features from tiny images"}],"container-title":["IEEE Journal on Selected Areas in Information Theory"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/8700143\/9319601\/9276447-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8700143\/9319601\/09276447.pdf?arnumber=9276447","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,2]],"date-time":"2022-12-02T01:55:13Z","timestamp":1669946113000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9276447\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,11]]},"references-count":51,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/jsait.2020.3042094","relation":{},"ISSN":["2641-8770"],"issn-type":[{"value":"2641-8770","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,11]]}}}