{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,15]],"date-time":"2026-01-15T11:55:01Z","timestamp":1768478101673,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,8,29]],"date-time":"2022-08-29T00:00:00Z","timestamp":1661731200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of Heilongjiang Province","award":["YQ2019F007"],"award-info":[{"award-number":["YQ2019F007"]}]},{"name":"Programs for Science and Technology Development of Heilongjiang Province","award":["2021ZXJ05A03"],"award-info":[{"award-number":["2021ZXJ05A03"]}]},{"name":"Key Science Technology Specific Projects of Heilongjiang Province","award":["2019ZX14A01"],"award-info":[{"award-number":["2019ZX14A01"]}]},{"name":"National Key R&D Program of China","award":["2021ZD0110905"],"award-info":[{"award-number":["2021ZD0110905"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62106061, 61972114, 62072137"],"award-info":[{"award-number":["62106061, 61972114, 62072137"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Fundamental Research Funds for the Central Universities","award":["AUGA5710010521"],"award-info":[{"award-number":["AUGA5710010521"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,8,29]]},"DOI":"10.1145\/3545008.3545013","type":"proceedings-article","created":{"date-parts":[[2023,1,15]],"date-time":"2023-01-15T01:04:08Z","timestamp":1673744648000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":10,"title":["Aperiodic Local SGD: Beyond Local SGD"],"prefix":"10.1145","author":[{"given":"Hao","family":"Zhang","sequence":"first","affiliation":[{"name":"Harbin Institute of Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tingting","family":"Wu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siyao","family":"Cheng","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Liu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology (Shenzhen), China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,1,13]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"QSGD: Communication-efficient SGD via gradient quantization and encoding. In Advances in Neural Information Processing Systems. 1709\u20131720.","author":"Alistarh Dan","year":"2017","unstructured":"Dan Alistarh, Demjan Grubic, Jerry Li, Ryota Tomioka, and Milan Vojnovic. 2017. QSGD: Communication-efficient SGD via gradient quantization and encoding. In Advances in Neural Information Processing Systems. 1709\u20131720."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3320060"},{"key":"e_1_3_2_1_3_1","unstructured":"Luke\u00a0N Darlow Elliot\u00a0J Crowley Antreas Antoniou and Amos\u00a0J Storkey. 2018. Cinic-10 is not imagenet or cifar-10. arXiv preprint arXiv:1810.03505(2018)."},{"key":"e_1_3_2_1_4_1","volume-title":"Ross Girshick, Pieter Noordhuis, Lukasz Wesolowski, Aapo Kyrola, Andrew Tulloch, Yangqing Jia, and Kaiming He.","author":"Goyal Priya","year":"2017","unstructured":"Priya Goyal, Piotr Dollar, Ross Girshick, Pieter Noordhuis, Lukasz Wesolowski, Aapo Kyrola, Andrew Tulloch, Yangqing Jia, and Kaiming He. 2017. Accurate, Large Minibatch SGD: Training ImageNet in 1 Hour. In arXiv preprint arXiv:1706.02677."},{"key":"e_1_3_2_1_5_1","first-page":"11082","article-title":"Local SGD with Periodic Averaging: Tighter Analysis and Adaptive Synchronization","volume":"32","author":"Haddadpour Farzin","year":"2019","unstructured":"Farzin Haddadpour, Mohammad\u00a0Mahdi Kamani, Mehrdad Mahdavi, and Viveck Cadambe. 2019. Local SGD with Periodic Averaging: Tighter Analysis and Adaptive Synchronization. Advances in Neural Information Processing Systems 32 (2019), 11082\u201311094.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_6_1","unstructured":"Farzin Haddadpour and Mehrdad Mahdavi. 2019. On the convergence of local descent methods in federated learning. arXiv preprint arXiv:1910.14425(2019)."},{"key":"e_1_3_2_1_7_1","unstructured":"Torsten Hoefler Dan Alistarh Tal Ben-Nun Nikoli Dryden and Alexandra Peste. 2021. Sparsity in Deep Learning: Pruning and growth for efficient inference and training in neural networks. arXiv preprint arXiv:2102.00554(2021)."},{"key":"e_1_3_2_1_8_1","volume-title":"International Conference on Artificial Intelligence and Statistics. PMLR, 4519\u20134529","author":"Khaled Ahmed","year":"2020","unstructured":"Ahmed Khaled, Konstantin Mishchenko, and Peter Richt\u00e1rik. 2020. Tighter theory for local SGD on identical and heterogeneous data. In International Conference on Artificial Intelligence and Statistics. PMLR, 4519\u20134529."},{"key":"e_1_3_2_1_9_1","unstructured":"Alex Krizhevsky Geoffrey Hinton 2009. Learning multiple layers of features from tiny images. (2009)."},{"key":"e_1_3_2_1_10_1","first-page":"19","article-title":"Communication efficient distributed machine learning with the parameter server","volume":"27","author":"Li Mu","year":"2014","unstructured":"Mu Li, David\u00a0G Andersen, Alexander\u00a0J Smola, and Kai Yu. 2014. Communication efficient distributed machine learning with the parameter server. Advances in Neural Information Processing Systems 27 (2014), 19\u201327.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_11_1","unstructured":"Qinbin Li Yiqun Diao Quan Chen and Bingsheng He. 2021. Federated Learning on Non-IID Data Silos: An Experimental Study. arXiv preprint arXiv:2102.02079(2021)."},{"key":"e_1_3_2_1_12_1","volume-title":"On the Convergence of FedAvg on Non-IID Data. In International Conference on Learning Representations.","author":"Li Xiang","year":"2019","unstructured":"Xiang Li, Kaixuan Huang, Wenhao Yang, Shusen Wang, and Zhihua Zhang. 2019. On the Convergence of FedAvg on Non-IID Data. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_13_1","unstructured":"Xiang Li Wenhao Yang Shusen Wang and Zhihua Zhang. 2019. Communication-efficient local decentralized SGD methods. arXiv preprint arXiv:1910.09126(2019)."},{"key":"e_1_3_2_1_14_1","volume-title":"Use Local SGD. In International Conference on Learning Representations.","author":"Lin Tao","year":"2019","unstructured":"Tao Lin, Sebastian\u00a0U Stich, Kumar\u00a0Kshitij Patel, and Martin Jaggi. 2019. Don\u2019t Use Large Mini-batches, Use Local SGD. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_15_1","volume-title":"Use Local SGD. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B1eyO1BFPr","author":"Lin Tao","year":"2020","unstructured":"Tao Lin, Sebastian\u00a0U. Stich, Kumar\u00a0Kshitij Patel, and Martin Jaggi. 2020. Don\u2019t Use Large Mini-batches, Use Local SGD. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B1eyO1BFPr"},{"key":"e_1_3_2_1_16_1","volume-title":"Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training. In International Conference on Learning Representations.","author":"Lin Yujun","year":"2018","unstructured":"Yujun Lin, Song Han, Huizi Mao, Yu Wang, and Bill Dally. 2018. Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_17_1","unstructured":"Brendan McMahan Eider Moore Daniel Ramage Seth Hampson and Blaise\u00a0Aguera y Arcas. 2017. Communication-Efficient Learning of Deep Networks from Decentralized Data. In Artificial Intelligence and Statistics. 1273\u20131282."},{"key":"e_1_3_2_1_18_1","unstructured":"Shuo Ouyang Dezun Dong Yemao Xu and Liquan Xiao. 2020. Communication Optimization Strategies for Distributed Deep Learning: A Survey. arXiv preprint arXiv:2003.03009(2020)."},{"key":"e_1_3_2_1_19_1","unstructured":"Razvan Pascanu Tomas Mikolov and Yoshua Bengio. 2012. Understanding the exploding gradient problem. ArXiv abs\/1211.5063(2012)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/637"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737367"},{"key":"e_1_3_2_1_22_1","volume-title":"Local SGD Converges Fast and Communicates Little. In International Conference on Learning Representations.","author":"Stich U","year":"2018","unstructured":"Sebastian\u00a0U Stich. 2018. Local SGD Converges Fast and Communicates Little. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_23_1","unstructured":"Sebastian\u00a0U Stich Jean-Baptiste Cordonnier and Martin Jaggi. 2018. Sparsified SGD with memory. In Advances in Neural Information Processing Systems. 4447\u20134458."},{"key":"e_1_3_2_1_24_1","first-page":"1","article-title":"The error-feedback framework: Better rates for sgd with delayed gradients and compressed updates","volume":"21","author":"Stich U","year":"2020","unstructured":"Sebastian\u00a0U Stich and Sai\u00a0Praneeth Karimireddy. 2020. The error-feedback framework: Better rates for sgd with delayed gradients and compressed updates. Journal of Machine Learning Research 21 (2020), 1\u201336.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Peng Sun Wansen Feng Ruobing Han Shengen Yan and Yonggang Wen. 2019. Optimizing network performance for distributed dnn training on gpu clusters: Imagenet\/alexnet training in 1.5 minutes. In arXiv preprint arXiv:1902.06855.","DOI":"10.1109\/TBDATA.2019.2957478"},{"key":"e_1_3_2_1_26_1","unstructured":"Zhenheng Tang Shaohuai Shi Xiaowen Chu Wei Wang and Bo Li. 2020. Communication-Efficient Distributed Deep Learning: A Comprehensive Survey. arXiv preprint arXiv:2003.06307(2020)."},{"key":"e_1_3_2_1_27_1","volume-title":"Deepesh Data","author":"Wang Jianyu","year":"2021","unstructured":"Jianyu Wang, Zachary Charles, Zheng Xu, Gauri Joshi, H\u00a0Brendan McMahan, Maruan Al-Shedivat, Galen Andrew, Salman Avestimehr, Katharine Daly, Deepesh Data, 2021. A Field Guide to Federated Optimization. arXiv preprint arXiv:2107.06917(2021)."},{"key":"e_1_3_2_1_28_1","first-page":"212","article-title":"Adaptive communication strategies to achieve the best error-runtime trade-off in local-update SGD","volume":"1","author":"Wang Jianyu","year":"2019","unstructured":"Jianyu Wang and Gauri Joshi. 2019. Adaptive communication strategies to achieve the best error-runtime trade-off in local-update SGD. Proceedings of Machine Learning and Systems 1 (2019), 212\u2013229.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_29_1","volume-title":"Cooperative SGD: A Unified Framework for the Design and Analysis of Communication-Efficient SGD Algorithms. In ICML Workshop on Coding Theory for Machine Learning.","author":"Wang Jianyu","year":"2019","unstructured":"Jianyu Wang and Gauri Joshi. 2019. Cooperative SGD: A Unified Framework for the Design and Analysis of Communication-Efficient SGD Algorithms. In ICML Workshop on Coding Theory for Machine Learning."},{"key":"e_1_3_2_1_30_1","volume-title":"International Conference on Machine Learning. PMLR, 10334\u201310343","author":"Woodworth Blake","year":"2020","unstructured":"Blake Woodworth, Kumar\u00a0Kshitij Patel, Sebastian Stich, Zhen Dai, Brian Bullins, Brendan Mcmahan, Ohad Shamir, and Nathan Srebro. 2020. Is local SGD better than minibatch SGD?. In International Conference on Machine Learning. PMLR, 10334\u201310343."},{"key":"e_1_3_2_1_31_1","volume-title":"International Conference on Machine Learning. PMLR, 5325\u20135333","author":"Wu Jiaxiang","year":"2018","unstructured":"Jiaxiang Wu, Weidong Huang, Junzhou Huang, and Tong Zhang. 2018. Error compensated quantized SGD and its applications to large-scale distributed optimization. In International Conference on Machine Learning. PMLR, 5325\u20135333."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_1"},{"key":"e_1_3_2_1_33_1","unstructured":"Han Xiao Kashif Rasul and Roland Vollgraf. 2017. Fashion-mnist: a novel image dataset for benchmarking machine learning algorithms. arXiv preprint arXiv:1708.07747(2017)."},{"key":"e_1_3_2_1_34_1","unstructured":"Xiang Yang. 2020. Shuffle-Exchange Brings Faster: Reduce the Idle Time During Communication for Decentralized Neural Network Training. arXiv preprint arXiv:2007.00433(2020)."},{"key":"e_1_3_2_1_35_1","volume-title":"International Conference on Artificial Intelligence and Statistics. PMLR","author":"Yin Dong","year":"2018","unstructured":"Dong Yin, Ashwin Pananjady, Max Lam, Dimitris Papailiopoulos, Kannan Ramchandran, and Peter Bartlett. 2018. Gradient diversity: a key ingredient for scalable distributed learning. In International Conference on Artificial Intelligence and Statistics. PMLR, 1998\u20132007."},{"key":"e_1_3_2_1_36_1","volume-title":"International Conference on Learning Representations.","author":"You Yang","year":"2020","unstructured":"Yang You, Jing Li, Sashank Reddi, Jonathan Hseu, Sanjiv Kumar, Srinadh Bhojanapalli, Xiaodan Song, James Demmel, Kurt Keutzer, and Cho-Jui Hsieh. 2020. Large Batch Optimization for Deep Learning: Training BERT in 76 minutes. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_37_1","volume-title":"International Conference on Machine Learning. PMLR, 7202\u20137212","author":"Yu Chen","year":"2019","unstructured":"Chen Yu, Hanlin Tang, Cedric Renggli, Simon Kassing, Ankit Singla, Dan Alistarh, Ce Zhang, and Ji Liu. 2019. Distributed learning over unreliable networks. In International Conference on Machine Learning. PMLR, 7202\u20137212."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33015693"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33015693"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.5555\/2567709.2567769"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/447"}],"event":{"name":"ICPP '22: 51st International Conference on Parallel Processing","location":"Bordeaux France","acronym":"ICPP '22"},"container-title":["Proceedings of the 51st International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3545008.3545013","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3545008.3545013","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:43Z","timestamp":1750186963000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3545008.3545013"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,29]]},"references-count":41,"alternative-id":["10.1145\/3545008.3545013","10.1145\/3545008"],"URL":"https:\/\/doi.org\/10.1145\/3545008.3545013","relation":{},"subject":[],"published":{"date-parts":[[2022,8,29]]},"assertion":[{"value":"2023-01-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}