{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T20:17:34Z","timestamp":1784233054377,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":122,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,4,17]],"date-time":"2024-04-17T00:00:00Z","timestamp":1713312000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Key R&D Program of China","award":["No.2021ZD0113001"],"award-info":[{"award-number":["No.2021ZD0113001"]}]},{"name":"National Nature Science Foundation of China","award":["62325201"],"award-info":[{"award-number":["62325201"]}]},{"name":"National Nature Science Foundation of China","award":["62102045"],"award-info":[{"award-number":["62102045"]}]},{"name":"National Nature Science Foundation of China","award":["62172008"],"award-info":[{"award-number":["62172008"]}]},{"name":"National Natural Science Fund for the Excellent Young Scientists Fund Program (Overseas)"},{"name":"Alibaba Group through Alibaba Innovative Research (AIR) Program"},{"name":"Beijing Outstanding Young Scientist Program","award":["BJJWZYJH01201910001004"],"award-info":[{"award-number":["BJJWZYJH01201910001004"]}]},{"name":"Center for Data Space Technology and System, Peking University"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,4,27]]},"DOI":"10.1145\/3617232.3624847","type":"proceedings-article","created":{"date-parts":[[2024,4,17]],"date-time":"2024-04-17T20:10:56Z","timestamp":1713384656000},"page":"368-385","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["SoCFlow: Efficient and Scalable DNN Training on SoC-Clustered Edge Servers"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6775-0688","authenticated-orcid":false,"given":"Daliang","family":"Xu","sequence":"first","affiliation":[{"name":"Key Laboratory of High Confidence Software Technologies (Peking University), Ministry of Education; School of Computer Science, Peking University, Beijing, China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6271-6993","authenticated-orcid":false,"given":"Mengwei","family":"Xu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1994-9947","authenticated-orcid":false,"given":"Chiheng","family":"Lou","sequence":"additional","affiliation":[{"name":"Key Laboratory of High Confidence Software Technologies (Peking University), Ministry of Education; School of Computer Science, Peking University, Beijing, China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0779-8310","authenticated-orcid":false,"given":"Li","family":"Zhang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4686-3181","authenticated-orcid":false,"given":"Gang","family":"Huang","sequence":"additional","affiliation":[{"name":"Key Laboratory of High Confidence Software Technologies (Peking University), Ministry of Education; School of Computer Science, Peking University, Beijing, China; National Key Laboratory of Data Space Technology and System, China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8741-5847","authenticated-orcid":false,"given":"Xin","family":"Jin","sequence":"additional","affiliation":[{"name":"Key Laboratory of High Confidence Software Technologies (Peking University), Ministry of Education; School of Computer Science, Peking University, Beijing, China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7908-8484","authenticated-orcid":false,"given":"Xuanzhe","family":"Liu","sequence":"additional","affiliation":[{"name":"Key Laboratory of High Confidence Software Technologies (Peking University), Ministry of Education; School of Computer Science, Peking University, Beijing, China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,4,17]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"https:\/\/en.wikipedia.org\/wiki\/Serial_Attached_SCSI","author":"Serial","year":"2019","unstructured":"Serial attached scsi. https:\/\/en.wikipedia.org\/wiki\/Serial_Attached_SCSI, 2019."},{"key":"e_1_3_2_1_2_1","volume-title":"https:\/\/www.qualcomm.com\/products\/application\/smartphones\/snapdragon-8-series-mobile-platforms\/snapdragon-865-5g-mobile-platform","author":"Snapdragon","year":"2019","unstructured":"Snapdragon 865 5g mobile platform. https:\/\/www.qualcomm.com\/products\/application\/smartphones\/snapdragon-8-series-mobile-platforms\/snapdragon-865-5g-mobile-platform, 2019."},{"key":"e_1_3_2_1_3_1","volume-title":"https:\/\/github.com\/XiaoMi\/mace","author":"Edge","year":"2021","unstructured":"Edge tpu. https:\/\/github.com\/XiaoMi\/mace, 2021."},{"key":"e_1_3_2_1_4_1","volume-title":"https:\/\/www.amazon.com\/luna\/landing-page","author":"Amazon","year":"2022","unstructured":"Amazon luna. https:\/\/www.amazon.com\/luna\/landing-page., 2022."},{"key":"e_1_3_2_1_5_1","volume-title":"meet facebook gaming. https:\/\/www.facebook.com\/fbgaminghome\/blog\/cloud-gaming-meetfacebook-gaming","author":"Cloud","year":"2022","unstructured":"Cloud gaming, meet facebook gaming. https:\/\/www.facebook.com\/fbgaminghome\/blog\/cloud-gaming-meetfacebook-gaming., 2022."},{"key":"e_1_3_2_1_6_1","volume-title":"https:\/\/www.nvidia.com\/en-us\/geforce-now\/","author":"Geforce","year":"2022","unstructured":"Geforce now. https:\/\/www.nvidia.com\/en-us\/geforce-now\/., 2022."},{"key":"e_1_3_2_1_7_1","volume-title":"https:\/\/stadia.google.com\/","author":"Google","year":"2022","unstructured":"Google stadia. https:\/\/stadia.google.com\/., 2022."},{"key":"e_1_3_2_1_8_1","volume-title":"http:\/\/www.cs.cornell.edu\/courses\/cs482\/2003su\/handouts\/greedy_ahead.pdf","author":"Greedy","year":"2022","unstructured":"Greedy stays ahead. http:\/\/www.cs.cornell.edu\/courses\/cs482\/2003su\/handouts\/greedy_ahead.pdf., 2022."},{"key":"e_1_3_2_1_9_1","volume-title":"https:\/\/www.technologyreview.com\/2019\/12\/11\/131629\/apple-ai-personalizes-sirifederated-learning\/","author":"How","year":"2022","unstructured":"How apple personalizes siri without hoovering up your data. https:\/\/www.technologyreview.com\/2019\/12\/11\/131629\/apple-ai-personalizes-sirifederated-learning\/, 2022."},{"key":"e_1_3_2_1_10_1","volume-title":"https:\/\/developer.nvidia.com\/blog\/massively-scale-deep-learning-training-nccl-2-4\/","author":"Massively","year":"2022","unstructured":"Massively scale your deep learning training with nccl 2.4. https:\/\/developer.nvidia.com\/blog\/massively-scale-deep-learning-training-nccl-2-4\/, 2022."},{"key":"e_1_3_2_1_11_1","volume-title":"https:\/\/cloud.google.com\/blog\/topics\/developers-practitioners\/optimize-training-performance-reduction-server-vertex-ai","author":"Optimize","year":"2022","unstructured":"Optimize training performance with reduction server on vertex ai. https:\/\/cloud.google.com\/blog\/topics\/developers-practitioners\/optimize-training-performance-reduction-server-vertex-ai, 2022."},{"key":"e_1_3_2_1_12_1","volume-title":"https:\/\/www.qualcomm.com\/products\/technology\/processors\/application-processors\/qcs603","author":"Smart","year":"2022","unstructured":"Smart camera. https:\/\/www.qualcomm.com\/products\/technology\/processors\/application-processors\/qcs603, 2022."},{"key":"e_1_3_2_1_13_1","volume-title":"https:\/\/www.xbox.com\/en-US\/xbox-game-pass\/cloud-gaming?xr=shellnav","year":"2022","unstructured":"X-cloud game pass. https:\/\/www.xbox.com\/en-US\/xbox-game-pass\/cloud-gaming?xr=shellnav., 2022."},{"key":"e_1_3_2_1_14_1","volume-title":"https:\/\/www.vclusters.com\/productinfo1.html","year":"2023","unstructured":"Soc-cluster. https:\/\/www.vclusters.com\/productinfo1.html, 2023."},{"key":"e_1_3_2_1_15_1","volume-title":"https:\/\/ai-benchmark.com\/ranking.html","year":"2023","unstructured":"Soc-cluster. https:\/\/ai-benchmark.com\/ranking.html, 2023."},{"key":"e_1_3_2_1_16_1","volume-title":"https:\/\/www.qualcomm.com\/products\/mobile\/snapdragon\/smartphones\/snapdragon-8-series-mobile-platforms\/snapdragon-8-gen-1-mobile-platform","year":"2023","unstructured":"Soc-cluster. https:\/\/www.qualcomm.com\/products\/mobile\/snapdragon\/smartphones\/snapdragon-8-series-mobile-platforms\/snapdragon-8-gen-1-mobile-platform, 2023."},{"key":"e_1_3_2_1_17_1","volume-title":"https:\/\/www.qualcomm.com\/products\/mobile\/snapdragon\/smartphones\/snapdragon-8-series-mobile-platforms\/snapdragon-8-gen-2-mobile-platform","year":"2023","unstructured":"Soc-cluster. https:\/\/www.qualcomm.com\/products\/mobile\/snapdragon\/smartphones\/snapdragon-8-series-mobile-platforms\/snapdragon-8-gen-2-mobile-platform, 2023."},{"key":"e_1_3_2_1_18_1","first-page":"265","volume-title":"12th USENIX Symposium on Operating Systems Design and Implementation","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, Michael Isard, et al. Tensorflow: a system for large-scale machine learning. In 12th USENIX Symposium on Operating Systems Design and Implementation, pages 265--283, 2016."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3519584"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477132.3483553"},{"key":"e_1_3_2_1_21_1","volume-title":"Scalable methods for 8-bit training of neural networks. Advances in neural information processing systems, 31","author":"Banner Ron","year":"2018","unstructured":"Ron Banner, Itay Hubara, Elad Hoffer, and Daniel Soudry. Scalable methods for 8-bit training of neural networks. Advances in neural information processing systems, 31, 2018."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1006\/jagm.1998.0938"},{"key":"e_1_3_2_1_23_1","volume-title":"Demystifying parallel and distributed deep learning: An in-depth concurrency analysis. ACM Computing Surveys (CSUR), 52(4):1--43","author":"Ben-Nun Tal","year":"2019","unstructured":"Tal Ben-Nun and Torsten Hoefler. Demystifying parallel and distributed deep learning: An in-depth concurrency analysis. ACM Computing Surveys (CSUR), 52(4):1--43, 2019."},{"key":"e_1_3_2_1_24_1","first-page":"374","article-title":"Towards federated learning at scale: System design","volume":"1","author":"Bonawitz Keith","year":"2019","unstructured":"Keith Bonawitz, Hubert Eichner, Wolfgang Grieskamp, Dzmitry Huba, Alex Ingerman, Vladimir Ivanov, Chloe Kiddon, Jakub Kone\u010dn\u1ef3, Stefano Mazzocchi, Brendan McMahan, et al. Towards federated learning at scale: System design. Proceedings of Machine Learning and Systems, 1:374--388, 2019.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3469116.3470009"},{"key":"e_1_3_2_1_26_1","volume-title":"Felix Xiaozhu Lin, and Mengwei Xu. Autofednlp: An efficient fednlp framework. arXiv preprint arXiv:2205.10162","author":"Cai Dongqi","year":"2022","unstructured":"Dongqi Cai, Yaozong Wu, Shangguang Wang, Felix Xiaozhu Lin, and Mengwei Xu. Autofednlp: An efficient fednlp framework. arXiv preprint arXiv:2205.10162, 2022."},{"key":"e_1_3_2_1_27_1","volume-title":"Peter Wu, Tian Li, Jakub Kone\u010dn\u1ef3, H Brendan McMahan, Virginia Smith, and Ameet Talwalkar. Leaf: A benchmark for federated settings. arXiv preprint arXiv:1812.01097","author":"Caldas Sebastian","year":"2018","unstructured":"Sebastian Caldas, Sai Meher Karthik Duddu, Peter Wu, Tian Li, Jakub Kone\u010dn\u1ef3, H Brendan McMahan, Virginia Smith, and Ameet Talwalkar. Leaf: A benchmark for federated settings. arXiv preprint arXiv:1812.01097, 2018."},{"key":"e_1_3_2_1_28_1","volume-title":"Hierarchical federated learning with privacy. arXiv preprint arXiv:2206.05209","author":"Chandrasekaran Varun","year":"2022","unstructured":"Varun Chandrasekaran, Suman Banerjee, Diego Perino, and Nicolas Kourtellis. Hierarchical federated learning with privacy. arXiv preprint arXiv:2206.05209, 2022."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2017.7966159"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421307"},{"key":"e_1_3_2_1_31_1","volume-title":"Binaryconnect: Training deep neural networks with binary weights during propagations. Advances in neural information processing systems, 28","author":"Courbariaux Matthieu","year":"2015","unstructured":"Matthieu Courbariaux, Yoshua Bengio, and Jean-Pierre David. Binaryconnect: Training deep neural networks with binary weights during propagations. Advances in neural information processing systems, 28, 2015."},{"key":"e_1_3_2_1_32_1","volume-title":"Cinic-10 is not imagenet or cifar-10. arXiv preprint arXiv:1810.03505","author":"Darlow Luke N","year":"2018","unstructured":"Luke N Darlow, Elliot J Crowley, Antreas Antoniou, and Amos J Storkey. Cinic-10 is not imagenet or cifar-10. arXiv preprint arXiv:1810.03505, 2018."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/PerComWorkshops53856.2022.9767442"},{"key":"e_1_3_2_1_34_1","volume-title":"Large scale distributed deep networks. Advances in neural information processing systems, 25","author":"Dean Jeffrey","year":"2012","unstructured":"Jeffrey Dean, Greg Corrado, Rajat Monga, Kai Chen, Matthieu Devin, Mark Mao, Marc'aurelio Ranzato, Andrew Senior, Paul Tucker, Ke Yang, et al. Large scale distributed deep networks. Advances in neural information processing systems, 25, 2012."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/MLHPC.2018.8638639"},{"issue":"76","key":"e_1_3_2_1_36_1","first-page":"1","article-title":"Fast and communication efficient framework for distributed machine learning","volume":"21","author":"Elgabli Anis","year":"2020","unstructured":"Anis Elgabli, Jihong Park, Amrit S Bedi, Mehdi Bennis, and Vaneet Aggarwal. Gadmm: Fast and communication efficient framework for distributed machine learning. J. Mach. Learn. Res., 21(76):1--39, 2020.","journal-title":"J. Mach. Learn. Res."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/2668332.2668349"},{"key":"e_1_3_2_1_38_1","volume-title":"large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677","author":"Goyal Priya","year":"2017","unstructured":"Priya Goyal, Piotr Doll\u00e1r, Ross Girshick, Pieter Noordhuis, Lukasz Wesolowski, Aapo Kyrola, Andrew Tulloch, Yangqing Jia, and Kaiming He. Accurate, large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677, 2017."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3449942"},{"key":"e_1_3_2_1_40_1","volume-title":"Federated learning for mobile keyboard prediction. arXiv preprint arXiv:1811.03604","author":"Hard Andrew","year":"2018","unstructured":"Andrew Hard, Kanishka Rao, Rajiv Mathews, Swaroop Ramaswamy, Fran\u00e7oise Beaufays, Sean Augenstein, Hubert Eichner, Chlo\u00e9 Kiddon, and Daniel Ramage. Federated learning for mobile keyboard prediction. arXiv preprint arXiv:1811.03604, 2018."},{"key":"e_1_3_2_1_41_1","first-page":"418","article-title":"Accelerating distributed deep learning with communication scheduling","volume":"1","author":"Hashemi Sayed Hadi","year":"2019","unstructured":"Sayed Hadi Hashemi, Sangeetha Abdu Jyothi, and Roy Campbell. Tictac: Accelerating distributed deep learning with communication scheduling. Proceedings of Machine Learning and Systems, 1:418--430, 2019.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_43_1","volume-title":"Phillip B Gibbons, Garth A Gibson, Greg Ganger, and Eric P Xing. More effective distributed ml via a stale synchronous parallel parameter server. Advances in neural information processing systems, 26","author":"Ho Qirong","year":"2013","unstructured":"Qirong Ho, James Cipar, Henggang Cui, Seunghak Lee, Jin Kyu Kim, Phillip B Gibbons, Garth A Gibson, Greg Ganger, and Eric P Xing. More effective distributed ml via a stale synchronous parallel parameter server. Advances in neural information processing systems, 26, 2013."},{"key":"e_1_3_2_1_44_1","volume-title":"Train longer, generalize better: closing the generalization gap in large batch training of neural networks. Advances in neural information processing systems, 30","author":"Hoffer Elad","year":"2017","unstructured":"Elad Hoffer, Itay Hubara, and Daniel Soudry. Train longer, generalize better: closing the generalization gap in large batch training of neural networks. Advances in neural information processing systems, 30, 2017."},{"key":"e_1_3_2_1_45_1","volume-title":"Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861","author":"Howard Andrew G","year":"2017","unstructured":"Andrew G Howard, Menglong Zhu, Bo Chen, Dmitry Kalenichenko, Weijun Wang, Tobias Weyand, Marco Andreetto, and Hartwig Adam. Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861, 2017."},{"key":"e_1_3_2_1_46_1","volume-title":"et al. Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems, 32","author":"Huang Yanping","year":"2019","unstructured":"Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Dehao Chen, Mia Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc V Le, Yonghui Wu, et al. Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems, 32, 2019."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3081333.3081360"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00286"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507778"},{"key":"e_1_3_2_1_50_1","volume-title":"Adaptive aggregation for federated learning. arXiv preprint arXiv:2203.12163","author":"Jayaram KR","year":"2022","unstructured":"KR Jayaram, Vinod Muthusamy, Gegi Thomas, Ashish Verma, and Mark Purcell. Adaptive aggregation for federated learning. arXiv preprint arXiv:2203.12163, 2022."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538932"},{"key":"e_1_3_2_1_52_1","first-page":"1","article-title":"Beyond data and model parallelism for deep neural networks","volume":"1","author":"Jia Zhihao","year":"2019","unstructured":"Zhihao Jia, Matei Zaharia, and Alex Aiken. Beyond data and model parallelism for deep neural networks. Proceedings of Machine Learning and Systems, 1:1--13, 2019.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_53_1","volume-title":"et al. Mnn: A universal and efficient inference engine. arXiv preprint arXiv:2002.12418","author":"Jiang Xiaotang","year":"2020","unstructured":"Xiaotang Jiang, Huan Wang, Yiliu Chen, Ziqi Wu, Lichuan Wang, Bin Zou, Yafeng Yang, Zongyang Cui, Yu Cai, Tianhang Yu, et al. Mnn: A universal and efficient inference engine. arXiv preprint arXiv:2002.12418, 2020."},{"key":"e_1_3_2_1_54_1","first-page":"463","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation","author":"Jiang Yimin","year":"2020","unstructured":"Yimin Jiang, Yibo Zhu, Chang Lan, Bairen Yi, Yong Cui, and Chuanxiong Guo. A unified architecture for accelerating distributed dnn training in heterogeneous gpu\/cpu clusters. In 14th USENIX Symposium on Operating Systems Design and Implementation, pages 463--479, 2020."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052707"},{"key":"e_1_3_2_1_56_1","first-page":"1","volume-title":"Proceedings of the Fourteenth EuroSys Conference 2019","author":"Kim Youngsok","year":"2019","unstructured":"Youngsok Kim, Joonsung Kim, Dongju Chae, Daehyun Kim, and Jangwoo Kim. \u03bclayer: Low latency on-device inference using cooperative single-layer acceleration and processor-friendly quantization. In Proceedings of the Fourteenth EuroSys Conference 2019, pages 1--15, 2019."},{"key":"e_1_3_2_1_57_1","volume-title":"Ananda Theertha Suresh, and Dave Bacon. Federated learning: Strategies for improving communication efficiency. arXiv preprint arXiv:1610.05492","author":"Kone\u010dn\u1ef3 Jakub","year":"2016","unstructured":"Jakub Kone\u010dn\u1ef3, H Brendan McMahan, Felix X Yu, Peter Richt\u00e1rik, Ananda Theertha Suresh, and Dave Bacon. Federated learning: Strategies for improving communication efficiency. arXiv preprint arXiv:1610.05492, 2016."},{"key":"e_1_3_2_1_58_1","volume-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky Alex","year":"2009","unstructured":"Alex Krizhevsky, Geoffrey Hinton, et al. Learning multiple layers of features from tiny images. 2009."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPSN.2016.7460664"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2973801"},{"key":"e_1_3_2_1_62_1","volume-title":"Lenet-5, convolutional neural networks. URL: http:\/\/yann.lecun.com\/exdb\/lenet, 20(5):14","author":"LeCun Yann","year":"2015","unstructured":"Yann LeCun et al. Lenet-5, convolutional neural networks. URL: http:\/\/yann.lecun.com\/exdb\/lenet, 20(5):14, 2015."},{"key":"e_1_3_2_1_64_1","first-page":"27","article-title":"Communication efficient distributed machine learning with the parameter server","author":"Li Mu","year":"2014","unstructured":"Mu Li, David G Andersen, Alexander J Smola, and Kai Yu. Communication efficient distributed machine learning with the parameter server. Advances in Neural Information Processing Systems, 27, 2014.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_65_1","volume-title":"Pytorch distributed: Experiences on accelerating data parallel training. arXiv preprint arXiv:2006.15704","author":"Li Shen","year":"2020","unstructured":"Shen Li, Yanli Zhao, Rohan Varma, Omkar Salpekar, Pieter Noordhuis, Teng Li, Adam Paszke, Jeff Smith, Brian Vaughan, Pritam Damania, et al. Pytorch distributed: Experiences on accelerating data parallel training. arXiv preprint arXiv:2006.15704, 2020."},{"key":"e_1_3_2_1_66_1","volume-title":"Vertical semi-federated learning for efficient online advertising. arXiv preprint arXiv:2209.15635","author":"Li Wenjie","year":"2022","unstructured":"Wenjie Li, Qiaolin Xia, Hao Cheng, Kouyin Xue, and Shu-Tao Xia. Vertical semi-federated learning for efficient online advertising. arXiv preprint arXiv:2209.15635, 2022."},{"key":"e_1_3_2_1_67_1","first-page":"3043","volume-title":"International Conference on Machine Learning","author":"Lian Xiangru","year":"2018","unstructured":"Xiangru Lian, Wei Zhang, Ce Zhang, and Ji Liu. Asynchronous decentralized parallel stochastic gradient descent. In International Conference on Machine Learning, pages 3043--3052. PMLR, 2018."},{"key":"e_1_3_2_1_68_1","volume-title":"Towards accurate binary convolutional neural network. Advances in neural information processing systems, 30","author":"Lin Xiaofan","year":"2017","unstructured":"Xiaofan Lin, Cong Zhao, and Wei Pan. Towards accurate binary convolutional neural network. Advances in neural information processing systems, 30, 2017."},{"key":"e_1_3_2_1_69_1","volume-title":"The International Conference on Learning Representations","author":"Lin Yujun","year":"2018","unstructured":"Yujun Lin, Song Han, Huizi Mao, Yu Wang, and William J Dally. Deep Gradient Compression: Reducing the communication bandwidth for distributed training. In The International Conference on Learning Representations, 2018."},{"key":"e_1_3_2_1_70_1","volume-title":"Neural networks with few multiplications. arXiv preprint arXiv:1510.03009","author":"Lin Zhouhan","year":"2015","unstructured":"Zhouhan Lin, Matthieu Courbariaux, Roland Memisevic, and Yoshua Bengio. Neural networks with few multiplications. arXiv preprint arXiv:1510.03009, 2015."},{"key":"e_1_3_2_1_71_1","volume-title":"Sharing a gpu between mpi processes: multiprocess service(mps). https:\/\/docs.nvidia.com\/deploy\/mps\/index.html","author":"Lite TensorFlow","year":"2019","unstructured":"TensorFlow Lite. Sharing a gpu between mpi processes: multiprocess service(mps). https:\/\/docs.nvidia.com\/deploy\/mps\/index.html, 2019."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.425"},{"key":"e_1_3_2_1_73_1","volume-title":"PMLR","author":"McMahan Brendan","year":"2017","unstructured":"Brendan McMahan, Eider Moore, Daniel Ramage, Seth Hampson, and Blaise Aguera y Arcas. Communication-efficient learning of deep networks from decentralized data. In Artificial intelligence and statistics, pages 1273--1282. PMLR, 2017."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2021.3053588"},{"key":"e_1_3_2_1_75_1","volume-title":"Kwang Choon Kim, Seon Heo, Yoonsang Kim, and Sungroh Yoon. Scalable smartphone cluster for deep learning. arXiv preprint arXiv:2110.12172","author":"Na Byunggook","year":"2021","unstructured":"Byunggook Na, Jaehee Jang, Seongsik Park, Seijoon Kim, Joonoo Kim, Moon Sik Jeong, Kwang Choon Kim, Seon Heo, Yoonsang Kim, and Sungroh Yoon. Scalable smartphone cluster for deep learning. arXiv preprint arXiv:2110.12172, 2021."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1145\/3372224.3419188"},{"key":"e_1_3_2_1_78_1","volume-title":"Gpt3-to-plan: Extracting plans from text using gpt-3. arXiv preprint arXiv:2106.07131","author":"Olmo Alberto","year":"2021","unstructured":"Alberto Olmo, Sarath Sreedharan, and Subbarao Kambhampati. Gpt3-to-plan: Extracting plans from text using gpt-3. arXiv preprint arXiv:2106.07131, 2021."},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4613-0303-9_16"},{"key":"e_1_3_2_1_80_1","volume-title":"et al. Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems, 32","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, et al. Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems, 32, 2019."},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_32"},{"key":"e_1_3_2_1_82_1","volume-title":"A lock-free approach to parallelizing stochastic gradient descent. Advances in neural information processing systems, 24","author":"Recht Benjamin","year":"2011","unstructured":"Benjamin Recht, Christopher Re, Stephen Wright, and Feng Niu. Hogwild!: A lock-free approach to parallelizing stochastic gradient descent. Advances in neural information processing systems, 24, 2011."},{"key":"e_1_3_2_1_83_1","first-page":"551","volume-title":"2021 USENIX Annual Technical Conference","author":"Ren Jie","year":"2021","unstructured":"Jie Ren, Samyam Rajbhandari, Reza Yazdani Aminabadi, Olatunji Ruwase, Shuangyan Yang, Minjia Zhang, Dong Li, and Yuxiong He. Zero-offload: Democratizing billion-scale model training. In 2021 USENIX Annual Technical Conference, pages 551--564, 2021."},{"key":"e_1_3_2_1_84_1","first-page":"785","volume-title":"18th USENIX Symposium on Networked Systems Design and Implementation","author":"Sapio Amedeo","year":"2021","unstructured":"Amedeo Sapio, Marco Canini, Chen-Yu Ho, Jacob Nelson, Panos Kalnis, Changhoon Kim, Arvind Krishnamurthy, Masoud Moshref, Dan Ports, and Peter Richt\u00e1rik. Scaling distributed machine learning with in-network aggregation. In 18th USENIX Symposium on Networked Systems Design and Implementation, pages 785--808, 2021."},{"key":"e_1_3_2_1_85_1","volume-title":"Horovod: fast and easy distributed deep learning in tensorflow. arXiv preprint arXiv:1802.05799","author":"Sergeev Alexander","year":"2018","unstructured":"Alexander Sergeev and Mike Del Balso. Horovod: fast and easy distributed deep learning in tensorflow. arXiv preprint arXiv:1802.05799, 2018."},{"key":"e_1_3_2_1_86_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556, 2014."},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575712"},{"key":"e_1_3_2_1_88_1","first-page":"400","volume-title":"Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","volume":"2","author":"Switzer Jennifer","year":"2023","unstructured":"Jennifer Switzer, Gabriel Marcano, Ryan Kastner, and Pat Pannuto. Junkyard computing: Repurposing discarded smartphones to minimize carbon. In Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2, pages 400--412, 2023."},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.5225"},{"key":"e_1_3_2_1_90_1","volume-title":"Attention is all you need. Advances in neural information processing systems, 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. Attention is all you need. Advances in neural information processing systems, 30, 2017."},{"key":"e_1_3_2_1_91_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2022.04.027"},{"key":"e_1_3_2_1_92_1","first-page":"172","article-title":"Fast and generic collectives for distributed ml","volume":"2","author":"Wang Guanhua","year":"2020","unstructured":"Guanhua Wang, Shivaram Venkataraman, Amar Phanishayee, Nikhil Devanur, Jorgen Thelin, and Ion Stoica. Blink: Fast and generic collectives for distributed ml. Proceedings of Machine Learning and Systems, 2:172--186, 2020.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_93_1","first-page":"212","article-title":"Adaptive communication strategies to achieve the best error-runtime trade-off in local-update sgd","volume":"1","author":"Wang Jianyu","year":"2019","unstructured":"Jianyu Wang and Gauri Joshi. Adaptive communication strategies to achieve the best error-runtime trade-off in local-update sgd. Proceedings of Machine Learning and Systems, 1:212--229, 2019.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_94_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3149787"},{"key":"e_1_3_2_1_95_1","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538928"},{"key":"e_1_3_2_1_96_1","volume-title":"Training and inference with integers in deep neural networks. arXiv preprint arXiv:1802.04680","author":"Wu Shuang","year":"2018","unstructured":"Shuang Wu, Guoqi Li, Feng Chen, and Luping Shi. Training and inference with integers in deep neural networks. arXiv preprint arXiv:1802.04680, 2018."},{"key":"e_1_3_2_1_97_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2020.2994391"},{"key":"e_1_3_2_1_98_1","volume-title":"Binarized neural networks on the imagenet classification task. arXiv preprint arXiv:1604.03058","author":"Wu Xundong","year":"2016","unstructured":"Xundong Wu, Yong Wu, and Yong Zhao. Binarized neural networks on the imagenet classification task. arXiv preprint arXiv:1604.03058, 2016."},{"key":"e_1_3_2_1_99_1","volume-title":"Fashion-mnist: a novel image dataset for benchmarking machine learning algorithms. arXiv preprint arXiv:1708.07747","author":"Xiao Han","year":"2017","unstructured":"Han Xiao, Kashif Rasul, and Roland Vollgraf. Fashion-mnist: a novel image dataset for benchmarking machine learning algorithms. arXiv preprint arXiv:1708.07747, 2017."},{"key":"e_1_3_2_1_100_1","first-page":"533","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation","author":"Xiao Wencong","year":"2020","unstructured":"Wencong Xiao, Shiru Ren, Yong Li, Yang Zhang, Pengyang Hou, Zhi Li, Yihui Feng, Wei Lin, and Yangqing Jia. Antman: Dynamic scaling on gpu clusters for deep learning. In 14th USENIX Symposium on Operating Systems Design and Implementation, pages 533--548, 2020."},{"key":"e_1_3_2_1_101_1","doi-asserted-by":"publisher","DOI":"10.1145\/3495243.3560545"},{"key":"e_1_3_2_1_102_1","doi-asserted-by":"publisher","DOI":"10.1145\/3487552.3487815"},{"key":"e_1_3_2_1_103_1","doi-asserted-by":"publisher","DOI":"10.1145\/3287075"},{"key":"e_1_3_2_1_104_1","doi-asserted-by":"publisher","DOI":"10.1109\/SEC54971.2022.00024"},{"key":"e_1_3_2_1_105_1","volume-title":"Shuffle-exchange brings faster: Reduce the idle time during communication for decentralized neural network training. arXiv preprint arXiv:2007.00433","author":"Yang Xiang","year":"2020","unstructured":"Xiang Yang. Shuffle-exchange brings faster: Reduce the idle time during communication for decentralized neural network training. arXiv preprint arXiv:2007.00433, 2020."},{"key":"e_1_3_2_1_106_1","volume-title":"The limit of the batch size. arXiv preprint arXiv:2006.08517","author":"You Yang","year":"2020","unstructured":"Yang You, Yuhui Wang, Huan Zhang, Zhao Zhang, James Demmel, and Cho-Jui Hsieh. The limit of the batch size. arXiv preprint arXiv:2006.08517, 2020."},{"key":"e_1_3_2_1_107_1","volume-title":"Hierarchical federated learning through lan-wan orchestration. arXiv preprint arXiv:2010.11612","author":"Yuan Jinliang","year":"2020","unstructured":"Jinliang Yuan, Mengwei Xu, Xiao Ma, Ao Zhou, Xuanzhe Liu, and Shangguang Wang. Hierarchical federated learning through lan-wan orchestration. arXiv preprint arXiv:2010.11612, 2020."},{"key":"e_1_3_2_1_108_1","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2018.1700105"},{"key":"e_1_3_2_1_109_1","first-page":"181","volume-title":"2017 USENIX Annual Technical Conference","author":"Zhang Hao","year":"2017","unstructured":"Hao Zhang, Zeyu Zheng, Shizhen Xu, Wei Dai, Qirong Ho, Xiaodan Liang, Zhiting Hu, Jinliang Wei, Pengtao Xie, and Eric P Xing. Poseidon: An efficient communication architecture for distributed deep learning on gpu clusters. In 2017 USENIX Annual Technical Conference, pages 181--193, 2017."},{"key":"e_1_3_2_1_110_1","volume-title":"Soc-cluster as an edge server: an application-driven measurement study. arXiv preprint arXiv:2212.12842","author":"Zhang Li","year":"2022","unstructured":"Li Zhang, Zhe Fu, Boqing Shi, Xiang Li, Rujin Lai, Chenyang Chen, Ao Zhou, Xiao Ma, Shangguang Wang, and Mengwei Xu. Soc-cluster as an edge server: an application-driven measurement study. arXiv preprint arXiv:2212.12842, 2022."},{"key":"e_1_3_2_1_111_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512148"},{"key":"e_1_3_2_1_112_1","first-page":"217","volume-title":"2022 USENIX Annual Technical Conference","author":"Zhang Wei","year":"2022","unstructured":"Wei Zhang, Binghao Chen, Zhenhua Han, Quan Chen, Peng Cheng, Fan Yang, Ran Shu, Yuqing Yang, and Minyi Guo. Pilotfish: Harvesting free cycles of cloud gaming with deep learning training. In 2022 USENIX Annual Technical Conference, pages 217--232, 2022."},{"key":"e_1_3_2_1_113_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737489"},{"key":"e_1_3_2_1_114_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00240"},{"key":"e_1_3_2_1_115_1","volume-title":"Scalable and low-latency federated learning with cooperative mobile edge networking. arXiv preprint arXiv:2205.13054","author":"Zhang Zhenxiao","year":"2022","unstructured":"Zhenxiao Zhang, Zhidong Gao, Yuanxiong Guo, and Yanmin Gong. Scalable and low-latency federated learning with cooperative mobile edge networking. arXiv preprint arXiv:2205.13054, 2022."},{"key":"e_1_3_2_1_116_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16462"},{"key":"e_1_3_2_1_117_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2019.00150"},{"key":"e_1_3_2_1_118_1","first-page":"177","volume-title":"2021 USENIX Annual Technical Conference","author":"Zhou Qihua","year":"2021","unstructured":"Qihua Zhou, Song Guo, Zhihao Qu, Jingcai Guo, Zhenda Xu, Jiewei Zhang, Tao Guo, Boyuan Luo, and Jingren Zhou. Octo:{INT8} training with loss-aware compensation and backward quantization for tiny on-device learning. In 2021 USENIX Annual Technical Conference, pages 177--191, 2021."},{"key":"e_1_3_2_1_119_1","volume-title":"Dorefa-net: Training low bitwidth convolutional neural networks with low bitwidth gradients. arXiv preprint arXiv:1606.06160","author":"Zhou Shuchang","year":"2016","unstructured":"Shuchang Zhou, Yuxin Wu, Zekun Ni, Xinyu Zhou, He Wen, and Yuheng Zou. Dorefa-net: Training low bitwidth convolutional neural networks with low bitwidth gradients. arXiv preprint arXiv:1606.06160, 2016."},{"key":"e_1_3_2_1_120_1","first-page":"713","volume-title":"Artificial Intelligence and Statistics","author":"Zhou Yi","year":"2016","unstructured":"Yi Zhou, Yaoliang Yu, Wei Dai, Yingbin Liang, and Eric Xing. On convergence of model parallel proximal gradient algorithm for stale synchronous parallel system. In Artificial Intelligence and Statistics, pages 713--722. PMLR, 2016."},{"key":"e_1_3_2_1_121_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11623"},{"key":"e_1_3_2_1_122_1","volume-title":"Trained ternary quantization. arXiv preprint arXiv:1612.01064","author":"Zhu Chenzhuo","year":"2016","unstructured":"Chenzhuo Zhu, Song Han, Huizi Mao, and William J Dally. Trained ternary quantization. arXiv preprint arXiv:1612.01064, 2016."},{"key":"e_1_3_2_1_123_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00204"}],"event":{"name":"ASPLOS '24: 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 1","location":"La Jolla CA USA","acronym":"ASPLOS '24","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture","SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3617232.3624847","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3617232.3624847","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:46:13Z","timestamp":1750178773000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3617232.3624847"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,17]]},"references-count":122,"alternative-id":["10.1145\/3617232.3624847","10.1145\/3617232"],"URL":"https:\/\/doi.org\/10.1145\/3617232.3624847","relation":{},"subject":[],"published":{"date-parts":[[2024,4,17]]},"assertion":[{"value":"2024-04-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}