{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:53:07Z","timestamp":1783439587490,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":95,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,14]],"date-time":"2022-10-14T00:00:00Z","timestamp":1665705600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the National Key R&D Program of China","award":["2020YFB1805500"],"award-info":[{"award-number":["2020YFB1805500"]}]},{"DOI":"10.13039\/501100005090","name":"Beijing Nova Program","doi-asserted-by":"publisher","award":["Z211100002121118"],"award-info":[{"award-number":["Z211100002121118"]}],"id":[{"id":"10.13039\/501100005090","id-type":"DOI","asserted-by":"publisher"}]},{"name":"the National Natural Science Foundation of China","award":["62172008 and 62102009"],"award-info":[{"award-number":["62172008 and 62102009"]}]},{"name":"Young Elite Scientists Sponsorship Program by CAST","award":["2021QNRC001"],"award-info":[{"award-number":["2021QNRC001"]}]},{"name":"the PKU-Baidu Fund Project","award":["2020BD007"],"award-info":[{"award-number":["2020BD007"]}]},{"name":"NSFC","award":["62102045"],"award-info":[{"award-number":["62102045"]}]},{"name":"the National Natural Science Fund for the Excellent Young Scientists Fund Program (Overseas)"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,14]]},"DOI":"10.1145\/3495243.3560545","type":"proceedings-article","created":{"date-parts":[[2022,10,14]],"date-time":"2022-10-14T15:38:33Z","timestamp":1665761913000},"page":"214-227","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":47,"title":["Mandheling"],"prefix":"10.1145","author":[{"given":"Daliang","family":"Xu","sequence":"first","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mengwei","family":"Xu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology (BUPT), Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qipeng","family":"Wang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shangguang","family":"Wang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology (BUPT), Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yun","family":"Ma","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kang","family":"Huang","sequence":"additional","affiliation":[{"name":"Linggui Tech Company, Beijing, China."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gang","family":"Huang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xin","family":"Jin","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuanzhe","family":"Liu","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,10,14]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Collaborative machine learning without centralized training data. https:\/\/ai.googleblog.com\/2017\/04\/federated-learning-collaborative.html","author":"Federated","year":"2017","unstructured":"Federated learning : Collaborative machine learning without centralized training data. https:\/\/ai.googleblog.com\/2017\/04\/federated-learning-collaborative.html , 2017 . Federated learning: Collaborative machine learning without centralized training data. https:\/\/ai.googleblog.com\/2017\/04\/federated-learning-collaborative.html, 2017."},{"key":"e_1_3_2_1_2_1","volume-title":"https:\/\/www.technologyreview.com\/2019\/12\/11\/131629\/apple-ai-personalizes-siri-federated-learning\/","author":"How","year":"2019","unstructured":"How apple personalizes siri without hoovering up your data. https:\/\/www.technologyreview.com\/2019\/12\/11\/131629\/apple-ai-personalizes-siri-federated-learning\/ , 2019 . How apple personalizes siri without hoovering up your data. https:\/\/www.technologyreview.com\/2019\/12\/11\/131629\/apple-ai-personalizes-siri-federated-learning\/, 2019."},{"key":"e_1_3_2_1_3_1","volume-title":"https:\/\/source.codeaurora.org\/quic\/hexagon_nn\/","author":"Qualcomm","year":"2020","unstructured":"Qualcomm hexagon nn offload framework. https:\/\/source.codeaurora.org\/quic\/hexagon_nn\/ , 2020 . Qualcomm hexagon nn offload framework. https:\/\/source.codeaurora.org\/quic\/hexagon_nn\/, 2020."},{"key":"e_1_3_2_1_4_1","volume-title":"https:\/\/developer.qualcomm.com\/software\/hexagon-dsp-sdk\/dsp-processor","year":"2021","unstructured":"dsp-processor. https:\/\/developer.qualcomm.com\/software\/hexagon-dsp-sdk\/dsp-processor , 2021 . dsp-processor. https:\/\/developer.qualcomm.com\/software\/hexagon-dsp-sdk\/dsp-processor, 2021."},{"key":"e_1_3_2_1_5_1","volume-title":"https:\/\/gdpr-info.eu\/","author":"General","year":"2021","unstructured":"General data protection regulation. https:\/\/gdpr-info.eu\/ , 2021 . General data protection regulation. https:\/\/gdpr-info.eu\/, 2021."},{"key":"e_1_3_2_1_6_1","volume-title":"https:\/\/genshin.mihoyo.com\/","year":"2021","unstructured":"Genshin. https:\/\/genshin.mihoyo.com\/ , 2021 . Genshin. https:\/\/genshin.mihoyo.com\/, 2021."},{"key":"e_1_3_2_1_7_1","volume-title":"https:\/\/developer.qualcomm.com\/software\/hexagon-dsp-sdk","year":"2021","unstructured":"\/hexagon-dsp-sdk. https:\/\/developer.qualcomm.com\/software\/hexagon-dsp-sdk , 2021 . \/hexagon-dsp-sdk. https:\/\/developer.qualcomm.com\/software\/hexagon-dsp-sdk, 2021."},{"key":"e_1_3_2_1_8_1","volume-title":"https:\/\/en.wikipedia.org\/wiki\/Qualcomm_Hexagon","author":"Qualcomm","year":"2021","unstructured":"Qualcomm hexagon. https:\/\/en.wikipedia.org\/wiki\/Qualcomm_Hexagon , 2021 . Qualcomm hexagon. https:\/\/en.wikipedia.org\/wiki\/Qualcomm_Hexagon, 2021."},{"key":"e_1_3_2_1_9_1","volume-title":"https:\/\/www.tensorflow.org\/guide\/function","author":"Tensorflow","year":"2021","unstructured":"Tensorflow graph reusing. https:\/\/www.tensorflow.org\/guide\/function , 2021 . Tensorflow graph reusing. https:\/\/www.tensorflow.org\/guide\/function, 2021."},{"key":"e_1_3_2_1_10_1","volume-title":"https:\/\/www.tiktok.com","year":"2021","unstructured":"Tiktok. https:\/\/www.tiktok.com , 2021 . Tiktok. https:\/\/www.tiktok.com, 2021."},{"key":"e_1_3_2_1_11_1","volume-title":"https:\/\/github.com\/Tencent\/TNN","year":"2021","unstructured":"Tnn. https:\/\/github.com\/Tencent\/TNN , 2021 . Tnn. https:\/\/github.com\/Tencent\/TNN, 2021."},{"key":"e_1_3_2_1_12_1","volume-title":"https:\/\/www.youtube.com","year":"2021","unstructured":"Youtube. https:\/\/www.youtube.com , 2021 . Youtube. https:\/\/www.youtube.com, 2021."},{"key":"e_1_3_2_1_13_1","first-page":"265","volume-title":"12th USENIX Symposium on Operating Systems Design and Implementation)","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi , Paul Barham , Jianmin Chen , Zhifeng Chen , Andy Davis , Jeffrey Dean , Matthieu Devin , Sanjay Ghemawat , Geoffrey Irving , Michael Isard , : A system for large-scale machine learning . In 12th USENIX Symposium on Operating Systems Design and Implementation) , pages 265 -- 283 , 2016 . Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, Michael Isard, et al. Tensorflow: A system for large-scale machine learning. In 12th USENIX Symposium on Operating Systems Design and Implementation), pages 265--283, 2016."},{"key":"e_1_3_2_1_14_1","volume-title":"Scalable methods for 8-bit training of neural networks. Advances in neural information processing systems, 31","author":"Banner Ron","year":"2018","unstructured":"Ron Banner , Itay Hubara , Elad Hoffer , and Daniel Soudry . Scalable methods for 8-bit training of neural networks. Advances in neural information processing systems, 31 , 2018 . Ron Banner, Itay Hubara, Elad Hoffer, and Daniel Soudry. Scalable methods for 8-bit training of neural networks. Advances in neural information processing systems, 31, 2018."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/1274971.1275000"},{"key":"e_1_3_2_1_16_1","volume-title":"Towards federated learning at scale: System design. arXiv preprint arXiv:1902.01046","author":"Bonawitz Keith","year":"2019","unstructured":"Keith Bonawitz , Hubert Eichner , Wolfgang Grieskamp , Dzmitry Huba , Alex Ingerman , Vladimir Ivanov , Chloe Kiddon , Jakub Kone\u010dn\u1ef3 , Stefano Mazzocchi , H Brendan McMahan , Towards federated learning at scale: System design. arXiv preprint arXiv:1902.01046 , 2019 . Keith Bonawitz, Hubert Eichner, Wolfgang Grieskamp, Dzmitry Huba, Alex Ingerman, Vladimir Ivanov, Chloe Kiddon, Jakub Kone\u010dn\u1ef3, Stefano Mazzocchi, H Brendan McMahan, et al. Towards federated learning at scale: System design. arXiv preprint arXiv:1902.01046, 2019."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3469116.3470009"},{"key":"e_1_3_2_1_19_1","first-page":"578","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen , Thierry Moreau , Ziheng Jiang , Lianmin Zheng , Eddie Yan , Haichen Shen , Meghan Cowan , Leyuan Wang , Yuwei Hu , Luis Ceze , : An automated end-to-end optimizing compiler for deep learning . In 13th USENIX Symposium on Operating Systems Design and Implementation , pages 578 -- 594 , 2018 . Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Haichen Shen, Meghan Cowan, Leyuan Wang, Yuwei Hu, Luis Ceze, et al. TVM: An automated end-to-end optimizing compiler for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation, pages 578--594, 2018."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2017.7966159"},{"key":"e_1_3_2_1_21_1","volume-title":"Binaryconnect: Training deep neural networks with binary weights during propagations. Advances in neural information processing systems, 28","author":"Courbariaux Matthieu","year":"2015","unstructured":"Matthieu Courbariaux , Yoshua Bengio , and Jean-Pierre David . Binaryconnect: Training deep neural networks with binary weights during propagations. Advances in neural information processing systems, 28 , 2015 . Matthieu Courbariaux, Yoshua Bengio, and Jean-Pierre David. Binaryconnect: Training deep neural networks with binary weights during propagations. Advances in neural information processing systems, 28, 2015."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3241539.3241559"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3081333.3081358"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2668332.2668349"},{"key":"e_1_3_2_1_25_1","volume-title":"large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677","author":"Goyal Priya","year":"2017","unstructured":"Priya Goyal , Piotr Doll\u00e1r , Ross Girshick , Pieter Noordhuis , Lukasz Wesolowski , Aapo Kyrola , Andrew Tulloch , Yangqing Jia , and Kaiming He. Accurate , large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677 , 2017 . Priya Goyal, Piotr Doll\u00e1r, Ross Girshick, Pieter Noordhuis, Lukasz Wesolowski, Aapo Kyrola, Andrew Tulloch, Yangqing Jia, and Kaiming He. Accurate, large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677, 2017."},{"key":"e_1_3_2_1_26_1","article-title":"Exploiting reuse for gpu subgraph enumeration","author":"Guo Wentian","year":"2020","unstructured":"Wentian Guo , Yuchen Li , and Kian-Lee Tan . Exploiting reuse for gpu subgraph enumeration . IEEE Transactions on Knowledge and Data Engineering , 2020 . Wentian Guo, Yuchen Li, and Kian-Lee Tan. Exploiting reuse for gpu subgraph enumeration. IEEE Transactions on Knowledge and Data Engineering, 2020.","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.3390\/s21072364"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT52795.2021.00014"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2019.00021"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447993.3483249"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/2906388.2906396"},{"key":"e_1_3_2_1_32_1","volume-title":"Federated learning for mobile keyboard prediction. arXiv preprint arXiv:1811.03604","author":"Hard Andrew","year":"2018","unstructured":"Andrew Hard , Kanishka Rao , Rajiv Mathews , Swaroop Ramaswamy , Fran\u00e7oise Beaufays , Sean Augenstein , Hubert Eichner , Chlo\u00e9 Kiddon , and Daniel Ramage . Federated learning for mobile keyboard prediction. arXiv preprint arXiv:1811.03604 , 2018 . Andrew Hard, Kanishka Rao, Rajiv Mathews, Swaroop Ramaswamy, Fran\u00e7oise Beaufays, Sean Augenstein, Hubert Eichner, Chlo\u00e9 Kiddon, and Daniel Ramage. Federated learning for mobile keyboard prediction. arXiv preprint arXiv:1811.03604, 2018."},{"key":"e_1_3_2_1_33_1","volume-title":"et al. Fedml: A research library and benchmark for federated machine learning. arXiv preprint arXiv:2007.13518","author":"He Chaoyang","year":"2020","unstructured":"Chaoyang He , Songze Li , Jinhyun So , Xiao Zeng , Mi Zhang , Hongyi Wang , Xiaoyang Wang , Praneeth Vepakomma , Abhishek Singh , Hang Qiu , et al. Fedml: A research library and benchmark for federated machine learning. arXiv preprint arXiv:2007.13518 , 2020 . Chaoyang He, Songze Li, Jinhyun So, Xiao Zeng, Mi Zhang, Hongyi Wang, Xiaoyang Wang, Praneeth Vepakomma, Abhishek Singh, Hang Qiu, et al. Fedml: A research library and benchmark for federated machine learning. arXiv preprint arXiv:2007.13518, 2020."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3081333.3081360"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00286"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538948"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413697"},{"key":"e_1_3_2_1_39_1","volume-title":"et al. Mnn: A universal and efficient inference engine. arXiv preprint arXiv:2002.12418","author":"Jiang Xiaotang","year":"2020","unstructured":"Xiaotang Jiang , Huan Wang , Yiliu Chen , Ziqi Wu , Lichuan Wang , Bin Zou , Yafeng Yang , Zongyang Cui , Yu Cai , Tianhang Yu , et al. Mnn: A universal and efficient inference engine. arXiv preprint arXiv:2002.12418 , 2020 . Xiaotang Jiang, Huan Wang, Yiliu Chen, Ziqi Wu, Lichuan Wang, Bin Zou, Yafeng Yang, Zongyang Cui, Yu Cai, Tianhang Yu, et al. Mnn: A universal and efficient inference engine. arXiv preprint arXiv:2002.12418, 2020."},{"key":"e_1_3_2_1_40_1","volume-title":"Kallista Bonawitz, Zachary Charles, Graham Cormode, Rachel Cummings, et al. Advances and open problems in federated learning. Foundations and Trends\u00ae in Machine Learning, 14(1--2):1--210","author":"Kairouz Peter","year":"2021","unstructured":"Peter Kairouz , H Brendan McMahan , Brendan Avent , Aur\u00e9lien Bellet , Mehdi Bennis , Arjun Nitin Bhagoji , Kallista Bonawitz, Zachary Charles, Graham Cormode, Rachel Cummings, et al. Advances and open problems in federated learning. Foundations and Trends\u00ae in Machine Learning, 14(1--2):1--210 , 2021 . Peter Kairouz, H Brendan McMahan, Brendan Avent, Aur\u00e9lien Bellet, Mehdi Bennis, Arjun Nitin Bhagoji, Kallista Bonawitz, Zachary Charles, Graham Cormode, Rachel Cummings, et al. Advances and open problems in federated learning. Foundations and Trends\u00ae in Machine Learning, 14(1--2):1--210, 2021."},{"key":"e_1_3_2_1_41_1","first-page":"1","volume-title":"Proceedings of the Fourteenth EuroSys Conference 2019","author":"Kim Youngsok","year":"2019","unstructured":"Youngsok Kim , Joonsung Kim , Dongju Chae , Daehyun Kim , and Jangwoo Kim . &mu;layer : Low latency on-device inference using cooperative single-layer acceleration and processor-friendly quantization . In Proceedings of the Fourteenth EuroSys Conference 2019 , pages 1 -- 15 , 2019 . Youngsok Kim, Joonsung Kim, Dongju Chae, Daehyun Kim, and Jangwoo Kim. &mu;layer: Low latency on-device inference using cooperative single-layer acceleration and processor-friendly quantization. In Proceedings of the Fourteenth EuroSys Conference 2019, pages 1--15, 2019."},{"key":"e_1_3_2_1_42_1","volume-title":"Federated optimization: Distributed machine learning for on-device intelligence. arXiv preprint arXiv:1610.02527","author":"Kone\u010dn\u1ef3 Jakub","year":"2016","unstructured":"Jakub Kone\u010dn\u1ef3 , H Brendan McMahan , Daniel Ramage , and Peter Richt\u00e1rik . Federated optimization: Distributed machine learning for on-device intelligence. arXiv preprint arXiv:1610.02527 , 2016 . Jakub Kone\u010dn\u1ef3, H Brendan McMahan, Daniel Ramage, and Peter Richt\u00e1rik. Federated optimization: Distributed machine learning for on-device intelligence. arXiv preprint arXiv:1610.02527, 2016."},{"key":"e_1_3_2_1_43_1","volume-title":"Ananda Theertha Suresh, and Dave Bacon. Federated learning: Strategies for improving communication efficiency. arXiv preprint arXiv:1610.05492","author":"Kone\u010dn\u1ef3 Jakub","year":"2016","unstructured":"Jakub Kone\u010dn\u1ef3 , H Brendan McMahan , Felix X Yu , Peter Richt\u00e1rik , Ananda Theertha Suresh, and Dave Bacon. Federated learning: Strategies for improving communication efficiency. arXiv preprint arXiv:1610.05492 , 2016 . Jakub Kone\u010dn\u1ef3, H Brendan McMahan, Felix X Yu, Peter Richt\u00e1rik, Ananda Theertha Suresh, and Dave Bacon. Federated learning: Strategies for improving communication efficiency. arXiv preprint arXiv:1610.05492, 2016."},{"key":"e_1_3_2_1_44_1","volume-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky Alex","year":"2009","unstructured":"Alex Krizhevsky , Geoffrey Hinton , Learning multiple layers of features from tiny images . 2009 . Alex Krizhevsky, Geoffrey Hinton, et al. Learning multiple layers of features from tiny images. 2009."},{"key":"e_1_3_2_1_45_1","volume-title":"Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems, 25","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky , Ilya Sutskever , and Geoffrey E Hinton . Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems, 25 , 2012 . Alex Krizhevsky, Ilya Sutskever, and Geoffrey E Hinton. Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems, 25, 2012."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPSN.2016.7460664"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/2750858.2804262"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3300061.3345455"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447993.3483278"},{"key":"e_1_3_2_1_50_1","volume-title":"Towards accurate binary convolutional neural network. Advances in neural information processing systems, 30","author":"Lin Xiaofan","year":"2017","unstructured":"Xiaofan Lin , Cong Zhao , and Wei Pan . Towards accurate binary convolutional neural network. Advances in neural information processing systems, 30 , 2017 . Xiaofan Lin, Cong Zhao, and Wei Pan. Towards accurate binary convolutional neural network. Advances in neural information processing systems, 30, 2017."},{"key":"e_1_3_2_1_51_1","volume-title":"Neural networks with few multiplications. arXiv preprint arXiv:1510.03009","author":"Lin Zhouhan","year":"2015","unstructured":"Zhouhan Lin , Matthieu Courbariaux , Roland Memisevic , and Yoshua Bengio . Neural networks with few multiplications. arXiv preprint arXiv:1510.03009 , 2015 . Zhouhan Lin, Matthieu Courbariaux, Roland Memisevic, and Yoshua Bengio. Neural networks with few multiplications. arXiv preprint arXiv:1510.03009, 2015."},{"key":"e_1_3_2_1_52_1","volume-title":"Deploy machine learning models on mobile and iot devices","author":"Lite TensorFlow","year":"2019","unstructured":"TensorFlow Lite . Deploy machine learning models on mobile and iot devices , 2019 . TensorFlow Lite. Deploy machine learning models on mobile and iot devices, 2019."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.5555\/3358807.3358895"},{"key":"e_1_3_2_1_54_1","volume-title":"Relaxed quantization for discretized neural networks. arXiv preprint arXiv:1810.01875","author":"Louizos Christos","year":"2018","unstructured":"Christos Louizos , Matthias Reisser , Tijmen Blankevoort , Efstratios Gavves , and Max Welling . Relaxed quantization for discretized neural networks. arXiv preprint arXiv:1810.01875 , 2018 . Christos Louizos, Matthias Reisser, Tijmen Blankevoort, Efstratios Gavves, and Max Welling. Relaxed quantization for discretized neural networks. arXiv preprint arXiv:1810.01875, 2018."},{"key":"e_1_3_2_1_55_1","volume-title":"PMLR","author":"McMahan Brendan","year":"2017","unstructured":"Brendan McMahan , Eider Moore , Daniel Ramage , Seth Hampson , and Blaise Aguera y Arcas . Communication-efficient learning of deep networks from decentralized data. In Artificial intelligence and statistics, pages 1273--1282 . PMLR , 2017 . Brendan McMahan, Eider Moore, Daniel Ramage, Seth Hampson, and Blaise Aguera y Arcas. Communication-efficient learning of deep networks from decentralized data. In Artificial intelligence and statistics, pages 1273--1282. PMLR, 2017."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/2907071"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458864.3466628"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3372224.3419188"},{"key":"e_1_3_2_1_59_1","volume-title":"et al. Pytorch: An imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke , Sam Gross , Francisco Massa , Adam Lerer , James Bradbury , Gregory Chanan , Trevor Killeen , Zeming Lin , Natalia Gimelshein , Luca Antiga , et al. Pytorch: An imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703 , 2019 . Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, et al. Pytorch: An imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703, 2019."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/1269843.1269850"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/MPRV.2016.36"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_32"},{"key":"e_1_3_2_1_63_1","volume-title":"Adaptive federated optimization. arXiv preprint arXiv:2003.00295","author":"Reddi Sashank","year":"2020","unstructured":"Sashank Reddi , Zachary Charles , Manzil Zaheer , Zachary Garrett , Keith Rush , Jakub Kone\u010dn\u1ef3 , Sanjiv Kumar , and H Brendan McMahan . Adaptive federated optimization. arXiv preprint arXiv:2003.00295 , 2020 . Sashank Reddi, Zachary Charles, Manzil Zaheer, Zachary Garrett, Keith Rush, Jakub Kone\u010dn\u1ef3, Sanjiv Kumar, and H Brendan McMahan. Adaptive federated optimization. arXiv preprint arXiv:2003.00295, 2020."},{"issue":"1","key":"e_1_3_2_1_64_1","first-page":"1","article-title":"Neon technology introduction","volume":"4","author":"Reddy Venu Gopal","year":"2008","unstructured":"Venu Gopal Reddy . Neon technology introduction . ARM Corporation , 4 ( 1 ): 1 -- 33 , 2008 . Venu Gopal Reddy. Neon technology introduction. ARM Corporation, 4(1):1--33, 2008.","journal-title":"ARM Corporation"},{"key":"e_1_3_2_1_65_1","first-page":"2021","volume-title":"International Conference on Artificial Intelligence and Statistics","author":"Reisizadeh Amirhossein","year":"2020","unstructured":"Amirhossein Reisizadeh , Aryan Mokhtari , Hamed Hassani , Ali Jadbabaie , and Ramtin Pedarsani . Fedpaq : A communication-efficient federated learning method with periodic averaging and quantization . In International Conference on Artificial Intelligence and Statistics , pages 2021 -- 2031 . PMLR, 2020 . Amirhossein Reisizadeh, Aryan Mokhtari, Hamed Hassani, Ali Jadbabaie, and Ramtin Pedarsani. Fedpaq: A communication-efficient federated learning method with periodic averaging and quantization. In International Conference on Artificial Intelligence and Statistics, pages 2021--2031. PMLR, 2020."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054168"},{"key":"e_1_3_2_1_67_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman . Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 , 2014 . Karen Simonyan and Andrew Zisserman. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556, 2014."},{"key":"e_1_3_2_1_68_1","volume-title":"Don't decay the learning rate, increase the batch size. arXiv preprint arXiv:1711.00489","author":"Smith Samuel L","year":"2017","unstructured":"Samuel L Smith , Pieter-Jan Kindermans , Chris Ying , and Quoc V Le . Don't decay the learning rate, increase the batch size. arXiv preprint arXiv:1711.00489 , 2017 . Samuel L Smith, Pieter-Jan Kindermans, Chris Ying, and Quoc V Le. Don't decay the learning rate, increase the batch size. arXiv preprint arXiv:1711.00489, 2017."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_1_70_1","first-page":"31","article-title":"Communication compression for decentralized training","author":"Tang Hanlin","year":"2018","unstructured":"Hanlin Tang , Shaoduo Gan , Ce Zhang , Tong Zhang , and Ji Liu . Communication compression for decentralized training . Advances in Neural Information Processing Systems , 31 , 2018 . Hanlin Tang, Shaoduo Gan, Ce Zhang, Tong Zhang, and Ji Liu. Communication compression for decentralized training. Advances in Neural Information Processing Systems, 31, 2018.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2010.5470428"},{"key":"e_1_3_2_1_72_1","volume-title":"A comprehensive survey on training acceleration for large machine learning models in iots","author":"Wang Haozhao","year":"2021","unstructured":"Haozhao Wang , Zhihao Qu , Qihua Zhou , Haobo Zhang , Boyuan Luo , Wenchao Xu , Song Guo , and Ruixuan Li . A comprehensive survey on training acceleration for large machine learning models in iots . IEEE Internet of Things Journal , 2021 . Haozhao Wang, Zhihao Qu, Qihua Zhou, Haobo Zhang, Boyuan Luo, Wenchao Xu, Song Guo, and Ruixuan Li. A comprehensive survey on training acceleration for large machine learning models in iots. IEEE Internet of Things Journal, 2021."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447993.3448625"},{"key":"e_1_3_2_1_74_1","volume-title":"Philip HW Leong, and Hayden KH So. Niti: Training integer neural networks using integer-only arithmetic. arXiv preprint arXiv:2009.13108","author":"Wang Maolin","year":"2020","unstructured":"Maolin Wang , Seyedramin Rasoulinezhad , Philip HW Leong, and Hayden KH So. Niti: Training integer neural networks using integer-only arithmetic. arXiv preprint arXiv:2009.13108 , 2020 . Maolin Wang, Seyedramin Rasoulinezhad, Philip HW Leong, and Hayden KH So. Niti: Training integer neural networks using integer-only arithmetic. arXiv preprint arXiv:2009.13108, 2020."},{"key":"e_1_3_2_1_75_1","volume-title":"Training deep neural networks with 8-bit floating point numbers. Advances in neural information processing systems, 31","author":"Wang Naigang","year":"2018","unstructured":"Naigang Wang , Jungwook Choi , Daniel Brand , Chia-Yu Chen , and Kailash Gopalakrishnan . Training deep neural networks with 8-bit floating point numbers. Advances in neural information processing systems, 31 , 2018 . Naigang Wang, Jungwook Choi, Daniel Brand, Chia-Yu Chen, and Kailash Gopalakrishnan. Training deep neural networks with 8-bit floating point numbers. Advances in neural information processing systems, 31, 2018."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538928"},{"key":"e_1_3_2_1_77_1","volume-title":"Training and inference with integers in deep neural networks. arXiv preprint arXiv:1802.04680","author":"Wu Shuang","year":"2018","unstructured":"Shuang Wu , Guoqi Li , Feng Chen , and Luping Shi . Training and inference with integers in deep neural networks. arXiv preprint arXiv:1802.04680 , 2018 . Shuang Wu, Guoqi Li, Feng Chen, and Luping Shi. Training and inference with integers in deep neural networks. arXiv preprint arXiv:1802.04680, 2018."},{"key":"e_1_3_2_1_78_1","volume-title":"Binarized neural networks on the imagenet classification task. arXiv preprint arXiv:1604.03058","author":"Wu Xundong","year":"2016","unstructured":"Xundong Wu , Yong Wu , and Yong Zhao . Binarized neural networks on the imagenet classification task. arXiv preprint arXiv:1604.03058 , 2016 . Xundong Wu, Yong Wu, and Yong Zhao. Binarized neural networks on the imagenet classification task. arXiv preprint arXiv:1604.03058, 2016."},{"key":"e_1_3_2_1_79_1","volume-title":"Proceedings of the ACM on Interactive, Mobile, Wearable and Ubiquitous Technologies.","author":"Xu Mengwei","unstructured":"Mengwei Xu , Feng Qian , Qiaozhu Mei , Kang Huang , and Xuanzhe Liu . Deep-type : On-device deep learning for input personalization service with minimal privacy concern . Proceedings of the ACM on Interactive, Mobile, Wearable and Ubiquitous Technologies. Mengwei Xu, Feng Qian, Qiaozhu Mei, Kang Huang, and Xuanzhe Liu. Deep-type: On-device deep learning for input personalization service with minimal privacy concern. Proceedings of the ACM on Interactive, Mobile, Wearable and Ubiquitous Technologies."},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2019.2893250"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3241539.3241563"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3449851"},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2019.12.027"},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/3372224.3419185"},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2021.3088910"},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.1145\/3384419.3430726"},{"key":"e_1_3_2_1_87_1","volume-title":"A comprehensive benchmark of deep learning libraries on mobile devices. arXiv preprint arXiv:2202.06512","author":"Zhang Qiyang","year":"2022","unstructured":"Qiyang Zhang , Xiang Li , Xiangying Che , Xiao Ma , Ao Zhou , Mengwei Xu , Shangguang Wang , Yun Ma , and Xuanzhe Liu . A comprehensive benchmark of deep learning libraries on mobile devices. arXiv preprint arXiv:2202.06512 , 2022 . Qiyang Zhang, Xiang Li, Xiangying Che, Xiao Ma, Ao Zhou, Mengwei Xu, Shangguang Wang, Yun Ma, and Xuanzhe Liu. A comprehensive benchmark of deep learning libraries on mobile devices. arXiv preprint arXiv:2202.06512, 2022."},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00240"},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"publisher","DOI":"10.1145\/1127577.1127584"},{"key":"e_1_3_2_1_90_1","volume-title":"Towards lower bit multiplication for convolutional neural network training. arXiv preprint arXiv:2006.02804, 3(4)","author":"Zhong Kai","year":"2020","unstructured":"Kai Zhong , Tianchen Zhao , Xuefei Ning , Shulin Zeng , Kaiyuan Guo , Yu Wang , and Huazhong Yang . Towards lower bit multiplication for convolutional neural network training. arXiv preprint arXiv:2006.02804, 3(4) , 2020 . Kai Zhong, Tianchen Zhao, Xuefei Ning, Shulin Zeng, Kaiyuan Guo, Yu Wang, and Huazhong Yang. Towards lower bit multiplication for convolutional neural network training. arXiv preprint arXiv:2006.02804, 3(4), 2020."},{"key":"e_1_3_2_1_91_1","first-page":"177","volume-title":"2021 USENIX Annual Technical Conference","author":"Zhou Qihua","year":"2021","unstructured":"Qihua Zhou , Song Guo , Zhihao Qu , Jingcai Guo , Zhenda Xu , Jiewei Zhang , Tao Guo , Boyuan Luo , and Jingren Zhou . Octo : {INT8} training with loss-aware compensation and backward quantization for tiny on-device learning . In 2021 USENIX Annual Technical Conference , pages 177 -- 191 , 2021 . Qihua Zhou, Song Guo, Zhihao Qu, Jingcai Guo, Zhenda Xu, Jiewei Zhang, Tao Guo, Boyuan Luo, and Jingren Zhou. Octo: {INT8} training with loss-aware compensation and backward quantization for tiny on-device learning. In 2021 USENIX Annual Technical Conference, pages 177--191, 2021."},{"key":"e_1_3_2_1_92_1","volume-title":"Dorefa-net: Training low bitwidth convolutional neural networks with low bitwidth gradients. arXiv preprint arXiv:1606.06160","author":"Zhou Shuchang","year":"2016","unstructured":"Shuchang Zhou , Yuxin Wu , Zekun Ni , Xinyu Zhou , He Wen , and Yuheng Zou . Dorefa-net: Training low bitwidth convolutional neural networks with low bitwidth gradients. arXiv preprint arXiv:1606.06160 , 2016 . Shuchang Zhou, Yuxin Wu, Zekun Ni, Xinyu Zhou, He Wen, and Yuheng Zou. Dorefa-net: Training low bitwidth convolutional neural networks with low bitwidth gradients. arXiv preprint arXiv:1606.06160, 2016."},{"key":"e_1_3_2_1_93_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11623"},{"key":"e_1_3_2_1_94_1","volume-title":"Trained ternary quantization. arXiv preprint arXiv:1612.01064","author":"Zhu Chenzhuo","year":"2016","unstructured":"Chenzhuo Zhu , Song Han , Huizi Mao , and William J Dally . Trained ternary quantization. arXiv preprint arXiv:1612.01064 , 2016 . Chenzhuo Zhu, Song Han, Huizi Mao, and William J Dally. Trained ternary quantization. arXiv preprint arXiv:1612.01064, 2016."},{"key":"e_1_3_2_1_95_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00204"}],"event":{"name":"ACM MobiCom '22: The 28th Annual International Conference on Mobile Computing and Networking","location":"Sydney NSW Australia","acronym":"ACM MobiCom '22","sponsor":["SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing"]},"container-title":["Proceedings of the 28th Annual International Conference on Mobile Computing And Networking"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3495243.3560545","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3495243.3560545","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:49:23Z","timestamp":1750182563000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3495243.3560545"}},"subtitle":["mixed-precision on-device DNN training with DSP offloading"],"short-title":[],"issued":{"date-parts":[[2022,10,14]]},"references-count":95,"alternative-id":["10.1145\/3495243.3560545","10.1145\/3495243"],"URL":"https:\/\/doi.org\/10.1145\/3495243.3560545","relation":{},"subject":[],"published":{"date-parts":[[2022,10,14]]},"assertion":[{"value":"2022-10-14","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}