{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T15:52:23Z","timestamp":1780674743520,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,8,17]],"date-time":"2020-08-17T00:00:00Z","timestamp":1597622400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100012659","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61502019, 61732002"],"award-info":[{"award-number":["61502019, 61732002"]}],"id":[{"id":"10.13039\/501100012659","id-type":"DOI","asserted-by":"publisher"}]},{"name":"SenseTime Research Fund for Young Scholars"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,8,17]]},"DOI":"10.1145\/3404397.3404407","type":"proceedings-article","created":{"date-parts":[[2020,8,9]],"date-time":"2020-08-09T03:54:26Z","timestamp":1596945266000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Extremely Low-bit Convolution Optimization for Quantized Neural Network on Modern Computer Architectures"],"prefix":"10.1145","author":[{"given":"Qingchang","family":"Han","sequence":"first","affiliation":[{"name":"Beihang University SenseTime Research, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongmin","family":"Hu","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fengwei","family":"Yu","sequence":"additional","affiliation":[{"name":"SenseTime Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hailong","family":"Yang","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bing","family":"Liu","sequence":"additional","affiliation":[{"name":"SenseTime Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Hu","sequence":"additional","affiliation":[{"name":"SenseTime Research Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ruihao","family":"Gong","sequence":"additional","affiliation":[{"name":"Beihang University SenseTime Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanfei","family":"Wang","sequence":"additional","affiliation":[{"name":"SenseTime Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rui","family":"Wang","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongzhi","family":"Luan","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Depei","family":"Qian","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2020,8,17]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the 12th USENIX conference on Operating Systems Design and Implementation. 579\u2013594","author":"Chen Tianqi","year":"2018"},{"key":"e_1_3_2_1_2_1","unstructured":"Sharan Chetlur Cliff Woolley Philippe Vandermersch Jonathan Cohen John Tran Bryan Catanzaro and Evan Shelhamer. 2014. cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759(2014).  Sharan Chetlur Cliff Woolley Philippe Vandermersch Jonathan Cohen John Tran Bryan Catanzaro and Evan Shelhamer. 2014. cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759(2014)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3368826.3377912"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/77626.79170"},{"key":"e_1_3_2_1_6_1","unstructured":"Marat Dukhan Yiming Wu and Hao Lu. 2018. QNNPACK: open source library for optimized mobile deep learning. https:\/\/github.com\/pytorch\/QNNPACK.  Marat Dukhan Yiming Wu and Hao Lu. 2018. QNNPACK: open source library for optimized mobile deep learning. https:\/\/github.com\/pytorch\/QNNPACK."},{"key":"e_1_3_2_1_7_1","volume-title":"LEARNED STEP SIZE QUANTIZATION. In International Conference on Learning Representations.","author":"Esser K.","year":"2020"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00495"},{"key":"e_1_3_2_1_9_1","unstructured":"Song Han Jeff Pool John Tran and William Dally. 2015. Learning both weights and connections for efficient neural network. In Advances in neural information processing systems. 1135\u20131143.  Song Han Jeff Pool John Tran and William Dally. 2015. Learning both weights and connections for efficient neural network. In Advances in neural information processing systems. 1135\u20131143."},{"key":"e_1_3_2_1_10_1","unstructured":"Babak Hassibi and David\u00a0G Stork. 1993. Second order derivatives for network pruning: Optimal brain surgeon. In Advances in neural information processing systems. 164\u2013171.  Babak Hassibi and David\u00a0G Stork. 1993. Second order derivatives for network pruning: Optimal brain surgeon. In Advances in neural information processing systems. 164\u2013171."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_1_13_1","unstructured":"Intel. 2016. Deep Neural Network Library. https:\/\/github.com\/intel\/mkl-dnn.  Intel. 2016. Deep Neural Network Library. https:\/\/github.com\/intel\/mkl-dnn."},{"key":"e_1_3_2_1_14_1","unstructured":"Benoit Jacob 2017. gemmlowp: a small self-contained low-precision GEMM library.(2017).  Benoit Jacob 2017. gemmlowp: a small self-contained low-precision GEMM library.(2017)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2017.7975270"},{"key":"e_1_3_2_1_16_1","volume-title":"Cmsis-nn: Efficient neural network kernels for arm cortex-m cpus. arXiv preprint arXiv:1801.06601(2018).","author":"Lai Liangzhen","year":"2018"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.435"},{"key":"e_1_3_2_1_18_1","volume-title":"Fully Quantized Network for Object Detection. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Li Rundong","year":"2019"},{"key":"e_1_3_2_1_19_1","unstructured":"Tsung-Yi Lin Michael Maire Serge Belongie Lubomir Bourdev Ross Girshick James Hays Pietro Perona Deva Ramanan C.\u00a0Lawrence Zitnick and Piotr Doll\u00e1r. 2014. Microsoft COCO: Common Objects in Context.  Tsung-Yi Lin Michael Maire Serge Belongie Lubomir Bourdev Ross Girshick James Hays Pietro Perona Deva Ramanan C.\u00a0Lawrence Zitnick and Piotr Doll\u00e1r. 2014. Microsoft COCO: Common Objects in Context."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2016.12.038"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2018.00091"},{"key":"e_1_3_2_1_22_1","volume-title":"GPU Technology Conference.","author":"Migacz Szymon","year":"2017"},{"key":"e_1_3_2_1_23_1","unstructured":"nihui 2017. NCNN. https:\/\/github.com\/Tencent\/ncnn.  nihui 2017. NCNN. https:\/\/github.com\/Tencent\/ncnn."},{"key":"e_1_3_2_1_24_1","unstructured":"NVIDIA. 2019. NVIDIA Nsight Compute. https:\/\/developer.nvidia.com\/nsight-compute.  NVIDIA. 2019. NVIDIA Nsight Compute. https:\/\/developer.nvidia.com\/nsight-compute."},{"key":"e_1_3_2_1_25_1","unstructured":"CUTLASS NVIDIA. 2017. CUDA Templates for Linear Algebra Subroutines. https:\/\/github.com\/NVIDIA\/cutlass.  CUTLASS NVIDIA. 2017. CUDA Templates for Linear Algebra Subroutines. https:\/\/github.com\/NVIDIA\/cutlass."},{"key":"e_1_3_2_1_26_1","volume-title":"Parallel Thread Execution ISA version 6.5","author":"PTX NVIDIA.","year":"2019"},{"key":"e_1_3_2_1_27_1","unstructured":"Vijay\u00a0Janapa Reddi Christine Cheng David Kanter Peter Mattson Guenther Schmuelling Carole-Jean Wu Brian Anderson Maximilien Breughe Mark Charlebois William Chou 2019. Mlperf inference benchmark. arXiv preprint arXiv:1911.02549(2019).  Vijay\u00a0Janapa Reddi Christine Cheng David Kanter Peter Mattson Guenther Schmuelling Carole-Jean Wu Brian Anderson Maximilien Breughe Mark Charlebois William Chou 2019. Mlperf inference benchmark. arXiv preprint arXiv:1911.02549(2019)."},{"key":"e_1_3_2_1_28_1","volume-title":"European Conference on Parallel Processing. Springer, 688\u2013699","author":"Schindler G\u00fcnther","year":"2017"},{"key":"e_1_3_2_1_29_1","unstructured":"SoftBank. 2017. Q4 2016 Roadshow Slides - Arm.(2017).  SoftBank. 2017. Q4 2016 Roadshow Slides - Arm.(2017)."},{"key":"e_1_3_2_1_30_1","unstructured":"Andrew Tulloch and Yangqing Jia. 2017. High performance ultra-low-precision convolutions on mobile devices. arXiv preprint arXiv:1712.02427(2017).  Andrew Tulloch and Yangqing Jia. 2017. High performance ultra-low-precision convolutions on mobile devices. arXiv preprint arXiv:1712.02427(2017)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3125501.3125528"},{"key":"e_1_3_2_1_32_1","unstructured":"Nicolas Vasilache Jeff Johnson Michael Mathieu Soumith Chintala Serkan Piantino and Yann LeCun. 2014. Fast convolutional nets with fbfft: A GPU performance evaluation. arXiv preprint arXiv:1412.7580(2014).  Nicolas Vasilache Jeff Johnson Michael Mathieu Soumith Chintala Serkan Piantino and Yann LeCun. 2014. Fast convolutional nets with fbfft: A GPU performance evaluation. arXiv preprint arXiv:1412.7580(2014)."},{"key":"e_1_3_2_1_33_1","volume-title":"Rotation Consistent Margin Loss for Efficient Low-bit Face Recognition. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Wu Yudong","year":"2020"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358269"}],"event":{"name":"ICPP '20: 49th International Conference on Parallel Processing","location":"Edmonton AB Canada","acronym":"ICPP '20"},"container-title":["49th International Conference on Parallel Processing - ICPP"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3404397.3404407","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3404397.3404407","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:31:42Z","timestamp":1750195902000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3404397.3404407"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,8,17]]},"references-count":34,"alternative-id":["10.1145\/3404397.3404407","10.1145\/3404397"],"URL":"https:\/\/doi.org\/10.1145\/3404397.3404407","relation":{},"subject":[],"published":{"date-parts":[[2020,8,17]]},"assertion":[{"value":"2020-08-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}