{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,25]],"date-time":"2025-07-25T10:36:43Z","timestamp":1753439803309,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,8,7]],"date-time":"2023-08-07T00:00:00Z","timestamp":1691366400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2037982"],"award-info":[{"award-number":["2037982"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,8,7]]},"DOI":"10.1145\/3605573.3605580","type":"proceedings-article","created":{"date-parts":[[2023,9,13]],"date-time":"2023-09-13T16:21:16Z","timestamp":1694622076000},"page":"645-654","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Output-Directed Dynamic Quantization for DNN Acceleration"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-3985-333X","authenticated-orcid":false,"given":"Beilei","family":"Jiang","sequence":"first","affiliation":[{"name":"University of North Texas, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3550-2647","authenticated-orcid":false,"given":"Xianwei","family":"Cheng","sequence":"additional","affiliation":[{"name":"University of North Texas, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9301-9369","authenticated-orcid":false,"given":"Yuan","family":"Li","sequence":"additional","affiliation":[{"name":"University of North Texas, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9828-115X","authenticated-orcid":false,"given":"Jocelyn","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of North Texas, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7705-0829","authenticated-orcid":false,"given":"Song","family":"Fu","sequence":"additional","affiliation":[{"name":"University of North Texas, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3495-370X","authenticated-orcid":false,"given":"Qing","family":"Yang","sequence":"additional","affiliation":[{"name":"University of North Texas, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5992-1221","authenticated-orcid":false,"given":"Mingxiong","family":"Liu","sequence":"additional","affiliation":[{"name":"Los Alamos National Laboratory, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9112-0734","authenticated-orcid":false,"given":"Alejandro","family":"Olvera","sequence":"additional","affiliation":[{"name":"University of North Texas, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,9,13]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2018. Pytorch: Tensors and dynamic neural networks in python with strong gpu acceleration. In https:\/\/github.com\/pytorch."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"David Bau and et. al. 2017. Network dissection: Quantifying interpretability of deep visual representations. In CVPR.","DOI":"10.1109\/CVPR.2017.354"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Sung-En Chang and et. al. 2021. Mix and Match: A novel FPGA-centric deep neural network quantization framework. In HPCA.","DOI":"10.1109\/HPCA51647.2021.00027"},{"key":"e_1_3_2_1_4_1","volume-title":"Diannao: A small-footprint high-throughput accelerator for ubiquitous machine-learning. ACM SIGARCH Computer Architecture News","author":"Tianshi Chen","year":"2014","unstructured":"Tianshi Chen and et. al. 2014. Diannao: A small-footprint high-throughput accelerator for ubiquitous machine-learning. ACM SIGARCH Computer Architecture News (2014)."},{"volume-title":"Proceedings of the IEEE conference on Computer Vision and Pattern Recognition.","author":"Xiaozhi","key":"e_1_3_2_1_5_1","unstructured":"Xiaozhi Chen and et. al. 2017. Multi-view 3d object detection network for autonomous driving. In Proceedings of the IEEE conference on Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_6_1","volume-title":"Binarized neural networks: Training deep neural networks with weights and activations constrained to+ 1 or-1. arXiv preprint arXiv:1602.02830","author":"Matthieu Courbariaux","year":"2016","unstructured":"Matthieu Courbariaux and et. al. 2016. Binarized neural networks: Training deep neural networks with weights and activations constrained to+ 1 or-1. arXiv preprint arXiv:1602.02830 (2016)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Li Deng and et. al. 2013. Recent Advances in Deep Learning for Speech Research at Microsoft. In ICASSP.","DOI":"10.1109\/ICASSP.2013.6639345"},{"volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition.","author":"Kaiming","key":"e_1_3_2_1_8_1","unstructured":"Kaiming He and et. al. 2016. Deep residual learning for image recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"G. Huang and et. al. 2017. Densely connected convolutional networks. In CVPR.","DOI":"10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_1_10_1","volume-title":"MLCNN: Cross-Layer Cooperative Optimization and Accelerator Architecture for Speeding Up Deep Learning Applications. In IPDPS.","author":"Beilei Jiang","year":"2022","unstructured":"Beilei Jiang and et. al. 2022. MLCNN: Cross-Layer Cooperative Optimization and Accelerator Architecture for Speeding Up Deep Learning Applications. In IPDPS."},{"key":"e_1_3_2_1_11_1","volume-title":"Jouppi and et. al","author":"Norman\u00a0 P.","year":"2017","unstructured":"Norman\u00a0P. Jouppi and et. al. 2017. In-Datacenter Performance Analysis of a Tensor Processing Unit. In ISCA \u201917."},{"key":"e_1_3_2_1_12_1","unstructured":"Alex Krizhevsky. 2010. CIFAR-10 and CIFAR-100 datasets. In https:\/\/www.cs.toronto.edu\/\u00a0kriz\/cifar.html."},{"key":"e_1_3_2_1_13_1","volume-title":"Convolutional neural networks using logarithmic data representation. arXiv preprint arXiv:1603.01025","author":"Daisuke Miyashita","year":"2016","unstructured":"Daisuke Miyashita and et. al. 2016. Convolutional neural networks using logarithmic data representation. arXiv preprint arXiv:1603.01025 (2016)."},{"key":"e_1_3_2_1_14_1","unstructured":"N. Muralimanohar and et. al. 2009. CACTI 6.0: A tool to model large caches. In HP laboratories."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Eunhyeok Park and et. al. 2018. Energy-efficient neural network accelerator based on outlier-aware low-precision computation. In ISCA.","DOI":"10.1109\/ISCA.2018.00063"},{"key":"e_1_3_2_1_16_1","unstructured":"S Preethi and et. al. 2020. Smart Healthcare Monitoring System for War-End Soldiers Using CNN. IGI Global."},{"key":"e_1_3_2_1_17_1","unstructured":"Marco Sandri and et. al. 2006. Variable selection using random forests. In Data analysis classification and the forward search."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Murugan Sankaradas and et. al. 2009. A massively parallel coprocessor for convolutional neural networks. In ASAP.","DOI":"10.1109\/ASAP.2009.25"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Hardik Sharma and et. al. 2018. Bit fusion: Bit-level dynamically composable architecture for accelerating deep neural network. In ISCA.","DOI":"10.1109\/ISCA.2018.00069"},{"volume-title":"Drq: dynamic region-based quantization for deep neural network acceleration","author":"Song 0.","key":"e_1_3_2_1_20_1","unstructured":"Z. Song and et. al. 2020. Drq: dynamic region-based quantization for deep neural network acceleration. In ISCA. IEEE."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Zhuoran Song and et. al. 2020. VR-DANN: Real-Time Video Recognition via Decoder-Assisted Neural Network Acceleration. In MICRO.","DOI":"10.1109\/MICRO50266.2020.00063"},{"key":"e_1_3_2_1_22_1","unstructured":"Mengshu Sun and et. al. 2022. FILM-QNN: Efficient FPGA Acceleration of Deep Neural Networks with Intra-Layer Mixed-Precision Quantization(FPGA \u201922)."},{"key":"e_1_3_2_1_23_1","unstructured":"Xilinx. 2020. https:\/\/www.xilinx.com\/support\/documentation\/sw_manuals\/ xilinx2020_2\/ug888-vivado-design-flows-overview-tutorial.pdf."},{"key":"e_1_3_2_1_24_1","unstructured":"Xiangyu Zhang and et. al. 2015. Accelerating Very Deep Convolutional Networks for Classification and Detection. In CoRR."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Shixuan Zheng and et. al. 2018. An efficient kernel transformation architecture for binary-and ternary-weight neural network inference. In DAC.","DOI":"10.1145\/3195970.3195988"},{"key":"e_1_3_2_1_26_1","volume-title":"Incremental network quantization: Towards lossless cnns with low-precision weights. arXiv preprint arXiv:1702.03044","author":"Aojun Zhou","year":"2017","unstructured":"Aojun Zhou and et. al. 2017. Incremental network quantization: Towards lossless cnns with low-precision weights. arXiv preprint arXiv:1702.03044 (2017)."},{"key":"e_1_3_2_1_27_1","volume-title":"Dorefa-net: Training low bitwidth convolutional neural networks with low bitwidth gradients. arXiv preprint arXiv:1606.06160","author":"Shuchang Zhou","year":"2016","unstructured":"Shuchang Zhou and et. al. 2016. Dorefa-net: Training low bitwidth convolutional neural networks with low bitwidth gradients. arXiv preprint arXiv:1606.06160 (2016)."}],"event":{"name":"ICPP 2023: 52nd International Conference on Parallel Processing","acronym":"ICPP 2023","location":"Salt Lake City UT USA"},"container-title":["Proceedings of the 52nd International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3605573.3605580","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3605573.3605580","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3605573.3605580","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:49:04Z","timestamp":1750182544000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3605573.3605580"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,7]]},"references-count":27,"alternative-id":["10.1145\/3605573.3605580","10.1145\/3605573"],"URL":"https:\/\/doi.org\/10.1145\/3605573.3605580","relation":{},"subject":[],"published":{"date-parts":[[2023,8,7]]},"assertion":[{"value":"2023-09-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}