{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T21:03:43Z","timestamp":1777928623734,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,6,22]],"date-time":"2021-06-22T00:00:00Z","timestamp":1624320000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,6,22]]},"DOI":"10.1145\/3453688.3461738","type":"proceedings-article","created":{"date-parts":[[2021,6,18]],"date-time":"2021-06-18T23:13:45Z","timestamp":1624058025000},"page":"157-162","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["3U-EdgeAI"],"prefix":"10.1145","author":[{"given":"Yao","family":"Chen","sequence":"first","affiliation":[{"name":"Advanced Digital Sciences Center, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cole","family":"Hawkins","sequence":"additional","affiliation":[{"name":"University of California, Santa Barbara, Santa Barbara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kaiqi","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of California, Santa Barbara, Santa Barbara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zheng","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of California, Santa Barbara, Santa Barbara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cong","family":"Hao","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,6,22]]},"reference":[{"key":"e_1_3_2_3_1_1","doi-asserted-by":"crossref","unstructured":"H. Alemdar et al. 2017. Ternary neural networks for resource-efficient AI applications. In IJCNN.","DOI":"10.1109\/IJCNN.2017.7966166"},{"key":"e_1_3_2_3_2_1","unstructured":"Yoshua Bengio et al. 2013. Estimating or propagating gradients through stochastic neurons for conditional computation. arXiv preprint arXiv:1308.3432 (2013)."},{"key":"e_1_3_2_3_3_1","unstructured":"Giuseppe G Calvi et al. 2019. Tucker tensor layer in fully connected neural networks. arXiv preprint arXiv:1903.06133 (2019)."},{"key":"e_1_3_2_3_4_1","doi-asserted-by":"crossref","unstructured":"Yao Chen et al. 2019. Cloud-DNN: An Open Framework for Mapping DNN Models to Cloud FPGAs. In FPGA.","DOI":"10.1145\/3289602.3293915"},{"key":"e_1_3_2_3_5_1","doi-asserted-by":"crossref","unstructured":"Yao Chen et al. 2019. T-DLA: An Open-source Deep Learning Acceleratorfor Ternarized DNN Models on Embedded FPGA. ISVLSI (2019).","DOI":"10.1109\/ISVLSI.2019.00012"},{"key":"e_1_3_2_3_6_1","doi-asserted-by":"crossref","unstructured":"Gong Cheng et al. 2019. \"\"L2Q: An Ultra-Low Loss Quantization Method for DNN Compression. (2019).","DOI":"10.1109\/IJCNN.2019.8851699"},{"key":"e_1_3_2_3_7_1","unstructured":"M. Courbariaux et al. 2016. Binarynet: training deep neural networks with weights and activations constrained to +1 or -1. arXiv (2016)."},{"key":"e_1_3_2_3_8_1","volume-title":"Binaryconnect: Training deep neural networks with binary weights during propagations. In Advances in neural information processing systems. 3123--3131.","author":"Courbariaux Matthieu","year":"2015","unstructured":"Matthieu Courbariaux, Yoshua Bengio, and Jean-Pierre David. 2015. Binaryconnect: Training deep neural networks with binary weights during propagations. In Advances in neural information processing systems. 3123--3131."},{"key":"e_1_3_2_3_9_1","unstructured":"Timur Garipov et al. 2016. Ultimate tensorization: compressing convolutional and fc layers alike. arXiv preprint arXiv:1611.03214 (2016)."},{"key":"e_1_3_2_3_10_1","doi-asserted-by":"crossref","unstructured":"Cheng Gong et al. 2020. VecQ: Minimal loss dnn model compression with vectorized weight quantization. IEEE Trans. Comput. (2020).","DOI":"10.1109\/TC.2020.2995593"},{"key":"e_1_3_2_3_11_1","volume-title":"Hardware-oriented Approximation of Convolutional Neural Networks. CoRR abs\/1604.03168","author":"Gysel Philipp","year":"2016","unstructured":"Philipp Gysel and other. 2016. Hardware-oriented Approximation of Convolutional Neural Networks. CoRR abs\/1604.03168 (2016)."},{"key":"e_1_3_2_3_12_1","volume-title":"Deep Compression: Compressing Deep Neural Network with Pruning, Trained Quantization and Huffman Coding.","author":"Song Han","year":"2016","unstructured":"Song Han et al. 2016. Deep Compression: Compressing Deep Neural Network with Pruning, Trained Quantization and Huffman Coding. (2016)."},{"key":"e_1_3_2_3_13_1","doi-asserted-by":"crossref","unstructured":"Cong Hao et al. 2019. FPGA\/DNN Co-Design: An Efficient Design Methodology for IoT Intelligence on the Edge. In DAC.","DOI":"10.1145\/3316781.3317829"},{"key":"e_1_3_2_3_14_1","unstructured":"C. Hao et al. 2021. Enabling Design Methodologies and Future Trends for Edge AI: Specialization and Co-design. IEEE Design Test (2021) 1--1."},{"key":"e_1_3_2_3_15_1","volume-title":"Towards Compact Neural Networks via End-to-End Training: A Bayesian Tensor Approach with Automatic Rank Determination. arXiv preprint arXiv:2010.08689","author":"Hawkins Cole","year":"2020","unstructured":"Cole Hawkins, Xing Liu, and Zheng Zhang. 2020. Towards Compact Neural Networks via End-to-End Training: A Bayesian Tensor Approach with Automatic Rank Determination. arXiv preprint arXiv:2010.08689 (2020)."},{"key":"e_1_3_2_3_16_1","volume-title":"Bayesian Tensorized Neural Networks with Automatic Rank Selection. arXiv preprint arXiv:1905.10478","author":"Hawkins Cole","year":"2019","unstructured":"Cole Hawkins and Zheng Zhang. 2019. Bayesian Tensorized Neural Networks with Automatic Rank Selection. arXiv preprint arXiv:1905.10478 (2019)."},{"key":"e_1_3_2_3_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/2512329"},{"key":"e_1_3_2_3_18_1","doi-asserted-by":"publisher","DOI":"10.5555\/2567709.2502622"},{"key":"e_1_3_2_3_19_1","volume-title":"Tensorized embedding layers for efficient model compression. arXiv preprint arXiv:1901.10787","author":"Khrulkov Valentin","year":"2019","unstructured":"Valentin Khrulkov, Oleksii Hrinchuk, Leyla Mirvakhabova, and Ivan Oseledets. 2019. Tensorized embedding layers for efficient model compression. arXiv preprint arXiv:1901.10787 (2019)."},{"key":"e_1_3_2_3_20_1","series-title":"SIAM review 51, 3","volume-title":"Tensor decompositions and applications","author":"Kolda Tamara G","year":"2009","unstructured":"Tamara G Kolda and Brett W Bader. 2009. Tensor decompositions and applications. SIAM review 51, 3 (2009), 455--500."},{"key":"e_1_3_2_3_21_1","volume-title":"Speeding-up convolutional neural networks using fine-tuned cp-decomposition. arXiv preprint arXiv:1412.6553","author":"Lebedev Vadim","year":"2014","unstructured":"Vadim Lebedev, Yaroslav Ganin, Maksim Rakhuba, Ivan Oseledets, and Victor Lempitsky. 2014. Speeding-up convolutional neural networks using fine-tuned cp-decomposition. arXiv preprint arXiv:1412.6553 (2014)."},{"key":"e_1_3_2_3_22_1","doi-asserted-by":"crossref","unstructured":"Cong Leng et al. 2018. Extremely Low Bit Neural Network: Squeeze the Last Bit Out With ADMM. In AAAI.","DOI":"10.1609\/aaai.v32i1.11713"},{"key":"e_1_3_2_3_23_1","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","volume":"32","author":"Cong","unstructured":"Cong Leng et al. 2018. Extremely low bit neural network: Squeeze the last bit out with admm. In Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 32."},{"key":"e_1_3_2_3_24_1","volume-title":"Proc. NIPS Workshop Efficient Methods Deep Neural Network","author":"Li F","year":"2016","unstructured":"F Li and B Liu. 2016. Ternary Weight Networks. Proc. NIPS Workshop Efficient Methods Deep Neural Network (2016)."},{"key":"e_1_3_2_3_25_1","volume-title":"Proceedings of the 30th International Conference on Neural Information Processing Systems. 2378--2386","author":"Liu Qiang","year":"2016","unstructured":"Qiang Liu and Dilin Wang. 2016. Stein variational Gradient descent: a general purpose Bayesian inference algorithm. In Proceedings of the 30th International Conference on Neural Information Processing Systems. 2378--2386."},{"key":"e_1_3_2_3_26_1","doi-asserted-by":"crossref","unstructured":"S. Liu et al. 2011. Real-time object tracking system on FPGAs. In SAAHPC.","DOI":"10.1109\/SAAHPC.2011.22"},{"key":"e_1_3_2_3_27_1","doi-asserted-by":"crossref","unstructured":"H. Nakahara et al. 2017. A fully connected layer elimination for a binarizec convolutional neural network on an FPGA. In FPL.","DOI":"10.23919\/FPL.2017.8056771"},{"key":"e_1_3_2_3_28_1","unstructured":"Maxim Naumov et al. 2019. Deep learning recommendation model for personalization and recommendation systems. arXiv preprint arXiv:1906.00091 (2019)."},{"key":"e_1_3_2_3_29_1","unstructured":"Alexander Novikov et al. 2015. Tensorizing neural networks. In Advances in neural information processing systems. 442--450."},{"key":"e_1_3_2_3_30_1","doi-asserted-by":"publisher","DOI":"10.1137\/090752286"},{"key":"e_1_3_2_3_31_1","volume-title":"FPGA Based Implementation of Deep Neural Networks Using On-chip Memory Only. CoRR","author":"Park Jinhwan","year":"2016","unstructured":"Jinhwan Park and Wonyong Sung. 2016. FPGA Based Implementation of Deep Neural Networks Using On-chip Memory Only. CoRR (2016)."},{"key":"e_1_3_2_3_32_1","volume-title":"Prost-Boucle et al","author":"A.","year":"2017","unstructured":"A. Prost-Boucle et al. 2017. Scalable high-performance architecture for convolutional ternary neural networks on FPGA. In FPL."},{"key":"e_1_3_2_3_33_1","unstructured":"Jiantao Qiu et al. 2016. Going deeper with embedded fpga platform for convolutional neural network. In FPGA. 26--35."},{"key":"e_1_3_2_3_34_1","doi-asserted-by":"crossref","unstructured":"Surat Teerapittayanon et al. 2017. Distributed Deep Neural Networks Over the Cloud the Edge and End Devices. In ICDCS. 328--339.","DOI":"10.1109\/ICDCS.2017.226"},{"key":"e_1_3_2_3_35_1","volume-title":"2017 International Joint Conference on Neural Networks (IJCNN). IEEE, 4451--4458","author":"Andros","unstructured":"Andros Tjandra et al. 2017. Compressing recurrent neural network with tensor train. In 2017 International Joint Conference on Neural Networks (IJCNN). IEEE, 4451--4458."},{"key":"e_1_3_2_3_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3020078.3021744"},{"key":"e_1_3_2_3_37_1","doi-asserted-by":"crossref","unstructured":"Peisong Wang et al. 2018. Two-step quantization for low-bit neural networks. In CVPR.","DOI":"10.1109\/CVPR.2018.00460"},{"key":"e_1_3_2_3_38_1","unstructured":"Yue Wang et al. 2019. E2-Train: Training State-of-the-art CNNs with Over 80% Energy Savings. In NeurIPS."},{"key":"e_1_3_2_3_39_1","unstructured":"Miao Yin et al. 2020. Compressing Recurrent Neural Networks Using Hierarchical Tucker Tensor Decomposition. arXiv preprint arXiv:2005.04366 (2020)."},{"key":"e_1_3_2_3_40_1","volume-title":"On-FPGA Training with Ultra Memory Reduction: A Low-Precision Tensor Method. ICLR Workshop of Hardware Aware Efficient Training","author":"Kaiqi","year":"2021","unstructured":"Kaiqi Zhang et al. 2021. On-FPGA Training with Ultra Memory Reduction: A Low-Precision Tensor Method. ICLR Workshop of Hardware Aware Efficient Training (2021)."},{"key":"e_1_3_2_3_41_1","doi-asserted-by":"crossref","unstructured":"Xiaofan Zhang et al. 2017. Machine learning on FPGAs to face the IoT revolution. In ICCAD.","DOI":"10.1109\/ICCAD.2017.8203875"},{"key":"e_1_3_2_3_42_1","doi-asserted-by":"crossref","unstructured":"Xiaofan Zhang et al. 2018. DNNBuilder: an automated tool for building high-performance Dnn hardware accelerators for FPGAs. In ICCAD.","DOI":"10.1145\/3240765.3240801"},{"key":"e_1_3_2_3_43_1","doi-asserted-by":"crossref","unstructured":"Ritchie Zhao et al. 2017. Accelerating Binarized Convolutional Neural Networks with Software-Programmable FPGAs. In FPGA.","DOI":"10.1145\/3020078.3021741"},{"key":"e_1_3_2_3_44_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.image.2018.03.017"}],"event":{"name":"GLSVLSI '21: Great Lakes Symposium on VLSI 2021","location":"Virtual Event USA","acronym":"GLSVLSI '21","sponsor":["SIGDA ACM Special Interest Group on Design Automation"]},"container-title":["Proceedings of the 2021 Great Lakes Symposium on VLSI"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3453688.3461738","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3453688.3461738","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:28:47Z","timestamp":1750195727000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3453688.3461738"}},"subtitle":["Ultra-Low Memory Training, Ultra-Low Bitwidth Quantization, and Ultra-Low Latency Acceleration"],"short-title":[],"issued":{"date-parts":[[2021,6,22]]},"references-count":44,"alternative-id":["10.1145\/3453688.3461738","10.1145\/3453688"],"URL":"https:\/\/doi.org\/10.1145\/3453688.3461738","relation":{},"subject":[],"published":{"date-parts":[[2021,6,22]]},"assertion":[{"value":"2021-06-22","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}