{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T22:42:25Z","timestamp":1781908945854,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,11,5]],"date-time":"2018-11-05T00:00:00Z","timestamp":1541376000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,11,5]]},"DOI":"10.1145\/3240765.3240856","type":"proceedings-article","created":{"date-parts":[[2018,11,6]],"date-time":"2018-11-06T13:36:57Z","timestamp":1541511417000},"page":"1-8","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":66,"title":["TGPA"],"prefix":"10.1145","author":[{"given":"Xuechao","family":"Wei","sequence":"first","affiliation":[{"name":"Peking University, China and Falcon Computing Solutions, Inc."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yun","family":"Liang","sequence":"additional","affiliation":[{"name":"Peking University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiuhong","family":"Li","sequence":"additional","affiliation":[{"name":"Peking University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cody Hao","family":"Yu","sequence":"additional","affiliation":[{"name":"University of California and Falcon Computing Solutions, Inc."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Falcon Computing Solutions, Inc."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jason","family":"Cong","sequence":"additional","affiliation":[{"name":"Peking University, China and University of California and Falcon Computing Solutions, Inc."}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2018,11,5]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","unstructured":"M. Alwani H. Chen M. Ferdman and P. Milder. 2016. Fused-layer CNN accelerators. In MICRO.","DOI":"10.5555\/3195638.3195664"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","unstructured":"U. Aydonat S. O'Connell D. Capalija A. Ling and G. Chiu. 2017. An OpenCL Deep Learning Accelerator on Arria 10. In FPGA. 10.1145\/3020078.3021738","DOI":"10.1145\/3020078.3021738"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","unstructured":"G. Chaitin. 2004. Register Allocation and Spilling via Graph Coloring. SIGPLAN Not. (2004). 10.1145\/989393.989403","DOI":"10.1145\/989393.989403"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","unstructured":"J. Cong P. Wei C. H. Yu and P. Zhou. 2017. Bandwidth Optimization Through On-Chip Memory Restructuring for HLS. In DAC. 10.1145\/3061639.3062208","DOI":"10.1145\/3061639.3062208"},{"key":"e_1_3_2_1_5_1","unstructured":"H. Gao Z. Liu K. Q. Weinberger and L. van der Maaten. 2017. Deep Residual Learning for Image Recognition. In CVPR."},{"key":"e_1_3_2_1_6_1","volume":"201","author":"Guan Y.","unstructured":"Y. Guan, H. Liang, N. Xu, W. Wang, S. Shi, X. Chen, G. Sun, W. Zhang, and J. Cong. 2017. FP-DNN: An Automated Framework for Mapping Deep Neural Networks onto FPGAs with RTL-HLS Hybrid Templates. In FCCM.","journal-title":"J. Cong."},{"key":"e_1_3_2_1_7_1","volume":"201","author":"He K.","unstructured":"K. He, X. Zhang, S. Ren, and J. Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR.","journal-title":"J. Sun."},{"key":"e_1_3_2_1_8_1","unstructured":"Intel. 2016. \"Not so fast FFT\": Winograd. https:\/\/ai.intel.com\/winograd\/. (2016)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080246"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","unstructured":"A. Krizhevsky I. Sutskever and G. E. Hinton. 2012. ImageNet Classification with Deep Convolutional Neural Networks. In NIPS.","DOI":"10.5555\/2999134.2999257"},{"key":"e_1_3_2_1_11_1","unstructured":"H. T. Kung and C. E. Leiserson. 1979. Algorithms for VLSI Processor Arrays."},{"key":"e_1_3_2_1_12_1","unstructured":"H. Li X. Fan L. Jiao W. Cao X. Zhou and L. Wang. 2016. A High Performance FPGA-based Accelerator for Large-scale Convolutional Neural Networks. In FPL."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","unstructured":"Liqiang Lu and Yun Liang. 2018. SpWA: An Efficient Sparse Winograd Convolutional Neural Networks Accelerator on FPGAs. In DAC. 10.1145\/3195970.3196120","DOI":"10.1145\/3195970.3196120"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"L. Lu Y. Liang Q. Xiao and S. Yan. 2017. Evaluating Fast Algorithms for Convolutional Neural Networks on FPGAs. In FCCM.","DOI":"10.1109\/FCCM.2017.64"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3020078.3021736"},{"key":"e_1_3_2_1_16_1","volume-title":"s. Seo","author":"Ma Y.","year":"2017","unstructured":"Y. Ma, M. Kim, Y. Cao, S. Vrudhula, and J. s. Seo. 2017. End-to-end scalable FPGA accelerator for deep residual networks. In ISCAS."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"K. Ovtcharov O. Ruwase J. Kim J. Fowers K. Strauss and E. Chung. 2015. Toward Accelerating Deep Learning at Scale Using Specialized Hardware in the Datacenter. In Hot Chips.","DOI":"10.1109\/HOTCHIPS.2015.7477459"},{"key":"e_1_3_2_1_18_1","unstructured":"PyTorch. 2018. https:\/\/pytorch.org. (2018)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","unstructured":"Y. Shen M. Ferdman and P. Milder. 2017. Maximizing CNN Accelerator Efficiency Through Resource Partitioning. In ISCA. 10.1145\/3079856.3080221","DOI":"10.1145\/3079856.3080221"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"C. Szegedy W. Liu Y. Jia P. Sermanet S. E. Reed D. Anguelov D. Erhan V. Vanhoucke and A. Rabinovich. 2014. Going Deeper with Convolutions. In CVPR.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3020078.3021744"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"S. I. Venieris and C. S. Bouganis. 2016. fpgaConvNet: A Framework for Mapping Convolutional Neural Networks on FPGAs. In FCCM.","DOI":"10.1109\/FCCM.2016.22"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3061639.3062207"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","unstructured":"Q. Xiao Y. Liang L. Lu S. Yan and Y. Tai. 2017. Exploring Heterogeneous Algorithms for Accelerating Deep Convolutional Neural Networks on FPGAs. In DAC. 10.1145\/3061639.3062244","DOI":"10.1145\/3061639.3062244"},{"key":"e_1_3_2_1_25_1","unstructured":"Xilinx. 2018. Large FPGA Methodology Guide. https:\/\/www.xilinx.com\/support\/documentation\/sw_manuals\/xilinx13_4\/ug872_largefpga.pdf. (2018)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2966986.2967011"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/2684746.2689060"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2934583.2934644"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3020078.3021698"}],"event":{"name":"ICCAD '18: IEEE\/ACM INTERNATIONAL CONFERENCE ON COMPUTER-AIDED DESIGN","location":"San Diego California","acronym":"ICCAD '18","sponsor":["IEEE-EDS Electronic Devices Society","IEEE CAS","IEEE CEDA"]},"container-title":["Proceedings of the International Conference on Computer-Aided Design"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240765.3240856","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3240765.3240856","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T00:57:33Z","timestamp":1750208253000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240765.3240856"}},"subtitle":["tile-grained pipeline architecture for low latency CNN inference"],"short-title":[],"issued":{"date-parts":[[2018,11,5]]},"references-count":29,"alternative-id":["10.1145\/3240765.3240856","10.1145\/3240765"],"URL":"https:\/\/doi.org\/10.1145\/3240765.3240856","relation":{},"subject":[],"published":{"date-parts":[[2018,11,5]]},"assertion":[{"value":"2018-11-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}