{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,28]],"date-time":"2026-08-28T09:47:41Z","timestamp":1787910461492,"version":"build-2784847793"},"reference-count":82,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2018,10,6]],"date-time":"2018-10-06T00:00:00Z","timestamp":1538784000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001843","name":"Science and Engineering Research Board","doi-asserted-by":"publisher","award":["ECR\/2017\/000622"],"award-info":[{"award-number":["ECR\/2017\/000622"]}],"id":[{"id":"10.13039\/501100001843","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2020,2]]},"DOI":"10.1007\/s00521-018-3761-1","type":"journal-article","created":{"date-parts":[[2018,10,6]],"date-time":"2018-10-06T03:14:51Z","timestamp":1538795691000},"page":"1109-1139","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":328,"title":["A survey of FPGA-based accelerators for convolutional neural networks"],"prefix":"10.1007","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2908-993X","authenticated-orcid":false,"given":"Sparsh","family":"Mittal","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2018,10,6]]},"reference":[{"key":"3761_CR1","unstructured":"Ovtcharov K, Ruwase O, Kim J-Y, Fowers J, Strauss K, Chung ES (2015) Accelerating deep convolutional neural networks using specialized hardware. Microsoft Research Whitepaper vol 2, no 11"},{"key":"3761_CR2","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1145\/2788396","volume":"47","author":"S Mittal","year":"2015","unstructured":"Mittal S, Vetter J (2015) A survey of methods for analyzing and improving GPU energy efficiency. ACM Comput Surv 47:19","journal-title":"ACM Comput Surv"},{"key":"3761_CR3","doi-asserted-by":"crossref","unstructured":"Zhao R, Song W, Zhang W, Xing T, Lin J-H, Srivastava M\u00a0B, Gupta R, Zhang Z (2017) Accelerating binarized convolutional neural networks with software-programmable FPGAs. In: FPGA, pp 15\u201324","DOI":"10.1145\/3020078.3021741"},{"key":"3761_CR4","doi-asserted-by":"crossref","unstructured":"Suda N, Chandra V, Dasika G, Mohanty A, Ma Y, Vrudhula S, Seo J-s, Cao Y (2016) Throughput-optimized OpenCL-based FPGA accelerator for large-scale convolutional neural networks. In International symposium on field-programmable gate arrays, pp 16\u201325","DOI":"10.1145\/2847263.2847276"},{"key":"3761_CR5","doi-asserted-by":"crossref","unstructured":"Zhang C, Prasanna V (2017) Frequency domain acceleration of convolutional neural networks on CPU-FPGA shared memory system. In: International symposium on field-programmable gate arrays, pp 35\u201344","DOI":"10.1145\/3020078.3021727"},{"key":"3761_CR6","unstructured":"Courbariaux M, Hubara I, Soudry D, El-Yaniv R, Bengio Y (2016) Binarized neural networks: training deep neural networks with weights and activations constrained to +\u00a01 or \u2212\u00a01. arXiv preprint \narXiv:1602.02830"},{"key":"3761_CR7","doi-asserted-by":"crossref","unstructured":"Zhang C, Fang Z, Zhou P, Pan P, Cong J (2016) Caffeine: towards uniformed representation and acceleration for deep convolutional neural networks. In: International conference on computer-aided design (ICCAD), pp 1\u20138","DOI":"10.1145\/2966986.2967011"},{"issue":"4","key":"3761_CR8","first-page":"62","volume":"13","author":"M Motamedi","year":"2017","unstructured":"Motamedi M, Gysel P, Ghiasi S (2017) PLACID: a platform for FPGA-based accelerator creation for DCNNs. ACM Trans Multimed Comput Commun Appl (TOMM) 13(4):62","journal-title":"ACM Trans Multimed Comput Commun Appl (TOMM)"},{"key":"3761_CR9","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In: IEEE conference on computer vision and pattern recognition, pp 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"3761_CR10","doi-asserted-by":"publisher","first-page":"1217","DOI":"10.1109\/TCSII.2017.2690919","volume":"64","author":"S Moini","year":"2017","unstructured":"Moini S, Alizadeh B, Emad M, Ebrahimpour R (2017) A resource-limited hardware accelerator for convolutional neural networks in embedded vision applications. IEEE Trans Circuits Syst II Express Briefs 64:1217\u20131221","journal-title":"IEEE Trans Circuits Syst II Express Briefs"},{"key":"3761_CR11","doi-asserted-by":"publisher","first-page":"113","DOI":"10.1109\/LES.2017.2743247","volume":"9","author":"K Abdelouahab","year":"2017","unstructured":"Abdelouahab K, Pelcat M, S\u00e9rot J, Bourrasset C, Berry F (2017) Tactics to directly map CNN graphs on embedded FPGAs. IEEE Embed Syst Lett 9:113\u2013116","journal-title":"IEEE Embed Syst Lett"},{"key":"3761_CR12","unstructured":"Xilinx (2015) Ultrascale architecture FPGAs memory interface solutions v7.0. Technical Report"},{"issue":"8","key":"3761_CR13","doi-asserted-by":"publisher","first-page":"1430002","DOI":"10.1142\/S0218126614300025","volume":"23","author":"S Mittal","year":"2014","unstructured":"Mittal S (2014) A survey of techniques for managing and leveraging caches in GPUs. J Circuits Syst Comput (JCSC) 23(8):1430002","journal-title":"J Circuits Syst Comput (JCSC)"},{"key":"3761_CR14","unstructured":"Chang AXM, Zaidy A, Gokhale V, Culurciello E (2017) Compiling deep learning models for custom hardware accelerators. arXiv preprint \narXiv:1708.00117"},{"issue":"4","key":"3761_CR15","doi-asserted-by":"publisher","first-page":"69:1","DOI":"10.1145\/2788396","volume":"47","author":"S Mittal","year":"2015","unstructured":"Mittal S, Vetter J (2015) A survey of CPU\u2013GPU heterogeneous computing techniques. ACM Comput Surv 47(4):69:1\u201369:35","journal-title":"ACM Comput Surv"},{"key":"3761_CR16","first-page":"18","volume":"14","author":"Y Li","year":"2018","unstructured":"Li Y, Liu Z, Xu K, Yu H, Ren F (2018) A GPU-outperforming FPGA accelerator architecture for binary convolutional neural networks. ACM J Emerg Technol Comput (JETC) 14:18","journal-title":"ACM J Emerg Technol Comput (JETC)"},{"key":"3761_CR17","doi-asserted-by":"crossref","unstructured":"Umuroglu Y, Fraser NJ, Gambardella G, Blott M, Leong P, Jahre M, Vissers K (2017) FINN: a framework for fast, scalable binarized neural network inference. In: International symposium on field-programmable gate arrays, pp 65\u201374","DOI":"10.1145\/3020078.3021744"},{"key":"3761_CR18","doi-asserted-by":"crossref","unstructured":"Park J, Sung W (2016) FPGA based implementation of deep neural networks using on-chip memory only. In: International conference on acoustics, speech and signal processing (ICASSP), pp 1011\u20131015","DOI":"10.1109\/ICASSP.2016.7471828"},{"key":"3761_CR19","doi-asserted-by":"crossref","unstructured":"Peemen M, Setio AA, Mesman B, Corporaal H (2013) Memory-centric accelerator design for convolutional neural networks. In: International conference on computer design (ICCD), pp 13\u201319","DOI":"10.1109\/ICCD.2013.6657019"},{"key":"3761_CR20","doi-asserted-by":"crossref","unstructured":"Rahman A, Oh S, Lee J, Choi K (2017) Design space exploration of FPGA accelerators for convolutional neural networks. In: 2017 Design, automation & test in Europe conference & exhibition (DATE). IEEE, pp 1147\u20131152","DOI":"10.23919\/DATE.2017.7927162"},{"key":"3761_CR21","doi-asserted-by":"crossref","unstructured":"Zhang C, Li P, Sun G, Guan Y, Xiao B, Cong J (2015) Optimizing FPGA-based accelerator design for deep convolutional neural networks. In: International symposium on field-programmable gate arrays, pp 161\u2013170","DOI":"10.1145\/2684746.2689060"},{"key":"3761_CR22","doi-asserted-by":"crossref","unstructured":"Guan Y, Liang H, Xu N, Wang W, Shi S, Chen X, Sun G, Zhang W, Cong J (2017) FP-DNN: an automated framework for mapping deep neural networks onto FPGAs with RTL-HLS hybrid templates. In: International symposium on field-programmable custom computing machines (FCCM), pp 152\u2013159","DOI":"10.1109\/FCCM.2017.25"},{"key":"3761_CR23","doi-asserted-by":"crossref","unstructured":"Moss DJ, Nurvitadhi E, Sim J, Mishra A, Marr D, Subhaschandra S, Leong PH (2017) High performance binary neural networks on the Xeon\u00a0+\u00a0FPGA platform. In: International conference on field programmable logic and applications (FPL), pp 1\u20134","DOI":"10.23919\/FPL.2017.8056823"},{"key":"3761_CR24","doi-asserted-by":"crossref","unstructured":"Ma Y, Cao Y, Vrudhula S, Seo J-s (2017) Optimizing loop operation and dataflow in FPGA acceleration of deep convolutional neural networks. In: International symposium on field-programmable gate arrays, pp 45\u201354","DOI":"10.1145\/3020078.3021736"},{"key":"3761_CR25","doi-asserted-by":"crossref","unstructured":"Yonekawa H, Nakahara H (2017) On-chip memory based binarized convolutional deep neural network applying batch normalization free technique on an FPGA. In: IEEE international parallel and distributed processing symposium workshops (IPDPSW), pp 98\u2013105","DOI":"10.1109\/IPDPSW.2017.95"},{"key":"3761_CR26","doi-asserted-by":"crossref","unstructured":"Nurvitadhi E, Sheffield D, Sim J, Mishra A, Venkatesh G, Marr D (2016) Accelerating binarized neural networks: comparison of FPGA, CPU, GPU, and ASIC. In: International conference on field-programmable technology (FPT), pp 77\u201384","DOI":"10.1109\/FPT.2016.7929192"},{"key":"3761_CR27","doi-asserted-by":"crossref","unstructured":"Zhang Y, Wang C, Gong L, Lu Y, Sun F, Xu C, Li X, Zhou X (2017) A power-efficient accelerator based on FPGAs for LSTM network. In: International conference on cluster computing (CLUSTER), pp 629\u2013630","DOI":"10.1109\/CLUSTER.2017.45"},{"key":"3761_CR28","unstructured":"Feng G, Hu Z, Chen S, Wu F (2016) Energy-efficient and high-throughput FPGA-based accelerator for convolutional neural networks. In: International conference on solid-state and integrated circuit technology (ICSICT), pp 624\u2013626"},{"key":"3761_CR29","unstructured":"Wang D, An J, Xu K (2016) PipeCNN: an OpenCL-based FPGA accelerator for large-scale convolution neuron networks. arXiv preprint \narXiv:1611.02450"},{"key":"3761_CR30","unstructured":"Liu Z, Dou Y, Jiang J, Xu J (2016) Automatic code generation of convolutional neural networks in FPGA implementation. In: International conference on field-programmable technology (FPT). IEEE, pp 61\u201368"},{"key":"3761_CR31","doi-asserted-by":"crossref","unstructured":"Samragh M, Ghasemzadeh M, Koushanfar F (2017) Customizing neural networks for efficient FPGA implementation. In: International symposium on field-programmable custom computing machines (FCCM), pp 85\u201392","DOI":"10.1109\/FCCM.2017.43"},{"key":"3761_CR32","doi-asserted-by":"crossref","unstructured":"Podili A, Zhang C, Prasanna V (2017) Fast and efficient implementation of convolutional neural networks on FPGA. In: International conference on application-specific systems, architectures and processors (ASAP), pp 11\u201318","DOI":"10.1109\/ASAP.2017.7995253"},{"key":"3761_CR33","doi-asserted-by":"crossref","unstructured":"Fraser NJ, Umuroglu Y, Gambardella G, Blott M, Leong P, Jahre M, Vissers K (2017) Scaling binarized neural networks on reconfigurable logic. In: Workshop on parallel programming and run-time management techniques for many-core architectures and design tools and architectures for multicore embedded computing platforms (PARMA-DITAM), pp 25\u201330","DOI":"10.1145\/3029580.3029586"},{"key":"3761_CR34","doi-asserted-by":"crossref","unstructured":"Xiao Q, Liang Y, Lu L, Yan S, Tai Y-W (2017) Exploring heterogeneous algorithms for accelerating deep convolutional neural networks on FPGAs. In: Design automation conference, p\u00a062","DOI":"10.1145\/3061639.3062244"},{"key":"3761_CR35","doi-asserted-by":"crossref","unstructured":"Ma Y, Cao Y, Vrudhula S, Seo J-s (2017) An automatic RTL compiler for high-throughput FPGA implementation of diverse deep convolutional neural networks. In: International conference on field programmable logic and applications (FPL), pp 1\u20138","DOI":"10.23919\/FPL.2017.8056824"},{"key":"3761_CR36","doi-asserted-by":"crossref","unstructured":"Rahman A, Lee J, Choi K (2016) Efficient FPGA acceleration of convolutional neural networks using logical-3D compute array. In: Design, automation & test in Europe(DATE), pp 1393\u20131398","DOI":"10.3850\/9783981537079_0833"},{"issue":"3","key":"3761_CR37","first-page":"17","volume":"10","author":"Z Liu","year":"2017","unstructured":"Liu Z, Dou Y, Jiang J, Xu J, Li S, Zhou Y, Xu Y (2017) Throughput-optimized FPGA accelerator for deep convolutional neural networks. ACM Trans Reconfig Technol Syst (TRETS) 10(3):17","journal-title":"ACM Trans Reconfig Technol Syst (TRETS)"},{"key":"3761_CR38","doi-asserted-by":"crossref","unstructured":"Zhang X, Liu X, Ramachandran A, Zhuge C, Tang S, Ouyang P, Cheng Z, Rupnow K, Chen D (2017) High-performance video content recognition with long-term recurrent convolutional network for FPGA. In: International conference on field programmable logic and applications (FPL), pp 1\u20134","DOI":"10.23919\/FPL.2017.8056833"},{"key":"3761_CR39","unstructured":"Ma Y, Suda N, Cao Y, Seo J-s, Vrudhula S (2016) Scalable and modularized RTL compilation of convolutional neural networks onto FPGA. In: International conference on field programmable logic and applications (FPL), pp 1\u20138"},{"issue":"2","key":"3761_CR40","doi-asserted-by":"publisher","first-page":"535","DOI":"10.1145\/3140659.3080221","volume":"45","author":"Yongming Shen","year":"2017","unstructured":"Shen Y, Ferdman M, Milder P (2017) Maximizing cnn accelerator efficiency through resource partitioning. In: International symposium on computer architecture, ser. ISCA \u201917, pp 535\u2013547","journal-title":"ACM SIGARCH Computer Architecture News"},{"key":"3761_CR41","doi-asserted-by":"crossref","unstructured":"Aydonat U, O\u2019Connell S, Capalija D, Ling AC, Chiu GR (2017) An OpenCL deep learning accelerator on Arria 10. In: FPGA","DOI":"10.1145\/3020078.3021738"},{"key":"3761_CR42","doi-asserted-by":"crossref","unstructured":"Kim JH, Grady B, Lian R, Brothers J, Anderson JH (2017) FPGA-based CNN inference accelerator synthesized from multi-threaded C software. In: IEEE SOCC","DOI":"10.1109\/SOCC.2017.8226056"},{"key":"3761_CR43","doi-asserted-by":"crossref","unstructured":"Wei X, Yu CH, Zhang P, Chen Y, Wang Y, Hu H, Liang Y, Cong J (2017) Automated systolic array architecture synthesis for high throughput CNN inference on FPGAs. In: Design automation conference (DAC), pp 1\u20136","DOI":"10.1145\/3061639.3062207"},{"key":"3761_CR44","doi-asserted-by":"crossref","unstructured":"Qiu J, Wang J, Yao S, Guo K, Li B, Zhou E, Yu J, Tang T, Xu N, Song S et\u00a0al (2016) Going deeper with embedded FPGA platform for convolutional neural network. In: International symposium on field-programmable gate arrays, pp 26\u201335","DOI":"10.1145\/2847263.2847265"},{"issue":"20","key":"3761_CR45","doi-asserted-by":"publisher","first-page":"e3850","DOI":"10.1002\/cpe.3850","volume":"29","author":"Y Qiao","year":"2017","unstructured":"Qiao Y, Shen J, Xiao T, Yang Q, Wen M, Zhang C (2017) FPGA-accelerated deep convolutional neural networks for high throughput and energy efficiency. Concurr Comput Pract Exp 29(20):e3850","journal-title":"Concurr Comput Pract Exp"},{"issue":"3","key":"3761_CR46","first-page":"31","volume":"13","author":"A Page","year":"2017","unstructured":"Page A, Jafari A, Shea C, Mohsenin T (2017) SPARCNet: a hardware accelerator for efficient deployment of sparse convolutional networks. ACM J Emerg Technol Comput Syst (JETC) 13(3):31","journal-title":"ACM J Emerg Technol Comput Syst (JETC)"},{"key":"3761_CR47","unstructured":"Zhao W, Fu H, Luk W, Yu T, Wang S, Feng B, Ma Y, Yang G (2016) F-CNN: an FPGA-based framework for training convolutional neural networks. In: International conference on application-specific systems, architectures and processors (ASAP), pp 107\u2013114"},{"key":"3761_CR48","doi-asserted-by":"publisher","first-page":"1072","DOI":"10.1016\/j.neucom.2017.09.046","volume":"275","author":"S Liang","year":"2018","unstructured":"Liang S, Yin S, Liu L, Luk W, Wei S (2018) FP-BNN: binarized neural network on FPGA. Neurocomputing 275:1072\u20131086","journal-title":"Neurocomputing"},{"key":"3761_CR49","doi-asserted-by":"crossref","unstructured":"Natale G, Bacis M, Santambrogio MD (2017) On how to design dataflow FPGA-based accelerators for convolutional neural networks. In: IEEE computer society annual symposium on VLSI (ISVLSI), pp 639\u2013644","DOI":"10.1109\/ISVLSI.2017.126"},{"key":"3761_CR50","doi-asserted-by":"crossref","unstructured":"Lu L, Liang Y, Xiao Q, Yan S (2017) Evaluating fast algorithms for convolutional neural networks on FPGAs. In: International symposium on field-programmable custom computing machines (FCCM), pp 101\u2013108","DOI":"10.1109\/FCCM.2017.64"},{"key":"3761_CR51","doi-asserted-by":"crossref","unstructured":"Zhang C, Wu D, Sun J, Sun G, Luo G, Cong J (2016) Energy-efficient CNN implementation on a deeply pipelined FPGA cluster. In: International symposium on low power electronics and design, pp 326\u2013331","DOI":"10.1145\/2934583.2934644"},{"key":"3761_CR52","doi-asserted-by":"crossref","unstructured":"DiCecco R, Lacey G, Vasiljevic J, Chow P, Taylor G, Areibi S (2016) Caffeinated FPGAs: FPGA framework for convolutional neural networks. In: International conference on field-programmable technology (FPT), pp 265\u2013268","DOI":"10.1109\/FPT.2016.7929549"},{"key":"3761_CR53","doi-asserted-by":"crossref","unstructured":"Venieris SI, Bouganis C-S (2016) fpgaConvNet: a framework for mapping convolutional neural networks on FPGAs. In: International symposium on field-programmable custom computing machines (FCCM), pp 40\u201347","DOI":"10.1109\/FCCM.2016.22"},{"key":"3761_CR54","unstructured":"Zeng H, Chen R, Prasanna VK (2017) Optimizing frequency domain implementation of CNNs on FPGAs. Technical report"},{"key":"3761_CR55","unstructured":"Li H, Fan X, Jiao L, Cao W, Zhou X, Wang L (2016) A high performance FPGA-based accelerator for large-scale convolutional neural networks. In: 2016 26th International conference on field programmable logic and applications (FPL). IEEE, pp 1\u20139"},{"key":"3761_CR56","doi-asserted-by":"crossref","unstructured":"Lin J-H, Xing T, Zhao R, Zhang Z, Srivastava M, Tu Z, Gupta RK (2017) Binarized convolutional neural networks with separable filters for efficient hardware acceleration. In: Computer vision and pattern recognition workshop (CVPRW)","DOI":"10.1109\/CVPRW.2017.48"},{"key":"3761_CR57","unstructured":"Nakahara H, Fujii T, Sato S (2017) A fully connected layer elimination for a binarized convolutional neural network on an FPGA. In: International conference on field programmable logic and applications (FPL), pp 1\u20134"},{"key":"3761_CR58","doi-asserted-by":"crossref","unstructured":"Jiao L, Luo C, Cao W, Zhou X, Wang L (2017) Accelerating low bit-width convolutional neural networks with embedded FPGA. In: International conference on field programmable logic and applications (FPL), pp 1\u20134","DOI":"10.23919\/FPL.2017.8056820"},{"key":"3761_CR59","doi-asserted-by":"crossref","unstructured":"Meloni P, Deriu G, Conti F, Loi I, Raffo L, Benini L (2016) Curbing the roofline: a scalable and flexible architecture for CNNs on FPGA. In: ACM international conference on computing frontiers, pp 376\u2013383","DOI":"10.1145\/2903150.2911715"},{"key":"3761_CR60","doi-asserted-by":"crossref","unstructured":"Abdelouahab K, Bourrasset C, Pelcat M, Berry F, Quinton J-C, Serot J (2016) A holistic approach for optimizing DSP block utilization of a CNN implementation on FPGA. In: International conference on distributed smart camera, pp 69\u201375","DOI":"10.1145\/2967413.2967430"},{"key":"3761_CR61","unstructured":"Gankidi PR, Thangavelautham J (2017) FPGA architecture for deep learning and its application to planetary robotics. In: IEEE aerospace conference, pp 1\u20139"},{"key":"3761_CR62","doi-asserted-by":"crossref","unstructured":"Venieris SI, Bouganis C-S (2017) Latency-driven design for FPGA-based convolutional neural networks. In: International conference on field programmable logic and applications (FPL), pp 1\u20138","DOI":"10.23919\/FPL.2017.8056828"},{"key":"3761_CR63","doi-asserted-by":"crossref","unstructured":"Shen Y, Ferdman M, Milder P (2017) Escher: a CNN accelerator with flexible buffering to minimize off-chip transfer. In: International symposium on field-programmable custom computing machines (FCCM)","DOI":"10.1109\/FCCM.2017.47"},{"key":"3761_CR64","doi-asserted-by":"crossref","unstructured":"Zhang J, Li J (2017) Improving the performance of OpenCL-based FPGA accelerator for convolutional neural network. In: FPGA, pp 25\u201334","DOI":"10.1145\/3020078.3021698"},{"key":"3761_CR65","doi-asserted-by":"crossref","unstructured":"Guo K, Sui L, Qiu J, Yao S, Han S, Wang Y, Yang H (2016) Angel-eye: a complete design flow for mapping CNN onto customized hardware. In: IEEE computer society annual symposium on VLSI (ISVLSI), pp 24\u201329","DOI":"10.1109\/ISVLSI.2016.129"},{"key":"3761_CR66","doi-asserted-by":"crossref","unstructured":"Han S, Kang J, Mao H, Hu Y, Li X, Li Y, Xie D, Luo H, Yao S, Wang Y et\u00a0al (2017) ESE: efficient speech recognition engine with sparse LSTM on FPGA. In: FPGA, pp 75\u201384","DOI":"10.1145\/3020078.3021745"},{"key":"3761_CR67","doi-asserted-by":"crossref","unstructured":"Wang Y, Xu J, Han Y, Li H, Li X (2016) DeepBurning: automatic generation of FPGA-based learning accelerators for the neural network family. In: Design automation conference (DAC). IEEE, pp 1\u20136","DOI":"10.1145\/2897937.2898003"},{"key":"3761_CR68","first-page":"14","volume-title":"Lecture Notes in Computer Science","author":"Yijin Guan","year":"2017","unstructured":"Guan Y, Xu N, Zhang C, Yuan Z, Cong J (2017) Using data compression for optimizing FPGA-based convolutional neural network accelerators. In: International workshop on advanced parallel processing technologies, pp 14\u201326"},{"key":"3761_CR69","doi-asserted-by":"crossref","unstructured":"Cadambi S, Majumdar A, Becchi M, Chakradhar S, Graf HP (2010) A programmable parallel accelerator for learning and classification. In: International conference on parallel architectures and compilation techniques, pp 273\u2013284","DOI":"10.1145\/1854273.1854309"},{"key":"3761_CR70","doi-asserted-by":"crossref","unstructured":"Motamedi M, Gysel P, Akella V, Ghiasi S (2016) Design space exploration of FPGA-based deep convolutional neural networks. In: Asia and South Pacific design automation conference (ASP-DAC), pp 575\u2013580","DOI":"10.1109\/ASPDAC.2016.7428073"},{"key":"3761_CR71","doi-asserted-by":"crossref","unstructured":"Han X, Zhou D, Wang S, Kimura S (2016) CNN-MERP: an FPGA-based memory-efficient reconfigurable processor for forward and backward propagation of convolutional neural networks. In: International conference on computer design (ICCD), pp 320\u2013327","DOI":"10.1109\/ICCD.2016.7753296"},{"key":"3761_CR72","doi-asserted-by":"crossref","unstructured":"Sharma H, Park J, Mahajan D, Amaro E, Kim JK, Shao C, Mishra A, Esmaeilzadeh H (2016) From high-level deep neural models to FPGAs. In: International symposium on microarchitecture (MICRO). IEEE, pp 1\u201312","DOI":"10.1109\/MICRO.2016.7783720"},{"key":"3761_CR73","unstructured":"Baskin C, Liss N, Mendelson A, Zheltonozhskii E (2017) Streaming architecture for large-scale quantized neural networks on an FPGA-based dataflow platform. arXiv preprint \narXiv:1708.00052"},{"key":"3761_CR74","doi-asserted-by":"crossref","unstructured":"Gokhale V, Zaidy A, Chang AXM, Culurciello E (2017) Snowflake: an efficient hardware accelerator for convolutional neural networks. In: IEEE international symposium on circuits and systems (ISCAS), pp 1\u20134","DOI":"10.1109\/ISCAS.2017.8050809"},{"key":"3761_CR75","doi-asserted-by":"crossref","unstructured":"Lee M, Hwang K, Park J, Choi S, Shin S, Sung W (2016) \u201cFPGA-based low-power speech recognition with recurrent neural networks. In: International workshop on signal processing systems (SiPS), pp 230\u2013235","DOI":"10.1109\/SiPS.2016.48"},{"key":"3761_CR76","doi-asserted-by":"crossref","unstructured":"Mahajan D, Park J, Amaro E, Sharma H, Yazdanbakhsh A, Kim JK, Esmaeilzadeh H (2016) Tabla: a unified template-based framework for accelerating statistical machine learning. In: International symposium on high performance computer architecture (HPCA). IEEE, pp 14\u201326","DOI":"10.1109\/HPCA.2016.7446050"},{"key":"3761_CR77","doi-asserted-by":"crossref","unstructured":"Prost-Boucle A, Bourge A, P\u00e9trot F, Alemdar H, Caldwell N, Leroy V (2017) Scalable high-performance architecture for convolutional ternary neural networks on FPGA. In: International conference on field programmable logic and applications (FPL), pp 1\u20137","DOI":"10.23919\/FPL.2017.8056850"},{"key":"3761_CR78","doi-asserted-by":"crossref","unstructured":"Alwani M, Chen H, Ferdman M, Milder P (2016) Fused-layer CNN accelerators. In: International symposium on microarchitecture (MICRO), pp 1\u201312","DOI":"10.1109\/MICRO.2016.7783725"},{"issue":"4","key":"3761_CR79","first-page":"62:1","volume":"48","author":"S Mittal","year":"2016","unstructured":"Mittal S (2016) A survey of techniques for approximate computing. ACM Comput Surv 48(4):62:1\u201362:33","journal-title":"ACM Comput Surv"},{"key":"3761_CR80","doi-asserted-by":"publisher","first-page":"1524","DOI":"10.1109\/TPDS.2015.2435788","volume":"27","author":"S Mittal","year":"2016","unstructured":"Mittal S, Vetter J (2016) A survey of architectural approaches for data compression in cache and main memory systems. IEEE Trans Parallel Distrib Syst (TPDS) 27:1524\u20131536","journal-title":"IEEE Trans Parallel Distrib Syst (TPDS)"},{"key":"3761_CR81","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611970364","volume-title":"Arithmetic complexity of computations","author":"S Winograd","year":"1980","unstructured":"Winograd S (1980) Arithmetic complexity of computations, vol 33. SIAM, Philadelphia"},{"issue":"1","key":"3761_CR82","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1016\/j.neucom.2006.11.029","volume":"71","author":"LP Maguire","year":"2007","unstructured":"Maguire LP, McGinnity TM, Glackin B, Ghani A, Belatreche A, Harkin J (2007) Challenges for large-scale implementations of spiking neural networks on FPGAs. Neurocomputing 71(1):13\u201329","journal-title":"Neurocomputing"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-018-3761-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s00521-018-3761-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-018-3761-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,1,30]],"date-time":"2020-01-30T04:13:14Z","timestamp":1580357594000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s00521-018-3761-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,10,6]]},"references-count":82,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2020,2]]}},"alternative-id":["3761"],"URL":"https:\/\/doi.org\/10.1007\/s00521-018-3761-1","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,10,6]]},"assertion":[{"value":"11 January 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 September 2018","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 October 2018","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with ethical standards"}},{"value":"The author has no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}