{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T08:11:11Z","timestamp":1781770271190,"version":"3.54.5"},"reference-count":83,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,3,29]],"date-time":"2024-03-29T00:00:00Z","timestamp":1711670400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,3,29]],"date-time":"2024-03-29T00:00:00Z","timestamp":1711670400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["IITP-2023-RS-2022-00156295"],"award-info":[{"award-number":["IITP-2023-RS-2022-00156295"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2022M3I7A1078936"],"award-info":[{"award-number":["NRF-2022M3I7A1078936"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Real-Time Image Proc"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s11554-024-01442-8","type":"journal-article","created":{"date-parts":[[2024,3,29]],"date-time":"2024-03-29T17:01:35Z","timestamp":1711731695000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":31,"title":["Survey of convolutional neural network accelerators on field-programmable gate array platforms: architectures and optimization techniques"],"prefix":"10.1007","volume":"21","author":[{"given":"Hyeonseok","family":"Hong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dahun","family":"Choi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Namjoon","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haein","family":"Lee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Beomjin","family":"Kang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huibeom","family":"Kang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hyun","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,3,29]]},"reference":[{"key":"1442_CR1","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"issue":"6","key":"1442_CR2","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Commun. ACM 60(6), 84\u201390 (2017)","journal-title":"Commun. ACM"},{"key":"1442_CR3","doi-asserted-by":"crossref","unstructured":"Choi, J., Chun, D., Kim, H., Lee, H.-J.: Gaussian yolov3: an accurate and fast object detector using localization uncertainty for autonomous driving. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 502\u2013511 (2019)","DOI":"10.1109\/ICCV.2019.00059"},{"key":"1442_CR4","unstructured":"Redmon, J., Farhadi, A.: Yolov3: An Incremental Improvement. arXiv preprint arXiv:1804.02767 (2018)"},{"key":"1442_CR5","doi-asserted-by":"crossref","unstructured":"Bolya, D., Zhou, C., Xiao, F., Lee, Y.J.: Yolact: real-time instance segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9157\u20139166 (2019)","DOI":"10.1109\/ICCV.2019.00925"},{"key":"1442_CR6","doi-asserted-by":"crossref","unstructured":"Lee, S.I., Kim, H.: Gaussianmask: uncertainty-aware instance segmentation based on Gaussian modeling. In: Proceedings of the 26th International Conference on Pattern Recognition (ICPR 2022) (2022)","DOI":"10.1109\/ICPR56361.2022.9956515"},{"key":"1442_CR7","unstructured":"Simonyan, K., Zisserman, A.: Very Deep Convolutional Networks for Large-Scale Image Recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"key":"1442_CR8","doi-asserted-by":"publisher","first-page":"5279","DOI":"10.1109\/TMM.2022.3189496","volume":"25","author":"NJ Kim","year":"2023","unstructured":"Kim, N.J., Kim, H.: FP-AGL: filter pruning with adaptive gradient learning for accelerating deep convolutional neural networks. IEEE Trans Multimed. 25, 5279\u20135290 (2023)","journal-title":"IEEE Trans Multimed."},{"key":"1442_CR9","doi-asserted-by":"crossref","first-page":"52812","DOI":"10.1109\/ACCESS.2023.3294993","volume":"11","author":"D Chun","year":"2023","unstructured":"Chun, D., Choi, J., Lee, H.-J., Kim, H.: CP-CNN: computational parallelization of CNN-based object detectors in heterogeneous embedded systems for autonomous driving. IEEE Access 11, 52812\u201352823 (2023)","journal-title":"IEEE Access"},{"key":"1442_CR10","unstructured":"Guo, K., Zeng, S., Yu, J., Wang, Y., Yang, H.: A Survey of FPGA-Based Neural Network Inference Accelerator. arXiv preprint arXiv:1712.08934 (2018)"},{"issue":"2","key":"1442_CR11","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1109\/MM.2021.3061394","volume":"41","author":"J Choquette","year":"2021","unstructured":"Choquette, J., Gandhi, W., Giroux, O., Stam, N., Krashinsky, R.: Nvidia a100 tensor core GPU: performance and innovation. IEEE Micro 41(2), 29\u201335 (2021). https:\/\/doi.org\/10.1109\/MM.2021.3061394","journal-title":"IEEE Micro"},{"issue":"2","key":"1442_CR12","doi-asserted-by":"publisher","first-page":"113","DOI":"10.5626\/JCSE.2022.16.2.113","volume":"16","author":"H Kim","year":"2022","unstructured":"Kim, H.: Review of optimal convolutional neural network accelerator platforms for mobile devices. J. Comput. Sci. Eng. 16(2), 113\u2013119 (2022)","journal-title":"J. Comput. Sci. Eng."},{"key":"1442_CR13","doi-asserted-by":"crossref","unstructured":"Nguyen, D.T., Nguyen, T.N., Kim, H., Lee, H.-J.: A high-throughput and power-efficient FPGA implementation of YOLO CNN for object detection. IEEE Trans. Very Large Scale Integr. Syst. 27(8) (2019) 1861\u20131873","DOI":"10.1109\/TVLSI.2019.2905242"},{"issue":"6","key":"1442_CR14","doi-asserted-by":"publisher","first-page":"2450","DOI":"10.1109\/TCSVT.2020.3020569","volume":"31","author":"DT Nguyen","year":"2021","unstructured":"Nguyen, D.T., Kim, H., Lee, H.-J.: Layer-specific optimization for mixed data flow with mixed precision in FPGA design for CNN-based object detectors. IEEE Trans. Circuits Syst. Video Technol. 31(6), 2450\u20132464 (2021)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1442_CR15","doi-asserted-by":"crossref","unstructured":"Rahman, A., Lee, J., Choi, K., Efficient FPGA acceleration of convolutional neural networks using logical-3d compute array. In: 2016 Design, Automation & Test in Europe Conference & Exhibition (DATE), pp. 1393\u20131398. IEEE (2016)","DOI":"10.3850\/9783981537079_0833"},{"key":"1442_CR16","doi-asserted-by":"crossref","unstructured":"Ma, Y., Suda, N., Cao, Y., Seo, J.-S., Vrudhula, S., Scalable and modularized RTL compilation of convolutional neural networks onto FPGA. In: 2016 26th International Conference on Field Programmable Logic and Applications (FPL), pp. 1\u20138. IEEE (2016)","DOI":"10.1109\/FPL.2016.7577356"},{"issue":"3","key":"1442_CR17","doi-asserted-by":"publisher","first-page":"367","DOI":"10.1145\/3007787.3001177","volume":"44","author":"Y-H Chen","year":"2016","unstructured":"Chen, Y.-H., Emer, J., Sze, V.: Eyeriss: a spatial architecture for energy-efficient dataflow for convolutional neural networks. ACM SIGARCH Comput. Archit. News 44(3), 367\u2013379 (2016)","journal-title":"ACM SIGARCH Comput. Archit. News"},{"key":"1442_CR18","doi-asserted-by":"crossref","unstructured":"Wang, J., Lou, Q., Zhang, X., Zhu, C., Lin, Y., Chen, D., Design flow of accelerating hybrid extremely low bit-width neural network in embedded FPGA. In: 2018 28th international conference on field programmable logic and applications (FPL), pp. 163\u20131636. IEEE (2018)","DOI":"10.1109\/FPL.2018.00035"},{"issue":"10","key":"1442_CR19","first-page":"3882","volume":"70","author":"S Ki","year":"2023","unstructured":"Ki, S., Park, J., Kim, H.: Dedicated FPGA implementation of the Gaussian TinyYOLOv3 accelerator. IEEE Trans. Circuits Syst. II Express Briefs 70(10), 3882\u20133886 (2023)","journal-title":"IEEE Trans. Circuits Syst. II Express Briefs"},{"issue":"4","key":"1442_CR20","doi-asserted-by":"publisher","first-page":"1109","DOI":"10.1007\/s00521-018-3761-1","volume":"32","author":"S Mittal","year":"2020","unstructured":"Mittal, S.: A survey of FPGA-based accelerators for convolutional neural networks. Neural Comput. Appl. 32(4), 1109\u20131139 (2020)","journal-title":"Neural Comput. Appl."},{"issue":"2","key":"1442_CR21","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1561\/1000000005","volume":"2","author":"I Kuon","year":"2008","unstructured":"Kuon, I., Tessier, R., Rose, J.: FPGA architecture: survey and challenges. Found. Trends Electron. Des. Autom. 2(2), 135\u2013253 (2008)","journal-title":"Found. Trends Electron. Des. Autom."},{"issue":"5","key":"1442_CR22","doi-asserted-by":"publisher","first-page":"322","DOI":"10.5573\/JSTS.2023.23.5.322","volume":"23","author":"J-H Jang","year":"2023","unstructured":"Jang, J.-H., Shin, J., Park, J.-T., Hwang, I.-S., Kim, H.: In-depth survey of processing-in-memory architectures for deep neural networks. J. Semicond. Technol. Sci. 23(5), 322\u2013339 (2023)","journal-title":"J. Semicond. Technol. Sci."},{"key":"1442_CR23","unstructured":"Howard, A.G., Zhu, M., Chen, B., Kalenichenko, D., Wang, W., Weyand, T., Andreetto, M., Adam, H.: Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861 (2017)"},{"key":"1442_CR24","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.-C.: Mobilenetv2: inverted residuals and linear bottlenecks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4510\u20134520 (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"1442_CR25","unstructured":"Iandola, F.N., Han, S., Moskewicz, M.W., Ashraf, K., Dally, W.J., Keutzer, K.: Squeezenet: Alexnet-Level Accuracy with 50x Fewer Parameters and $$< 0.5$$ mb Model Size. arXiv preprint arXiv:1602.07360 (2016)"},{"key":"1442_CR26","doi-asserted-by":"crossref","unstructured":"Kim, J., Lee, J.K., Lee,, K.M.: Accurate image super-resolution using very deep convolutional networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1646\u20131654 (2016)","DOI":"10.1109\/CVPR.2016.182"},{"key":"1442_CR27","doi-asserted-by":"crossref","unstructured":"Park, J., Bin, K., Lee, K.: mGEMM: low-latency convolution with minimal memory overhead optimized for mobile devices. In: Proceedings of the 20th Annual International Conference on Mobile Systems, Applications and Services, pp. 222\u2013234 (2022)","DOI":"10.1145\/3498361.3538940"},{"key":"1442_CR28","doi-asserted-by":"crossref","unstructured":"Papaphilippou, P., Luk, W.: Accelerating database systems using FPGAs: a survey. In: 2018 28th International Conference on Field Programmable Logic and Applications (FPL), pp. 125\u20131255. IEEE (2018)","DOI":"10.1109\/FPL.2018.00030"},{"key":"1442_CR29","unstructured":"Xilinx, Getting started with Alveo data center accelerator cards, bit.ly\/48gwXiT, pDF document (2022)"},{"key":"1442_CR30","unstructured":"Intel, Intel acceleration stack quick start guide for intel programmable acceleration card with Intel Arria 10 gx FPGA, bit.ly\/48gwXiT, PDF document (2018)"},{"issue":"8","key":"1442_CR31","doi-asserted-by":"publisher","first-page":"895","DOI":"10.3390\/electronics10080895","volume":"10","author":"KP Seng","year":"2021","unstructured":"Seng, K.P., Lee, P.J., Ang, L.M.: Embedded intelligence on FPGA: survey, applications and challenges. Electronics 10(8), 895 (2021)","journal-title":"Electronics"},{"key":"1442_CR32","doi-asserted-by":"publisher","first-page":"7823","DOI":"10.1109\/ACCESS.2018.2890150","volume":"7","author":"A Shawahna","year":"2018","unstructured":"Shawahna, A., Sait, S.M., El-Maleh, A.: FPGA-based accelerators of deep learning networks for learning and classification: a review. IEEE Access 7, 7823\u20137859 (2018)","journal-title":"IEEE Access"},{"key":"1442_CR33","doi-asserted-by":"crossref","unstructured":"Jinghong, D., Yaling, D., Kun, L.: Development of image processing system based on DSP and FPGA. In: 2007 8th International Conference on Electronic Measurement and Instruments, pp. 2\u2013791. IEEE (2007)","DOI":"10.1109\/ICEMI.2007.4350799"},{"key":"1442_CR34","doi-asserted-by":"crossref","unstructured":"Ryu, S., Oh, Y., Kim, J.-J., Mobileware: a high-performance mobilenet accelerator with channel stationary dataflow. In: 2021 IEEE\/ACM International Conference On Computer Aided Design (ICCAD), pp. 1\u20139. IEEE (2021)","DOI":"10.1109\/ICCAD51958.2021.9643497"},{"issue":"20","key":"1442_CR35","doi-asserted-by":"publisher","first-page":"2514","DOI":"10.3390\/electronics10202514","volume":"10","author":"T Pacini","year":"2021","unstructured":"Pacini, T., Rapuano, E., Dinelli, G., Fanucci, L.: A multi-cache system for on-chip memory optimization in FPGA-based CNN accelerators. Electronics 10(20), 2514 (2021)","journal-title":"Electronics"},{"key":"1442_CR36","doi-asserted-by":"crossref","unstructured":"Motamedi, M., Gysel, P., Akella, V., Ghiasi, S., Design space exploration of FPGA-based deep convolutional neural networks. In: 2016 21st Asia and South Pacific Design Automation Conference (ASP-DAC), pp. 575\u2013580. IEEE (2016)","DOI":"10.1109\/ASPDAC.2016.7428073"},{"key":"1442_CR37","doi-asserted-by":"crossref","unstructured":"Li, H., Fan, X., Jiao, L., Cao, W., Zhou, X., Wang, L.: A high performance FPGA-based accelerator for large-scale convolutional neural networks. In: 2016 26th International Conference on Field Programmable Logic and Applications (FPL), pp. 1\u20139. IEEE (2016)","DOI":"10.1109\/FPL.2016.7577308"},{"key":"1442_CR38","doi-asserted-by":"crossref","unstructured":"Jia, X., Zhang, Y., Liu, G., Yang, X., Zhang, T., Zheng, J., Xu, D., Wang, H., Zheng, R., Pareek, S., et\u00a0al.: XVDPU: a high performance CNN accelerator on the versal platform powered by the AI engine. In: 2022 32nd International Conference on Field-Programmable Logic and Applications (FPL), pp. 01\u201309. IEEE (2022)","DOI":"10.1109\/FPL57034.2022.00041"},{"key":"1442_CR39","doi-asserted-by":"crossref","unstructured":"Podili, A., Zhang, C., Prasanna, V., Fast and efficient implementation of convolutional neural networks on FPGA. In: 2017 IEEE 28th International Conference on Application-Specific Systems, Architectures and Processors (ASAP), pp. 11\u201318. IEEE (2017)","DOI":"10.1109\/ASAP.2017.7995253"},{"issue":"5","key":"1442_CR40","doi-asserted-by":"publisher","first-page":"1436","DOI":"10.1109\/TCAD.2021.3082868","volume":"41","author":"G Li","year":"2021","unstructured":"Li, G., Liu, Z., Li, F., Cheng, J.: Block convolution: toward memory-efficient inference of large-scale CNNs on FPGA. IEEE Trans. Comput.-Aid. Des. Integr. Circuits Syst. 41(5), 1436\u20131447 (2021)","journal-title":"IEEE Trans. Comput.-Aid. Des. Integr. Circuits Syst."},{"issue":"10","key":"1442_CR41","first-page":"1415","volume":"65","author":"L Bai","year":"2018","unstructured":"Bai, L., Zhao, Y., Huang, X.: A CNN accelerator on FPGA using depthwise separable convolution. IEEE Trans. Circuits Syst. II Express Briefs 65(10), 1415\u20131419 (2018)","journal-title":"IEEE Trans. Circuits Syst. II Express Briefs"},{"key":"1442_CR42","doi-asserted-by":"crossref","unstructured":"Fan, H., Ferianc, M., Que, Z., Li, H., Liu, S., Niu, X., Luk, W.: Algorithm and hardware co-design for reconfigurable CNN accelerator. In: 2022 27th Asia and South Pacific Design Automation Conference (ASP-DAC), pp. 250\u2013255. IEEE (2022)","DOI":"10.1109\/ASP-DAC52403.2022.9712541"},{"key":"1442_CR43","doi-asserted-by":"crossref","unstructured":"Ma, Y., Cao, Y., Vrudhula, S., Seo, J.-s.: Optimizing loop operation and dataflow in FPGA acceleration of deep convolutional neural networks. In: Proceedings of the 2017 ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays, pp. 45\u201354 (2017)","DOI":"10.1145\/3020078.3021736"},{"issue":"11","key":"1442_CR44","doi-asserted-by":"publisher","first-page":"2072","DOI":"10.1109\/TCAD.2017.2785257","volume":"38","author":"C Zhang","year":"2018","unstructured":"Zhang, C., Sun, G., Fang, Z., Zhou, P., Pan, P., Cong, J.: Caffeine: toward uniformed representation and acceleration for deep convolutional neural networks. IEEE Trans. Comput.-Aid. Des. Integr. Circuits Syst. 38(11), 2072\u20132085 (2018)","journal-title":"IEEE Trans. Comput.-Aid. Des. Integr. Circuits Syst."},{"issue":"2","key":"1442_CR45","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3570928","volume":"16","author":"S Basalama","year":"2023","unstructured":"Basalama, S., Sohrabizadeh, A., Wang, J., Guo, L., Cong, J.: FlexCNN: an end-to-end framework for composing CNN accelerators on FPGA. ACM Trans. Reconfig. Technol. Syst. 16(2), 1\u201332 (2023)","journal-title":"ACM Trans. Reconfig. Technol. Syst."},{"key":"1442_CR46","doi-asserted-by":"crossref","unstructured":"Gao, M., Yang, X., Pu, J., Horowitz, M., Kozyrakis, C., Tangram: optimized coarse-grained dataflow for scalable NN accelerators. In: Proceedings of the Twenty-Fourth International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 807\u2013820 (2019)","DOI":"10.1145\/3297858.3304014"},{"key":"1442_CR47","doi-asserted-by":"crossref","unstructured":"Aydonat, U., O\u2019Connell, S., Capalija, D., Ling, A.C., Chiu, G.R.: An opencl$$^{\\rm TM}$$ deep learning accelerator on Arria 10. In: Proceedings of the 2017 ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays, pp. 55\u201364 (2017)","DOI":"10.1145\/3020078.3021738"},{"key":"1442_CR48","doi-asserted-by":"crossref","unstructured":"Song, Y., Wu, B., Yuan, T., Liu, W.: A high-speed CNN hardware accelerator with regular pruning. In: 2022 23rd International Symposium on Quality Electronic Design (ISQED), pp. 1\u20135. IEEE (2022)","DOI":"10.1109\/ISQED54688.2022.9806216"},{"issue":"1","key":"1442_CR49","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1109\/TCAD.2017.2705069","volume":"37","author":"K Guo","year":"2017","unstructured":"Guo, K., Sui, L., Qiu, J., Yu, J., Wang, J., Yao, S., Han, S., Wang, Y., Yang, H.: Angel-eye: a complete design flow for mapping CNN onto embedded FPGA. IEEE Trans. Comput.-Aid. Des. Integr. Circuits Syst. 37(1), 35\u201347 (2017)","journal-title":"IEEE Trans. Comput.-Aid. Des. Integr. Circuits Syst."},{"key":"1442_CR50","doi-asserted-by":"crossref","unstructured":"Park, J., Sung, W.: FPGA based implementation of deep neural networks using on-chip memory only. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1011\u20131015. IEEE (2016)","DOI":"10.1109\/ICASSP.2016.7471828"},{"key":"1442_CR51","doi-asserted-by":"crossref","unstructured":"Vogel, S., Liang, M., Guntoro, A., Stechele, W., Ascheid, G.: Efficient hardware acceleration of CNNs using logarithmic data representation with arbitrary log-base. In: 2018 IEEE\/ACM International Conference on Computer-Aided Design (ICCAD), pp. 1\u20138. ACM (2018)","DOI":"10.1145\/3240765.3240803"},{"key":"1442_CR52","doi-asserted-by":"crossref","unstructured":"Lee, S., Sim, H., Choi, J., Lee, J.: Successive log quantization for cost-efficient neural networks using stochastic computing. In: Proceedings of the 56th Annual Design Automation Conference 2019, pp. 1\u20136 (2019)","DOI":"10.1145\/3316781.3317916"},{"key":"1442_CR53","doi-asserted-by":"crossref","unstructured":"Qiu, J., Wang, J., Yao, S., Guo, K., Li, B., Zhou, E., Yu, J., Tang, T., Xu, N., Song, S. et\u00a0al.: Going deeper with embedded FPGA platform for convolutional neural network. In: Proceedings of the 2016 ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays, pp. 26\u201335 (2016)","DOI":"10.1145\/2847263.2847265"},{"key":"1442_CR54","doi-asserted-by":"crossref","unstructured":"Sun, M., Li, Z., Lu, A., Li, Y., Chang, S.-E., Ma, X., Lin, X., Fang, Z.: FILM-QNN: Efficient FPGA acceleration of deep neural networks with intra-layer, mixed-precision quantization. In: Proceedings of the 2022 ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays, pp. 134\u2013145 (2022)","DOI":"10.1145\/3490422.3502364"},{"key":"1442_CR55","doi-asserted-by":"crossref","unstructured":"Meng, J., Venkataramanaiah, S.K., Zhou, C., Hansen, P., Whatmough, P., Seo, J.-s.: FIXYFPGA: Efficient fpga accelerator for deep neural networks with high element-wise sparsity and without external memory access. In: 2021 31st International Conference on Field-Programmable Logic and Applications (FPL), pp. 9\u201316. IEEE (2021)","DOI":"10.1109\/FPL53798.2021.00010"},{"issue":"3","key":"1442_CR56","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1145\/3007787.3001163","volume":"44","author":"S Han","year":"2016","unstructured":"Han, S., Liu, X., Mao, H., Pu, J., Pedram, A., Horowitz, M.A., Dally, W.J.: EIE: efficient inference engine on compressed deep neural network. ACM SIGARCH Comput. Archit. News 44(3), 243\u2013254 (2016)","journal-title":"ACM SIGARCH Comput. Archit. News"},{"key":"1442_CR57","doi-asserted-by":"crossref","unstructured":"Pellauer, M., Shao, Y.S., Clemons, J., Crago, N., Hegde, K., Venkatesan, R., Keckler, S.W., Fletcher, C.W., Emer, J.: Buffets: an efficient and composable storage idiom for explicit decoupled data orchestration. In: Proceedings of the Twenty-Fourth International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 137\u2013151 (2019)","DOI":"10.1145\/3297858.3304025"},{"issue":"12","key":"1442_CR58","doi-asserted-by":"publisher","first-page":"7084","DOI":"10.1109\/TCSVT.2023.3274964","volume":"33","author":"M Liu","year":"2023","unstructured":"Liu, M., Zhou, C., Qiu, S., He, Y., Jiao, H.: CNN accelerator at the edge with adaptive zero skipping and sparsity-driven data flow. IEEE Trans. Circuits Syst. Video Technol. 33(12), 7084\u20137095 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1442_CR59","doi-asserted-by":"publisher","first-page":"5588","DOI":"10.1109\/TMM.2023.3338052","volume":"26","author":"NJ Kim","year":"2023","unstructured":"Kim, N.J., Kim, H.: Trunk pruning: highly compatible channel pruning for convolutional neural networks without fine-tuning. IEEE Trans. Multimed. 26, 5588\u20135599 (2023)","journal-title":"IEEE Trans. Multimed."},{"key":"1442_CR60","doi-asserted-by":"publisher","unstructured":"Wang, H., Lu, J., Lin, J., Wang, Z.: An FPGA-based reconfigurable CNN training accelerator using decomposable Winograd. In: 2023 IEEE Computer Society Annual Symposium on VLSI (ISVLSI), pp. 1\u20136 (2023). https:\/\/doi.org\/10.1109\/ISVLSI59464.2023.10238574","DOI":"10.1109\/ISVLSI59464.2023.10238574"},{"key":"1442_CR61","doi-asserted-by":"publisher","first-page":"20828","DOI":"10.1109\/ACCESS.2021.3054879","volume":"9","author":"S Kim","year":"2021","unstructured":"Kim, S., Kim, H.: Zero-centered fixed-point quantization with iterative retraining for deep convolutional neural network-based object detectors. IEEE Access 9, 20828\u201320839 (2021)","journal-title":"IEEE Access"},{"key":"1442_CR62","doi-asserted-by":"crossref","unstructured":"Gholami, A., Kim, S., Dong, Z., Yao, Z., Mahoney, M.W, Keutzer, K.: A Survey of Quantization Methods for Efficient Neural Network Inference, arXiv preprint arXiv:2103.13630 (2021)","DOI":"10.1201\/9781003162810-13"},{"key":"1442_CR63","doi-asserted-by":"crossref","unstructured":"Alwani, M., Chen, H., Ferdman, M., Milder, P.: Fused-layer CNN accelerators. In: 2016 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO), pp. 1\u201312. IEEE (2016)","DOI":"10.1109\/MICRO.2016.7783725"},{"key":"1442_CR64","doi-asserted-by":"crossref","unstructured":"Erdem, A., Babic, D., Silvano, C.: A tile-based fused-layer approach to accelerate DCNNs on low-density FPGAs. In: 2019 26th IEEE International Conference on Electronics, Circuits and Systems (ICECS), pp. 37\u201340. IEEE(2019)","DOI":"10.1109\/ICECS46596.2019.8964870"},{"key":"1442_CR65","doi-asserted-by":"crossref","unstructured":"Indirli, F., Erdem, A., Silvano, C.: A tile-based fused-layer CNN accelerator for FPGAs. In: 2020 27th IEEE International Conference on Electronics, Circuits and Systems (ICECS), pp. 1\u20134. IEEE (2020)","DOI":"10.1109\/ICECS49266.2020.9294981"},{"key":"1442_CR66","doi-asserted-by":"crossref","unstructured":"Wu, C.-B., Wu, R.-F., Chan, T.-W.: Hetero layer fusion based architecture design and implementation for of deep learning accelerator. In: 2022 IEEE International Conference on Consumer Electronics-Taiwan, pp. 63\u201364. IEEE (2022)","DOI":"10.1109\/ICCE-Taiwan55306.2022.9869072"},{"issue":"2","key":"1442_CR67","doi-asserted-by":"publisher","first-page":"535","DOI":"10.1145\/3140659.3080221","volume":"45","author":"Y Shen","year":"2017","unstructured":"Shen, Y., Ferdman, M., Milder, P.: Maximizing CNN accelerator efficiency through resource partitioning. ACM SIGARCH Comput. Archit. News 45(2), 535\u2013547 (2017)","journal-title":"ACM SIGARCH Comput. Archit. News"},{"key":"1442_CR68","doi-asserted-by":"crossref","unstructured":"Wu, D., Zhang, Y., Jia, X., Tian, L., Li, T., Sui, L., Xie, D., Shan, Y.: A high-performance CNN processor based on FPGA for mobilenets. In: 2019 29th International Conference on Field Programmable Logic and Applications (FPL), pp. 136\u2013143. IEEE (2019)","DOI":"10.1109\/FPL.2019.00030"},{"key":"1442_CR69","doi-asserted-by":"crossref","unstructured":"Qararyah, F., Azhar, M.W., Trancoso, P., Fibha: fixed budget hybrid CNN accelerator. In: 2022 IEEE 34th International Symposium on Computer Architecture and High Performance Computing (SBAC-PAD), pp. 180\u2013190. IEEE (2022)","DOI":"10.1109\/SBAC-PAD55451.2022.00029"},{"key":"1442_CR70","doi-asserted-by":"crossref","unstructured":"Wei, X., Yu, C.H., Zhang, P., Chen, Y., Wang, Y., Hu, H., Liang, Y., Cong, J.: Automated systolic array architecture synthesis for high throughput CNN inference on FPGAs. In: Proceedings of the 54th Annual Design Automation Conference 2017, pp. 1\u20136 (2017)","DOI":"10.1145\/3061639.3062207"},{"key":"1442_CR71","doi-asserted-by":"crossref","unstructured":"Selvam, S., Ganesan, V., Kumar, P., FuSeConv: fully separable convolutions for fast inference on systolic arrays. In: 2021 Design, Automation & Test in Europe Conference & Exhibition (DATE), pp. 651\u2013656. IEEE (2021)","DOI":"10.23919\/DATE51398.2021.9473985"},{"issue":"20","key":"1442_CR72","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.3850","volume":"29","author":"Y Qiao","year":"2017","unstructured":"Qiao, Y., Shen, J., Xiao, T., Yang, Q., Wen, M., Zhang, C.: FPGA-accelerated deep convolutional neural networks for high throughput and energy efficiency. Concurr. Comput. Pract. Exp. 29(20), e3850 (2017)","journal-title":"Concurr. Comput. Pract. Exp."},{"key":"1442_CR73","doi-asserted-by":"publisher","first-page":"116569","DOI":"10.1109\/ACCESS.2020.3004198","volume":"8","author":"Z Wang","year":"2020","unstructured":"Wang, Z., Xu, K., Wu, S., Liu, L., Liu, L., Wang, D.: Sparse-YOLO: hardware\/software co-design of an FPGA accelerator for YOLOv2. IEEE Access 8, 116569\u2013116585 (2020)","journal-title":"IEEE Access"},{"issue":"3","key":"1442_CR74","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3284357","volume":"11","author":"P Meloni","year":"2018","unstructured":"Meloni, P., Capotondi, A., Deriu, G., Brian, M., Conti, F., Rossi, D., Raffo, L., Benini, L.: Neuraghe: Exploiting CPU-FPGA synergies for efficient and flexible CNN inference acceleration on ZYNQ SOCs. ACM Trans. Reconfig. Technol. Syst (TRETS) 11(3), 1\u201324 (2018)","journal-title":"ACM Trans. Reconfig. Technol. Syst (TRETS)"},{"key":"1442_CR75","doi-asserted-by":"crossref","unstructured":"Liu, W., Li, Y., Yang, Y., Zhu, J., Liu, L., Design an efficient DNN inference framework with PS-PL synergies in FPGA for edge computing. In: 2022 China Automation Congress (CAC), pp. 4186\u20134190. IEEE (2022)","DOI":"10.1109\/CAC57257.2022.10055526"},{"key":"1442_CR76","doi-asserted-by":"crossref","unstructured":"Zhang, C., Li, P., Sun, G., Guan, Y., Xiao, B., Cong, J.: Optimizing FPGA-based accelerator design for deep convolutional neural networks. In: Proceedings of the 2015 ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays, pp. 161\u2013170 (2015)","DOI":"10.1145\/2684746.2689060"},{"key":"1442_CR77","doi-asserted-by":"crossref","unstructured":"Zhang, J., Zhang, W., Luo, G., Wei, X., Liang, Y., Cong, J.: Frequency improvement of systolic array-based CNNs on FPGAS. In: 2019 IEEE International Symposium on Circuits and Systems (ISCAS), pp. 1\u20134. IEEE (2019)","DOI":"10.1109\/ISCAS.2019.8702071"},{"issue":"3","key":"1442_CR78","doi-asserted-by":"publisher","first-page":"295","DOI":"10.3390\/electronics8030295","volume":"8","author":"M Zhang","year":"2019","unstructured":"Zhang, M., Li, L., Wang, H., Liu, Y., Qin, H., Zhao, W.: Optimized compression for implementing convolutional neural networks on FPGA. Electronics 8(3), 295 (2019)","journal-title":"Electronics"},{"key":"1442_CR79","doi-asserted-by":"crossref","unstructured":"Liu, Z., Dou, Y., Jiang, J., Xu, J., Automatic code generation of convolutional neural networks in FPGA implementation, In: 2016 International Conference on Field-Programmable Technology (FPT), pp. 61\u201368. IEEE (2016)","DOI":"10.1109\/FPT.2016.7929190"},{"key":"1442_CR80","doi-asserted-by":"crossref","unstructured":"Li, X., Cai, Y., Han, J., Zeng, X., A high utilization FPGA-based accelerator for variable-scale convolutional neural network. In: 2017 IEEE 12th International Conference on ASIC (ASICON), pp. 944\u2013947. IEEE (2017)","DOI":"10.1109\/ASICON.2017.8252633"},{"key":"1442_CR81","doi-asserted-by":"crossref","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C.-Y., Berg, A.C.: SSD: single shot multibox detector. In: Computer Vision\u2013ECCV 2016: 14th European Conference. Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part I 14, pp. 21\u201337. Springer, Berlin (2016)","DOI":"10.1007\/978-3-319-46448-0_2"},{"issue":"1","key":"1442_CR82","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1007\/s11554-023-01378-5","volume":"21","author":"X Sang","year":"2024","unstructured":"Sang, X., Ruan, T., Li, C., Li, H., Yang, R., Liu, Z.: A real-time and high-performance mobilenet accelerator based on adaptive dataflow scheduling for image classification. J. Real-Time Image Process. 21(1), 4 (2024)","journal-title":"J. Real-Time Image Process."},{"issue":"14","key":"1442_CR83","doi-asserted-by":"publisher","first-page":"2205","DOI":"10.3390\/rs12142205","volume":"12","author":"G Giuffrida","year":"2020","unstructured":"Giuffrida, G., Diana, L., de Gioia, F., Benelli, G., Meoni, G., Donati, M., Fanucci, L.: Cloudscout: a deep neural network for on-board cloud detection on hyperspectral images. Remote Sens. 12(14), 2205 (2020)","journal-title":"Remote Sens."}],"container-title":["Journal of Real-Time Image Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11554-024-01442-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11554-024-01442-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11554-024-01442-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,15]],"date-time":"2024-11-15T08:36:56Z","timestamp":1731659816000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11554-024-01442-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3,29]]},"references-count":83,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["1442"],"URL":"https:\/\/doi.org\/10.1007\/s11554-024-01442-8","relation":{},"ISSN":["1861-8200","1861-8219"],"issn-type":[{"value":"1861-8200","type":"print"},{"value":"1861-8219","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,3,29]]},"assertion":[{"value":"23 November 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 February 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 March 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"64"}}