{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T22:15:30Z","timestamp":1780438530029,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,5,25]],"date-time":"2020-05-25T00:00:00Z","timestamp":1590364800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100003246","name":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek","doi-asserted-by":"publisher","award":["P15-06"],"award-info":[{"award-number":["P15-06"]}],"id":[{"id":"10.13039\/501100003246","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,5,25]]},"DOI":"10.1145\/3378678.3391882","type":"proceedings-article","created":{"date-parts":[[2020,5,26]],"date-time":"2020-05-26T00:21:35Z","timestamp":1590452495000},"page":"48-53","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":22,"title":["Reviewing inference performance of state-of-the-art deep learning frameworks"],"prefix":"10.1145","author":[{"given":"Berk","family":"Ulker","sequence":"first","affiliation":[{"name":"Eindhoven University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sander","family":"Stuijk","sequence":"additional","affiliation":[{"name":"Eindhoven University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Henk","family":"Corporaal","sequence":"additional","affiliation":[{"name":"Eindhoven University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rob","family":"Wijnhoven","sequence":"additional","affiliation":[{"name":"ViNotion Eindhoven, The Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2020,5,25]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"[n.d.]. https:\/\/git.ics.ele.tue.nl\/Public0\/inference-benchmark  [n.d.]. https:\/\/git.ics.ele.tue.nl\/Public0\/inference-benchmark"},{"key":"e_1_3_2_1_2_1","volume-title":"Pytorch autograd. https:\/\/pytorch.org\/docs\/stable\/autograd.html. Accessed","year":"2019","unstructured":"[n.d.]. Pytorch autograd. https:\/\/pytorch.org\/docs\/stable\/autograd.html. Accessed : 2019 . [n.d.]. Pytorch autograd. https:\/\/pytorch.org\/docs\/stable\/autograd.html. Accessed: 2019."},{"key":"e_1_3_2_1_3_1","volume-title":"Intel Math Kernel Library. Reference Manual","unstructured":"2009. Intel Math Kernel Library. Reference Manual . Intel Corporation , Santa Clara, USA. ISBN 630813-054US. 2009. Intel Math Kernel Library. Reference Manual. Intel Corporation, Santa Clara, USA. ISBN 630813-054US."},{"key":"e_1_3_2_1_4_1","unstructured":"2018. cuBLAS. https:\/\/developer.nvidia.com\/cublas  2018. cuBLAS. https:\/\/developer.nvidia.com\/cublas"},{"key":"e_1_3_2_1_5_1","unstructured":"Mart\u00edn Abadi et al. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems. https:\/\/www.tensorflow.org\/  Mart\u00edn Abadi et al. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems. https:\/\/www.tensorflow.org\/"},{"key":"e_1_3_2_1_6_1","unstructured":"Soheil Bahrampour et al. 2015. Comparative study of deep learning software frameworks. arXiv:1511.06435 (2015).  Soheil Bahrampour et al. 2015. Comparative study of deep learning software frameworks. arXiv:1511.06435 (2015)."},{"key":"e_1_3_2_1_7_1","volume-title":"Demystifying parallel and distributed deep learning: An in-depth concurrency analysis. arXiv:1802.09941","author":"Ben-Nun Tal","year":"2018","unstructured":"Tal Ben-Nun and Torsten Hoefler . 2018. Demystifying parallel and distributed deep learning: An in-depth concurrency analysis. arXiv:1802.09941 ( 2018 ). Tal Ben-Nun and Torsten Hoefler. 2018. Demystifying parallel and distributed deep learning: An in-depth concurrency analysis. arXiv:1802.09941 (2018)."},{"key":"e_1_3_2_1_8_1","unstructured":"Chetlur et al. 2014. cudnn: Efficient primitives for deep learning. arXiv:1410.0759 (2014).  Chetlur et al. 2014. cudnn: Efficient primitives for deep learning. arXiv:1410.0759 (2014)."},{"key":"e_1_3_2_1_9_1","volume-title":"Imagenet: A large-scale hierarchical image database. In CVPR.","author":"Deng","year":"2009","unstructured":"Deng et al. 2009 . Imagenet: A large-scale hierarchical image database. In CVPR. Deng et al. 2009. Imagenet: A large-scale hierarchical image database. In CVPR."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1134\/S1054661816010065"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10278-017-9965-6"},{"key":"e_1_3_2_1_12_1","unstructured":"Ga\u00ebl Guennebaud et al. 2010. Eigen v3. http:\/\/eigen.tuxfamily.org.  Ga\u00ebl Guennebaud et al. 2010. Eigen v3. http:\/\/eigen.tuxfamily.org."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Hanhirova et al. 2018. Latency and throughput characterization of convolutional neural networks for mobile computer vision. In MMSys. ACM.  Hanhirova et al. 2018. Latency and throughput characterization of convolutional neural networks for mobile computer vision. In MMSys. ACM.","DOI":"10.1145\/3204949.3204975"},{"key":"e_1_3_2_1_14_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR.  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Hinton et al. 2012. Deep neural networks for acoustic modeling in speech recognition. IEEE Signal processing magazine 29 (2012).  Hinton et al. 2012. Deep neural networks for acoustic modeling in speech recognition. IEEE Signal processing magazine 29 (2012).","DOI":"10.1109\/MSP.2012.2205597"},{"key":"e_1_3_2_1_16_1","volume-title":"Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv:1704.04861","author":"Howard","year":"2017","unstructured":"Howard et al. 2017 . Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv:1704.04861 (2017). Howard et al. 2017. Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv:1704.04861 (2017)."},{"key":"e_1_3_2_1_17_1","unstructured":"Iandola et al. 2016. SqueezeNet: AlexNet-level accuracy with 50x fewer parameters and&lt; 0.5 MB model size. arXiv:1602.07360 (2016).  Iandola et al. 2016. SqueezeNet: AlexNet-level accuracy with 50x fewer parameters and&lt; 0.5 MB model size. arXiv:1602.07360 (2016)."},{"key":"e_1_3_2_1_18_1","unstructured":"Alex Krizhevsky et al. 2012. Imagenet classification with deep convolutional neural networks. In Advances in neural information processing systems.  Alex Krizhevsky et al. 2012. Imagenet classification with deep convolutional neural networks. In Advances in neural information processing systems."},{"key":"e_1_3_2_1_19_1","volume-title":"Deep learning on fpgas: Past, present, and future. arXiv.1602.04283","author":"Lacey Griffin","year":"2016","unstructured":"Griffin Lacey , Graham W Taylor , and Shawki Areibi . 2016. Deep learning on fpgas: Past, present, and future. arXiv.1602.04283 ( 2016 ). Griffin Lacey, Graham W Taylor, and Shawki Areibi. 2016. Deep learning on fpgas: Past, present, and future. arXiv.1602.04283 (2016)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Ma et al. 2018. Shufflenet v2: Practical guidelines for efficient cnn architecture design. In ECCV.  Ma et al. 2018. Shufflenet v2: Practical guidelines for efficient cnn architecture design. In ECCV.","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"e_1_3_2_1_21_1","volume-title":"A comprehensive classification of deep learning libraries","author":"Pandey Hari Mohan","unstructured":"Hari Mohan Pandey and David Windridge . 2019. A comprehensive classification of deep learning libraries . In ICICT. Springer . Hari Mohan Pandey and David Windridge. 2019. A comprehensive classification of deep learning libraries. In ICICT. Springer."},{"key":"e_1_3_2_1_22_1","unstructured":"Paszke et al. 2019. PyTorch: An Imperative Style High-Performance Deep Learning Library. In Advances in Neural Information Processing Systems 32 H. Wallach H. Larochelle A. Beygelzimer F. d'Alch\u00e9-Buc E. Fox and R. Garnett (Eds.). Curran Associates Inc. 8024--8035.  Paszke et al. 2019. PyTorch: An Imperative Style High-Performance Deep Learning Library. In Advances in Neural Information Processing Systems 32 H. Wallach H. Larochelle A. Beygelzimer F. d'Alch\u00e9-Buc E. Fox and R. Garnett (Eds.). Curran Associates Inc. 8024--8035."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Sandler et al. 2018. Mobilenetv2: Inverted residuals and linear bottlenecks. In CVPR.  Sandler et al. 2018. Mobilenetv2: Inverted residuals and linear bottlenecks. In CVPR.","DOI":"10.1109\/CVPR.2018.00474"},{"key":"e_1_3_2_1_24_1","volume-title":"Benchmarking state-of-the-art deep learning software tools","author":"Shi Shaohuai","unstructured":"Shaohuai Shi , Qiang Wang , Pengfei Xu , and Xiaowen Chu . 2016. Benchmarking state-of-the-art deep learning software tools . In CCBD. IEEE. Shaohuai Shi, Qiang Wang, Pengfei Xu, and Xiaowen Chu. 2016. Benchmarking state-of-the-art deep learning software tools. In CCBD. IEEE."},{"key":"e_1_3_2_1_25_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman . 2014. Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556 ( 2014 ). Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Szegedy et al. 2015. Going deeper with convolutions. In CVPR.  Szegedy et al. 2015. Going deeper with convolutions. In CVPR.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"B. Ulker S. Stuijk H. Corporaal and R. Wijnhoven. 2020. Reviewing Inference Performance of State-of-the-Art Deep Learning Frameworks. Technical Report. TU Eindhoven. http:\/\/www.es.ele.tue.nl\/esreports\/esr-2020-02.pdf  B. Ulker S. Stuijk H. Corporaal and R. Wijnhoven. 2020. Reviewing Inference Performance of State-of-the-Art Deep Learning Frameworks. Technical Report. TU Eindhoven. http:\/\/www.es.ele.tue.nl\/esreports\/esr-2020-02.pdf","DOI":"10.1145\/3378678.3391882"},{"key":"e_1_3_2_1_28_1","first-page":"513","article-title":"DLAU: A scalable deep learning accelerator unit on FPGA","volume":"36","author":"Wang","year":"2017","unstructured":"Wang et al. 2017 . DLAU: A scalable deep learning accelerator unit on FPGA . TCAD 36 , 3 (2017), 513 -- 517 . Wang et al. 2017. DLAU: A scalable deep learning accelerator unit on FPGA. TCAD 36, 3 (2017), 513--517.","journal-title":"TCAD"},{"key":"e_1_3_2_1_29_1","volume-title":"Shufflenet: An extremely efficient convolutional neural network for mobile devices. In CVPR.","author":"Zhang","year":"2018","unstructured":"Zhang et al. 2018 . Shufflenet: An extremely efficient convolutional neural network for mobile devices. In CVPR. Zhang et al. 2018. Shufflenet: An extremely efficient convolutional neural network for mobile devices. In CVPR."},{"key":"e_1_3_2_1_30_1","unstructured":"Xingzhou Zhang et al. 2018. pcamp: Performance comparison of machine learning packages on the edges. In {USENIX} Workshop on HotEdge 18.  Xingzhou Zhang et al. 2018. pcamp: Performance comparison of machine learning packages on the edges. In { USENIX } Workshop on HotEdge 18."}],"event":{"name":"SCOPES '20: 23rd International Workshop on Software and Compilers for Embedded Systems","location":"St. Goar Germany","acronym":"SCOPES '20","sponsor":["SIGBED ACM Special Interest Group on Embedded Systems","EDAA European Design Automation Association"]},"container-title":["Proceedings of the 23th International Workshop on Software and Compilers for Embedded Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3378678.3391882","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3378678.3391882","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T22:41:19Z","timestamp":1750200079000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3378678.3391882"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,5,25]]},"references-count":30,"alternative-id":["10.1145\/3378678.3391882","10.1145\/3378678"],"URL":"https:\/\/doi.org\/10.1145\/3378678.3391882","relation":{},"subject":[],"published":{"date-parts":[[2020,5,25]]},"assertion":[{"value":"2020-05-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}