{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T00:09:09Z","timestamp":1779322149434,"version":"3.51.4"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032238702","type":"print"},{"value":"9783032238719","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-23871-9_7","type":"book-chapter","created":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T23:43:00Z","timestamp":1779320580000},"page":"82-94","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Benchmarking Deep Learning Convolutions on\u00a0Energy-Constrained CPUs"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-9675-4049","authenticated-orcid":false,"given":"Enrique","family":"Galvez","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6741-5329","authenticated-orcid":false,"given":"Adrien","family":"Cassagne","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2170-6366","authenticated-orcid":false,"given":"Alix","family":"Munier","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Manuel","family":"Bouyer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,5,1]]},"reference":[{"key":"7_CR1","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1007\/978-3-319-41321-1_2","volume-title":"High Performance Computing","author":"A Abdelfattah","year":"2016","unstructured":"Abdelfattah, A., Haidar, A., Tomov, S., Dongarra, J.: Performance, design, and autotuning of batched GEMM for GPUs. In: Kunkel, J.M., Balaji, P., Dongarra, J. (eds.) High Performance Computing, pp. 21\u201338. Springer International Publishing, Cham (2016)"},{"key":"7_CR2","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part I, pp. 213\u2013229. Springer-Verlag (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"7_CR3","unstructured":"Cassagne, A., Amiot, N., Bouyer, M.: Dalek: An unconventional and energy-aware heterogeneous cluster (2025). https:\/\/doi.org\/10.48550\/arXiv.2508.10481"},{"issue":"9","key":"7_CR4","doi-asserted-by":"publisher","first-page":"9819","DOI":"10.1007\/s11227-023-05050-4","volume":"79","author":"MF Dolz","year":"2023","unstructured":"Dolz, M.F., et al.: Performance\u2013energy trade-offs of deep learning convolution algorithms on ARM processors. J. Supercomput. 79(9), 9819\u20139836 (2023)","journal-title":"J. Supercomput."},{"key":"7_CR5","doi-asserted-by":"crossref","unstructured":"Dolz, M.F., Martinez, H., Alonso, P., Quintana-Orti, E.S.: Convolution operators for deep learning inference on the Fujitsu A64FX Processor. In: 2022 IEEE 34th International Symposium on Computer Architecture and High Performance Computing (SBAC-PAD) (2022)","DOI":"10.1109\/SBAC-PAD55451.2022.00027"},{"key":"7_CR6","doi-asserted-by":"crossref","unstructured":"Du, J.: Understanding of object detection based on CNN family and YOLO. J. Physics: Conf. Series 1004(1) (2018)","DOI":"10.1088\/1742-6596\/1004\/1\/012029"},{"key":"7_CR7","doi-asserted-by":"publisher","unstructured":"Georganas, E., et al.: Anatomy of high-performance deep learning convolutions on SIMD architectures (2018). https:\/\/doi.org\/10.48550\/ARXIV.1808.05567","DOI":"10.48550\/ARXIV.1808.05567"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"7_CR9","unstructured":"Intel: OneAPI Deep Neural Network Library (OneDNN). https:\/\/oneapi-src.github.io\/oneDNN\/"},{"key":"7_CR10","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid57682.2023.00020","author":"M Jay","year":"2023","unstructured":"Jay, M., Ostapenco, V., Lefevre, L., Trystram, D., Orgerie, A.C., Fichel, B.: An experimental comparison of software-based power meters: focus on CPU and GPU (2023). https:\/\/doi.org\/10.1109\/CCGrid57682.2023.00020","journal-title":"An experimental comparison of software-based power meters: focus on CPU and GPU"},{"key":"7_CR11","doi-asserted-by":"publisher","unstructured":"Khan, K., Hirki, M., Niemi, T., Nurminen, J., Ou, Z.: RAPL in action: Experiences in using RAPL for power measurements. ACM Trans. Modeling Performance Eval. Comput. Syst. (TOMPECS) 3 (2018). https:\/\/doi.org\/10.1145\/3177754","DOI":"10.1145\/3177754"},{"key":"7_CR12","unstructured":"Khanam, R., Hussain, M.: YOLOv11: An overview of the key architectural enhancements (2024). https:\/\/doi.org\/10.48550\/arXiv.2410.17725"},{"issue":"3","key":"7_CR13","doi-asserted-by":"publisher","first-page":"268","DOI":"10.1145\/292395.292412","volume":"24","author":"B K\u00e5gstr\u00f6m","year":"1998","unstructured":"K\u00e5gstr\u00f6m, B., Ling, P., van Loan, C.: GEMM-based level 3 BLAS: high-performance model implementations and performance evaluation benchmark. ACM Trans. Math. Softw. 24(3), 268\u2013302 (1998)","journal-title":"ACM Trans. Math. Softw."},{"issue":"6","key":"7_CR14","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Commun. ACM 60(6), 84\u201390 (2017)","journal-title":"Commun. ACM"},{"key":"7_CR15","doi-asserted-by":"publisher","unstructured":"Lavin, A., Gray, S.: Fast algorithms for convolutional neural networks (2015). https:\/\/doi.org\/10.48550\/ARXIV.1509.09308","DOI":"10.48550\/ARXIV.1509.09308"},{"key":"7_CR16","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision - ECCV 2014","author":"TY Lin","year":"2014","unstructured":"Lin, T.Y., et al.: Microsoft coco: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) Computer Vision - ECCV 2014, pp. 740\u2013755. Springer International Publishing, Cham (2014)"},{"key":"7_CR17","doi-asserted-by":"crossref","unstructured":"Lym, S., Lee, D., O\u2019Connor, M., Chatterjee, N., Erez, M.: DeLTA: GPU Performance Model for Deep Learning Applications with In-depth Memory System Traffic Analysis. In: 2019 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS), pp. 293\u2013303 (2019)","DOI":"10.1109\/ISPASS.2019.00041"},{"key":"7_CR18","unstructured":"Ma, Y., et al.: Revisiting DETR pre-training for object detection (2023). https:\/\/arxiv.org\/abs\/2308.01300"},{"key":"7_CR19","doi-asserted-by":"publisher","unstructured":"Meng, L., Brothers, J.: Efficient winograd convolution via integer arithmetic (2019). https:\/\/doi.org\/10.48550\/ARXIV.1901.01965","DOI":"10.48550\/ARXIV.1901.01965"},{"key":"7_CR20","doi-asserted-by":"publisher","unstructured":"Park, H., Kim, D., Ahn, J., Yoo, S.: Zero and data reuse-aware fast convolution for deep neural networks on GPU. In: Proceedings of the Eleventh IEEE\/ACM\/IFIP International Conference on Hardware\/Software Codesign and System Synthesis. CODES \u201916, Association for Computing Machinery, New York, NY, USA (2016). https:\/\/doi.org\/10.1145\/2968456.2968476","DOI":"10.1145\/2968456.2968476"},{"key":"7_CR21","doi-asserted-by":"crossref","unstructured":"Santana, A.d.L., Armejach, A., Casas, M.: Efficient direct convolution using long SIMD instructions. In: Proceedings of the 28th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming. PPoPP \u201923, ACM (2023)","DOI":"10.1145\/3572848.3577435"},{"key":"7_CR22","doi-asserted-by":"publisher","unstructured":"Sze, V., Chen, Y.H., Yang, T.J., Emer, J.: Efficient Processing of Deep Neural Networks: A Tutorial and Survey. Tech. rep. (2017). https:\/\/doi.org\/10.48550\/arXiv.1703.09039","DOI":"10.48550\/arXiv.1703.09039"},{"key":"7_CR23","doi-asserted-by":"crossref","unstructured":"Wei, H., Liu, E., Zhao, Y., Yu, H.: Efficient non-fused winograd on GPUs. In: Advances in Computer Graphics. pp. 411\u2013418. Springer International Publishing, Cham (2020)","DOI":"10.1007\/978-3-030-61864-3_35"},{"key":"7_CR24","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611970364","author":"S Winograd","year":"1980","unstructured":"Winograd, S.: Arithmetic complexity of computations. Soc. Industrial Appl. Math. (1980). https:\/\/doi.org\/10.1137\/1.9781611970364","journal-title":"Soc. Industrial Appl. Math."},{"key":"7_CR25","doi-asserted-by":"crossref","unstructured":"Yan, D., Wang, W., Chu, X.: Optimizing batched winograd convolution on GPUs. In: Proceedings of the 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp. 32\u201344. New York, NY, USA (2020)","DOI":"10.1145\/3332466.3374520"},{"key":"7_CR26","doi-asserted-by":"publisher","unstructured":"Zhang, J., Franchetti, F., Low, T.M.: High performance zero-memory overhead direct convolutions (2018). https:\/\/doi.org\/10.48550\/ARXIV.1809.10170","DOI":"10.48550\/ARXIV.1809.10170"},{"key":"7_CR27","doi-asserted-by":"publisher","unstructured":"Zhou, Y., et al.: Characterizing and demystifying the implicit convolution algorithm on commercial matrix-multiplication accelerators (2021). https:\/\/doi.org\/10.48550\/ARXIV.2110.03901","DOI":"10.48550\/ARXIV.2110.03901"},{"key":"7_CR28","doi-asserted-by":"publisher","first-page":"47653","DOI":"10.1109\/ACCESS.2025.3550947","volume":"13","author":"S Zhuo","year":"2025","unstructured":"Zhuo, S., et al.: SCL-YOLOv11: a lightweight object detection network for low-illumination environments. IEEE Access 13, 47653\u201347662 (2025)","journal-title":"IEEE Access"}],"container-title":["Lecture Notes in Computer Science","Design and Architecture for Signal and Image Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-23871-9_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T23:43:04Z","timestamp":1779320584000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-23871-9_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032238702","9783032238719"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-23871-9_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"1 May 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"DASIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Workshop on Design and Architectures for Signal and Image Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Krakow","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Poland","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 January 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 January 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"dasip2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/dasip-2026.github.io\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}