{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,10]],"date-time":"2026-02-10T07:18:45Z","timestamp":1770707925193,"version":"3.49.0"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"17","license":[{"start":{"date-parts":[[2023,5,30]],"date-time":"2023-05-30T00:00:00Z","timestamp":1685404800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,5,30]],"date-time":"2023-05-30T00:00:00Z","timestamp":1685404800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Key-Area Research and Development Program of Guangdong Province","award":["2022B0101070001"],"award-info":[{"award-number":["2022B0101070001"]}]},{"name":"Key-Area Research and Development Program of Guangdong Province","award":["2022B0101070001"],"award-info":[{"award-number":["2022B0101070001"]}]},{"name":"Key-Area Research and Development Program of Guangdong Province","award":["2022B0101070001"],"award-info":[{"award-number":["2022B0101070001"]}]},{"name":"Major scientific and technological projects in Zhongshan","award":["210602103890051"],"award-info":[{"award-number":["210602103890051"]}]},{"name":"Major scientific and technological projects in Zhongshan","award":["210602103890051"],"award-info":[{"award-number":["210602103890051"]}]},{"name":"Major scientific and technological projects in Zhongshan","award":["210602103890051"],"award-info":[{"award-number":["210602103890051"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2023,11]]},"DOI":"10.1007\/s11227-023-05399-6","type":"journal-article","created":{"date-parts":[[2023,5,30]],"date-time":"2023-05-30T14:02:36Z","timestamp":1685455356000},"page":"19547-19573","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Novel accelerated methods for convolution neural network with matrix core"],"prefix":"10.1007","volume":"79","author":[{"given":"Yijie","family":"Guo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lu","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Songxiang","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,5,30]]},"reference":[{"issue":"4","key":"5399_CR1","doi-asserted-by":"publisher","first-page":"611","DOI":"10.1007\/s13244-018-0639-9","volume":"9","author":"R Yamashita","year":"2018","unstructured":"Yamashita R, Nishio M, Do RKG, Togashi K (2018) Convolutional neural networks: an overview and application in radiology. Insights Imaging 9(4):611\u2013629","journal-title":"Insights Imaging"},{"issue":"10","key":"5399_CR2","doi-asserted-by":"publisher","first-page":"4843","DOI":"10.1109\/TIP.2017.2725580","volume":"26","author":"H Lee","year":"2017","unstructured":"Lee H, Kwon H (2017) Going deeper with contextual cnn for hyperspectral image classification. IEEE Trans Image Process 26(10):4843\u20134855","journal-title":"IEEE Trans Image Process"},{"key":"5399_CR3","doi-asserted-by":"crossref","unstructured":"Salvador A, Gir\u00f3-i-Nieto X, Marqu\u00e9s F, Satoh S (2016) Faster r-cnn features for instance search. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops pp 9\u201316","DOI":"10.1109\/CVPRW.2016.56"},{"key":"5399_CR4","doi-asserted-by":"crossref","unstructured":"Bao L, Wu B, Liu W (2018) Cnn in mrf: Video object segmentation via inference in a cnn-based higher-order spatio-temporal mrf. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition pp 5977\u20135986","DOI":"10.1109\/CVPR.2018.00626"},{"key":"5399_CR5","doi-asserted-by":"crossref","unstructured":"Sharma S, Shanmugasundaram K, Ramasamy SK (2016) Farec-cnn based efficient face recognition technique using dlib. In: 2016 International Conference on Advanced Communication Control and Computing Technologies (ICACCCT) pp 192\u2013195. IEEE","DOI":"10.1109\/ICACCCT.2016.7831628"},{"issue":"16","key":"5399_CR6","doi-asserted-by":"publisher","first-page":"7519","DOI":"10.1007\/s00500-021-06519-1","volume":"26","author":"A Saranya","year":"2022","unstructured":"Saranya A, Kottursamy K, AlZubi AA, Bashir AK (2022) Analyzing fibrous tissue pattern in fibrous dysplasia bone images using deep r-cnn networks for segmentation. Soft Comput 26(16):7519\u20137533","journal-title":"Soft Comput"},{"issue":"2","key":"5399_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3178454","volume":"14","author":"TE Potok","year":"2018","unstructured":"Potok TE, Schuman C, Young S, Patton R, Spedalieri F, Liu J, Yao K-T, Rose G, Chakma G (2018) A study of complex deep learning networks on high-performance, neuromorphic, and quantum computers. ACM J Emerg Technol Comput Syst (JETC) 14(2):1\u201321","journal-title":"ACM J Emerg Technol Comput Syst (JETC)"},{"key":"5399_CR8","doi-asserted-by":"crossref","unstructured":"Chang M-C, Pan Z-G, Chen J-L (2017) Hardware accelerator for boosting convolution computation in image classification applications. In: 2017 IEEE 6th Global Conference on Consumer Electronics (GCCE) pp. 1\u20132 IEEE","DOI":"10.1109\/GCCE.2017.8229395"},{"key":"5399_CR9","unstructured":"Khan J, Fultz P, Tamazov A, Lowell D, Liu C, Melesse M, Nandhimandalam M, Nasyrov K, Perminov I, Shah T, et al (2019) Miopen: An open source library for deep learning primitives. arXiv preprint arXiv:1910.00078"},{"key":"5399_CR10","unstructured":"Chetlur S, Woolley C, Vandermersch P, Cohen J, Tran J, Catanzaro B (2019) Shelhamer, e. cudnn: Efficient primitives for deep learning. arxiv 2014. arXiv preprint arXiv:1410.0759"},{"key":"5399_CR11","unstructured":"NVIDIA: cutlass. https:\/\/github.com\/NVIDIA\/cutlass (2022)"},{"key":"5399_CR12","doi-asserted-by":"crossref","unstructured":"Georganas E, Avancha S, Banerjee K, Kalamkar D, Henry G, Pabst H, Heinecke A (2018) Anatomy of high-performance deep learning convolutions on simd architectures. In: SC18: International Conference for High Performance Computing, Networking Storage and Analysis, pp 830\u2013841. IEEE","DOI":"10.1109\/SC.2018.00069"},{"key":"5399_CR13","unstructured":"Mathieu M, Henaff M, LeCun Y (2013) Fast training of convolutional networks through ffts. arXiv preprint arXiv:1312.5851"},{"key":"5399_CR14","doi-asserted-by":"crossref","unstructured":"Lavin A, Gray S (2016) Fast algorithms for convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4013\u20134021","DOI":"10.1109\/CVPR.2016.435"},{"key":"5399_CR15","unstructured":"INTEL\/oneapi-src: oneDNN. https:\/\/github.com\/oneapi-src\/oneDNN (2021)"},{"key":"5399_CR16","unstructured":"Tencent: ncnn. https:\/\/github.com\/Tencent\/ncnn (2022)"},{"key":"5399_CR17","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611970364","volume-title":"Arithmetic complexity of computations","author":"S Winograd","year":"1980","unstructured":"Winograd S (1980) Arithmetic complexity of computations, vol 33. Siam, India"},{"key":"5399_CR18","doi-asserted-by":"crossref","unstructured":"Kuo L-W, Yang C-C, Lee J-K, Tseng S-Y (2014) The design of llvm-based shader compiler for embedded architecture. In: 2014 20th IEEE International Conference on Parallel and Distributed Systems (ICPADS), pp 961\u2013968. IEEE","DOI":"10.1109\/PADSW.2014.7097916"},{"key":"5399_CR19","doi-asserted-by":"crossref","unstructured":"Horn RA (1990) The hadamard product. In: Proc Symp Appl Math vol 40: pp 87\u2013169","DOI":"10.1090\/psapm\/040\/1059485"},{"key":"5399_CR20","unstructured":"ROCmSoftwarePlatform: rocWMMA. https:\/\/github.com\/ROCmSoftwarePlatform\/rocWMMA (2022)"},{"issue":"2","key":"5399_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s42979-020-0114-9","volume":"1","author":"D Theckedath","year":"2020","unstructured":"Theckedath D, Sedamkar R (2020) Detecting affect states using vgg16, resnet50 and se-resnet50 networks. SN Comput Sci 1(2):1\u20137","journal-title":"SN Comput Sci"},{"key":"5399_CR22","doi-asserted-by":"crossref","unstructured":"Vasudevan A, Anderson A, Gregg D (2017) Parallel multi channel convolution using general matrix multiplication. In: 2017 IEEE 28th International Conference on Application-specific Systems, Architectures and Processors (ASAP) pp 19\u201324. IEEE","DOI":"10.1109\/ASAP.2017.7995254"},{"key":"5399_CR23","doi-asserted-by":"crossref","unstructured":"Chikin V, Kryzhanovskiy V (2022) Channel balancing for accurate quantization of winograd convolutions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition pp 12507\u201312516","DOI":"10.1109\/CVPR52688.2022.01218"},{"key":"5399_CR24","doi-asserted-by":"crossref","unstructured":"Yan D, Wang W, Chu X (2020) Optimizing batched winograd convolution on gpus. In: Proceedings of the 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp 32\u201344","DOI":"10.1145\/3332466.3374520"},{"issue":"17","key":"5399_CR25","doi-asserted-by":"publisher","first-page":"2033","DOI":"10.3390\/math9172033","volume":"9","author":"RL Castro","year":"2021","unstructured":"Castro RL, Andrade D, Fraguela BB (2021) Opencnn: a winograd minimal filtering algorithm implementation in cuda. Mathematics 9(17):2033","journal-title":"Mathematics"},{"key":"5399_CR26","doi-asserted-by":"crossref","unstructured":"Markidis S, Der\u00a0Chien SW, Laure E, Peng IB, Vetter JS (2018) Nvidia tensor core programmability, performance & precision. In: 2018 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW), pp 522\u2013531 IEEE","DOI":"10.1109\/IPDPSW.2018.00091"},{"issue":"7","key":"5399_CR27","first-page":"986","volume":"69","author":"L Jia","year":"2020","unstructured":"Jia L, Liang Y, Li X, Lu L, Yan S (2020) Enabling efficient fast convolution algorithms on gpus via megakernels. IEEE Trans Comput 69(7):986\u2013997","journal-title":"IEEE Trans Comput"},{"key":"5399_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2022.102954","volume":"113","author":"J Jiang","year":"2022","unstructured":"Jiang J, Huang D, Du J, Lu Y, Liao X (2022) Optimizing small channel 3d convolution on gpu with tensor core. Parallel Comput 113:102954","journal-title":"Parallel Comput"},{"key":"5399_CR29","doi-asserted-by":"crossref","unstructured":"Jia Z, Zlateski A, Durand F, Li K (2018) Optimizing n-dimensional, winograd-based convolution for manycore cpus. In: Proceedings of the 23rd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp 109\u2013123","DOI":"10.1145\/3178487.3178496"},{"key":"5399_CR30","doi-asserted-by":"crossref","unstructured":"Ma Y, Cao Y, Vrudhula S, Seo J-s (2018) Optimizing the convolution operation to accelerate deep neural networks on fpga. IEEE Trans Very Large Scale Int (VLSI) Syst 26(7): 1354\u20131367","DOI":"10.1109\/TVLSI.2018.2815603"},{"key":"5399_CR31","doi-asserted-by":"crossref","unstructured":"Kala S, Jose BR, Mathew J, Nalesh S (2019) High-performance cnn accelerator on fpga using unified winograd-gemm architecture. IEEE Trans Very Large Scale Int (VLSI) Syst 27(12): 2816\u20132828","DOI":"10.1109\/TVLSI.2019.2941250"},{"key":"5399_CR32","doi-asserted-by":"crossref","unstructured":"Coppersmith D, Winograd S (1987) Matrix multiplication via arithmetic progressions. In: Proceedings of the Nineteenth Annual ACM Symposium on Theory of Computing, pp 1\u20136","DOI":"10.1145\/28395.28396"},{"key":"5399_CR33","doi-asserted-by":"crossref","unstructured":"Smith A, James N (2022) Amd instinct$$^{{\\rm TM}}$$ mi200 series accelerator and node architectures. In: 2022 IEEE Hot Chips 34 Symposium (HCS), pp 1\u201323. IEEE Computer Society","DOI":"10.1109\/HCS55958.2022.9895477"},{"key":"5399_CR34","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"issue":"2","key":"5399_CR35","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1109\/MM.2020.2974843","volume":"40","author":"P Mattson","year":"2020","unstructured":"Mattson P, Reddi VJ, Cheng C, Coleman C, Diamos G, Kanter D, Micikevicius P, Patterson D, Schmuelling G, Tang H et al (2020) Mlperf: an industry standard benchmark suite for machine learning performance. IEEE Micro 40(2):8\u201316","journal-title":"IEEE Micro"},{"key":"5399_CR36","doi-asserted-by":"crossref","unstructured":"Wei H, Liu E, Zhao Y, Yu H (2020) Efficient non-fused winograd on gpus. In: Computer Graphics International Conference pp 411\u2013418. Springer","DOI":"10.1007\/978-3-030-61864-3_35"},{"key":"5399_CR37","unstructured":"Abadi M, Barham P, Chen J, Chen Z, Davis A, Dean J, Devin M, Ghemawat S, Irving G, Isard M, et al. (2016) $$\\{$$TensorFlow$$\\}$$: a system for $$\\{$$Large-Scale$$\\}$$ machine learning. In: 12th USENIX Symposium on Operating Systems Design and Implementation (OSDI 16) pp 265\u2013283"},{"key":"5399_CR38","unstructured":"NervanaSystems: neon. https:\/\/github.com\/NervanaSystems\/neon (2019)"},{"key":"5399_CR39","unstructured":"ROCmSoftwarePlatform: rocRAND. https:\/\/github.com\/ROCmSoftwarePlatform\/rocRAND (2019)"},{"key":"5399_CR40","doi-asserted-by":"crossref","unstructured":"Sun Y, Mukherjee S, Baruah T, Dong S, Gutierrez J, Mohan P, Kaeli D (2018) Evaluating performance tradeoffs on the radeon open compute platform. In: 2018 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS) pp 209\u2013218. IEEE","DOI":"10.1109\/ISPASS.2018.00034"},{"key":"5399_CR41","doi-asserted-by":"crossref","unstructured":"Zhou Y, Yang M, Guo C, Leng J, Liang Y, Chen Q, Guo M, Zhu Y (2021) Characterizing and demystifying the implicit convolution algorithm on commercial matrix-multiplication accelerators. In: 2021 IEEE International Symposium on Workload Characterization (IISWC) pp 214\u2013225. IEEE","DOI":"10.1109\/IISWC53511.2021.00029"},{"key":"5399_CR42","unstructured":"Tsai YM, Cojean T, Anzt H (2020) Evaluating the performance of nvidia\u2019s a100 ampere gpu for sparse linear algebra computations. arXiv preprint arXiv:2008.08478"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-023-05399-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-023-05399-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-023-05399-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,10,2]],"date-time":"2023-10-02T08:10:54Z","timestamp":1696234254000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-023-05399-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,30]]},"references-count":42,"journal-issue":{"issue":"17","published-print":{"date-parts":[[2023,11]]}},"alternative-id":["5399"],"URL":"https:\/\/doi.org\/10.1007\/s11227-023-05399-6","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,5,30]]},"assertion":[{"value":"15 May 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 May 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}}]}}