{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T07:15:30Z","timestamp":1779174930639,"version":"3.51.4"},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2022,3,19]],"date-time":"2022-03-19T00:00:00Z","timestamp":1647648000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,3,19]],"date-time":"2022-03-19T00:00:00Z","timestamp":1647648000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"name":"The School of Computer Science and Engineering, South China University of Technology","award":["210602103890051"],"award-info":[{"award-number":["210602103890051"]}]},{"name":"Major Project on the Integration of Industry, Education and Research of Zhongshan","award":["210610173898370"],"award-info":[{"award-number":["210610173898370"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2022,7]]},"DOI":"10.1007\/s11227-022-04336-3","type":"journal-article","created":{"date-parts":[[2022,3,19]],"date-time":"2022-03-19T05:02:57Z","timestamp":1647666177000},"page":"13393-13408","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["A batched GEMM optimization framework for deep learning"],"prefix":"10.1007","volume":"78","author":[{"given":"Zhiwei","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lu","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruimin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,3,19]]},"reference":[{"key":"4336_CR1","doi-asserted-by":"crossref","unstructured":"Abdelfattah A, Haidar A, Tomov S, Dongarra J (2016) Performance, design, and autotuning of batched GEMM for GPUs. In: International Conference on High Performance Computing, Springer, pp 21\u201338","DOI":"10.1007\/978-3-319-41321-1_2"},{"key":"4336_CR2","doi-asserted-by":"crossref","unstructured":"Abdelfattah A, Haidar A, Tomov S, Dongarra J (2017) Novel hpc techniques to batch execution of many variable size blas computations on GPUs. In: Proceedings of the International Conference on Supercomputing, pp 1\u201310","DOI":"10.1145\/3079079.3079103"},{"key":"4336_CR3","unstructured":"AMD (2021a) INTRODUCING AMD CDNA ARCHITECTURE The All-New AMD GPU Architecture for the Modern Era of HPC & AI. https:\/\/www.amd.com\/system\/files\/documents\/amd-cdna-whitepaper.pdf"},{"key":"4336_CR4","unstructured":"AMD (2021b) INTRODUCING RDNA ARCHITECTURE The all new Radeon gaming architecture powering \u201cNavi\u201d. https:\/\/www.amd.com\/system\/files\/documents\/rdna-whitepaper.pdf"},{"key":"4336_CR5","unstructured":"AMD (2021c) rocBLAS Documentation. https:\/\/rocblas.readthedocs.io\/_\/downloads\/en\/rocm-4.5.2\/pdf\/"},{"key":"4336_CR6","unstructured":"AMD (2021d) ROCm Documentation. https:\/\/rocmdocs.amd.com\/_\/downloads\/en\/latest\/pdf\/"},{"key":"4336_CR7","unstructured":"Bao W, Chang LW, Chen Y, Deng K, Agarwal A, Barsoum E, Taha A (2019) Ngemm: Optimizing gemm for deep learning via compiler-based techniques. arXiv preprint arXiv:1910.00178"},{"key":"4336_CR8","unstructured":"Chellapilla K, Puri S, Simard P (2006) High performance convolutional neural networks for document processing. In: Tenth international workshop on frontiers in handwriting recognition, Suvisoft"},{"key":"4336_CR9","unstructured":"Chetlur S, Woolley C, Vandermersch P, Cohen J, Tran J, Catanzaro B, Shelhamer E (2014) cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759"},{"key":"4336_CR10","unstructured":"Intel (2021) Intel oneAPI Programming Guide. https:\/\/www.intel.com\/content\/dam\/develop\/external\/us\/en\/documents\/oneapi-programming-guide.pdf"},{"key":"4336_CR11","volume-title":"Learning semantic image representations at a large scale","author":"Y Jia","year":"2014","unstructured":"Jia Y (2014) Learning semantic image representations at a large scale. University of California, Berkeley"},{"key":"4336_CR12","unstructured":"Khan J, Fultz P, Tamazov A, Lowell D, Liu C, Melesse M, Nandhimandalam M, Nasyrov K, Perminov I, Shah T, Filippov V, Zhang J, Zhou J, Natarajan B, Daga M (2019) Miopen: An open source library for deep learning primitives. arXiv:1910.00078"},{"key":"4336_CR13","doi-asserted-by":"crossref","unstructured":"Kim R, Choi J, Lee M (2019) Optimizing parallel gemm routines using auto-tuning with intel avx-512. In: Proceedings of the International Conference on High Performance Computing in Asia-Pacific Region, pp 101\u2013110","DOI":"10.1145\/3293320.3293334"},{"key":"4336_CR14","first-page":"1097","volume":"25","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. Adv Neural Inf Process Syst 25:1097\u20131105","journal-title":"Adv Neural Inf Process Syst"},{"issue":"11","key":"4336_CR15","doi-asserted-by":"publisher","first-page":"2045","DOI":"10.1109\/TPDS.2011.311","volume":"23","author":"J Kurzak","year":"2012","unstructured":"Kurzak J, Tomov S, Dongarra J (2012) Autotuning GEMM kernels for the fermi GPU. IEEE Trans Parallel Distrib Syst 23(11):2045\u20132057","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"4336_CR16","unstructured":"Lai J, Seznec A (2013) Performance upper bound analysis and optimization of sgemm on fermi and kepler GPUs. In: Proceedings of the 2013 IEEE\/ACM international symposium on code generation and optimization (CGO), IEEE, pp 1\u201310"},{"key":"4336_CR17","doi-asserted-by":"crossref","unstructured":"Li X, Zhang G, Huang HH, Wang Z, Zheng W (2016) Performance analysis of GPU-based convolutional neural networks. In: 2016 45th International Conference on Parallel Processing (ICPP), IEEE, pp 67\u201376","DOI":"10.1109\/ICPP.2016.15"},{"key":"4336_CR18","doi-asserted-by":"crossref","unstructured":"Li X, Liang Y, Yan S, Jia L, Li Y (2019) A coordinated tiling and batching framework for efficient GEMM on GPUs. In: Proceedings of the 24th symposium on principles and practice of parallel programming, pp 229\u2013241","DOI":"10.1145\/3293883.3295734"},{"key":"4336_CR19","doi-asserted-by":"crossref","unstructured":"Lym S, Lee D, O\u2019Connor M, Chatterjee N, Erez M (2019) Delta: GPU performance model for deep learning applications with in-depth memory system traffic analysis. In: 2019 IEEE International symposium on performance analysis of systems and software (ISPASS), IEEE, pp 293\u2013303","DOI":"10.1109\/ISPASS.2019.00041"},{"issue":"4","key":"4336_CR20","doi-asserted-by":"publisher","first-page":"511","DOI":"10.1177\/1094342010385729","volume":"24","author":"R Nath","year":"2010","unstructured":"Nath R, Tomov S, Dongarra J (2010) An improved magma GEMM for fermi graphics processing units. Int J High Perform Comput Appl 24(4):511\u2013515","journal-title":"Int J High Perform Comput Appl"},{"key":"4336_CR21","unstructured":"NVIDIA (2018) CUTLASS: Fast Linear Algebra in CUDA C++. https:\/\/devblogs.nvidia.com\/cutlass-linear-algebra-cuda\/"},{"key":"4336_CR22","unstructured":"NVIDIA (2021a) cuBLAS. https:\/\/docs.nvidia.com\/cuda\/cublas\/"},{"key":"4336_CR23","unstructured":"NVIDIA (2021b) CUDA Occupancy Calculator. https:\/\/docs.nvidia.com\/cuda\/cuda-occupancy-calculator\/index.html"},{"key":"4336_CR24","unstructured":"Ren\u00e9 van Oostrum DMea Noel Chalmers (2019) AMD GPU Hardware Basics. https:\/\/www.olcf.ornl.gov\/wp-content\/uploads\/2019\/10\/ORNL_Application_Readiness_Workshop-AMD_GPU_Basics.pdf"},{"issue":"3","key":"4336_CR25","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky O, Deng J, Su H, Krause J, Satheesh S, Ma S, Huang Z, Karpathy A, Khosla A, Bernstein M, Berg AC, Fei-Fei L (2015) ImageNet large scale visual recognition challenge. Int J Comput Vis (IJCV) 115(3):211\u2013252. https:\/\/doi.org\/10.1007\/s11263-015-0816-y","journal-title":"Int J Comput Vis (IJCV)"},{"key":"4336_CR26","doi-asserted-by":"crossref","unstructured":"Shi S, Wang Q, Xu P, Chu X (2016) Benchmarking state-of-the-art deep learning software tools. In: 2016 7th International Conference on Cloud Computing and Big Data (CCBD), IEEE, pp 99\u2013104","DOI":"10.1109\/CCBD.2016.029"},{"key":"4336_CR27","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556"},{"key":"4336_CR28","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"4336_CR29","doi-asserted-by":"crossref","unstructured":"Tan G, Li L, Triechle S, Phillips E, Bao Y, Sun N (2011) Fast implementation of dgemm on fermi GPU. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis, pp 1\u201311","DOI":"10.1145\/2063384.2063431"},{"key":"4336_CR30","doi-asserted-by":"crossref","unstructured":"Vasudevan A, Anderson A, Gregg D (2017) Parallel multi channel convolution using general matrix multiplication. In: 2017 IEEE 28th International Conference on Application-Specific Systems, Architectures and Processors (ASAP), IEEE, pp 19\u201324","DOI":"10.1109\/ASAP.2017.7995254"},{"key":"4336_CR31","doi-asserted-by":"crossref","unstructured":"Xie S, Girshick R, Doll\u00e1r P, Tu Z, He K (2017) Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1492\u20131500","DOI":"10.1109\/CVPR.2017.634"},{"key":"4336_CR32","doi-asserted-by":"crossref","unstructured":"Yan D, Wang W, Chu X (2020) Optimizing batched winograd convolution on GPUs. In: Proceedings of the 25th ACM SIGPLAN symposium on principles and practice of parallel programming, pp 32\u201344","DOI":"10.1145\/3332466.3374520"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-022-04336-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-022-04336-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-022-04336-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,4]],"date-time":"2022-07-04T14:09:48Z","timestamp":1656943788000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-022-04336-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,3,19]]},"references-count":32,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2022,7]]}},"alternative-id":["4336"],"URL":"https:\/\/doi.org\/10.1007\/s11227-022-04336-3","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,3,19]]},"assertion":[{"value":"24 January 2022","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 March 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}