{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T12:47:34Z","timestamp":1770814054809,"version":"3.50.1"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2025,4,15]],"date-time":"2025-04-15T00:00:00Z","timestamp":1744675200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,4,15]],"date-time":"2025-04-15T00:00:00Z","timestamp":1744675200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Major Science and Technology Special Projects in Henan Province","award":["221100210600"],"award-info":[{"award-number":["221100210600"]}]},{"name":"Basic Research Projects of Key Scientific Research Projects Plan in Henan Higher Education Institutions","award":["25ZX013"],"award-info":[{"award-number":["25ZX013"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-025-07195-w","type":"journal-article","created":{"date-parts":[[2025,4,15]],"date-time":"2025-04-15T06:01:55Z","timestamp":1744696915000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["VBATS: an adaptive strategy for grouped GEMM on GPUs"],"prefix":"10.1007","volume":"81","author":[{"given":"Jiandong","family":"Shang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhuxin","family":"Wen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haobo","family":"Hua","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hengliang","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gang","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenlong","family":"Fan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangsheng","family":"Qin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,4,15]]},"reference":[{"key":"7195_CR1","unstructured":"Jia Y (2014) Learning semantic image representations at a large scale. https:\/\/api.semanticscholar.org\/CorpusID:15611603"},{"key":"7195_CR2","doi-asserted-by":"crossref","unstructured":"Chen Z, Wang J, He H, Huang X (2014) A fast deep learning system using gpu. In: 2014 IEEE International Symposium on Circuits and Systems (ISCAS), pp. 1552\u20131555 IEEE","DOI":"10.1109\/ISCAS.2014.6865444"},{"key":"7195_CR3","doi-asserted-by":"crossref","unstructured":"Jouppi NP, Young C, Patil N, Patterson D, Agrawal G, Bajwa R, Bates S, Bhatia S, Boden N, Borchers A, etal (2017) In-datacenter performance analysis of a tensor processing unit. In: Proceedings of the 44th Annual International Symposium on Computer Architecture, pp. 1\u201312","DOI":"10.1145\/3079856.3080246"},{"key":"7195_CR4","doi-asserted-by":"crossref","unstructured":"Markidis S, Der\u00a0Chien SW, Laure E, Peng IB, Vetter JS (2018) Nvidia tensor core programmability, performance & precision. In: 2018 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW), pp. 522\u2013531. IEEE","DOI":"10.1109\/IPDPSW.2018.00091"},{"key":"7195_CR5","doi-asserted-by":"crossref","unstructured":"Kim R, Choi J, Lee M (2019) Optimizing parallel gemm routines using auto-tuning with intel avx-512. In: Proceedings of the International Conference on High Performance Computing in Asia-Pacific Region, pp. 101\u2013110","DOI":"10.1145\/3293320.3293334"},{"issue":"11","key":"7195_CR6","doi-asserted-by":"publisher","first-page":"2045","DOI":"10.1109\/TPDS.2011.311","volume":"23","author":"J Kurzak","year":"2012","unstructured":"Kurzak J, Tomov S, Dongarra J (2012) Autotuning gemm kernels for the fermi gpu. IEEE Trans Parallel Distrib Syst 23(11):2045\u20132057","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"7195_CR7","doi-asserted-by":"crossref","unstructured":"Tan G, Li L, Triechle S, Phillips E, Bao Y, Sun N (2011) Fast implementation of dgemm on fermi gpu. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201311","DOI":"10.1145\/2063384.2063431"},{"key":"7195_CR8","doi-asserted-by":"crossref","unstructured":"Jiang C, Snir M (2005) Automatic tuning matrix multiplication performance on graphics hardware. In: 14th International Conference on Parallel Architectures and Compilation Techniques (PACT\u201905), pp. 185\u2013194 IEEE","DOI":"10.1109\/PACT.2005.10"},{"key":"7195_CR9","doi-asserted-by":"crossref","unstructured":"Lai J, Seznec A (2013) Performance upper bound analysis and optimization of sgemm on fermi and kepler gpus. In: Proceedings of the 2013 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO), pp. 1\u201310 IEEE","DOI":"10.1109\/CGO.2013.6494986"},{"issue":"4","key":"7195_CR10","doi-asserted-by":"publisher","first-page":"511","DOI":"10.1177\/1094342010385729","volume":"24","author":"R Nath","year":"2010","unstructured":"Nath R, Tomov S, Dongarra J (2010) An improved magma gemm for fermi graphics processing units. Int J High Perform Comput Appl 24(4):511\u2013515","journal-title":"Int J High Perform Comput Appl"},{"key":"7195_CR11","unstructured":"CUTLASS (2024) Fast Linear Algebra in CUDA C++: https:\/\/devblogs.nvidia.com\/cutlass-linear-algebra-cuda\/"},{"key":"7195_CR12","unstructured":"Bao W, Chang L-W, Chen Y, Deng K, Agarwal A, Barsoum E, Taha A (2019) Ngemm: Optimizing gemm for deep learning via compiler-based techniques. arXiv preprint arXiv:1910.00178"},{"key":"7195_CR13","doi-asserted-by":"crossref","unstructured":"Fatahalian K, Sugerman J, Hanrahan P (2004) Understanding the efficiency of gpu algorithms for matrix-matrix multiplication. In: Proceedings of the ACM SIGGRAPH\/EUROGRAPHICS Conference on Graphics Hardware, pp. 133\u2013137","DOI":"10.1145\/1058129.1058148"},{"key":"7195_CR14","unstructured":"AMD rocBLAS Next generation BLAS implementation for ROCm platform: https:\/\/github.com\/ROCm\/rocBLAS (2024)"},{"key":"7195_CR15","unstructured":"NVIDIA CUDA Documentation: https:\/\/docs.nvidia.com\/cuda\/cublas\/index.html (2024)"},{"key":"7195_CR16","unstructured":"Inter oneMKL:Math Kernel Library: https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/tools\/oneapi\/onemkl.html (2024)"},{"key":"7195_CR17","doi-asserted-by":"crossref","unstructured":"Abdelfattah A, Haidar A, Tomov S, Dongarra J (2017) Novel hpc techniques to batch execution of many variable size blas computations on gpus. In: Proceedings of the International Conference on Supercomputing, pp. 1\u201310","DOI":"10.1145\/3079079.3079103"},{"issue":"3","key":"7195_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3431921","volume":"47","author":"A Abdelfattah","year":"2021","unstructured":"Abdelfattah A, Costa T, Dongarra J, Gates M, Haidar A, Hammarling S, Higham NJ, Kurzak J, Luszczek P, Tomov S et al (2021) A set of batched basic linear algebra subprograms and lapack routines. ACM Trans Math Softw 47(3):1\u201323","journal-title":"ACM Trans Math Softw"},{"key":"7195_CR19","doi-asserted-by":"crossref","unstructured":"Vasudevan A, Anderson A, Gregg D (2017) Parallel multi channel convolution using general matrix multiplication. In: 2017 IEEE 28th International Conference on Application-specific Systems, Architectures and Processors (ASAP), pp. 19\u201324 IEEE","DOI":"10.1109\/ASAP.2017.7995254"},{"key":"7195_CR20","doi-asserted-by":"crossref","unstructured":"Anderson MJ, Sheffield D, Keutzer K (2012) A predictive model for solving small linear algebra problems in gpu registers. In: 2012 IEEE 26th International Parallel and Distributed Processing Symposium, pp. 2\u201313 IEEE","DOI":"10.1109\/IPDPS.2012.11"},{"key":"7195_CR21","doi-asserted-by":"crossref","unstructured":"Messer OB, Harris JA, Parete-Koon S, Chertkow MA (2013) Multicore and accelerator development for a leadership-class stellar astrophysics code. In: Applied Parallel and Scientific Computing: 11th International Conference, PARA 2012, Helsinki, Finland, June 10-13, 2012, Revised Selected Papers 11, pp. 92\u2013106 Springer","DOI":"10.1007\/978-3-642-36803-5_6"},{"key":"7195_CR22","doi-asserted-by":"crossref","unstructured":"Li X, Liang Y, Yan S, Jia L, Li Y (2019) A coordinated tiling and batching framework for efficient gemm on gpus. In: Proceedings of the 24th Symposium on Principles and Practice of Parallel Programming, pp. 229\u2013241","DOI":"10.1145\/3293883.3295734"},{"issue":"2","key":"7195_CR23","doi-asserted-by":"publisher","first-page":"1741","DOI":"10.1007\/s11227-021-03936-9","volume":"78","author":"R Wang","year":"2022","unstructured":"Wang R, Yang Z, Xu H, Lu L (2022) A high-performance batched matrix multiplication framework for gpus under unbalanced input distribution. J Supercomput 78(2):1741\u20131758","journal-title":"J Supercomput"},{"key":"7195_CR24","doi-asserted-by":"publisher","unstructured":"Yang Z, Lu L, Wang R (2022) A batched gemm optimization framework for deep learning. The Journal of Supercomputing 78 https:\/\/doi.org\/10.1007\/s11227-022-04336-3","DOI":"10.1007\/s11227-022-04336-3"},{"key":"7195_CR25","doi-asserted-by":"crossref","unstructured":"Hayes AB, Li L, Chavarr\u00eda-Miranda D, Song SL, Zhang EZ (2016) Orion: A framework for gpu occupancy tuning. In: Proceedings of the 17th International Middleware Conference, pp. 1\u201313","DOI":"10.1145\/2988336.2988355"},{"key":"7195_CR26","doi-asserted-by":"publisher","unstructured":"Zhang Y, Wang Y, Mo Z, Zhou Y, Sun T, Xu G, Xing C, Yang L (2022) Accelerating small matrix multiplications by adaptive batching strategy on gpu, pp. 882\u2013887 https:\/\/doi.org\/10.1109\/HPCC-DSS-SmartCity-DependSys57074.2022.00143","DOI":"10.1109\/HPCC-DSS-SmartCity-DependSys57074.2022.00143"},{"key":"7195_CR27","unstructured":"Chetlur S, Woolley C, Vandermersch P, Cohen J, Tran J, Catanzaro B, Shelhamer E (2014) cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759"},{"key":"7195_CR28","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"7195_CR29","unstructured":"Iandola FN, Han S, Moskewicz MW, Ashraf K, Dally WJ, Keutzer K (2016) Squeezenet: Alexnet-level accuracy with 50x fewer parameters and< 0.5 mb model size. arXiv preprint arXiv:1602.07360"},{"key":"7195_CR30","doi-asserted-by":"crossref","unstructured":"Xie S, Girshick R, Doll\u00e1r P, Tu Z, He K (2017) Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1492\u20131500","DOI":"10.1109\/CVPR.2017.634"},{"key":"7195_CR31","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07195-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-025-07195-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07195-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,15]],"date-time":"2025-04-15T06:02:14Z","timestamp":1744696934000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-025-07195-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,15]]},"references-count":31,"journal-issue":{"issue":"5","published-online":{"date-parts":[[2025,4]]}},"alternative-id":["7195"],"URL":"https:\/\/doi.org\/10.1007\/s11227-025-07195-w","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,4,15]]},"assertion":[{"value":"13 March 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 April 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"No potential conflict of interest was reported by the authors.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}],"article-number":"745"}}