{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T12:09:12Z","timestamp":1780056552658,"version":"3.54.0"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T00:00:00Z","timestamp":1780012800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T00:00:00Z","timestamp":1780012800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"crossref","award":["NRF-RS-2024-00359076"],"award-info":[{"award-number":["NRF-RS-2024-00359076"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-026-08606-2","type":"journal-article","created":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T11:41:46Z","timestamp":1780054906000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Block-structured matrix reordering for efficient SDDMM on tensor cores"],"prefix":"10.1007","volume":"82","author":[{"given":"Chengxing","family":"Zou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Changwan","family":"Hong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gordon Euhyun","family":"Moon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinsung","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,29]]},"reference":[{"key":"8606_CR1","doi-asserted-by":"crossref","unstructured":"Bharadwaj V, Bulu\u00e7 A, Demmel J (2022) Distributed-memory sparse kernels for machine learning. In: 2022 IEEE International Parallel and Distributed Processing Symposium (IPDPS). IEEE, pp 47\u201358","DOI":"10.1109\/IPDPS53621.2022.00014"},{"key":"8606_CR2","doi-asserted-by":"crossref","unstructured":"Hegde K, Asghari-Moghaddam H, Pellauer M, Crago N, Jaleel A, Solomonik E, Emer J, Fletcher CW (2019) Extensor: an accelerator for sparse tensor algebra. In: Proceedings of the 52nd Annual IEEE\/ACM International Symposium on Microarchitecture, pp 319\u2013333","DOI":"10.1145\/3352460.3358275"},{"key":"8606_CR3","unstructured":"Liu J, Cai Z, Chen Z, Wang M (2024) Df-gnn: Dynamic fusion framework for attention graph neural networks on gpus. Preprint at arXiv:2411.16127"},{"key":"8606_CR4","unstructured":"Kipf TN, Welling M (2016) Semi-supervised classification with graph convolutional networks. Preprint at arXiv:1609.02907"},{"key":"8606_CR5","doi-asserted-by":"publisher","first-page":"1617","DOI":"10.1016\/j.ins.2022.06.075","volume":"607","author":"S Kumar","year":"2022","unstructured":"Kumar S, Mallik A, Khetarpal A, Panda BS (2022) Influence maximization in social networks using graph embedding and graph neural network. Inf Sci 607:1617\u20131636","journal-title":"Inf Sci"},{"key":"8606_CR6","doi-asserted-by":"crossref","unstructured":"Liu B, Wang M, Foroosh H, Tappen M, Pensky M (2015) Sparse convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 806\u2013814","DOI":"10.1109\/CVPR.2015.7298681"},{"issue":"1","key":"8606_CR7","doi-asserted-by":"publisher","first-page":"61","DOI":"10.1109\/TNN.2008.2005605","volume":"20","author":"F Scarselli","year":"2008","unstructured":"Scarselli F, Gori M, Tsoi AC, Hagenbuchner M, Monfardini G (2008) The graph neural network model. IEEE Trans Neural Netw 20(1):61\u201380","journal-title":"IEEE Trans Neural Netw"},{"issue":"12","key":"8606_CR8","first-page":"3165","volume":"71","author":"L Liu","year":"2022","unstructured":"Liu L, Qu Z, Chen Z, Tu F, Ding Y, Xie Y (2022) Dynamic sparse attention for scalable transformer acceleration. IEEE Trans Comput 71(12):3165\u20133178","journal-title":"IEEE Trans Comput"},{"key":"8606_CR9","unstructured":"Thekumparampil KK, Wang C, Oh S, Li L-J (2018) Attention-based graph neural network for semi-supervised learning. Preprint at arXiv:1803.03735"},{"key":"8606_CR10","unstructured":"Veli\u010dkovi\u0107 P, Cucurull G, Casanova A, Romero A, Lio P, Bengio Y (2017) Graph attention networks. Preprint at arXiv:1710.10903"},{"key":"8606_CR11","unstructured":"Hu Y, Li J, Yu Z, Zhang Z (2022) Analysis and optimization of gnn-based recommender systems on persistent memory. Preprint at arXiv:2207.11918"},{"key":"8606_CR12","doi-asserted-by":"crossref","unstructured":"Rahman MK, Sujon MH, Azad A (2021) Fusedmm: A unified sddmm-spmm kernel for graph embedding and graph neural networks. In: 2021 IEEE International Parallel and Distributed Processing Symposium (IPDPS). IEEE, pp 256\u2013266","DOI":"10.1109\/IPDPS49936.2021.00034"},{"issue":"3","key":"8606_CR13","doi-asserted-by":"publisher","first-page":"793","DOI":"10.1007\/s10115-013-0682-2","volume":"41","author":"H-F Yu","year":"2014","unstructured":"Yu H-F, Hsieh C-J, Si S, Dhillon IS (2014) Parallel matrix factorization for recommender systems. Knowl Inf Syst 41(3):793\u2013819","journal-title":"Knowl Inf Syst"},{"key":"8606_CR14","doi-asserted-by":"crossref","unstructured":"Canny J (2002) Collaborative filtering with privacy. In: Proceedings 2002 IEEE Symposium on Security and Privacy. IEEE, pp 45\u201357","DOI":"10.1109\/SECPRI.2002.1004361"},{"key":"8606_CR15","doi-asserted-by":"crossref","unstructured":"Lan AS, Waters AE, Studer C, Baraniuk RG (2013) Sparse factor analysis for learning and content analytics. Preprint at arXiv:1303.5685","DOI":"10.1109\/ICASSP.2013.6639380"},{"key":"8606_CR16","doi-asserted-by":"crossref","unstructured":"Zhao H, Jiang B, Canny JF, Jaros B (2015) Same but different: Fast and high quality gibbs parameter estimation. In: Proceedings of the 21th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, pp 1495\u20131502","DOI":"10.1145\/2783258.2783416"},{"key":"8606_CR17","first-page":"993","volume":"3","author":"DM Blei","year":"2003","unstructured":"Blei DM, Ng AY, Jordan MI (2003) Latent Dirichlet allocation. J Mach Learn Res 3:993\u20131022","journal-title":"J Mach Learn Res"},{"key":"8606_CR18","doi-asserted-by":"crossref","unstructured":"Canny J (2004) Gap: a factor model for discrete data. In: Proceedings of the 27th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval, pp 122\u2013129","DOI":"10.1145\/1008992.1009016"},{"key":"8606_CR19","unstructured":"Titsias M (2007) The infinite Gamma-Poisson feature model. Adv Neural Inf Process Syst 20"},{"key":"8606_CR20","doi-asserted-by":"crossref","unstructured":"Hong C, Sukumaran-Rajam A, Nisa I, Singh K, Sadayappan P (2019) Adaptive sparse tiling for sparse matrix multiplication. In: Proceedings of the 24th Symposium on Principles and Practice of Parallel Programming, pp 300\u2013314","DOI":"10.1145\/3293883.3295712"},{"key":"8606_CR21","doi-asserted-by":"crossref","unstructured":"Pang M, Fei X, Qu P, Zhang Y, Li Z (2024) A row decomposition-based approach for sparse matrix multiplication on gpus. In: Proceedings of the 29th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming, pp 377\u2013389","DOI":"10.1145\/3627535.3638470"},{"key":"8606_CR22","unstructured":"Wang Y, Feng B, Wang Z, Huang G, Ding Y (2023) $$\\{$$TC-GNN$$\\}$$: Bridging sparse $$\\{$$GNN$$\\}$$ computation and dense tensor cores on $$\\{$$GPUs$$\\}$$. In: 2023 USENIX Annual Technical Conference (USENIX ATC 23), pp 149\u2013164"},{"key":"8606_CR23","doi-asserted-by":"crossref","unstructured":"Shi J, Li S, Xu Y, Fu R, Wang X, Wu T (2025) Flashsparse: Minimizing computation redundancy for fast sparse matrix multiplications on tensor cores. In: Proceedings of the 30th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming. pp 312\u2013325","DOI":"10.1145\/3710848.3710858"},{"issue":"2","key":"8606_CR24","doi-asserted-by":"publisher","first-page":"428","DOI":"10.1109\/TPDS.2015.2401575","volume":"27","author":"D Langr","year":"2015","unstructured":"Langr D, Tvrdik P (2015) Evaluation criteria for sparse matrix storage formats. IEEE Trans Parallel Distrib Syst 27(2):428\u2013440","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"8606_CR25","doi-asserted-by":"crossref","unstructured":"Luebke D, Harris M, Govindaraju N, Lefohn A, Houston M, Owens J, Segal M, Papakipos M, Buck I (2006) Gpgpu: general-purpose computation on graphics hardware. In: Proceedings of the 2006 ACM\/IEEE Conference on Supercomputing, pp 208","DOI":"10.1145\/1188455.1188672"},{"issue":"5","key":"8606_CR26","doi-asserted-by":"publisher","first-page":"879","DOI":"10.1109\/JPROC.2008.917757","volume":"96","author":"JD Owens","year":"2008","unstructured":"Owens JD, Houston M, Luebke D, Green S, Stone JE, Phillips JC (2008) GPU computing. Proc IEEE 96(5):879\u2013899","journal-title":"Proc IEEE"},{"key":"8606_CR27","unstructured":"Xiang L, Asudeh O, Sabin G, Sukumaran-Rajam A, Sadayappan P (2025) cutespmm: Accelerating sparse-dense matrix multiplication using GPU tensor cores. Preprint at arXiv:2504.06443"},{"key":"8606_CR28","doi-asserted-by":"crossref","unstructured":"Nisa I, Sukumaran-Rajam A, Kurt SE, Hong C, Sadayappan P (2018) Sampled dense matrix multiplication for high-performance machine learning. In: IEEE 25th International Conference on High Performance Computing (HiPC). IEEE, pp 32\u201341","DOI":"10.1109\/HiPC.2018.00013"},{"key":"8606_CR29","unstructured":"Bradley T (2012) GPU performance analysis and optimisation. NVIDIA Corporation. pp 1\u2013117"},{"key":"8606_CR30","unstructured":"NVIDIA T (2017) V100 GPU architecture whitepaper. WP-08608-001"},{"key":"8606_CR31","unstructured":"NVIDIA: cuBLAS. https:\/\/developer.nvidia.com\/cublas. Accessed 30 July 2025 (n.d.)"},{"key":"8606_CR32","doi-asserted-by":"crossref","unstructured":"Gale T, Zaharia M, Young C, Elsen E (2020) Sparse gpu kernels for deep learning. In: SC20: International Conference for High Performance Computing, Networking, Storage and Analysis. IEEE, pp 1\u201314","DOI":"10.1109\/SC41405.2020.00021"},{"key":"8606_CR33","doi-asserted-by":"crossref","unstructured":"Jiang P, Hong C, Agrawal G (2020) A novel data transformation and execution strategy for accelerating sparse matrix multiplication on gpus. In: Proceedings of the 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp 376\u2013388","DOI":"10.1145\/3332466.3374546"},{"key":"8606_CR34","doi-asserted-by":"crossref","unstructured":"Lee E, Han Y, Moon GE (2024) Accelerated block-sparsity-aware matrix reordering for leveraging tensor cores in sparse matrix-multivector multiplication. In: European Conference on Parallel Processing, Springer, pp 3\u201316","DOI":"10.1007\/978-3-031-69583-4_1"},{"key":"8606_CR35","unstructured":"Shi J, Li S, Xu Y, Wang X, Fu R, Ma Z, Wu T (2025) Libra: Synergizing cuda and tensor cores for high-performance sparse matrix multiplication. Preprint at arXiv:2506.22714"},{"key":"8606_CR36","doi-asserted-by":"crossref","unstructured":"Palmer B, Ghosh S, M\u00e1rquez A (2024) Distributed-memory sparse deep neural network inference using global arrays. In: 2024 IEEE High Performance Extreme Computing Conference (HPEC). IEEE, pp 1\u20137","DOI":"10.1109\/HPEC62836.2024.10938434"},{"issue":"1","key":"8606_CR37","first-page":"1","volume":"38","author":"TA Davis","year":"2011","unstructured":"Davis TA, Hu Y (2011) The university of Florida sparse matrix collection. ACM Trans on Mat Softw (TOMS) 38(1):1\u201325","journal-title":"ACM Trans on Mat Softw (TOMS)"},{"key":"8606_CR38","unstructured":"Naumov M, Chien L, Vandermersch P, Kapasi U (2010) Cusparse library. In: GPU Technology Conference, vol. 12"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08606-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-026-08606-2","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08606-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T11:42:09Z","timestamp":1780054929000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-026-08606-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,29]]},"references-count":38,"journal-issue":{"issue":"8","published-online":{"date-parts":[[2026,6]]}},"alternative-id":["8606"],"URL":"https:\/\/doi.org\/10.1007\/s11227-026-08606-2","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,29]]},"assertion":[{"value":"30 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 May 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"465"}}