{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T19:40:20Z","timestamp":1782934820918,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","funder":[{"name":"the Strategic Priority Research Program of Chinese Academy of Sciences","award":["XDB0500102"],"award-info":[{"award-number":["XDB0500102"]}]},{"name":"National Science Foundation of China","award":["62032023, 92270206, T2125013, 62372435, 61972377, T2293702, 62322201"],"award-info":[{"award-number":["62032023, 92270206, T2125013, 62372435, 61972377, T2293702, 62322201"]}]},{"name":"CAS Project for Young Scientists in Basic Research","award":["YSBR-005"],"award-info":[{"award-number":["YSBR-005"]}]},{"name":"Beijing Natural Science Foundation","award":["4254087"],"award-info":[{"award-number":["4254087"]}]},{"name":"China National Postdoctoral Program for Innovative Talents","award":["BX20240383"],"award-info":[{"award-number":["BX20240383"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,28]]},"DOI":"10.1145\/3719276.3725173","type":"proceedings-article","created":{"date-parts":[[2025,7,4]],"date-time":"2025-07-04T05:00:46Z","timestamp":1751605246000},"page":"195-204","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["FastSpMM: Leveraging Tensor Cores for Sparse Matrix Multiplication"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7846-1881","authenticated-orcid":false,"given":"Hongyu","family":"Wang","sequence":"first","affiliation":[{"name":"School of Advanced Interdisciplinary Sciences, University of Chinese Academy of Sciences, Beijing, China and SKLP, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4115-9072","authenticated-orcid":false,"given":"Mingzhen","family":"Li","sequence":"additional","affiliation":[{"name":"SKLP, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China and University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8539-8326","authenticated-orcid":false,"given":"Weile","family":"Jia","sequence":"additional","affiliation":[{"name":"SKLP, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China and University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1101-7927","authenticated-orcid":false,"given":"Hailong","family":"Yang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6361-5948","authenticated-orcid":false,"given":"Guangming","family":"Tan","sequence":"additional","affiliation":[{"name":"SKLP, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China and University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,7,4]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"Peter Benner and Thomas Mach. 2011. Locally Optimal Block Preconditioned Conjugate Gradient Method for Hierarchical Matrices. PAMM 11 (2011). https:\/\/api.semanticscholar.org\/CorpusID:121137642","DOI":"10.1002\/pamm.201110360"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607087"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476182"},{"key":"e_1_3_3_1_5_2","unstructured":"Rewon Child Scott Gray Alec Radford and Ilya Sutskever. 2019. Generating Long Sequences with Sparse Transformers. ArXiv abs\/1904.10509 (2019). https:\/\/api.semanticscholar.org\/CorpusID:129945531"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651378"},{"key":"e_1_3_3_1_7_2","series-title":"(SC \u201920)","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Gale Trevor","year":"2020","unstructured":"Trevor Gale, Matei Zaharia, Cliff Young, and Erich Elsen. 2020. Sparse GPU kernels for deep learning. In Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis (Atlanta, Georgia) (SC \u201920). IEEE Press, Article 17, 14\u00a0pages."},{"key":"e_1_3_3_1_8_2","unstructured":"Martin\u00a0H. Gutknecht. 2005. BLOCK KRYLOV SPACE METHODS FOR LINEAR SYSTEMS WITH MULTIPLE RIGHT-HAND SIDES : AN. https:\/\/api.semanticscholar.org\/CorpusID:10923448"},{"key":"e_1_3_3_1_9_2","unstructured":"Torsten Hoefler Dan Alistarh Tal Ben-Nun Nikoli Dryden and Alexandra Peste. 2021. Sparsity in deep learning: pruning and growth for efficient inference and training in neural networks. J. Mach. Learn. Res. 22 1 Article 241 (jan 2021) 124\u00a0pages."},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3293883.3295712"},{"key":"e_1_3_3_1_11_2","unstructured":"Weihua Hu Matthias Fey Hongyu Ren Maho Nakata Yuxiao Dong and Jure Leskovec. 2021. OGB-LSC: A Large-Scale Challenge for Machine Learning on Graphs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2103.09430 (2021)."},{"key":"e_1_3_3_1_12_2","series-title":"(SC \u201920)","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Huang Guyue","year":"2020","unstructured":"Guyue Huang, Guohao Dai, Yu Wang, and Huazhong Yang. 2020. GE-SpMM: general-purpose sparse matrix-matrix multiplication on GPUs for graph neural networks. In Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis (Atlanta, Georgia) (SC \u201920). IEEE Press, Article 72, 12\u00a0pages."},{"key":"e_1_3_3_1_13_2","unstructured":"Omid Jafari Preeti Maurya Parth Nagarkar Khandker\u00a0Mushfiqul Islam and Chidambaram Crushev. 2021. A Survey on Locality Sensitive Hashing Algorithms and their Applications. ArXiv abs\/2102.08942 (2021). https:\/\/api.semanticscholar.org\/CorpusID:231942424"},{"key":"e_1_3_3_1_14_2","unstructured":"Andrew Kerr Duane Merrill Julien Demouth and John Tran. 2017. CUTLASS: Fast Linear Algebra in CUDA C++. https:\/\/developer.nvidia.com\/blog\/cutlass-linear-algebra-cuda\/."},{"key":"e_1_3_3_1_15_2","unstructured":"Thomas\u00a0N. Kipf and Max Welling. 2016. Semi-Supervised Classification with Graph Convolutional Networks. CoRR abs\/1609.02907 (2016). arXiv:https:\/\/arXiv.org\/abs\/1609.02907http:\/\/arxiv.org\/abs\/1609.02907"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Jure Leskovec and Rok Sosi\u010d. 2016. SNAP: A General-Purpose Network Analysis and Graph-Mining Library. ACM Trans. Intell. Syst. Technol. 8 1 Article 1 (jul 2016) 20\u00a0pages. https:\/\/doi.org\/10.1145\/2898361","DOI":"10.1145\/2898361"},{"key":"e_1_3_3_1_17_2","series-title":"(SC \u201922)","volume-title":"Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis","author":"Li Shigang","year":"2022","unstructured":"Shigang Li, Kazuki Osawa, and Torsten Hoefler. 2022. Efficient quantized sparse matrix operations on tensor cores. In Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis (Dallas, Texas) (SC \u201922). IEEE Press, Article 37, 15\u00a0pages."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441581"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607051"},{"key":"e_1_3_3_1_20_2","volume-title":"GPU Technology Conference","author":"Naumov Maxim","year":"2010","unstructured":"Maxim Naumov, L. Chien, Philippe Vandermersch, and Ujval Kapasi. 2010. Cusparse library. In GPU Technology Conference."},{"key":"e_1_3_3_1_21_2","unstructured":"NVIDIA Corporation. 2020. NVIDIA A100 Tensor Core GPU Architecture. https:\/\/www.nvidia.com\/content\/dam\/en-zz\/Solutions\/Data-Center\/nvidia-ampere-architecture-whitepaper.pdf NVIDIA Whitepaper."},{"key":"e_1_3_3_1_22_2","volume-title":"CUDA C Programming Guide","author":"Corporation NVIDIA","year":"2024","unstructured":"NVIDIA Corporation. 2024. CUDA C Programming Guide. https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html Accessed: 2024-08-16."},{"key":"e_1_3_3_1_23_2","volume-title":"NVIDIA CUDA Parallel Thread Execution (PTX) ISA Documentation","author":"Corporation NVIDIA","year":"2024","unstructured":"NVIDIA Corporation. 2024. NVIDIA CUDA Parallel Thread Execution (PTX) ISA Documentation. https:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/ Accessed: 2024-08-16."},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3627535.3638470"},{"key":"e_1_3_3_1_25_2","unstructured":"S.\u00a0Roger Qiu You Liang and Zheng Wang. 2021. Optimizing Sparse Matrix Multiplications for Graph Neural Networks. ArXiv abs\/2111.00352 (2021). https:\/\/api.semanticscholar.org\/CorpusID:240353823"},{"key":"e_1_3_3_1_26_2","unstructured":"Minjie Wang Da Zheng Zihao Ye Quan Gan Mufei Li Xiang Song Jinjing Zhou Chao Ma Lingfan Yu Yu Gai Tianjun Xiao Tong He George Karypis Jinyang Li and Zheng Zhang. 2019. Deep Graph Library: A Graph-Centric Highly-Performant Package for Graph Neural Networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1909.01315 (2019)."},{"key":"e_1_3_3_1_27_2","first-page":"149","volume-title":"2023 USENIX Annual Technical Conference (USENIX ATC 23)","author":"Wang Yuke","year":"2023","unstructured":"Yuke Wang, Boyuan Feng, Zheng Wang, Guyue Huang, and Yufei Ding. 2023. TC-GNN: Bridging Sparse GNN Computation and Dense Tensor Cores on GPUs. In 2023 USENIX Annual Technical Conference (USENIX ATC 23). USENIX Association, Boston, MA, 149\u2013164. https:\/\/www.usenix.org\/conference\/atc23\/presentation\/wang-yuke"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"crossref","unstructured":"Haojun Xia Zhen Zheng Yuchao Li Donglin Zhuang Zhongzhu Zhou Xiafei Qiu Yong Li Wei Lin and Shuaiwen\u00a0Leon Song. 2023. Flash-LLM: Enabling Cost-Effective and Highly-Efficient Large Generative Model Inference with Unstructured Sparsity. Proc. VLDB Endow. 17 2 (oct 2023) 211\u2013224. https:\/\/doi.org\/10.14778\/3626292.3626303","DOI":"10.14778\/3626292.3626303"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"crossref","unstructured":"Zeyu Xue Mei Wen Zhaoyun Chen Yang Shi Minjin Tang Jianchao Yang and Zhongdi Luo. 2023. Releasing the Potential of Tensor Core for Unstructured SpMM using Tiled-CSR Format. 2023 IEEE 41st International Conference on Computer Design (ICCD) (2023) 457\u2013464. https:\/\/api.semanticscholar.org\/CorpusID:266494546","DOI":"10.1109\/ICCD58817.2023.00076"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582047"}],"event":{"name":"CF '25: 22nd ACM International Conference on Computing Frontiers","location":"Cagliari Italy","acronym":"CF '25","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"]},"container-title":["Proceedings of the 22nd ACM International Conference on Computing Frontiers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3719276.3725173","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,21]],"date-time":"2025-07-21T09:49:56Z","timestamp":1753091396000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3719276.3725173"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,28]]},"references-count":29,"alternative-id":["10.1145\/3719276.3725173","10.1145\/3719276"],"URL":"https:\/\/doi.org\/10.1145\/3719276.3725173","relation":{},"subject":[],"published":{"date-parts":[[2025,5,28]]},"assertion":[{"value":"2025-07-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}