{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T03:30:26Z","timestamp":1769657426197,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","funder":[{"name":"National Key Research and Development Program of China","award":["2023YFB3001801"],"award-info":[{"award-number":["2023YFB3001801"]}]},{"name":"National Natural Science Foundation of China","award":["62322201, U23B2020, U22A2028, 92373110"],"award-info":[{"award-number":["62322201, U23B2020, U22A2028, 92373110"]}]},{"name":"Fundamental Research Funds for the Central Universities","award":["JKF-2025012343648, JKF-20240598"],"award-info":[{"award-number":["JKF-2025012343648, JKF-20240598"]}]},{"name":"State Key Laboratory of Complex & Critical Software Environment","award":["SKLCCSE-2025ZX-04"],"award-info":[{"award-number":["SKLCCSE-2025ZX-04"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,1,28]]},"DOI":"10.1145\/3774934.3786441","type":"proceedings-article","created":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T15:25:57Z","timestamp":1769613957000},"page":"245-258","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Exploiting Efficient Mapping and Pipelined Execution for Accelerating SpMV on Tensor Cores"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-3261-3483","authenticated-orcid":false,"given":"Kaige","family":"Zhang","sequence":"first","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1101-7927","authenticated-orcid":false,"given":"Hailong","family":"Yang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5163-4607","authenticated-orcid":false,"given":"Xin","family":"You","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4830-5482","authenticated-orcid":false,"given":"Tianyu","family":"Feng","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7787-6460","authenticated-orcid":false,"given":"Yufan","family":"Xu","sequence":"additional","affiliation":[{"name":"Independent Researcher, Cupertino, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7186-0556","authenticated-orcid":false,"given":"Zhongzhi","family":"Luan","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1829-2817","authenticated-orcid":false,"given":"Yi","family":"Liu","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5382-1473","authenticated-orcid":false,"given":"Depei","family":"Qian","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,1,28]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477141"},{"key":"e_1_3_2_2_2_1","volume-title":"Auto-spmv: Automated optimizing spmv kernels on gpu. arXiv preprint arXiv:2302.05662.","author":"Ashoury Mina","year":"2023","unstructured":"Mina Ashoury, Mohammad Loni, Farshad Khunjush, and Masoud Daneshtalab. 2023. Auto-spmv: Automated optimizing spmv kernels on gpu. arXiv preprint arXiv:2302.05662."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607087"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3673038.3673055"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3572848.3577500"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2021.3061394"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00071"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651378"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441599"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2023.104799"},{"key":"e_1_3_2_2_11_1","unstructured":"Jianhua Gao Bingjie Liu Weixing Ji and Hua Huang. 2024. A systematic literature survey of sparse matrix-vector multiplication. arXiv preprint arXiv:2404.06047."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11075-005-1526-2"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2024.3477431"},{"key":"e_1_3_2_2_14_1","volume-title":"Open graph benchmark: Datasets for machine learning on graphs. Advances in neural information processing systems, 33","author":"Hu Weihua","year":"2020","unstructured":"Weihua Hu, Matthias Fey, Marinka Zitnik, Yuxiao Dong, Hongyu Ren, Bowen Liu, Michele Catasta, and Jure Leskovec. 2020. Open graph benchmark: Datasets for machine learning on graphs. Advances in neural information processing systems, 33 (2020), 22118\u201322133."},{"key":"e_1_3_2_2_15_1","unstructured":"Zhe Jia Marco Maggioni Benjamin Staiger and Daniele P Scarpazza. 2018. Dissecting the NVIDIA volta GPU architecture via microbenchmarking. arXiv preprint arXiv:1804.06826."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2022.102954"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3332466.3374546"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.01244"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/2898361"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.5555\/3571885.3571934"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3472456.3472473"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607051"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00058"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"crossref","unstructured":"Weile Luo Ruibo Fan Zeyu Li Dayou Du Hongyuan Liu Qiang Wang and Xiaowen Chu. 2025. Dissecting the NVIDIA Hopper Architecture through Microbenchmarking and Multiple Level Analysis. arXiv preprint arXiv:2501.12084.","DOI":"10.1109\/IPDPS57955.2024.00064"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS57955.2024.00064"},{"key":"e_1_3_2_2_26_1","volume-title":"Jeff Pool, Darko Stosic, Dusan Stosic, Ganesh Venkatesh, Chong Yu, and Paulius Micikevicius.","author":"Mishra Asit","year":"2021","unstructured":"Asit Mishra, Jorge Albericio Latorre, Jeff Pool, Darko Stosic, Dusan Stosic, Ganesh Venkatesh, Chong Yu, and Paulius Micikevicius. 2021. Accelerating sparse deep neural networks. arXiv preprint arXiv:2104.08378."},{"key":"e_1_3_2_2_27_1","volume-title":"GPU Technology Conference. 12","author":"Naumov Maxim","year":"2010","unstructured":"Maxim Naumov, L Chien, Philippe Vandermersch, and Ujval Kapasi. 2010. Cusparse library. In GPU Technology Conference. 12."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS49936.2021.00016"},{"key":"e_1_3_2_2_29_1","unstructured":"NVIDIA. 2017. Programming tensor cores in cuda 9. https:\/\/devblogs.nvidia.com\/programming-tensor-cores-cuda-9.."},{"key":"e_1_3_2_2_30_1","unstructured":"NVIDIA. 2021-1-30. NVIDIA NVIDIA Ampere Architecture Whitepaper. online. https:\/\/www.nvidia.com\/en-us\/data-center\/a100\/"},{"key":"e_1_3_2_2_31_1","unstructured":"NVIDIA. 2025. cuSPARSELt: A High-Performance CUDA Library for Sparse Matrix-Matrix Multiplication. https:\/\/docs.nvidia.com\/cuda\/cusparselt\/.."},{"key":"e_1_3_2_2_32_1","unstructured":"NVIDIA. 2025. NVIDIA Nsight Compute. https:\/\/developer.nvidia.com\/nsight-compute\/.."},{"key":"e_1_3_2_2_33_1","unstructured":"NVIDIA. 2025-2-26. NVIDIA Parallel Thread Execution ISA Version 8.7. online. https:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/index.html"},{"key":"e_1_3_2_2_34_1","volume-title":"High Performance Unstructured SpMM Computation Using Tensor Cores. In SC24: International Conference for High Performance Computing, Networking, Storage and Analysis. 1\u201314","author":"Okanovic Patrik","year":"2024","unstructured":"Patrik Okanovic, Grzegorz Kwasniewski, Paolo Sylos Labini, Maciej Besta, Flavio Vella, and Torsten Hoefler. 2024. High Performance Unstructured SpMM Computation Using Tensor Cores. In SC24: International Conference for High Performance Computing, Networking, Storage and Analysis. 1\u201314."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3217824"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jsb.2010.01.021"},{"key":"e_1_3_2_2_37_1","volume-title":"ICLR workshop on representation learning on graphs and manifolds.","author":"Wang Minjie Yu","year":"2019","unstructured":"Minjie Yu Wang. 2019. Deep graph library: Towards efficient and scalable deep learning on graphs. In ICLR workshop on representation learning on graphs and manifolds."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498765.1498785"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS47924.2020.00071"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3572848.3577506"},{"key":"e_1_3_2_2_41_1","article-title":"DSTC: Dual-Side Sparsity Tensor Core for DNNs Acceleration on Modern GPU Architectures","author":"Zhang Chen","year":"2024","unstructured":"Chen Zhang, Yang Wang, Zhiqiang Xie, Cong Guo, Yunxin Liu, Jingwen Leng, Guangyu Sun, Zhigang Ji, Runsheng Wang, Yuan Xie, et al. 2024. DSTC: Dual-Side Sparsity Tensor Core for DNNs Acceleration on Modern GPU Architectures. IEEE Trans. Comput..","journal-title":"IEEE Trans. Comput.."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3225058.3225100"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","unstructured":"Kaige Zhang. 2025. PPoPP26_AE_DRAWLOOM_CODE. https:\/\/doi.org\/10.5281\/zenodo.17709956 Accessed: 2026-01-06 10.5281\/zenodo.17709956","DOI":"10.5281\/zenodo.17709956"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3673038.3673108"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3725798.3725803"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"crossref","unstructured":"Haisha Zhao San Li Jiaheng Wang Chunbao Zhou Jue Wang Zhikuang Xin Shunde Li Zhiqiang Liang Zhijie Pan Fang Liu et al. 2025. Acc-SpMM: Accelerating General-purpose Sparse Matrix-Matrix Multiplication with GPU Tensor Cores. arXiv preprint arXiv:2501.09251.","DOI":"10.1145\/3710848.3710888"}],"event":{"name":"PPoPP '26: 31st ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming","location":"Sydney NSW Australia","acronym":"PPoPP '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 31st ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774934.3786441","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T15:27:08Z","timestamp":1769614028000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774934.3786441"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,28]]},"references-count":46,"alternative-id":["10.1145\/3774934.3786441","10.1145\/3774934"],"URL":"https:\/\/doi.org\/10.1145\/3774934.3786441","relation":{},"subject":[],"published":{"date-parts":[[2026,1,28]]},"assertion":[{"value":"2026-01-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}