{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T10:05:25Z","timestamp":1772791525670,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":58,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,8,29]],"date-time":"2022-08-29T00:00:00Z","timestamp":1661731200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,8,29]]},"DOI":"10.1145\/3545008.3545028","type":"proceedings-article","created":{"date-parts":[[2023,1,15]],"date-time":"2023-01-15T01:04:08Z","timestamp":1673744648000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":12,"title":["TileSpMSpV: A Tiled Algorithm for Sparse Matrix-Sparse Vector Multiplication on GPUs"],"prefix":"10.1145","author":[{"given":"Haonan","family":"Ji","sequence":"first","affiliation":[{"name":"Super Scientific Software Laboratory, China University of Petroleum-Beijing, China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huimin","family":"Song","sequence":"additional","affiliation":[{"name":"Super Scientific Software Laboratory, China University of Petroleum-Beijing, China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shibo","family":"Lu","sequence":"additional","affiliation":[{"name":"Northeastern University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhou","family":"Jin","sequence":"additional","affiliation":[{"name":"Super Scientific Software Laboratory, China University of Petroleum-Beijing, China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangming","family":"Tan","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Computer Architecture, Institute of Computing Technology, Chinese Academy of Sciences, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weifeng","family":"Liu","sequence":"additional","affiliation":[{"name":"Super Scientific Software Laboratory, China University of Petroleum-Beijing, China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,1,13]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Junwhan Ahn Sungpack Hong Sungjoo Yoo Onur Mutlu and Kiyoung Choi. 2015. A Scalable Processing-in-Memory Accelerator for Parallel Graph Processing. In ISCA \u201915."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Michael\u00a0J. Anderson Narayanan Sundaram Nadathur Satish Md. Mostofa\u00a0Ali Patwary Theodore\u00a0L. Willke and Pradeep Dubey. 2016. GraphPad: Optimized Graph Primitives for Parallel and Distributed Platforms. In IPDPS \u201916.","DOI":"10.1109\/IPDPS.2016.86"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Ariful Azad and Aydin Bulu\u00e7. 2017. A Work-Efficient Parallel Sparse Matrix-Sparse Vector Multiplication Algorithm. In IPDPS \u201917.","DOI":"10.1109\/IPDPS.2017.76"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Ariful Azad Mathias Jacquelin Aydin Bulu\u00e7 and Esmond\u00a0G. Ng. 2017. The Reverse Cuthill-McKee Algorithm in Distributed-Memory. In IPDPS \u201917.","DOI":"10.1109\/IPDPS.2017.85"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3094091"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Scott Beamer Krste Asanovi\u0107 and David Patterson. 2012. Direction-optimizing Breadth-First Search. In SC \u201912.","DOI":"10.1109\/SC.2012.50"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Maciej Besta Micha\u0142 Podstawski Linus Groner Edgar Solomonik and Torsten Hoefler. 2017. To Push or To Pull: On Reducing Communication and Synchronization in Graph Computations. In HPDC \u201917.","DOI":"10.1145\/3078597.3078616"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Aydin Bulu\u00e7 and Kamesh Madduri. 2011. Parallel Breadth-First Search on Distributed Memory Systems. In SC \u201911.","DOI":"10.1145\/2063384.2063471"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342011403516"},{"key":"e_1_3_2_1_10_1","volume-title":"Optimal Algebraic Breadth-First Search for Sparse Graphs. ACM Transactions on Knowledge Discovery from Data 15, 5","author":"Burkhardt Paul","year":"2021","unstructured":"Paul Burkhardt. 2021. Optimal Algebraic Breadth-First Search for Sparse Graphs. ACM Transactions on Knowledge Discovery from Data 15, 5 (2021)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Rong Chen Jiaxin Shi Yanzhe Chen and Haibo Chen. 2015. PowerLyra: Differentiated Graph Computation and Partitioning on Skewed Graphs. In EuroSys \u201915.","DOI":"10.1145\/2741948.2741970"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Yuze Chi Guohao Dai Yu Wang Guangyu Sun Guoliang Li and Huazhong Yang. 2016. NXgraph: An efficient graph processing system on a single machine. In ICDE \u201916.","DOI":"10.1109\/ICDE.2016.7498258"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3322125"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2049662.2049663"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"S.\u00a0M. Faisal Srinivasan Parthasarathy and P. Sadayappan. 2014. Global graphs: A middleware for large scale graph processing. In Big Data \u201914.","DOI":"10.1109\/BigData.2014.7004369"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.14778\/3384345.3384358"},{"key":"e_1_3_2_1_17_1","unstructured":"John\u00a0R. Gilbert Steve Reinhardt and Viral\u00a0B. Shah. 2007. High-Performance Graph Algorithms from Parallel Sparse Matrices. In LLC \u201907."},{"key":"e_1_3_2_1_18_1","unstructured":"Joseph\u00a0E. Gonzalez Yucheng Low Haijie Gu Danny Bickson and Carlos Guestrin. 2012. PowerGraph: Distributed Graph-Parallel Computation on Natural Graphs. In OSDI \u201912."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/355791.355796"},{"key":"e_1_3_2_1_20_1","volume-title":"Graphicionado: A high-performance and energy-efficient accelerator for graph analytics. In MICRO \u201916.","author":"Ham Tae\u00a0Jun","year":"2016","unstructured":"Tae\u00a0Jun Ham, Lisa Wu, Narayanan Sundaram, Nadathur Satish, and Margaret Martonosi. 2016. Graphicionado: A high-performance and energy-efficient accelerator for graph analytics. In MICRO \u201916."},{"key":"e_1_3_2_1_21_1","unstructured":"Wook-Shin Han Sangyeon Lee Kyungyeol Park Jeong-Hoon Lee Min-Soo Kim Jinha Kim and Hwanjo Yu. 2013. TurboGraph: A Fast Parallel Graph Engine Handling Billion-Scale Graphs in a Single PC. In KDD \u201913."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Changwan Hong Aravind Sukumaran-Rajam Jinsung Kim and P. Sadayappan. 2017. MultiGraph: Efficient Graph Processing on GPUs. In PACT \u201917.","DOI":"10.1109\/PACT.2017.48"},{"key":"e_1_3_2_1_23_1","volume-title":"Merijn Verstraaten, and Hassan Chafi.","author":"Hong Sungpack","year":"2015","unstructured":"Sungpack Hong, Siegfried Depner, Thomas Manhardt, Jan Van Der\u00a0Lugt, Merijn Verstraaten, and Hassan Chafi. 2015. PGX.D: A Fast Distributed Graph Processing Engine. In SC \u201915."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Sungpack Hong Tayo Oguntebi and Kunle Olukotun. 2011. Efficient Parallel Graph Exploration on Multi-Core CPU and GPU. In PACT \u201911.","DOI":"10.1109\/PACT.2011.14"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10766-021-00695-1"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2015.05.353"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Farzad Khorasani Keval Vora Rajiv Gupta and Laxmi\u00a0N. Bhuyan. 2014. CuSha: Vertex-Centric Graph Processing on GPUs. In HPDC \u201914.","DOI":"10.1145\/2600212.2600227"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2015.2401575"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Da Li and Michela Becchi. 2013. Deploying Graph Algorithms on GPUs: An Adaptive Solution. In IPDPS \u201913.","DOI":"10.1109\/IPDPS.2013.101"},{"key":"e_1_3_2_1_30_1","unstructured":"Haoran Li Harumichi Yokoyama and Takuya Araki. 2018. Merge-Based Parallel Sparse Matrix-Sparse Vector Multiplication with a Vector Architecture. In HPCC\/SmartCity\/DSS \u201918."},{"key":"e_1_3_2_1_31_1","first-page":"1842","article-title":"Adaptive SpMV\/SpMSpV on GPUs for Input Vectors of Varied Sparsity","volume":"32","author":"Li Min","year":"2021","unstructured":"Min Li, Yulong Ao, and Chao Yang. 2021. Adaptive SpMV\/SpMSpV on GPUs for Input Vectors of Varied Sparsity. IEEE Transactions on Parallel and Distributed Systems 32, 7 (2021), 1842\u20131853.","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/2807591.2807594"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10766-018-0604-8"},{"key":"e_1_3_2_1_34_1","unstructured":"Weifeng Liu and Brian Vinter. 2014. An Efficient GPU General Sparse Matrix-Matrix Multiplication for Irregular Data. In IPDPS \u201914."},{"key":"e_1_3_2_1_35_1","unstructured":"Weifeng Liu and Brian Vinter. 2015. CSR5: An Efficient Storage Format for Cross-Platform Sparse Matrix-Vector Multiplication. In ICS \u201915."},{"key":"e_1_3_2_1_36_1","unstructured":"Yucheng Low Joseph Gonzalez Aapo Kyrola Danny Bickson Carlos Guestrin and Joseph Hellerstein. 2010. GraphLab: A New Framework for Parallel Machine Learning. In UAI\u201910."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Ke Meng Jiajia Li Guangming Tan and Ninghui Sun. 2019. A Pattern Based Algorithmic Autotuner for Graph Processing on GPUs. In PPoPP \u201919.","DOI":"10.1145\/3293883.3295716"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Duane Merrill Michael Garland and Andrew Grimshaw. 2012. Scalable GPU Graph Traversal. In PPoPP \u201912.","DOI":"10.1145\/2145816.2145832"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Donald Nguyen Andrew Lenharth and Keshav Pingali. 2013. A Lightweight Infrastructure for Graph Analytics. In SOSP \u201913.","DOI":"10.1145\/2517349.2522739"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"Yuyao Niu Zhengyang Lu Meichen Dong Zhou Jin Weifeng Liu and Guangming Tan. 2021. TileSpMV: A Tiled Algorithm for Sparse Matrix-Vector Multiplication on GPUs. In IPDPS \u201921.","DOI":"10.1109\/IPDPS49936.2021.00016"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Yuyao Niu Zhengyang Lu Haonan Ji Shuhui Song Zhou Jin and Weifeng Liu. 2022. TileSpGEMM: A Tiled Algorithm for Parallel Sparse General Matrix-Matrix Multiplication on GPUs. In PPoPP \u201922.","DOI":"10.1145\/3503221.3508431"},{"key":"e_1_3_2_1_42_1","unstructured":"Subhankar Pal Aporva Amarnath Siying Feng Michael O\u2019Boyle Ronald Dreslinski and Christophe Dubach. 2021. SparseAdapt: Runtime Control for Sparse Linear Algebra on a Reconfigurable Accelerator. In MICRO \u201921."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3243176.3243205"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"crossref","unstructured":"Dipanjan Sengupta Shuaiwen\u00a0Leon Song Kapil Agarwal and Karsten Schwan. 2015. GraphReduce: processing large-scale graphs on accelerator-based systems. In SC \u201915.","DOI":"10.1145\/2807591.2807655"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/2442516.2442530"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"Edgar Solomonik Maciej Besta Flavio Vella and Torsten Hoefler. 2017. Scaling Betweenness Centrality Using Communication-Efficient Sparse Matrix Multiplication. In SC \u201917.","DOI":"10.1145\/3126908.3126971"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.14778\/2809974.2809983"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/2851141.2851145"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Hao Wen and Wei Zhang. 2019. Improving Parallelism of Breadth First Search (BFS) Algorithm for Accelerated Performance on GPUs. In HPEC \u201919.","DOI":"10.1109\/HPEC.2019.8916551"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3090328"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3466795"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"crossref","unstructured":"Carl Yang Ayd\u0131n Bulu\u00e7 and John\u00a0D. Owens. 2018. Implementing Push-Pull Efficiently in GraphBLAS. In ICPP \u201918.","DOI":"10.1145\/3225058.3225122"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","unstructured":"Carl Yang Yangzihao Wang and John\u00a0D. Owens. 2015. Fast Sparse Matrix and Sparse Vector Multiplication Algorithm on the GPU. In IPDPSW \u201915.","DOI":"10.1109\/IPDPSW.2015.77"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.14778\/1938545.1938548"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2017.2714667"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11227-018-2525-0"},{"key":"e_1_3_2_1_57_1","volume-title":"Performance Evaluation and Analysis of Sparse Matrix and Graph Kernels on Heterogeneous Processors. CCF Transactions on High Performance Computing","author":"Zhang Feng","year":"2019","unstructured":"Feng Zhang, Weifeng Liu, Ningxuan Feng, Jidong Zhai, and Xiaoyong Du. 2019. Performance Evaluation and Analysis of Sparse Matrix and Graph Kernels on Heterogeneous Processors. CCF Transactions on High Performance Computing (2019)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"crossref","unstructured":"Yue Zhao Jiajia Li Chunhua Liao and Xipeng Shen. 2018. Bridging the Gap between Deep Learning and Sparse Matrix Format Selection. In PPoPP \u201918. 94\u2013108.","DOI":"10.1145\/3178487.3178495"}],"event":{"name":"ICPP '22: 51st International Conference on Parallel Processing","location":"Bordeaux France","acronym":"ICPP '22"},"container-title":["Proceedings of the 51st International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3545008.3545028","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3545008.3545028","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:43Z","timestamp":1750186963000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3545008.3545028"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,29]]},"references-count":58,"alternative-id":["10.1145\/3545008.3545028","10.1145\/3545008"],"URL":"https:\/\/doi.org\/10.1145\/3545008.3545028","relation":{},"subject":[],"published":{"date-parts":[[2022,8,29]]},"assertion":[{"value":"2023-01-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}