{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T22:05:38Z","timestamp":1781215538513,"version":"3.54.1"},"reference-count":23,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015,7]]},"DOI":"10.1109\/asap.2015.7245713","type":"proceedings-article","created":{"date-parts":[[2015,9,11]],"date-time":"2015-09-11T04:34:48Z","timestamp":1441946088000},"page":"82-89","source":"Crossref","is-referenced-by-count":45,"title":["LightSpMV: Faster CSR-based sparse matrix-vector multiplication on CUDA-enabled GPUs"],"prefix":"10.1109","author":[{"given":"Yongchao","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bertil","family":"Schmidt","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"crossref","first-page":"781","DOI":"10.1109\/SC.2014.69","article-title":"Fast Sparse Matrix-Vector Multiplication on GPUs for Graph Applications","author":"ashari","year":"2014","journal-title":"Proceedings of the International Conference for High Performance Computing Networking Storage and Analysis"},{"key":"ref11","doi-asserted-by":"crossref","first-page":"769","DOI":"10.1109\/SC.2014.68","article-title":"Efficient Sparse Matrix-Vector Multiplication on GPUs using the CSR Storage Format","author":"greathouse","year":"2014","journal-title":"Proceedings of the International Conference for High Performance Computing Networking Storage and Analysis"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2013.09.005"},{"key":"ref13","article-title":"The Open Standard for Parallel Programming of Heterogeneous Systems","author":"group","year":"2014"},{"key":"ref14","article-title":"ViennaCL &#x2013; A High Level Linear Algebra Library for GPUs and Multi-Core CPUs","author":"rupp","year":"2010","journal-title":"International Workshop on GPUs and Scientific Applications"},{"key":"ref15","article-title":"NVIDIA, The NVIDIA CUDA Sparse Matrix Library (cuSPARSE)","year":"2014","journal-title":"CUDA 6 5 toolkit"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/1693453.1693471"},{"key":"ref17","first-page":"111125","article-title":"Automatically Tuning Sparse Matrix-Vector Multiplication for GPU architectures","author":"monakov","year":"2010","journal-title":"HiPEAC LNCS"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/2304576.2304624"},{"key":"ref19","first-page":"1","article-title":"Efficient Sparse Matrix-Vector Multiplication on Cache-based GPUs","author":"reguly","year":"2012","journal-title":"Innovative Parallel Computing"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.14778\/1938545.1938548"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-75755-9_32"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2008.5214359"},{"key":"ref5","doi-asserted-by":"crossref","first-page":"739","DOI":"10.1007\/978-3-540-85451-7_79","article-title":"Solving Dense Linear Systems on Graphics Processors","volume":"5168","author":"barrachina","year":"2008","journal-title":"Lecture Notes in Computer Science"},{"key":"ref8","article-title":"Optimizing Sparse Matrix-Vector Multiplication on GPUs","author":"baskaran","year":"2009","journal-title":"IBM Research Report RC24704 IBM"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/1654059.1654078"},{"key":"ref2","article-title":"Optimization of Sparse Matrix Kernels for Data Mining","author":"im","year":"2001","journal-title":"Proc Workshop on Text Mining"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898718003"},{"key":"ref9","article-title":"CUSP: Generic Parallel Algorithms for Sparse Matrix and Graph Computations (v0.4)","author":"bell","year":"2014"},{"key":"ref20","article-title":"NVIDIA, NVIDIAs Next Generation CUDA Compute Architecture: Kepler GK110","year":"2013","journal-title":"NVIDIA white paper"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-5041-2940-4_9"},{"key":"ref21","article-title":"Faster Parallel Reductions on Kepler","author":"luitjens","year":"2014"},{"key":"ref23","first-page":"1085","article-title":"The University of Florida Sparse Matrix Collection","volume":"38","author":"davis","year":"2013","journal-title":"ACM Transactions on Mathematical Software"}],"event":{"name":"2015 IEEE 26th International Conference on Application-specific Systems, Architectures and Processors (ASAP)","location":"Toronto, ON, Canada","start":{"date-parts":[[2015,7,27]]},"end":{"date-parts":[[2015,7,29]]}},"container-title":["2015 IEEE 26th International Conference on Application-specific Systems, Architectures and Processors (ASAP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7227129\/7245687\/07245713.pdf?arnumber=7245713","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2017,6,23]],"date-time":"2017-06-23T19:17:55Z","timestamp":1498245475000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7245713\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,7]]},"references-count":23,"URL":"https:\/\/doi.org\/10.1109\/asap.2015.7245713","relation":{},"subject":[],"published":{"date-parts":[[2015,7]]}}}