{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T08:27:56Z","timestamp":1759134476026,"version":"3.37.3"},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2019,4,10]],"date-time":"2019-04-10T00:00:00Z","timestamp":1554854400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"crossref","award":["2017YFB0202104"],"award-info":[{"award-number":["2017YFB0202104"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2020,3]]},"DOI":"10.1007\/s11227-019-02835-4","type":"journal-article","created":{"date-parts":[[2019,4,10]],"date-time":"2019-04-10T20:28:28Z","timestamp":1554928108000},"page":"2063-2081","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":17,"title":["VBSF: a new storage format for SIMD sparse matrix\u2013vector multiplication on modern processors"],"prefix":"10.1007","volume":"76","author":[{"given":"Yishui","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peizhen","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinhai","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shengguo","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunye","family":"Gong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinbiao","family":"Gan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Han","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,4,10]]},"reference":[{"unstructured":"Blelloch GE, Heroux MA, Zagha M (1993) Segmented operations for sparse matrix computation on vector multiprocessors. Technical reports, Pittsburgh, PA, USA","key":"2835_CR1"},{"unstructured":"Chen S, Fang J, Chen D, Xu C, Wang Z (2018) Optimizing sparse matrix\u2013vector multiplication on emerging many-core architectures. ArXiv preprint \narXiv:1805.11938","key":"2835_CR2"},{"issue":"23","key":"2835_CR3","doi-asserted-by":"publisher","first-page":"e4800","DOI":"10.1002\/cpe.4800","volume":"30","author":"X Chen","year":"2018","unstructured":"Chen X, Xie P, Chi L et al (2018) An efficient SIMD compression format for sparse matrix-vector multiplication. Concurr Comput Pract Exp 30(23):e4800","journal-title":"Concurr Comput Pract Exp"},{"issue":"1","key":"2835_CR4","first-page":"1:1","volume":"38","author":"TA Davis","year":"2011","unstructured":"Davis TA, Hu Y (2011) The University of Florida sparse matrix collection. ACM Trans Math Softw 38(1):1:1\u20131:25","journal-title":"ACM Trans Math Softw"},{"doi-asserted-by":"crossref","unstructured":"DAzevedo EF, Fahey MR, Mills RT (2005) Vectorized sparse matrix multiply for compressed row storage format. In: Proceedings of the 5th International Conference on Computational Science-Volume Part I, ICCS\u201905. Springer, Berlin, pp 99\u2013106","key":"2835_CR5","DOI":"10.1007\/11428831_13"},{"issue":"1","key":"2835_CR6","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1007\/s11227-008-0251-8","volume":"50","author":"G Goumas","year":"2009","unstructured":"Goumas G, Kourtis K, Anastopoulos N, Karakasis V, Koziris N (2009) Performance evaluation of the sparse matrix\u2013vector multiplication on modern architectures. J Supercomput 50(1):36\u201377","journal-title":"J Supercomput"},{"issue":"1","key":"2835_CR7","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1177\/1094342004041296","volume":"18","author":"EJ Im","year":"2004","unstructured":"Im EJ, Yelick K, Vuduc R (2004) Sparsity: optimization framework for sparse matrix kernels. Int J High Perform Comput Appl 18(1):135\u2013158","journal-title":"Int J High Perform Comput Appl"},{"unstructured":"Im EJ, Yelick KA (2001) Optimizing sparse matrix computations for register reuse in SPARSITY. In: Proceedings of the International Conference on Computational Sciences-Part I, ICCS \u201901. Springer, Berlin, pp 127\u2013136","key":"2835_CR8"},{"doi-asserted-by":"crossref","unstructured":"Karakasis V, Goumas G, Koziris N (2009) A comparative study of blocking storage methods for sparse matrices on multicore architectures. In: Proceedings of the 2009 International Conference On Computational Science And Engineering-Volume 01, CSE \u201909. IEEE Computer Society, Washington, DC, pp 247\u2013256","key":"2835_CR9","DOI":"10.1109\/CSE.2009.223"},{"doi-asserted-by":"crossref","unstructured":"Karakasis V, Goumas G, Koziris N (2009) Perfomance models for blocked sparse matrix\u2013vector multiplication kernels. In: Proceedings of the 2009 International Conference on Parallel Processing, ICPP \u201909. IEEE Computer Society, Washington, DC, pp 356\u2013364","key":"2835_CR10","DOI":"10.1109\/ICPP.2009.21"},{"issue":"5","key":"2835_CR11","doi-asserted-by":"publisher","first-page":"C401","DOI":"10.1137\/130930352","volume":"36","author":"M Kreutzer","year":"2013","unstructured":"Kreutzer M, Hager G, Wellein G, Fehske H, Bishop A (2013) A unified sparse matrix data format for efficient general sparse matrix\u2013vector multiplication on modern processors with wide SIMD units. SIAM J Sci Comput 36(5):C401\u2013C423","journal-title":"SIAM J Sci Comput"},{"issue":"2","key":"2835_CR12","doi-asserted-by":"publisher","first-page":"428","DOI":"10.1109\/TPDS.2015.2401575","volume":"27","author":"D Langr","year":"2015","unstructured":"Langr D, Tvrdik P (2015) Evaluation criteria for sparse matrix storage formats. IEEE Trans Parallel Distrib Syst 27(2):428\u2013440","journal-title":"IEEE Trans Parallel Distrib Syst"},{"issue":"6","key":"2835_CR13","doi-asserted-by":"publisher","first-page":"117","DOI":"10.1145\/2499370.2462181","volume":"48","author":"J Li","year":"2013","unstructured":"Li J, Tan G, Chen M, Sun N (2013) SMAT: an input adaptive auto-tuner for sparse matrix\u2013vector multiplication. SIGPLAN Not. 48(6):117\u2013126","journal-title":"SIGPLAN Not."},{"issue":"4","key":"2835_CR14","first-page":"882","volume":"51","author":"J Li","year":"2014","unstructured":"Li J, Zhang X, Tan G, Chen M (2014) Study of choosing the optimal storage format of sparse matrix vector multiplication. J Comput Res Dev 51(4):882\u2013894","journal-title":"J Comput Res Dev"},{"issue":"4","key":"2835_CR15","first-page":"269","volume":"35","author":"F Liu","year":"2014","unstructured":"Liu F, Yang C (2014) A new sparse matrix storage format for improving SpMV performance by SIMD. J Numer Methods Comput Appl 35(4):269\u2013276","journal-title":"J Numer Methods Comput Appl"},{"doi-asserted-by":"crossref","unstructured":"Liu W, Vinter B (2015) CSR5: an efficient storage format for cross-platform sparse matrix\u2013vector multiplication. In: Proceedings of the 29th ACM on International Conference on Supercomputing, ICS \u201915. New York, pp 339\u2013350","key":"2835_CR16","DOI":"10.1145\/2751205.2751209"},{"doi-asserted-by":"crossref","unstructured":"Liu X, Smelyanskiy M, Chow E, Dubey P (2013) Efficient sparse matrix\u2013vector multiplication on x86-based many-core processors. In: Proceedings of the 27th International ACM Conference on International Conference on Supercomputing. ACM, pp 273\u2013282","key":"2835_CR17","DOI":"10.1145\/2464996.2465013"},{"doi-asserted-by":"crossref","unstructured":"Patterson DA (2007) The parallel computing landscape: a Berkeley view. In: ACM\/IEEE International Symposium on Low Power Electronics and Design","key":"2835_CR18","DOI":"10.1145\/1283780.1283829"},{"doi-asserted-by":"crossref","unstructured":"Pinar A, Heath M.T (1999) Improving performance of sparse matrix\u2013vector multiplication. In: Proceedings of the 1999 ACM\/IEEE Conference on Supercomputing, SC \u201999. ACM, New York","key":"2835_CR19","DOI":"10.1145\/331532.331562"},{"doi-asserted-by":"crossref","unstructured":"Saad Y (1990) SPARSKIT: a basic tool kit for sparse matrix computations NASA Ames Research Center TR 90-20\u00a0","key":"2835_CR20","DOI":"10.1145\/77726.255162"},{"key":"2835_CR21","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898718003","volume-title":"Iterative methods for sparse linear systems","author":"Y Saad","year":"2003","unstructured":"Saad Y (2003) Iterative methods for sparse linear systems, 2nd edn. Society for Industrial and Applied Mathematics, Philadelphia","edition":"2"},{"doi-asserted-by":"crossref","unstructured":"Sedaghati N, Mu T, Pouchet L.N, Parthasarathy S, Sadayappan P (2015) Automatic selection of sparse matrix representation on GPUs. In: Proceedings of the 29th ACM on International Conference on Supercomputing, ICS \u201915. New York, pp 99\u2013108","key":"2835_CR22","DOI":"10.1145\/2751205.2751244"},{"doi-asserted-by":"crossref","unstructured":"Shalf J, Dosanjh S, Morrison J (2011) Exascale computing technology challenges. In: Proceedings of the 9th International Conference on High Performance Computing for Computational Science, VECPAR\u201910. Springer, Berlin, pp 1\u201325","key":"2835_CR23","DOI":"10.1007\/978-3-642-19328-6_1"},{"doi-asserted-by":"crossref","unstructured":"Shen J, Varbanescu AL, Zou P, Lu Y, Sips H (2014) Improving performance by matching imbalanced workloads with heterogeneous platforms. In: Proceedings of the 28th ACM International Conference on Supercomputing, ICS \u201914. ACM, New York, pp 241\u2013250","key":"2835_CR24","DOI":"10.1145\/2597652.2597675"},{"doi-asserted-by":"crossref","unstructured":"Sun X, Zhang Y, Wang T, Long G, Zhang X, Li Y (2011) CRSD: application specific auto-tuning of SpMV for diagonal sparse matrices. In: Proceedings of the 17th International Conference on Parallel Processing-Volume Part II, Euro-Par\u201911. Springer, pp 316\u2013327","key":"2835_CR25","DOI":"10.1007\/978-3-642-23397-5_32"},{"unstructured":"Vuduc R.W, Moon H.J (2005) Fast sparse matrix\u2013vector multiplication by exploiting variable block structure. In: Proceedings of the First International Conference on High Performance Computing and Communications, HPCC\u201905. Springer, Berlin, pp 807\u2013816","key":"2835_CR26"},{"key":"2835_CR27","doi-asserted-by":"publisher","first-page":"275","DOI":"10.1016\/j.jcp.2014.08.024","volume":"278","author":"C Xu","year":"2014","unstructured":"Xu C, Deng X, Zhang L, Fang J, Wang G, Jiang Y, Cao W, Che Y, Wang Y, Wang Z (2014) Collaborating CPU and GPU for large-scale high-order CFD simulations with complex grids on the TianHe-1A supercomputer. J Comput Phys 278:275\u2013297","journal-title":"J Comput Phys"},{"unstructured":"Yelick K (2008) pOSKI: an extensible autotuning framework to perform optimized SpMVs on Multicore Architectures. Ph.D. Thesis, Department of Electrical Engineering and Computer Sciences, University of California at Berkeley","key":"2835_CR28"},{"issue":"4","key":"2835_CR29","first-page":"818","volume":"37","author":"A Zhang","year":"2016","unstructured":"Zhang A, An H, Yao W, Liang W, Jiang X, Li F (2016) Efficient sparse matrix\u2013vector multiplication on intel xeon phi. J Chin Comput Syst 37(4):818\u2013823","journal-title":"J Chin Comput Syst"},{"issue":"1","key":"2835_CR30","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1145\/3200691.3178495","volume":"53","author":"Y Zhao","year":"2018","unstructured":"Zhao Y, Li J, Liao C, Shen X (2018) Bridging the gap between deep learning and sparse matrix format selection. SIGPLAN Not. 53(1):94\u2013108","journal-title":"SIGPLAN Not."}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-019-02835-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-019-02835-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-019-02835-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,4,8]],"date-time":"2020-04-08T23:13:24Z","timestamp":1586387604000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-019-02835-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,4,10]]},"references-count":30,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2020,3]]}},"alternative-id":["2835"],"URL":"https:\/\/doi.org\/10.1007\/s11227-019-02835-4","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"type":"print","value":"0920-8542"},{"type":"electronic","value":"1573-0484"}],"subject":[],"published":{"date-parts":[[2019,4,10]]},"assertion":[{"value":"10 April 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}