{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T19:40:41Z","timestamp":1782934841146,"version":"3.54.5"},"reference-count":23,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2021,3,26]],"date-time":"2021-03-26T00:00:00Z","timestamp":1616716800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,3,26]],"date-time":"2021-03-26T00:00:00Z","timestamp":1616716800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100010244","name":"Science Foundation of China University of Petroleum, Beijing","doi-asserted-by":"publisher","award":["No. 2462019YJRC004"],"award-info":[{"award-number":["No. 2462019YJRC004"]}],"id":[{"id":"10.13039\/501100010244","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010244","name":"Science Foundation of China University of Petroleum, Beijing","doi-asserted-by":"publisher","award":["2462020XKJS03"],"award-info":[{"award-number":["2462020XKJS03"]}],"id":[{"id":"10.13039\/501100010244","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013287","name":"Science Challenge Project","doi-asserted-by":"publisher","award":["TZZT2016002"],"award-info":[{"award-number":["TZZT2016002"]}],"id":[{"id":"10.13039\/501100013287","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. 61972415"],"award-info":[{"award-number":["No. 61972415"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2021,10]]},"DOI":"10.1007\/s10766-021-00695-1","type":"journal-article","created":{"date-parts":[[2021,3,26]],"date-time":"2021-03-26T18:02:27Z","timestamp":1616781747000},"page":"732-744","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Segmented Merge: A New Primitive for Parallel Sparse Matrix Computations"],"prefix":"10.1007","volume":"49","author":[{"given":"Haonan","family":"Ji","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shibo","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kaixi","family":"Hou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0632-9494","authenticated-orcid":false,"given":"Zhou","family":"Jin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weifeng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Brian","family":"Vinter","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,3,26]]},"reference":[{"key":"695_CR1","unstructured":"Blelloch, G.E., Heroux, M.A., Zagha, M.: Segmented Operations for Sparse Matrix Computation on Vector Multiprocessors. CMU Tech. Rep. (1993)"},{"key":"695_CR2","doi-asserted-by":"publisher","first-page":"179","DOI":"10.1016\/j.parco.2015.04.004","volume":"49","author":"W Liu","year":"2015","unstructured":"Liu, W., Vinter, B.: Speculative segmented sum for sparse matrix-vector multiplication on heterogeneous processors. Parallel Comput. 49, 179\u2013193 (2015)","journal-title":"Parallel Comput."},{"key":"695_CR3","doi-asserted-by":"crossref","unstructured":"Dotsenko, Y., Govindaraju, N.K., Sloan, P.-P., Boyd, C., Manferdelli, J.: Fast scan algorithms on graphics processors. In: Proceedings of the 22nd Annual International Conference on Supercomputing, ser. ICS\u201908, pp. 205\u2013213 (2008)","DOI":"10.1145\/1375527.1375559"},{"key":"695_CR4","doi-asserted-by":"crossref","unstructured":"Hou, K., Liu, W., Wang, H., Feng, W.-C.: Fast segmented sort on GPUs. In: Proceedings of the International Conference on Supercomputing, ser. ICS\u201917 (2017)","DOI":"10.1145\/3079079.3079105"},{"issue":"4","key":"695_CR5","doi-asserted-by":"publisher","first-page":"C429","DOI":"10.1137\/17M1121378","volume":"40","author":"F Gremse","year":"2018","unstructured":"Gremse, F., K\u00fcpper, K., Naumann, U.: Memory-efficient sparse matrix\u2013matrix multiplication by row merging on many-core architectures. SIAM J. Sci. Comput. 40(4), C429\u2013C449 (2018)","journal-title":"SIAM J. Sci. Comput."},{"key":"695_CR6","doi-asserted-by":"publisher","first-page":"403","DOI":"10.1007\/s10766-018-0604-8","volume":"47","author":"J Liu","year":"2019","unstructured":"Liu, J., He, X., Liu, W., Tan, G.: Register-aware optimizations for parallel sparse matrix\u2013matrix multiplication. Int. J. Parallel Program. 47, 403\u2013417 (2019)","journal-title":"Int. J. Parallel Program."},{"key":"695_CR7","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1016\/j.jpdc.2015.06.010","volume":"85","author":"W Liu","year":"2015","unstructured":"Liu, W., Vinter, B.: A framework for general sparse matrix\u2013matrix multiplication on GPUs and heterogeneous processors. J. Parallel Distrib. Comput. 85, 47\u201361 (2015)","journal-title":"J. Parallel Distrib. Comput."},{"issue":"4","key":"695_CR8","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2699470","volume":"41","author":"S Dalton","year":"2015","unstructured":"Dalton, S., Olson, L., Bell, N.: Optimizing sparse matrix\u2013matrix multiplication for the GPU. ACM Trans. Math. Softw. 41(4), 1\u201320 (2015)","journal-title":"ACM Trans. Math. Softw."},{"key":"695_CR9","doi-asserted-by":"crossref","unstructured":"Winter, M., Mlakar, D., Zayer, R., Seidel, H.-P., Steinberger, M.: Adaptive sparse matrix\u2013matrix multiplication on the GPU. In: Proceedings of the 24th Symposium on Principles and Practice of Parallel Programming, ser. PPoPP\u201919, pp. 68\u201381 (2019)","DOI":"10.1145\/3293883.3295701"},{"issue":"2","key":"695_CR10","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1007\/s42514-019-00008-6","volume":"1","author":"F Zhang","year":"2019","unstructured":"Zhang, F., Liu, W., Feng, N., Zhai, J., Du, X.: Performance evaluation and analysis of sparse matrix and graph kernels on heterogeneous processors. CCF Trans. High Perform. Comput. 1(2), 131\u2013143 (2019)","journal-title":"CCF Trans. High Perform. Comput."},{"issue":"1","key":"695_CR11","doi-asserted-by":"publisher","first-page":"C54","DOI":"10.1137\/130948811","volume":"37","author":"F Gremse","year":"2015","unstructured":"Gremse, F., Hofter, A., Schwen, L.O., Kiessling, F., Naumann, U.: GPU-accelerated sparse matrix\u2013matrix multiplication by iterative row merging. SIAM J. Sci. Comput. 37(1), C54\u2013C71 (2015)","journal-title":"SIAM J. Sci. Comput."},{"issue":"1","key":"695_CR12","first-page":"1:1","volume":"38","author":"TA Davis","year":"2011","unstructured":"Davis, T.A., Hu, Y.: The University of Florida sparse matrix collection. ACM Trans. Math. Softw. 38(1), 1:1-1:25 (2011)","journal-title":"ACM Trans. Math. Softw."},{"key":"695_CR13","doi-asserted-by":"crossref","unstructured":"Liu, W., Vinter, B.: CSR5: an efficient storage format for cross-platform sparse matrix-vector Multiplication. In: Proceedings of the 29th ACM on International Conference on Supercomputing, ser. ICS\u201915, pp. 339\u2013350 (2015)","DOI":"10.1145\/2751205.2751209"},{"key":"695_CR14","unstructured":"Liu, W.: Parallel and scalable sparse basic linear algebra subprograms. Ph.D. dissertation, University of Copenhagen (2015)"},{"key":"695_CR15","doi-asserted-by":"crossref","unstructured":"Green, O., McColl, R., Bader, D.A.: GPU merge path: a GPU merging algorithm. In: Proceedings of the 26th ACM International Conference on Supercomputing, ser. ICS\u201912, pp. 331\u2013340 (2012)","DOI":"10.1145\/2304576.2304621"},{"key":"695_CR16","doi-asserted-by":"crossref","unstructured":"Wang, H., Liu, W., Hou, K., Feng, W.-C.: Parallel transposition of sparse data structures. In: Proceedings of the 2016 International Conference on Supercomputing, ser. ICS\u201916, pp. 33:1\u201333:13 (2016)","DOI":"10.1145\/2925426.2926291"},{"issue":"8","key":"695_CR17","doi-asserted-by":"publisher","first-page":"193","DOI":"10.1145\/2692916.2555253","volume":"49","author":"B Catanzaro","year":"2014","unstructured":"Catanzaro, B., Keller, A., Garland, M.: A decomposition for in-place matrix transposition. ACM SIGPLAN Not. 49(8), 193\u2013206 (2014)","journal-title":"ACM SIGPLAN Not."},{"key":"695_CR18","doi-asserted-by":"crossref","unstructured":"Bulu\u00e7, A., Fineman, J.T., Frigo, M., Gilbert, J.R., Leiserson, C.E.: Parallel sparse matrix-vector and matrix-transpose-vector multiplication using compressed sparse blocks. In: Proceedings of the Twenty-First Annual Symposium on Parallelism in Algorithms and Architectures, ser. SPAA\u201909, pp. 233\u2013244 (2009)","DOI":"10.1145\/1583991.1584053"},{"issue":"3","key":"695_CR19","doi-asserted-by":"publisher","first-page":"250","DOI":"10.1145\/355791.355796","volume":"4","author":"FG Gustavson","year":"1978","unstructured":"Gustavson, F.G.: Two fast algorithms for sparse matrices: multiplication and permuted transposition. ACM Trans. Math. Softw. 4(3), 250\u2013269 (1978)","journal-title":"ACM Trans. Math. Softw."},{"key":"695_CR20","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1016\/j.parco.2018.06.009","volume":"78","author":"M Deveci","year":"2018","unstructured":"Deveci, M., Trott, C., Rajamanickam, S.: Multithreaded sparse matrix\u2013matrix multiplication for many-core and GPU architectures. Parallel Comput. 78, 33\u201346 (2018)","journal-title":"Parallel Comput."},{"key":"695_CR21","doi-asserted-by":"crossref","unstructured":"Liu, J., He, X., Liu, W., Tan, G.: Register-based implementation of the sparse general matrix\u2013matrix multiplication on GPUs. In: Proceedings of the 23rd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, ser. PPoPP\u201918, pp. 407\u2013408 (2018)","DOI":"10.1145\/3178487.3178529"},{"key":"695_CR22","doi-asserted-by":"crossref","unstructured":"Nagasaka, Y., Nukada, A., Matsuoka, S.: High-performance and memory-saving sparse general matrix\u2013matrix multiplication for NVIDIA Pascal GPU. In: 2017 46th International Conference on Parallel Processing (ICPP), pp. 101\u2013110 (2017)","DOI":"10.1109\/ICPP.2017.19"},{"key":"695_CR23","doi-asserted-by":"crossref","unstructured":"Xie, Z., Tan, G., Liu, W., Sun, N.: IA-SpGEMM: an input-aware auto-tuning framework for parallel sparse matrix\u2013matrix multiplication. In: Proceedings of the ACM International Conference on Supercomputing, ser. ICS\u201919, pp. 94\u2013105 (2019)","DOI":"10.1145\/3330345.3330354"}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-021-00695-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10766-021-00695-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-021-00695-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,8,24]],"date-time":"2021-08-24T21:07:35Z","timestamp":1629839255000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10766-021-00695-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,3,26]]},"references-count":23,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2021,10]]}},"alternative-id":["695"],"URL":"https:\/\/doi.org\/10.1007\/s10766-021-00695-1","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"value":"0885-7458","type":"print"},{"value":"1573-7640","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,3,26]]},"assertion":[{"value":"26 November 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 February 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 March 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}