{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T12:16:13Z","timestamp":1763468173170},"reference-count":17,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2014,1,23]],"date-time":"2014-01-23T00:00:00Z","timestamp":1390435200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2014,11]]},"DOI":"10.1007\/s11227-014-1102-4","type":"journal-article","created":{"date-parts":[[2014,1,22]],"date-time":"2014-01-22T19:59:50Z","timestamp":1390420790000},"page":"577-587","source":"Crossref","is-referenced-by-count":15,"title":["Performance evaluation of kernel fusion BLAS routines on the GPU: iterative solvers as case study"],"prefix":"10.1007","volume":"70","author":[{"given":"S.","family":"Tabik","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"G.","family":"Ortega","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"E. M.","family":"Garz\u00f3n","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,1,23]]},"reference":[{"issue":"5","key":"1102_CR1","doi-asserted-by":"crossref","first-page":"1162","DOI":"10.1109\/TMAG.2010.2081662","volume":"47","author":"MM Dehnavi","year":"2011","unstructured":"Dehnavi MM, Fernandez DM, Giannacopoulos D (2011) Enhancing the performance of conjugate gradient solvers on graphic processing units. IEEE Trans Magn 47(5):1162\u20131165","journal-title":"IEEE Trans Magn"},{"key":"1102_CR2","unstructured":"Filipovi\u010d J, Madzin M, Fousek J, Matyska L (2013) Optimizing cuda code by kernel fusion\u2014application on BLAS. CoRR abs\/1305.1183"},{"key":"1102_CR3","doi-asserted-by":"crossref","unstructured":"Gaikwad A, Toke IM (2010) Parallel iterative linear solvers on GPU: a financial engineering case. In: Proceediongs of PDP, pp 607\u2013614","DOI":"10.1109\/PDP.2010.55"},{"key":"1102_CR4","doi-asserted-by":"crossref","unstructured":"Garcia N (2010) Parallel power flow solutions using a biconjugate gradient algorithm and a newton method: a GPU-based approach. In: IEEE Power and Energy Society general meeting, pp 1\u20134","DOI":"10.1109\/PES.2010.5589682"},{"key":"1102_CR5","unstructured":"Golub GH, van Van Loan CF (1996) Matrix computations (Johns Hopkins studies in mathematical sciences), 3rd edn. The Johns Hopkins University Press. Baltimore, MD"},{"key":"1102_CR6","doi-asserted-by":"crossref","unstructured":"Haidar A, Ltaief H, Luszczek P, Dongarra J (2012) A comprehensive study of task coalescing for selecting parallelism granularity in a two-stage bidiagonal reduction. In: Proceedings of of IEEE IPDPS, pp 25\u201335","DOI":"10.1109\/IPDPS.2012.13"},{"key":"1102_CR7","volume-title":"Computing Gems Jade Edition. Applications of GPU computing series","author":"W Hwu","year":"2011","unstructured":"Hwu W (2011) Computing Gems Jade Edition. Applications of GPU computing series, Jade edn. Elsevier Science, Amsterdam","edition":"Jade"},{"key":"1102_CR8","doi-asserted-by":"crossref","first-page":"33","DOI":"10.6028\/jres.049.006","volume":"49","author":"C Lanczos","year":"1952","unstructured":"Lanczos C (1952) Solution of systems of linear equations by minimized iterations. J Res Natl Bur Stand 49:33\u201353","journal-title":"J Res Natl Bur Stand"},{"issue":"3","key":"1102_CR9","doi-asserted-by":"crossref","first-page":"308","DOI":"10.1145\/355841.355847","volume":"5","author":"CL Lawson","year":"1979","unstructured":"Lawson CL, Hanson RJ, Kincaid DR, Krogh FT (1979) Basic linear algebra subprograms for fortran usage. ACM Trans Math Softw 5(3):308\u2013323","journal-title":"ACM Trans Math Softw"},{"key":"1102_CR10","doi-asserted-by":"crossref","unstructured":"Navarro AG, Asenjo R, Tabik S, Cascaval C (2009) Analytical modeling of pipeline parallelism. In: Proceedings of PACT, pp 281\u2013290. IEEE Computer Society","DOI":"10.1109\/PACT.2009.28"},{"key":"1102_CR11","unstructured":"NVIDIA (2013) Du-06702-001\\_v5.5 CUBLAS user guide. Technical report. http:\/\/docs.nvidia.com\/cuda\/pdf\/CUBLAS_Library.pdf"},{"key":"1102_CR12","unstructured":"NVIDIA (2013) Du-06709-001\\_v5.5 CUSPARSE library. Technical report. http:\/\/docs.nvidia.com\/cuda\/pdf\/CUSPARSE_Library.pdf"},{"key":"1102_CR13","doi-asserted-by":"crossref","first-page":"49","DOI":"10.1007\/s11227-012-0761-2","volume":"64","author":"G Ortega","year":"2013","unstructured":"Ortega G, Garz\u00f3n EM, V\u00e1zquez F, Garc\u00eda I (2013) The biconjugate gradient method on GPUs. J Supercomput 64:49\u201358","journal-title":"J Supercomput"},{"key":"1102_CR14","doi-asserted-by":"crossref","first-page":"408","DOI":"10.1016\/j.parco.2011.08.003","volume":"38","author":"F V\u00e1zquez","year":"2012","unstructured":"V\u00e1zquez F, Fern\u00e1ndez JJ, Garz\u00f3n EM (2012) Automatic tuning of the sparse matrix vector product on GPUs based on the ELLR-T approach. Parallel Comput 38:408\u2013420","journal-title":"Parallel Comput"},{"key":"1102_CR15","doi-asserted-by":"crossref","unstructured":"V\u00e1zquez F, Ortega G, Fern\u00e1ndez JJ, Garz\u00f3n EM (2010) Improving the performance of the sparse matrix vector product with GPUs. In: Proceedings of IEEE CIT, pp 1146\u20131151. IEEE Computer Society","DOI":"10.1109\/CIT.2010.208"},{"key":"1102_CR16","doi-asserted-by":"crossref","unstructured":"Wozniak M, Olas T, Wyrzykowski R (2010) Parallel implementation of conjugate gradient method on graphics processors. In: Parallel processing and applied mathematics, LNCS vol 6067, pp 125\u2013135","DOI":"10.1007\/978-3-642-14390-8_14"},{"key":"1102_CR17","doi-asserted-by":"crossref","unstructured":"Wu H, Diamos G, Wang J, Cadambi S, Yalamanchili S, Chakradhar S (2012) Optimizing data warehousing applications for GPUs using kernel fusion\/fission. In: Proceedings of IEEE IPDPSW, pp 2433\u20132442","DOI":"10.1109\/IPDPSW.2012.300"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1102-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-014-1102-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1102-4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,6]],"date-time":"2019-08-06T22:47:05Z","timestamp":1565131625000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-014-1102-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,1,23]]},"references-count":17,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2014,11]]}},"alternative-id":["1102"],"URL":"https:\/\/doi.org\/10.1007\/s11227-014-1102-4","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,1,23]]}}}