{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T10:31:33Z","timestamp":1752229893996},"publisher-location":"Berlin, Heidelberg","reference-count":11,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642387173"},{"type":"electronic","value":"9783642387180"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-38718-0_10","type":"book-chapter","created":{"date-parts":[[2013,5,23]],"date-time":"2013-05-23T20:57:50Z","timestamp":1369342670000},"page":"72-79","source":"Crossref","is-referenced-by-count":6,"title":["Optimizing Memory-Bound SYMV Kernel on GPU Hardware Accelerators"],"prefix":"10.1007","author":[{"given":"Ahmad","family":"Abdelfattah","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jack","family":"Dongarra","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David","family":"Keyes","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hatem","family":"Ltaief","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"10_CR1","unstructured":"Matrix Algebra on GPU and Multicore Architectures. Innovative Computing Laboratory, University of Tennessee, \n                    \n                      http:\/\/icl.cs.utk.edu\/magma\/"},{"key":"10_CR2","unstructured":"Nvidia visual profiler, \n                    \n                      http:\/\/developer.nvidia.com\/nvidia-visual-profiler"},{"key":"10_CR3","unstructured":"Performance Application Programming Interface (PAPI). Innovative Computing Laboratory, University of Tennessee, \n                    \n                      http:\/\/icl.cs.utk.edu\/papi\/"},{"key":"10_CR4","unstructured":"Datta, K., Williams, S., Volkov, V., Carter, J., Oliker, L., Shalf, J., Yelick, K.: Auto-tuning the 27-Point Stencil for Multicore. In: Proc. iWAPT 2009: The Fourth International Workshop on Automatic Performance Tuning (2009)"},{"key":"10_CR5","unstructured":"Glaskowsky, P.N.: nVidia\u2019s Fermi: The first complete gpu computing architecture. Technical report (2009)"},{"key":"10_CR6","unstructured":"Kirk, D., Mei Hwu, W.: Programming Massively Parallel Processors, A Hands-on Approach. Morgan Kaufmann (2010)"},{"issue":"9","key":"10_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TPDS.2007.70813","volume":"19","author":"J. Kurzak","year":"2008","unstructured":"Kurzak, J., Buttari, A., Dongarra, J.J.: Solving systems of linear equations on the CELL processor using Cholesky factorization. IEEE Transactions on Parallel and Distributed Systems\u00a019(9), 1\u201311 (2008)","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"10_CR8","unstructured":"McCalpin, J.: Stream: Sustainable memory bandwidth in high performance computers, \n                    \n                      http:\/\/www.cs.virginia.edu\/stream\/"},{"key":"10_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2063384.2063392","volume-title":"Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2011","author":"R. Nath","year":"2011","unstructured":"Nath, R., Tomov, S., Dong, T., Dongarra, J.: Optimizing symmetric dense matrix-vector multiplication on gpus. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2011, pp. 6:1\u20136:10. ACM, New York (2011)"},{"key":"10_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1007\/978-3-642-19328-6_10","volume-title":"High Performance Computing for Computational Science \u2013 VECPAR 2010","author":"R. Nath","year":"2011","unstructured":"Nath, R., Tomov, S., Dongarra, J.: Accelerating GPU Kernels for Dense Linear Algebra. In: Palma, J.M.L.M., Dayd\u00e9, M., Marques, O., Lopes, J.C. (eds.) VECPAR 2010. LNCS, vol.\u00a06449, pp. 83\u201392. Springer, Heidelberg (2011)"},{"key":"10_CR11","doi-asserted-by":"crossref","unstructured":"Volkov, V., Demmel, J.W.: Benchmarking GPUs to Tune Dense Linear Algebra. In: Proceedings of the 2008 ACM\/IEEE Conference on Supercomputing, SC 2008, pp. 31:1\u201331:11. IEEE Press, Piscataway (2008)","DOI":"10.1109\/SC.2008.5214359"}],"container-title":["Lecture Notes in Computer Science","High Performance Computing for Computational Science - VECPAR 2012"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-38718-0_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,13]],"date-time":"2019-05-13T02:10:57Z","timestamp":1557713457000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-38718-0_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642387173","9783642387180"],"references-count":11,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-38718-0_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2013]]}}}