{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,6]],"date-time":"2024-09-06T20:03:50Z","timestamp":1725653030331},"publisher-location":"London","reference-count":26,"publisher":"Springer London","isbn-type":[{"type":"print","value":"9781447124368"},{"type":"electronic","value":"9781447124375"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2012]]},"DOI":"10.1007\/978-1-4471-2437-5_5","type":"book-chapter","created":{"date-parts":[[2012,1,16]],"date-time":"2012-01-16T12:58:27Z","timestamp":1326718707000},"page":"123-146","source":"Crossref","is-referenced-by-count":2,"title":["Dense Linear Algebra on Accelerated Multicore Hardware"],"prefix":"10.1007","author":[{"given":"Jack","family":"Dongarra","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jakub","family":"Kurzak","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Piotr","family":"Luszczek","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stanimire","family":"Tomov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"5_CR1","volume-title":"Dependence Analysis (Loop Transformation for Restructuring Compilers)","author":"U. Banerjee","year":"1996","unstructured":"Banerjee, U.: Dependence Analysis (Loop Transformation for Restructuring Compilers). Springer, Berlin (1996)"},{"key":"5_CR2","first-page":"340","volume-title":"International Conference on Supercomputing","author":"J. Bilmes","year":"1997","unstructured":"Bilmes, J., Asanovic, K., Chin, C.W., Demmel, J.: Optimizing matrix multiply using PHiPAC: a portable, high-performance, ANSI C coding methodology. In: International Conference on Supercomputing, pp. 340\u2013347 (1997). \n                citeseer.ist.psu.edu\/article\/bilmes97optimizing.html"},{"key":"5_CR3","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898719642","volume-title":"ScaLAPACK Users\u2019 Guide","author":"L.S. Blackford","year":"1997","unstructured":"Blackford, L.S., Choi, J., Cleary, A., D\u2019Azevedo, E., Demmel, J., Dhillon, I., Dongarra, J., Hammarling, S., Henry, G., Petitet, A., Stanley, K., Walker, D., Whaley, R.C.: ScaLAPACK Users\u2019 Guide. Society for Industrial and Applied Mathematics, Philadelphia (1997)"},{"issue":"4","key":"5_CR4","doi-asserted-by":"publisher","first-page":"481","DOI":"10.1177\/1094342006070078","volume":"20","author":"R. Bolze","year":"2006","unstructured":"Bolze, R., Cappello, F., Caron, E., Dayd\u00e9, M., Desprez, F., Jeannot, E., J\u00e9gou, Y., Lanteri, S., Leduc, J., Melab, N., Mornet, G., Namyst, R., Primet, P., Quetier, B., Richard, O., Talbi, E.G., Touche, I.: Grid\u20195000: A large scale and highly reconfigurable experimental grid testbed. Int. J. High Perform. Comput. Appl. 20(4), 481\u2013494 (2006)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"5_CR5","doi-asserted-by":"crossref","unstructured":"Bosilca, G., Bouteiller, A., Danalis, A., Faverge, M., Haidar, A., Herault, T., Kurzak, J., Langou, J., Lemarinier, P., Ltaief, H., Luszczek, P., YarKhan, A., Dongarra, J.: Flexible development of dense linear algebra algorithms on massively parallel architectures with DPLASMA. In: IEEE International Symposium on Parallel and Distributed Processing Workshops and PhD Forum, pp.\u00a01432\u20131444 (2011)","DOI":"10.1109\/IPDPS.2011.299"},{"key":"5_CR6","unstructured":"Bosilca, G., Bouteiller, A., Danalis, A., Faverge, M., Haidar, H., Herault, T., Kurzak, J., Langou, J., Lemarinier, P., Ltaief, H., Luszczek, P., YarKhan, A., Dongarra, J.: Distributed-memory task execution and dependence tracking within DAGuE and the DPLASMA project. Tech. Rep. 232, LAPACK Working Note (2010). \n                http:\/\/www.netlib.org\/lapack\/lawnspdf\/lawn232.pdf"},{"key":"5_CR7","unstructured":"Bosilca, G., Bouteiller, A., Danalis, A., Herault, T., Lemarinier, P., Dongarra, J.: DAGuE: A generic distributed DAG engine for high performance computing. Tech. Rep. 231, LAPACK Working Note (2010). \n                http:\/\/www.netlib.org\/lapack\/lawnspdf\/lawn231.pdf"},{"key":"5_CR8","doi-asserted-by":"crossref","unstructured":"Bosilca, G., Bouteiller, A., Danalis, A., Herault, T., Lemarinier, P., Dongarra, J.: DAGuE: A generic distributed DAG engine for high performance computing. In: Proceedings of the 16th International Workshop on High-Level Parallel Programming Models and Supportive Environments (HIPS\u201911), Anchorage, AL, USA (2011)","DOI":"10.1109\/IPDPS.2011.281"},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"Bosilca, G., Bouteiller, A., Danalis, A., Herault, T., Lemarinier, P., Dongarra, J.: DAGuE: A generic distributed DAG engine for high performance computing. In: 16th International Workshop on High-Level Parallel Programming Models and Supportive Environments (HIPS-11), Anchorage, AK (2011)","DOI":"10.1109\/IPDPS.2011.281"},{"key":"5_CR10","unstructured":"CUDA CUBLAS Library. \n                http:\/\/developer.download.nvidia.com"},{"issue":"5","key":"5_CR11","doi-asserted-by":"publisher","first-page":"256","DOI":"10.1109\/JSSC.1974.1050511","volume":"9","author":"R.H. Dennard","year":"1974","unstructured":"Dennard, R.H., Gaensslen, F.H., Rideout, V.L., Bassous, E., LeBlanc, A.R.: Design of ion-implanted MOSFET\u2019s with very small physical dimensions. IEEE J. Solid-State Circuits 9(5), 256\u2013268 (1974). doi:\n                10.1109\/JSSC.1974.1050511","journal-title":"IEEE J. Solid-State Circuits"},{"issue":"9","key":"5_CR12","doi-asserted-by":"publisher","first-page":"803","DOI":"10.1002\/cpe.728","volume":"15","author":"J.J. Dongarra","year":"2003","unstructured":"Dongarra, J.J., Luszczek, P., Petitet, A.: The LINPACK benchmark: Past, present and future. Concurr. Comput. 15(9), 803\u2013820 (2003). doi:\n                10.1002\/cpe.728","journal-title":"Concurr. Comput."},{"issue":"2","key":"5_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1499096.1499100","volume":"36","author":"F.G. Gustavson","year":"2009","unstructured":"Gustavson, F.G., Karlsson, L., K\u00e5gstr\u00f6m, B.: Distributed SBP Cholesky factorization algorithms with near-optimal scheduling. ACM Trans. Math. Softw. 36(2), 1\u201325 (2009). \n                http:\/\/doi.acm.org\/10.1145\/1499096.1499100","journal-title":"ACM Trans. Math. Softw."},{"key":"5_CR14","unstructured":"Kogge, P., Bergman, K., Borkar, S., Campbell, D., Carlson, W., Dally, W., Denneau, M., Franzon, P., Harrod, W., Hill, K., Hiller, J., Karp, S., Keckler, S., Klein, D., Lucas, R., Richards, M., Scarpelli, A., Scott, S., Snavely, A., Sterling, T., Williams, R.S., Yelick, K.: ExaScale computing study: Technology challenges in achieving exascale systems. Tech. Rep. TR-2008-13, Department of Computer Science and Engineering, University of Notre Dame (2008)"},{"key":"5_CR15","doi-asserted-by":"publisher","first-page":"884","DOI":"10.1007\/978-3-642-01970-8_89","volume-title":"ICCS \u201909: Proceedings of the 9th International Conference on Computational Science","author":"Y. Li","year":"2009","unstructured":"Li, Y., Dongarra, J., Tomov, S.: A note on auto-tuning GEMM for GPUs. In: ICCS \u201909: Proceedings of the 9th International Conference on Computational Science, pp. 884\u2013892. Springer, Berlin (2009). doi:\n                10.1007\/978-3-642-01970-8_89"},{"key":"5_CR16","unstructured":"Moore, G.E.: Cramming more components onto integrated circuits. Electronics 38(8) (1965)"},{"key":"5_CR17","volume-title":"Proc. of High Performance Computing for Computational Science (VECPAR\u201910)","author":"R. Nath","year":"2010","unstructured":"Nath, R., Tomov, S., Dongarra, J.: Accelerating GPU kernels for dense linear algebra. In: Proc. of High Performance Computing for Computational Science (VECPAR\u201910), June 22\u201325, 2010"},{"key":"5_CR18","volume-title":"The Potential Impact of High-End Capability Computing on Four Illustrative Fields of Science and Engineering","author":"National Research Council Committee on the Potential Impact of High-End Computing on Illustrative Fields of Science and Engineering","year":"2008","unstructured":"National Research Council Committee on the Potential Impact of High-End Computing on Illustrative Fields of Science and Engineering: The Potential Impact of High-End Capability Computing on Four Illustrative Fields of Science and Engineering. Academies Press, Washington (2008)"},{"key":"5_CR19","unstructured":"NVIDIA: NVIDIA\u2019s Next Generation CUDA Compute Architecture: Fermi (2009). \n                http:\/\/www.nvidia.com\/object\/fermi_architecture.html"},{"key":"5_CR20","unstructured":"NVIDIA: NVIDIA CUDA\u2122 Best Practices Guide Version 3.0. NVIDIA Corporation (2010)"},{"key":"5_CR21","unstructured":"NVIDIA: NVIDIA CUDA\u2122 Programming Guide Version 3.0. NVIDIA Corporation (2010)"},{"key":"5_CR22","unstructured":"Sutter, H.: The free lunch is over: A fundamental turn toward concurrency in software. Dr. Dobb\u2019s Journal 30(3) (2005). \n                http:\/\/www.ddj.com\/184405990"},{"key":"5_CR23","unstructured":"Tomov, S., Nath, R., Du, P., Dongarra, J.: MAGMA version 0.2 User Guide (11\/2009). \n                http:\/\/icl.cs.utk.edu\/magma"},{"key":"5_CR24","unstructured":"University of Tennessee: PLASMA Users\u2019 Guide, Parallel Linear Algebra Software for Multicore Architectures, Version 2.2 (2009)"},{"key":"5_CR25","first-page":"1","volume-title":"SC \u201908: Proceedings of the 2008 ACM\/IEEE Conference on Supercomputing","author":"V. Volkov","year":"2008","unstructured":"Volkov, V., Demmel, J.: Benchmarking GPUs to tune dense linear algebra. In: SC \u201908: Proceedings of the 2008 ACM\/IEEE Conference on Supercomputing, pp. 1\u201311. IEEE Press, Piscataway (2008). \n                http:\/\/doi.acm.org\/10.1145\/1413370.1413402"},{"issue":"1\u20132","key":"5_CR26","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1016\/S0167-8191(00)00087-9","volume":"27","author":"R.C. Whaley","year":"2001","unstructured":"Whaley, R.C., Petitet, A., Dongarra, J.: Automated empirical optimizations of software and the ATLAS project. Parallel Comput. 27(1\u20132), 3\u201335 (2001)","journal-title":"Parallel Comput."}],"container-title":["High-Performance Scientific Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-1-4471-2437-5_5.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,5,1]],"date-time":"2021-05-01T00:56:38Z","timestamp":1619830598000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-1-4471-2437-5_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012]]},"ISBN":["9781447124368","9781447124375"],"references-count":26,"URL":"https:\/\/doi.org\/10.1007\/978-1-4471-2437-5_5","relation":{},"subject":[],"published":{"date-parts":[[2012]]}}}