{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T12:17:54Z","timestamp":1763468274896,"version":"3.40.3"},"publisher-location":"Cham","reference-count":45,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319201184"},{"type":"electronic","value":"9783319201191"}],"license":[{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015]]},"DOI":"10.1007\/978-3-319-20119-1_3","type":"book-chapter","created":{"date-parts":[[2015,6,19]],"date-time":"2015-06-19T10:36:48Z","timestamp":1434710208000},"page":"31-47","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":18,"title":["A Framework for Batched and GPU-Resident Factorization Algorithms Applied to Block Householder Transformations"],"prefix":"10.1007","author":[{"given":"Azzam","family":"Haidar","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tingxing Tim","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stanimire","family":"Tomov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Piotr","family":"Luszczek","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jack","family":"Dongarra","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,6,20]]},"reference":[{"issue":"1","key":"3_CR1","first-page":"012037","volume":"180","author":"E Agullo","year":"2009","unstructured":"Agullo, E., Demmel, J., Dongarra, J., Hadri, B., Kurzak, J., Langou, J., Ltaief, H., Luszczek, P., Tomov, S.: Numerical linear algebra on emerging architectures: the PLASMA and MAGMA projects. J. Phys.: Conf. Ser. 180(1), 012037 (2009)","journal-title":"J. Phys.: Conf. Ser."},{"key":"3_CR2","volume-title":"GPU Computing Gems","author":"E Agullo","year":"2010","unstructured":"Agullo, E., Augonnet, C., Dongarra, J., Ltaief, H., Namyst, R., Thibault, S., Tomov, S.: Faster, cheaper, better - a hybridization methodology to develop linear algebra software for GPUS. In: Hwu, W.W. (ed.) GPU Computing Gems. Morgan Kaufmann, California (2010)"},{"key":"3_CR3","doi-asserted-by":"crossref","unstructured":"Agullo, E., Dongarra, J., Nath, R.,Tomov, S.: Fully empirical autotuned qr factorization for multicore architectures (2011). CoRR, abs\/1102.5328","DOI":"10.1007\/978-3-642-23397-5_19"},{"key":"3_CR4","unstructured":"ACML - AMD Core Math Library (2014). http:\/\/developer.amd.com\/tools-and-sdks\/cpu-development\/amd-core-math-library-acml"},{"key":"3_CR5","doi-asserted-by":"crossref","unstructured":"Anderson, M.J., Sheffield, D., Keutzer. K.: A predictive model for solving small linear algebra problems in gpu registers. In: IEEE 26th International Parallel Distributed Processing Symposium (IPDPS) (2012)","DOI":"10.1109\/IPDPS.2012.11"},{"key":"3_CR6","series-title":"Lecture Notes in Computer Science","first-page":"1","volume-title":"Applied Parallel Computing","author":"A Buttari","year":"2007","unstructured":"Buttari, A., Dongarra, J., Kurzak, J., Langou, J., Luszczek, P., Tomov, S.: The impact of multicore on math software. In: K\u00e5gstr\u00f6m, B., Elmroth, E., Dongarra, J., Wa\u015bniewski, J. (eds.) PARA 2006. LNCS, vol. 4699, pp. 1\u201310. Springer, Heidelberg (2007)"},{"key":"3_CR7","unstructured":"Cao, C., Dongarra, J., Du, P., Gates, M., Luszczek, P., Tomov, S.: clMAGMA: high performance dense linear algebra with OpenCL. In: The ACM International Conference Series, Atlanta, May 13\u201314 (2013). (submitted)"},{"key":"3_CR8","doi-asserted-by":"crossref","unstructured":"Dong, T., Haidar, A., Luszczek, P., Harris, A., Tomov, S., Dongarra, J.: LU factorization of small matrices: accelerating batched DGETRF on the GPU. In: Proceedings of 16th IEEE International Conference on High Performance and Communications (HPCC 2014), August 2014","DOI":"10.1109\/HPCC.2014.30"},{"key":"3_CR9","doi-asserted-by":"crossref","unstructured":"Dong, T., Haidar, A., Tomov, S., Dongarra, J.: A fast batched cholesky factorization on a GPU. In: Proceedings of 2014 International Conference on Parallel Processing (ICPP-2014), September 2014","DOI":"10.1109\/ICPP.2014.52"},{"key":"3_CR10","doi-asserted-by":"crossref","unstructured":"Dong, T., Dobrev, V., Kolev, T., Rieben, R., Tomov, S., Dongarra, J.: A step towards energy efficient computing: redesigning a hydrodynamic application on CPU-GPU. In: IEEE 28th International Parallel Distributed Processing Symposium (IPDPS) (2014)","DOI":"10.1109\/IPDPS.2014.103"},{"issue":"1","key":"3_CR11","first-page":"85","volume":"1","author":"J Dongarra","year":"2014","unstructured":"Dongarra, J., Haidar, A., Kurzak, J., Luszczek, P., Tomov, S., YarKhan, A.: Model-driven one-sided factorizations on multicore accelerated systems. Int. J.Supercomputing Frontiers Innovations 1(1), 85 (2014)","journal-title":"Int. J.Supercomputing Frontiers Innovations"},{"issue":"8","key":"3_CR12","doi-asserted-by":"publisher","first-page":"391","DOI":"10.1016\/j.parco.2011.10.002","volume":"38","author":"D Peng","year":"2012","unstructured":"Peng, D., Weber, R., Luszczek, P., Tomov, S., Peterson, G., Dongarra, J.: From CUDA to OpenCL: towards a performance-portable solution for multi-platform GPU programming. Parallel Comput. 38(8), 391\u2013407 (2012)","journal-title":"Parallel Comput."},{"key":"3_CR13","unstructured":"Oak Ridge Leadership Computing Facility. Annual report 2013\u20132014 (2014). https:\/\/www.olcf.ornl.gov\/wp-content\/uploads\/2015\/01\/AR_2014_Small.pdf"},{"issue":"5","key":"3_CR14","doi-asserted-by":"publisher","first-page":"532","DOI":"10.1145\/42411.42415","volume":"31","author":"JL Gustafson","year":"1988","unstructured":"Gustafson, J.L.: Reevaluating Amdahl\u2019s law. Commun. ACM 31(5), 532\u2013533 (1988)","journal-title":"Commun. ACM"},{"issue":"2","key":"3_CR15","doi-asserted-by":"publisher","first-page":"196","DOI":"10.1177\/1094342013502097","volume":"28","author":"A Haidar","year":"2012","unstructured":"Haidar, A., Tomov, S., Dongarra, J., Solca, R., Schulthess, T.: A novel hybrid CPU-GPU generalized eigensolver for electronic structure calculations based on fine grained memory aware tasks. Int. J. High Perform. Comput. Appl. 28(2), 196\u2013209 (2012)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"3_CR16","doi-asserted-by":"crossref","unstructured":"Haidar, A., Cao, C., Yarkhan, A., Luszczek, P., Tomov, S., Kabir, K., Dongarra, J.: Unified development for mixed multi-gpu and multi-coprocessor environments using a lightweight runtime environment. In: IPDPS 2014 Proceedings of the 2014 IEEE 28th International Parallel and Distributed Processing Symposium, pp. 491\u2013500. IEEE Computer Society, Washington, (2014)","DOI":"10.1109\/IPDPS.2014.58"},{"issue":"1","key":"3_CR17","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1177\/1094342014567546","volume":"18","author":"A Haidar","year":"2015","unstructured":"Haidar, A., Dong, T., Luszczek, P., Tomov, S., Dongarra, J.: Batched matrix computations on hardware accelerators based on GPUs. Int. J. High Performance Comput. Appl. 18(1), 135\u2013158 (2015). doi:10.1177\/1094342014567546","journal-title":"Int. J. High Performance Comput. Appl."},{"key":"3_CR18","doi-asserted-by":"crossref","unstructured":"Haidar, A., Luszczek, P., Tomov, S., Dongarra, J.: Optimization for performance and energy for batched matrix computations on GPUs. In: PPoPP 2015 8th Workshop on General Purpose Processing Using GPUs (GPGPU 8) co-located with PPOPP 2015, ACM, San Francisco, February 2015","DOI":"10.1145\/2716282.2716288"},{"key":"3_CR19","doi-asserted-by":"crossref","unstructured":"Haidar, A., Luszczek, P., Tomov, S., Dongarra, J.: Towards batched linear solvers on accelerated hardware platforms. In: PPoPP 2015 Proceedings of the 20th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, ACM, San Francisco, February 2015","DOI":"10.1145\/2688500.2688534"},{"issue":"1","key":"3_CR20","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1177\/1094342004041296","volume":"18","author":"E-J Im","year":"2004","unstructured":"Im, E.-J., Yelick, K., Vuduc, R.: Sparsity: optimization framework for sparse matrix kernels. Int. J. High Perform. Comput. Appl. 18(1), 135\u2013158 (2004)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"3_CR21","unstructured":"Matrix algebra on GPU and multicore architectures (MAGMA), MAGMA Release 1.6.1 (2015). http:\/\/icl.cs.utk.edu\/magma\/"},{"key":"3_CR22","unstructured":"Intel Pentium III Processor - Small Matrix Library (1999). http:\/\/www.intel.com\/design\/pentiumiii\/sml\/"},{"key":"3_CR23","unstructured":"Intel Math Kernel Library (2014). http:\/\/software.intel.com\/intel-mkl\/"},{"key":"3_CR24","unstructured":"Intel 64 and IA-32 architectures software developer\u2019s manual, July 20 (2014). http:\/\/download.intel.com\/products\/processor\/manual\/"},{"key":"3_CR25","unstructured":"Keyes, D., Taylor, V.: NSF-ACCI task force on software for science and engineering, December 2010"},{"key":"3_CR26","first-page":"50","volume":"25C","author":"JC Liao","year":"2014","unstructured":"Liao, J.C., Khodayari, A., Zomorrodi, A.R., Maranas, C.D.: A kinetic model of escherichia coli core metabolism satisfying multiple sets of mutant flux data. Metab. Eng. 25C, 50\u201362 (2014)","journal-title":"Metab. Eng."},{"key":"3_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"884","DOI":"10.1007\/978-3-642-01970-8_89","volume-title":"Computational Science \u2013 ICCS 2009","author":"Y Li","year":"2009","unstructured":"Li, Y., Dongarra, J., Tomov, S.: A note on auto-tuning GEMM for GPUs. In: Allen, G., Nabrzyski, J., Seidel, E., van Albada, G.D., Dongarra, J., Sloot, P.M.A. (eds.) ICCS 2009, Part I. LNCS, vol. 5544, pp. 884\u2013892. Springer, Heidelberg (2009)"},{"key":"3_CR28","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1007\/978-3-642-36803-5_6","volume-title":"Applied Parallel and Scientific Computing","author":"OEB Messer","year":"2013","unstructured":"Messer, O.E.B., Harris, J.A., Parete-Koon, S., Chertkow, M.A.: Multicore and accelerator development for a leadership-class stellar astrophysics code. In: Manninen, P., \u00d6ster, P. (eds.) PARA. LNCS, vol. 7782, pp. 92\u2013106. Springer, Heidelberg (2013)"},{"key":"3_CR29","unstructured":"Molero, J.M., Garz\u00f3n, E.M., Garc\u00eda, I., Quintana-Ort\u00ed, E.S, Plaza, A.: Poster: a batched Cholesky solver for local RX anomaly detection on GPUs. In: PUMPS (2013)"},{"key":"3_CR30","doi-asserted-by":"crossref","unstructured":"Nath, R., Tomov,S., Dong, T., Dongarra, T.: Optimizing symmetric dense matrix-vectormultiplication on GPUs. In: Proceedings of 2011 International Conference for High PerformanceComputing, Networking, Storage and Analysis, November 2011","DOI":"10.1145\/2063384.2063392"},{"key":"3_CR31","unstructured":"Nath, R., Tomov, S., Dongarra, T.: Accelerating GPU kernels for dense linear algebra. In: VECPAR 2010 Proceedings of the 2009 International Meeting on High Performance Computing for Computational Science, pp. 22\u201325. Springer, Berkeley, June 2010"},{"issue":"4","key":"3_CR32","doi-asserted-by":"publisher","first-page":"511","DOI":"10.1177\/1094342010385729","volume":"24","author":"R Nath","year":"2010","unstructured":"Nath, R., Tomov, S., Dongarra, J.: An improved magma gemm for fermi graphics processing units. Int. J. High Perform. Comput. Appl. 24(4), 511\u2013515 (2010)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"3_CR33","unstructured":"Nvidia visual profiler"},{"key":"3_CR34","unstructured":"https:\/\/developer.nvidia.com\/nvidia-management-library-nvml (2014)"},{"key":"3_CR35","unstructured":"CUBLAS (2014). http:\/\/docs.nvidia.com\/cuda\/cublas\/"},{"key":"3_CR36","unstructured":"CUBLAS 6.5, January 2015. http:\/\/docs.nvidia.com\/cuda\/cublas\/"},{"key":"3_CR37","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"813","DOI":"10.1007\/978-3-642-40047-6_81","volume-title":"Euro-Par 2013 Parallel Processing","author":"O Villa","year":"2013","unstructured":"Villa, O., Fatica, M., Gawande, N., Tumeo, A.: Power\/performance trade-offs of small batched LU based solvers on GPUs. In: Wolf, F., Mohr, B., an Mey, D. (eds.) Euro-Par 2013. LNCS, vol. 8097, pp. 813\u2013825. Springer, Heidelberg (2013)"},{"key":"3_CR38","unstructured":"Nitin, V.O., Gawande, A., Tumeo, A.: Accelerating subsurface transport simulation on heterogeneous clusters. In: IEEE International Conference on Cluster Computing (CLUSTER 2013), pp. 23\u201327, Indiana, September 2013"},{"issue":"2","key":"3_CR39","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1109\/MM.2012.12","volume":"32","author":"E Rotem","year":"2012","unstructured":"Rotem, E., Naveh, A., Rajwan, D., Ananthakrishnan, A., Weissmann, E.: Power-management architecture of the intel microarchitecture code-named sandy bridge. IEEE Micro. 32(2), 20\u201327 (2012). doi:10.1109\/MM.2012.12. ISSN: 0272\u20131732","journal-title":"IEEE Micro."},{"issue":"5\u20136","key":"3_CR40","doi-asserted-by":"publisher","first-page":"232","DOI":"10.1016\/j.parco.2009.12.005","volume":"36","author":"S Tomov","year":"2010","unstructured":"Tomov, S., Dongarra, J., Baboulin, M.: Towards dense linear algebra for hybrid gpu accelerated manycore systems. Parellel Comput. Syst. Appl. 36(5\u20136), 232\u2013240 (2010). doi:10.1016\/j.parco.2009.12.005","journal-title":"Parellel Comput. Syst. Appl."},{"key":"3_CR41","doi-asserted-by":"publisher","unstructured":"Tomov, S., Nath, R., Ltaief, H., Dongarra, J.: Dense linear algebra solvers for multicore with GPU accelerators. In: Proceedings of the IEEE IPDPS 2010, pp. 1\u20138. IEEE Computer Society, Atlanta, 19\u201323 April 2010. doi:10.1109\/IPDPSW.2010.5470941","DOI":"10.1109\/IPDPSW.2010.5470941"},{"key":"3_CR42","volume-title":"Scientific Computing with Multicore and Accelerators","author":"S Tomov","year":"2010","unstructured":"Tomov, S., Dongarra, J.: Dense linear algebra for hybrid gpu-based systems. In: Kurzak, J., Bader, D.A., Dongarra, J. (eds.) Scientific Computing with Multicore and Accelerators. Chapman and Hall\/CRC, UK (2010)"},{"key":"3_CR43","unstructured":"Wainwright, I .: Optimized LU-decomposition with full pivot for small batched matrices, GTC 2013 - ID S3069. April 2013"},{"key":"3_CR44","doi-asserted-by":"crossref","unstructured":"Yamazaki, I., Tomov, S., Dongarra, J.: One-sided dense matrix factorizations on a multicore with multiple GPU accelerators. In: Proceedings of the International Conference on Computational Science, ICCS 2012, pp. 37\u201346. Procedia Computer Science, 9(0):37 (2012)","DOI":"10.1016\/j.procs.2012.04.005"},{"key":"3_CR45","unstructured":"Yeralan, S.N., Davis, T.A., Ranka, S.: Sparse mulitfrontal QR on the GPU. Technical report, University of Florida Technical report (2013)"}],"container-title":["Lecture Notes in Computer Science","High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-20119-1_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,10]],"date-time":"2023-02-10T10:34:18Z","timestamp":1676025258000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-20119-1_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015]]},"ISBN":["9783319201184","9783319201191"],"references-count":45,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-20119-1_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2015]]},"assertion":[{"value":"20 June 2015","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}