{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T21:42:08Z","timestamp":1784842928241,"version":"3.55.0"},"reference-count":29,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015,9]]},"DOI":"10.1109\/hpec.2015.7322444","type":"proceedings-article","created":{"date-parts":[[2015,11,12]],"date-time":"2015-11-12T23:12:34Z","timestamp":1447369954000},"page":"1-6","source":"Crossref","is-referenced-by-count":14,"title":["MAGMA embedded: Towards a dense linear algebra library for energy efficient extreme computing"],"prefix":"10.1109","author":[{"given":"Azzam","family":"Haidar","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stanimire","family":"Tomov","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Piotr","family":"Luszczek","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jack","family":"Dongarra","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"crossref","first-page":"46","DOI":"10.1145\/1513895.1513901","author":"fatica","year":"2009","journal-title":"Accelerating LINPACK with CUDA on heterogenous clusters In Proceedings of 2nd Workshop on General Purpose Processing on Graphics Processing Units (GPGPU-2)"},{"key":"ref11","author":"haidar","year":"2012","journal-title":"A novel hybrid CPU-GPU generalized eigensolver for electronic structure calculations based on fine grained memory aware tasks International Journal of High Performance Computing Applications"},{"key":"ref12","first-page":"491","author":"haidar","year":"2014","journal-title":"Unified development for mixed multi-gpu and multi-coprocessor environments using a lightweight runtime environment In Proceedings of the 2014 IEEE 28th International Parallel and Distributed Processing Symposium IPDPS '14"},{"key":"ref13","author":"haidar","year":"2015","journal-title":"Optimization for performance and energy for batched matrix computations on gpus In 8th Workshop on General Purpose Processing Using GPUs (GPGPU 8) co-located with PPOPP 2015 PPoPP 2015 San Francisco CA 02\/2015"},{"key":"ref14","author":"haidar","year":"2015","journal-title":"Towards batched linear solvers on accelerated hardware platforms In Proceedings of the 20th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming PPoPP 2015 San Francisco CA 02\/2015"},{"key":"ref15","year":"2015","journal-title":"Matrix algebra on GPU and multicore architectures (MAGMA) MAGMA Release 1 6 1"},{"key":"ref16","year":"2014","journal-title":"Intel Math Kernel Library"},{"key":"ref17","year":"2014","journal-title":"Intel&#x00AE; 64 and IA-32 architectures software de -veloper's manual"},{"key":"ref18","author":"li","year":"2009","journal-title":"A note on auto-tuning GEMM for GPUs In Proceedings of the 2009 International Conference on Computational Science ICCS'09"},{"key":"ref19","author":"nath","year":"2011","journal-title":"Optimizing symmetric dense matrix-vector multiplication on GPUs In Proceedings of 2011 International Conference for High Performance Computing Networking Storage and Analysis"},{"key":"ref28","author":"tomov","year":"2010","journal-title":"Scientific Computing with Multicore and Accelerators chapter Dense Linear Algebra for Hybrid GPU-based Systems Chapman and Hall\/CRC"},{"key":"ref4","first-page":"1","author":"buttari","year":"2006","journal-title":"The Impact of Multicore on Math Software"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2010.5470941"},{"key":"ref3","author":"agullo","year":"2011","journal-title":"Fully empirical autotuned qr factorization for multicore architectures CoRR abs\/1102 5328"},{"key":"ref6","author":"dong","year":"2012","journal-title":"Matrix-vector multiplication and tridiagonalization of a dense symmetric matrix on multiple GPUs and its application to symmetric eigenvalue problems Parallel Comput"},{"key":"ref29","article-title":"QUARK Users' Guide: QUeueing And Runtime for Kernels","author":"yarkhan","year":"2011","journal-title":"University of Tennessee Innovative Computing Laboratory Technical Report ICL-UT-11&#x2013;02"},{"key":"ref5","author":"cao","year":"2013","journal-title":"clmagma High performance dense linear algebra with opencl In The ACM International Conference Series"},{"key":"ref8","first-page":"1","author":"dongarra","year":"2012","journal-title":"Anatomy of a globally recursive embedded linpack benchmark In HPEC"},{"key":"ref7","volume":"1","author":"dongarra","year":"2014","journal-title":"Model-driven one-sided factorizations on multicore accelerated systems International Journal on Supercomputing Frontiers and Innovations"},{"key":"ref2","author":"agullo c\u00e9dric augonnet","year":"2010","journal-title":"Faster Cheaper Better A Hybridization Methodology to Develop Linear Algebra Software for GPUs"},{"key":"ref9","first-page":"391","volume":"38","author":"du","year":"2012","journal-title":"From CUDA to OpenCL Towards a Performance-Portable Solution for Multi-Platform GPU Programming"},{"key":"ref1","volume":"180","author":"aguilo","year":"2009","journal-title":"Numerical linear algebra on emerging architectures The PLASMA and MAGMA projects J Phys Conf Ser"},{"key":"ref20","author":"nath","year":"2010","journal-title":"Accelerating GPU kernels for dense linear algebra In Proceedings of the 2009 International Meeting on High Performance Computing for Computational Science VECPAR'10"},{"key":"ref22","year":"0","journal-title":"NVIDIA Visual Profiler"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1177\/1094342010385729"},{"key":"ref24","volume":"32","author":"rotem","year":"2012","journal-title":"Power-management architecture of the intel microarchitecture code-named sandy bridge IEEE Micro"},{"key":"ref23","year":"2014","journal-title":"Nvidia Management Library"},{"key":"ref26","first-page":"232","volume":"36","author":"tomov","year":"2010","journal-title":"Towards dense linear algebra for hybrid GPU accelerated manycore systems Parellel Comput Syst Appl"},{"key":"ref25","first-page":"53","volume":"10","author":"schreiber","year":"1989","journal-title":"A storage efficient WY representation for products of Householder transformations SIAM J Sci Stat Comput"}],"event":{"name":"2015 IEEE High Performance Extreme Computing Conference (HPEC)","location":"Waltham, MA, USA","start":{"date-parts":[[2015,9,15]]},"end":{"date-parts":[[2015,9,17]]}},"container-title":["2015 IEEE High Performance Extreme Computing Conference (HPEC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7311169\/7322434\/07322444.pdf?arnumber=7322444","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,26]],"date-time":"2022-05-26T00:59:06Z","timestamp":1653526746000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7322444\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,9]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/hpec.2015.7322444","relation":{},"subject":[],"published":{"date-parts":[[2015,9]]}}}