{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T12:17:37Z","timestamp":1763468257988,"version":"3.41.0"},"publisher-location":"New York, New York, USA","reference-count":22,"publisher":"ACM Press","license":[{"start":{"date-parts":[[2014,1,1]],"date-time":"2014-01-01T00:00:00Z","timestamp":1388534400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"AMD"},{"name":"National Science Foundation through the Keeneland: National Institute for Experimental Computing grant","award":["0910735"],"award-info":[{"award-number":["0910735"]}]},{"name":"Department of Energy"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2014]]},"DOI":"10.1145\/2664666.2664667","type":"proceedings-article","created":{"date-parts":[[2015,2,23]],"date-time":"2015-02-23T16:02:15Z","timestamp":1424707335000},"page":"1-9","source":"Crossref","is-referenced-by-count":9,"title":["clMAGMA"],"prefix":"10.1145","author":[{"given":"Chongxiao","family":"Cao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jack","family":"Dongarra","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peng","family":"Du","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mark","family":"Gates","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Piotr","family":"Luszczek","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stanimire","family":"Tomov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","reference":[{"key":"key-10.1145\/2664666.2664667-1","unstructured":"E. Agullo, J. Demmel, J. Dongarra, B. Hadri, J. Kurzak, J. Langou, H. Ltaief, P. Luszczek, and S. Tomov. Numerical linear algebra on emerging architectures: The PLASMA and MAGMA projects.J. Phys.: Conf. Ser., 180(1), 2009."},{"key":"key-10.1145\/2664666.2664667-2","unstructured":"AMD. Accelerated Parallel Processing Math Libraries (APPML). Available at http: \/\/developer.amd.com\/tools\/heterogeneous-computing\/."},{"key":"key-10.1145\/2664666.2664667-3","unstructured":"AMD. AMD Core Math Library (ACML). Available at http:\/\/developer.amd.com\/tools\/."},{"key":"key-10.1145\/2664666.2664667-4","unstructured":"AMD. AMD Accelerated Parallel Processing OpenCL Programming Guide (v2.4), 2012. Available at: http:\/\/developer.amd.com."},{"key":"key-10.1145\/2664666.2664667-5","doi-asserted-by":"crossref","unstructured":"E. Anderson, Z. Bai, C. Bischof, S. L. Blackford, J. W. Demmel, J. J. Dongarra, J. D. Croz, A. Greenbaum, S. Hammarling, A. McKenney, and D. C. Sorensen.LAPACK User's Guide. SIAM, Philadelphia, Third edition, 1999.","DOI":"10.1137\/1.9780898719604"},{"key":"key-10.1145\/2664666.2664667-6","doi-asserted-by":"crossref","unstructured":"C. Augonnet, S. Thibault, R. Namyst, and P. Wacrenier. StarPU: A unified platform for task scheduling on heterogeneous multicore architectures.Concurrency Computat. Pract. Exper., 2010. (to appear).","DOI":"10.1002\/cpe.1631"},{"key":"key-10.1145\/2664666.2664667-7","unstructured":"Barcelona Supercomputing Center.SMP Superscalar (SMPSs) User's Manual, Version 2.0, 2008. http:\/\/www.bsc.es\/media\/1002.pdf."},{"key":"key-10.1145\/2664666.2664667-8","doi-asserted-by":"crossref","unstructured":"L. S. Blackford, J. Choi, A. Cleary, E. D'Azevedo, J. Demmel, I. Dhillon, J. J. Dongarra, S. Hammarling, G. Henry, A. Petitet, K. Stanley, D. Walker, and R. C. Whaley.ScaLAPACK Users' Guide. SIAM, Philadelphia, PA, 1997. http:\/\/www.netlib.org\/scalapack\/slug\/.","DOI":"10.1137\/1.9780898719642"},{"key":"key-10.1145\/2664666.2664667-9","unstructured":"Software distribution of clMAGMA version 1.0. http:\/\/icl.cs.utk.edu\/magma\/software\/, October 24 2012."},{"key":"key-10.1145\/2664666.2664667-10","doi-asserted-by":"crossref","unstructured":"J. Dongarra, J. Bunch, C. Moler, and G. Stewart.LINPACK Users' Guide. SIAM, Philadelphia, PA, 1979.","DOI":"10.1137\/1.9781611971811"},{"key":"key-10.1145\/2664666.2664667-11","doi-asserted-by":"crossref","unstructured":"P. Du, R. Weber, P. Luszczek, S. Tomov, G. Peterson, and J. Dongarra. From CUDA to OpenCL: Towards a performance-portable solution for multi-platform GPU programming.Parallel Comput., 38(8):391--407, Aug. 2012.","DOI":"10.1016\/j.parco.2011.10.002"},{"key":"key-10.1145\/2664666.2664667-12","doi-asserted-by":"crossref","unstructured":"A. Haidar, S. Tomov, J. Dongarra, R. Solca, and T. Schulthess. A novel hybrid CPU-GPU generalized eigensolver for electronic structure calculations based on fine grained memory aware tasks.International Journal of High Performance Computing Applications, September 2012. (accepted).","DOI":"10.1177\/1094342013502097"},{"key":"key-10.1145\/2664666.2664667-13","unstructured":"Intel. Math Kernel Library. Available at http:\/\/software.intel.com\/en-us\/articles\/intel-mkl\/."},{"key":"key-10.1145\/2664666.2664667-14","unstructured":"Khronos OpenCL Working Group. The opencl specification, version: 1.0 document revision: 48, 2009."},{"key":"key-10.1145\/2664666.2664667-15","doi-asserted-by":"crossref","unstructured":"K. Matsumoto, N. Nakasato, S. Sedukhin, I. Tsuruga, and A. City. Implementing a code generator for fast matrix multiplication in OpenCL on the GPU. Technical Report 2012-002, The University of Aizu, July 2012.","DOI":"10.1109\/MCSoC.2012.30"},{"key":"key-10.1145\/2664666.2664667-16","doi-asserted-by":"crossref","unstructured":"N. Nakasato. A fast GEMM implementation on the Cypress GPU.SIGMETRICS Performance Evaluation Review, 38(4):50--55, 2011.","DOI":"10.1145\/1964218.1964227"},{"key":"key-10.1145\/2664666.2664667-17","unstructured":"K. Rupp, F. Rudolf, and J. Weinbub. ViennaCL - A High Level Linear Algebra Library for GPUs and Multi-Core CPUs. InIntl. Workshop on GPUs and Scientific Applications, pages 51--56, 2010."},{"key":"key-10.1145\/2664666.2664667-18","unstructured":"P. Tillet, K. Rupp, S. Selberherr, and C.-T. Lin. Towards performance-portable, scalable, and convenient linear algebra. InPresented as part of the 5th USENIX Workshop on Hot Topics in Parallelism, Berkeley, CA, 2013. USENIX."},{"key":"key-10.1145\/2664666.2664667-19","doi-asserted-by":"crossref","unstructured":"S. Tomov and J. Dongarra. Dense linear algebra for hybrid GPU-based systems. In J. Kurzak, D. A. Bader, and J. Dongarra, editors,Scientific Computing with Multicore and Accelerators. Chapman and Hall\/CRC, 2010.","DOI":"10.1201\/b10376-5"},{"key":"key-10.1145\/2664666.2664667-20","doi-asserted-by":"crossref","unstructured":"V. Volkov and J. Demmel. Benchmarking GPUs to tune dense linear algebra. InSupercomputing 08. IEEE, 2008.","DOI":"10.1109\/SC.2008.5214359"},{"key":"key-10.1145\/2664666.2664667-21","doi-asserted-by":"crossref","unstructured":"R. C. Whaley, A. Petitet, and J. J. Dongarra. Automated empirical optimizations of software and the ATLAS project.Parallel Computing, 27(1-2):3--35, 2001.","DOI":"10.1016\/S0167-8191(00)00087-9"},{"key":"key-10.1145\/2664666.2664667-22","unstructured":"A. YarKhan, J. Kurzak, and J. Dongarra. QUARK Users' Guide: QUeueing And Runtime for Kernels.University of Tennessee Innovative Computing Laboratory Technical Report ICL-UT-11-02, 2011."}],"event":{"number":"2014","sponsor":["ARM","Intel","StreamComputing, StreamComputing BV","Altera Corp., Altera Corporation","AMD","Codeplay, Codeplay Software Ltd.","SAMSUNG","Imagination, Imagination Technologies Limited","QI, Qualcomm Inc."],"acronym":"IWOCL '14","name":"the International Workshop","start":{"date-parts":[[2014,5,12]]},"location":"Bristol, United Kingdom","end":{"date-parts":[[2014,5,13]]}},"container-title":["Proceedings of the International Workshop on OpenCL 2013 &amp; 2014 - IWOCL '14"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2664666.2664667","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/dl.acm.org\/ft_gateway.cfm?id=2664667&amp;ftid=1545576&amp;dwn=1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T06:13:24Z","timestamp":1750227204000},"score":1,"resource":{"primary":{"URL":"http:\/\/dl.acm.org\/citation.cfm?doid=2664666.2664667"}},"subtitle":["high performance dense linear algebra with OpenCL"],"proceedings-subject":"OpenCL 2013 & 2014","short-title":[],"issued":{"date-parts":[[2014]]},"references-count":22,"URL":"https:\/\/doi.org\/10.1145\/2664666.2664667","relation":{},"subject":[],"published":{"date-parts":[[2014]]}}}