{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,5]],"date-time":"2024-09-05T01:52:27Z","timestamp":1725501147066},"publisher-location":"Berlin, Heidelberg","reference-count":25,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540757542"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.1007\/978-3-540-75755-9_66","type":"book-chapter","created":{"date-parts":[[2007,9,22]],"date-time":"2007-09-22T02:44:54Z","timestamp":1190429094000},"page":"540-549","source":"Crossref","is-referenced-by-count":5,"title":["Minimal Data Copy for Dense Linear Algebra Factorization"],"prefix":"10.1007","author":[{"given":"Fred G.","family":"Gustavson","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John A.","family":"Gunnels","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"James C.","family":"Sexton","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"issue":"5","key":"66_CR1","doi-asserted-by":"crossref","first-page":"563","DOI":"10.1147\/rd.385.0563","volume":"38","author":"R.C. Agarwal","year":"1994","unstructured":"Agarwal, R.C., Gustavson, F.G., Zubair, M.: Exploiting functional parallelism of POWER2 to design high-performance numerical algorithms. IBM Journal of Research and Development\u00a038(5), 563\u2013576 (1994)","journal-title":"IBM Journal of Research and Development"},{"issue":"2","key":"66_CR2","doi-asserted-by":"publisher","first-page":"214","DOI":"10.1145\/383738.383741","volume":"27","author":"B.S. Andersen","year":"2001","unstructured":"Andersen, B.S., Gustavson, F.G., Wasnieski, J.: A Recursive Formulation of Cholesky Factorization of a Matrix in Packed Storage. ACM TOMS\u00a027(2), 214\u2013244 (2001)","journal-title":"ACM TOMS"},{"issue":"2","key":"66_CR3","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1145\/1067967.1067969","volume":"31","author":"B.S. Andersen","year":"2005","unstructured":"Andersen, B.S., Gunnels, J.A., Gustavson, F.G., Reid, J.K., Wasnieski, J.: A Fully Portable High Performance Minimal Storage Hybrid Cholesky Algorithm. ACM TOMS\u00a031(2), 201\u2013227 (2005)","journal-title":"ACM TOMS"},{"doi-asserted-by":"crossref","unstructured":"Anderson, E., Bai, Z., Bischof, C., Demmel, J., Dongarra, J., Du Croz, J., Greenbaum, A., Hammarling, S., McKenney, A., Ostrouchov, S., Sorensen, D.: LAPACK Users\u2019 Guide Release 3.0, SIAM, Philadelphia (1999), http:\/\/www.netlib.org\/lapack\/lug\/lapack_lug.html","key":"66_CR4","DOI":"10.1137\/1.9780898719604"},{"doi-asserted-by":"crossref","unstructured":"Bilmes, J., Asanovic, K., Whye Chin, C., Demmel, J.: Optimizing Matrix Multiply Using PHiPAC: A Portable, High-Performance, ANSI C Coding Methodology. In: Proceedings of International Conference on Supercomputing, Vienna, Austria (1997)","key":"66_CR5","DOI":"10.1145\/263580.263662"},{"issue":"2-3","key":"66_CR6","doi-asserted-by":"publisher","first-page":"377","DOI":"10.1147\/rd.492.0377","volume":"49","author":"S. Chatterjee","year":"2005","unstructured":"Chatterjee, S., et al.: Design and Exploitation of a High-performance SIMD Floating-point Unit for Blue Gene\/L. IBM Journal of Research and Development\u00a049(2-3), 377\u2013391 (2005)","journal-title":"IBM Journal of Research and Development"},{"doi-asserted-by":"crossref","unstructured":"Dongarra, J.J., Moler, C.B., Bunch, J.R., Stewart, G.W.: LINPACK Users\u2019 Guide Release 2.0. SIAM, Philadelphia (1979)","key":"66_CR7","DOI":"10.1137\/1.9781611971811"},{"issue":"1","key":"66_CR8","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1137\/1026003","volume":"26","author":"J.J. Dongarra","year":"1984","unstructured":"Dongarra, J.J., Gustavson, F.G., Karp, A.: Implementing Linear Algebra Algorithms for Dense Matrices on a Vector Pipeline Machine. SIAM Review\u00a026(1), 91\u2013112 (1984)","journal-title":"SIAM Review"},{"issue":"1","key":"66_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/42288.42291","volume":"14","author":"J.J. Dongarra","year":"1988","unstructured":"Dongarra, J.J., Du Croz, J., Hammarling, S., Hanson, R.J.: An Extended Set of FORTRAN Basic Linear Algebra Subprograms. TOMS\u00a014(1), 1\u201317 (1988)","journal-title":"TOMS"},{"issue":"1","key":"66_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/77626.79170","volume":"16","author":"J.J. Dongarra","year":"1990","unstructured":"Dongarra, J.J., Du Croz, J., Hammarling, S., Duff, I.: A Set of Level 3 Basic Linear Algebra Subprograms. TOMS\u00a016(1), 1\u201317 (1990)","journal-title":"TOMS"},{"issue":"1","key":"66_CR11","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1137\/S0036144503428693","volume":"46","author":"E. Elmroth","year":"2004","unstructured":"Elmroth, E., Gustavson, F.G., Kagstrom, B., Jonsson, I.: Recursive Blocked Algorithms and Hybrid Data Structures for Dense Matrix Library Software. SIAM Review\u00a046(1), 3\u201345 (2004)","journal-title":"SIAM Review"},{"issue":"4","key":"66_CR12","doi-asserted-by":"publisher","first-page":"422","DOI":"10.1145\/504210.504213","volume":"27","author":"J. Gunnels","year":"2001","unstructured":"Gunnels, J., Gustavson, F.G., Henry, G., van de Geijn, R.: Formal linear algebra methods environment (FLAME). ACM TOMS\u00a027(4), 422\u2013455 (2001)","journal-title":"ACM TOMS"},{"key":"66_CR13","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"247","DOI":"10.1007\/11558958_29","volume-title":"Applied Parallel Computing","author":"J.A. Gunnels","year":"2006","unstructured":"Gunnels, J.A., Gustavson, F.G.: A New Array Format for Symmetric and Triangular Matrices. In: Dongarra, J.J., Madsen, K., Wa\u015bniewski, J. (eds.) PARA 2004. LNCS, vol.\u00a03732, pp. 247\u2013255. Springer, Heidelberg (2006)"},{"key":"66_CR14","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"256","DOI":"10.1007\/11558958_30","volume-title":"Applied Parallel Computing","author":"J.A. Gunnels","year":"2006","unstructured":"Gunnels, J.A., Gustavson, F.G., Henry, G.M., van de Geijn, R.A.: A Family of High-Performance Matrix Multiplication Algorithms. In: Dongarra, J.J., Madsen, K., Wa\u015bniewski, J. (eds.) PARA 2004. LNCS, vol.\u00a03732, pp. 256\u2013265. Springer, Heidelberg (2006)"},{"issue":"6","key":"66_CR15","doi-asserted-by":"crossref","first-page":"737","DOI":"10.1147\/rd.416.0737","volume":"41","author":"F.G. Gustavson","year":"1997","unstructured":"Gustavson, F.G.: Recursion Leads to Automatic Variable Blocking for Dense Linear-Algebra Algorithms. IBM Journal of Research and Development\u00a041(6), 737\u2013755 (1997)","journal-title":"IBM Journal of Research and Development"},{"issue":"6","key":"66_CR16","doi-asserted-by":"crossref","first-page":"823","DOI":"10.1147\/rd.446.0823","volume":"44","author":"F.G. Gustavson","year":"2000","unstructured":"Gustavson, F.G., Jonsson, I.: Minimal Storage High Performance Cholesky via Blocking and Recursion. IBM Journal of Research and Development\u00a044(6), 823\u2013849 (2000)","journal-title":"IBM Journal of Research and Development"},{"issue":"1","key":"66_CR17","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1147\/rd.471.0031","volume":"47","author":"F.G. Gustavson","year":"2003","unstructured":"Gustavson, F.G.: High Performance Linear Algebra Algorithms using New Generalized Data Structures for Matrices. IBM Journal of Research and Development\u00a047(1), 31\u201355 (2003)","journal-title":"IBM Journal of Research and Development"},{"key":"66_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1007\/11558958_2","volume-title":"Applied Parallel Computing","author":"F.G. Gustavson","year":"2006","unstructured":"Gustavson, F.G.: New Generalized Data Structures for Matrices Lead to a Variety of High performance Dense Linear Algorithms. In: Dongarra, J.J., Madsen, K., Wa\u015bniewski, J. (eds.) PARA 2004. LNCS, vol.\u00a03732, pp. 11\u201320. Springer, Heidelberg (2006)"},{"doi-asserted-by":"crossref","unstructured":"Gustavson, F.G., Wasniewski, J.: Rectangular Full Packed Format for LAPACK Algorithms Timings on Several Computers. In: Kagstom, B., Elmroth, E. (eds.) Para 2006. LNCS, vol.\u00a0xxxx, pp. 570\u2013579. Springer, Heidelberg (2006)","key":"66_CR19","DOI":"10.1007\/978-3-540-75755-9_69"},{"unstructured":"IBM: IBM Engineering and Scientific Subroutine Library for AIX Version 3, Release 3. IBM Pub. No. SA22-7272-04 (December 2001)","key":"66_CR20"},{"unstructured":"Kalla, R., Sinharoy, B., Tendler, J.: Power 5. HotChips-15, August 17-19, 2003, Stanford, CA (2003)","key":"66_CR21"},{"issue":"3","key":"66_CR22","doi-asserted-by":"publisher","first-page":"308","DOI":"10.1145\/355841.355847","volume":"5","author":"C.L. Lawson","year":"1979","unstructured":"Lawson, C.L., Hanson, R.J., Kincaid, D.R., Krogh, F.T.: Basic Linear Algebra Subprograms for Fortran Usage. TOMS\u00a05(3), 308\u2013323 (1979)","journal-title":"TOMS"},{"issue":"7","key":"66_CR23","doi-asserted-by":"publisher","first-page":"640","DOI":"10.1109\/TPDS.2003.1214317","volume":"14","author":"N. Park","year":"2003","unstructured":"Park, N., Hong, B., Prasanna, V.K.: Tiling, Block Data Layout, and Memory Hierarchy Performance. IEEE Trans. Parallel and Distributed Systems\u00a014(7), 640\u2013654 (2003)","journal-title":"IEEE Trans. Parallel and Distributed Systems"},{"issue":"4\/5","key":"66_CR24","doi-asserted-by":"crossref","first-page":"505","DOI":"10.1147\/rd.494.0505","volume":"49","author":"B. Sinharoy","year":"2005","unstructured":"Sinharoy, B., Kalla, R.N., Tendler, J.M, Kovacs, R.G., Eickemeyer, R.J., Joyner, J.B.: POWER5 System Microarchitecture. IBM Journal of Research and Development\u00a049(4\/5), 505\u2013521 (2005)","journal-title":"IBM Journal of Research and Development"},{"doi-asserted-by":"crossref","unstructured":"Whaley, R.C., Petitet, A., Dongarra, J.J.: Automated Empirical Optimization of Software and the ATLAS Project. Parallel Computing (1-2), 3\u201335 (2001)","key":"66_CR25","DOI":"10.1016\/S0167-8191(00)00087-9"}],"container-title":["Lecture Notes in Computer Science","Applied Parallel Computing. State of the Art in Scientific Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-540-75755-9_66.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,4,27]],"date-time":"2021-04-27T10:31:22Z","timestamp":1619519482000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-540-75755-9_66"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[null]]},"ISBN":["9783540757542"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-540-75755-9_66","relation":{},"subject":[]}}