{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,1,22]],"date-time":"2025-01-22T05:04:14Z","timestamp":1737522254794,"version":"3.33.0"},"publisher-location":"Berlin, Heidelberg","reference-count":17,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540757542"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.1007\/978-3-540-75755-9_109","type":"book-chapter","created":{"date-parts":[[2007,9,22]],"date-time":"2007-09-22T02:44:54Z","timestamp":1190429094000},"page":"919-928","source":"Crossref","is-referenced-by-count":18,"title":["Is Cache-Oblivious DGEMM Viable?"],"prefix":"10.1007","author":[{"given":"John A.","family":"Gunnels","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fred G.","family":"Gustavson","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keshav","family":"Pingali","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kamen","family":"Yotov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"issue":"5","key":"109_CR1","doi-asserted-by":"crossref","first-page":"563","DOI":"10.1147\/rd.385.0563","volume":"38","author":"R.C. Agarwal","year":"1994","unstructured":"Agarwal, R.C., Gustavson, F.G., Zubair, M.: Exploiting functional parallelism of POWER2 to design high-performance numerical algorithms. IBM Journal of Research and Development\u00a038(5), 563\u2013576 (1994)","journal-title":"IBM Journal of Research and Development"},{"issue":"2","key":"109_CR2","doi-asserted-by":"crossref","first-page":"78","DOI":"10.1147\/sj.52.0078","volume":"5","author":"L.A. Belady","year":"1966","unstructured":"Belady, L.A.: A study of replacement algorithms for a virtual-storage computer. IBM Systems Journal\u00a05(2), 78\u2013101 (1966)","journal-title":"IBM Systems Journal"},{"issue":"2-3","key":"109_CR3","doi-asserted-by":"publisher","first-page":"377","DOI":"10.1147\/rd.492.0377","volume":"49","author":"S. Chatterjee","year":"2005","unstructured":"Chatterjee, S., et al.: Design and Exploitation of a High-performance SIMD Floating-point Unit for Blue Gene\/L. IBM Journal of Research and Development\u00a049(2-3), 377\u2013391 (2005)","journal-title":"IBM Journal of Research and Development"},{"key":"109_CR4","first-page":"285","volume-title":"FOCS 1999: Proceedings of the 40th Annual Symposium on Foundations of Computer Science","author":"M. Frigo","year":"1999","unstructured":"Frigo, M., Leiserson, C., Prokop, H., Ramachandran, S.: Cache-oblivious Algorithms. In: FOCS 1999: Proceedings of the 40th Annual Symposium on Foundations of Computer Science, p. 285. IEEE Computer Society Press, Los Alamitos (1999)"},{"issue":"1","key":"109_CR5","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1137\/1026003","volume":"26","author":"J.J. Dongarra","year":"1984","unstructured":"Dongarra, J.J., Gustavson, F.G., Karp, A.: Implementing Linear Algebra Algorithms for Dense Matrices on a Vector Pipeline Machine. SIAM Review\u00a026(1), 91\u2013112 (1984)","journal-title":"SIAM Review"},{"issue":"1","key":"109_CR6","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1137\/S0036144503428693","volume":"46","author":"E. Elmroth","year":"2004","unstructured":"Elmroth, E., Gustavson, F.G., K\u00e5gstr\u00f6m, B., Jonsson, I.: Recursive Blocked Algorithms and Hybrid Data Structures for Dense Matrix Library Software. SIAM Review\u00a046(1), 3\u201345 (2004)","journal-title":"SIAM Review"},{"key":"109_CR7","doi-asserted-by":"crossref","unstructured":"Hong, J.-W., Kung, H.T.: I\/O complexity: The red-blue pebble game. In: Proc. of the thirteenth annual ACM symposium on Theory of computing, pp. 326\u2013333 (1981)","DOI":"10.1145\/800076.802486"},{"key":"109_CR8","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"256","DOI":"10.1007\/11558958_30","volume-title":"Applied Parallel Computing","author":"J.A. Gunnels","year":"2006","unstructured":"Gunnels, J.A., Gustavson, F.G., Henry, G.M., van de Geijn, R.A.: A Family of High-Performance Matrix Multiplication Algorithms. In: Dongarra, J.J., Madsen, K., Wa\u015bniewski, J. (eds.) PARA 2004. LNCS, vol.\u00a03732, pp. 256\u2013265. Springer, Heidelberg (2006)"},{"issue":"6","key":"109_CR9","doi-asserted-by":"crossref","first-page":"737","DOI":"10.1147\/rd.416.0737","volume":"41","author":"F.G. Gustavson","year":"1997","unstructured":"Gustavson, F.G.: Recursion Leads to Automatic Variable Blocking for Dense Linear-Algebra Algorithms. IBM Journal of Research and Development\u00a041(6), 737\u2013755 (1997)","journal-title":"IBM Journal of Research and Development"},{"issue":"1","key":"109_CR10","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1147\/rd.471.0031","volume":"47","author":"F.G. Gustavson","year":"2003","unstructured":"Gustavson, F.G.: High Performance Linear Algebra Algorithms using New Generalized Data Structures for Matrices. IBM Journal of Research and Development\u00a047(1), 31\u201355 (2003)","journal-title":"IBM Journal of Research and Development"},{"key":"109_CR11","series-title":"Lecture Notes in Computer Science","first-page":"540","volume-title":"Computational Science - Para 2006","author":"F.G. Gustavson","year":"2006","unstructured":"Gustavson, F.G., Gunnels, J.A., Sexton, J.C.: Minimal Data Copy for Dense Linear Algebra Factorization. In: K\u00e5gstr\u00f6m, B., Elmroth, E. (eds.) Computational Science - Para 2006. LNCS, vol.\u00a0xxxx, pp. 540\u2013549. Springer, Heidelberg (2006)"},{"key":"109_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1007\/BFb0095337","volume-title":"Applied Parallel Computing. Large Scale Scientific and Industrial Problems","author":"F.G. Gustavson","year":"1998","unstructured":"Gustavson, F.G., Henriksson, A., Jonsson, I., K\u00e5gstr\u00f6m, B., Ling, P.: Recursive blocked data formats and BLAS\u2019s for dense linear algebra algorithms. In: Kagstr\u00f6m, B., Elmroth, E., Wa\u015bniewski, J., Dongarra, J.J. (eds.) PARA 1998. LNCS, vol.\u00a01541, pp. 195\u2013206. Springer, Heidelberg (1998)"},{"key":"109_CR13","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1007\/BFb0095338","volume-title":"Applied Parallel Computing. Large Scale Scientific and Industrial Problems","author":"F.G. Gustavson","year":"1998","unstructured":"Gustavson, F.G., Henriksson, A., Jonsson, I., K\u00e5gstr\u00f6m, B., Ling, P.: Superscalar GEMM-based level 3 BLAS\u2014the on-going evolution of a portable and high-performance library. In: Kagstr\u00f6m, B., Elmroth, E., Wa\u015bniewski, J., Dongarra, J.J. (eds.) PARA 1998. LNCS, vol.\u00a01541, pp. 207\u2013215. Springer, Heidelberg (1998)"},{"issue":"7","key":"109_CR14","doi-asserted-by":"publisher","first-page":"640","DOI":"10.1109\/TPDS.2003.1214317","volume":"14","author":"N. Park","year":"2003","unstructured":"Park, N., Hong, B., Prasanna, V.K.: Tiling, Block Data Layout, and Memory Hierarchy Performance. IEEE Trans. Parallel and Distributed Systems\u00a014(7), 640\u2013654 (2003)","journal-title":"IEEE Trans. Parallel and Distributed Systems"},{"key":"109_CR15","unstructured":"Roeder, T., Yotov, K., Pingali, K., Gunnels, J., Gustavson, F.: The Price of Cache Obliviousness. Department of Computer Science, University of Texas, Austin Technical Report CS-TR-06-43 (September 2006)"},{"issue":"4\/5","key":"109_CR16","doi-asserted-by":"crossref","first-page":"505","DOI":"10.1147\/rd.494.0505","volume":"49","author":"B. Sinharoy","year":"2005","unstructured":"Sinharoy, B., Kalla, R.N., Tendler, J.M, Kovacs, R.G., Eickemeyer, R.J., Joyner, J.B.: POWER5 System Microarchitecture. IBM Journal of Research and Development\u00a049(4\/5), 505\u2013521 (2005)","journal-title":"IBM Journal of Research and Development"},{"key":"109_CR17","doi-asserted-by":"crossref","unstructured":"Whaley, R.C., Petitet, A., Dongarra, J.J.: Automated Empirical Optimization of Software and the ATLAS Project. Parallel Computing\u00a0(1-2), 3\u201335 (2001)","DOI":"10.1016\/S0167-8191(00)00087-9"}],"container-title":["Lecture Notes in Computer Science","Applied Parallel Computing. State of the Art in Scientific Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-540-75755-9_109.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T04:21:47Z","timestamp":1737433307000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-540-75755-9_109"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[null]]},"ISBN":["9783540757542"],"references-count":17,"URL":"https:\/\/doi.org\/10.1007\/978-3-540-75755-9_109","relation":{},"subject":[]}}