{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T11:32:27Z","timestamp":1774870347506,"version":"3.50.1"},"publisher-location":"Berlin, Heidelberg","reference-count":40,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783540757542","type":"print"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.1007\/978-3-540-75755-9_70","type":"book-chapter","created":{"date-parts":[[2007,9,22]],"date-time":"2007-09-22T02:44:54Z","timestamp":1190429094000},"page":"580-588","source":"Crossref","is-referenced-by-count":2,"title":["Using Non-canonical Array Layouts in Dense Matrix Operations"],"prefix":"10.1007","author":[{"given":"Jos\u00e9 R.","family":"Herrero","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Juan J.","family":"Navarro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"70_CR1","doi-asserted-by":"crossref","first-page":"563","DOI":"10.1147\/rd.385.0563","volume":"38","author":"R.C. Agarwal","year":"1994","unstructured":"Agarwal, R.C., Gustavson, F.G., Zubair, M.: Exploiting functional parallelism of POWER2 to design high-performance numerical algorithms. IBM J. Res. Dev.\u00a038, 563\u2013576 (1994)","journal-title":"IBM J. Res. Dev."},{"key":"70_CR2","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1137\/S0036144503428693","volume":"46","author":"E. Elmroth","year":"2004","unstructured":"Elmroth, E., Gustavson, F., Jonsson, I., K\u00e5gstr\u00f6m, B.: Recursive blocked algorithms and hybrid data structures for dense matrix library software. SIAM Review\u00a046, 3\u201345 (2004)","journal-title":"SIAM Review"},{"key":"70_CR3","unstructured":"IBM: ESSL Guide and Reference for IBM ES\/3090 Vector Multiprocessors (1986) Order No. SA 22-9220 (Febuary 1986)"},{"key":"70_CR4","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1177\/109434208800200103","volume":"2","author":"K. Gallivan","year":"1988","unstructured":"Gallivan, K., Jalby, W., Meier, U., Sameh, A.: Impact of hierarchical memory systems on linear algebra algorithm design. Int. J. of Supercomputer Appl.\u00a02, 12\u201348 (1988)","journal-title":"Int. J. of Supercomputer Appl."},{"key":"70_CR5","doi-asserted-by":"publisher","first-page":"319","DOI":"10.1145\/73560.73588","volume-title":"POPL 1988: Proceedings of the 15th ACM SIGPLAN-SIGACT symposium on Principles of programming languages","author":"F. Irigoin","year":"1988","unstructured":"Irigoin, F., Triolet, R.: Supernode partitioning. In: POPL 1988: Proceedings of the 15th ACM SIGPLAN-SIGACT symposium on Principles of programming languages, pp. 319\u2013329. ACM Press, New York (1988)"},{"key":"70_CR6","doi-asserted-by":"publisher","first-page":"655","DOI":"10.1145\/76263.76337","volume-title":"Supercomputing 1989","author":"M. Wolfe","year":"1989","unstructured":"Wolfe, M.: More iteration space tiling. In: ACM (ed.) Supercomputing 1989, Reno, Nevada, November 13-17, 1989, pp. 655\u2013664. ACM Press, New York (1989)"},{"key":"70_CR7","doi-asserted-by":"crossref","unstructured":"Lam, M., Rothberg, E., Wolf, M.: The cache performance and optimizations of blocked algorithms. In: Proceedings of ASPLOS 1991, pp. 67\u201374 (1991)","DOI":"10.1145\/106972.106981"},{"key":"70_CR8","doi-asserted-by":"crossref","unstructured":"Temam, O., Granston, E.D., Jalby, W.: To copy or not to copy: a compile-time technique for assessing when data copying should be used to eliminate cache conflicts. In: Supercomputing, pp. 410\u2013419 (1993)","DOI":"10.1145\/169627.169762"},{"key":"70_CR9","doi-asserted-by":"publisher","first-page":"153","DOI":"10.1145\/362875.362879","volume":"12","author":"A.C. McKellar","year":"1969","unstructured":"McKellar, A.C., Coffman, J.E.G.: Organizing matrices and matrix operations for paged memory systems. Communications of the ACM\u00a012, 153\u2013165 (1969)","journal-title":"Communications of the ACM"},{"key":"70_CR10","doi-asserted-by":"publisher","first-page":"737","DOI":"10.1147\/rd.416.0737","volume":"41","author":"F.G. Gustavson","year":"1997","unstructured":"Gustavson, F.G.: Recursion leads to automatic variable blocking for dense linear-algebra algorithms. IBM J. Res. Dev.\u00a041, 737\u2013756 (1997)","journal-title":"IBM J. Res. Dev."},{"key":"70_CR11","doi-asserted-by":"publisher","first-page":"1065","DOI":"10.1137\/S0895479896297744","volume":"18","author":"S. Toledo","year":"1997","unstructured":"Toledo, S.: Locality of reference in LU decomposition with partial pivoting. SIAM J. Matrix Anal. Appl.\u00a018, 1065\u20131081 (1997)","journal-title":"SIAM J. Matrix Anal. Appl."},{"key":"70_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"368","DOI":"10.1007\/3-540-44520-X_48","volume-title":"Euro-Par 2000 Parallel Processing","author":"N. Ahmed","year":"2000","unstructured":"Ahmed, N., Pingali, K.: Automatic generation of block-recursive codes. In: Bode, A., Ludwig, T., Karl, W.C., Wism\u00fcller, R. (eds.) Euro-Par 2000. LNCS, vol.\u00a01900, pp. 368\u2013378. Springer, Heidelberg (2000)"},{"key":"70_CR13","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1007\/BFb0095337","volume-title":"Applied Parallel Computing. Large Scale Scientific and Industrial Problems","author":"F. Gustavson","year":"1998","unstructured":"Gustavson, F., Henriksson, A., Jonsson, I., K\u00e5gstr\u00f6m, B.: Recursive blocked data formats and BLAS\u2019s for dense linear algebra algorithms. In: Kagstr\u00f6m, B., Elmroth, E., Wa\u015bniewski, J., Dongarra, J.J. (eds.) PARA 1998. LNCS, vol.\u00a01541, pp. 195\u2013206. Springer, Heidelberg (1998)"},{"key":"70_CR14","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1007\/3-540-70734-4_7","volume-title":"Applied Parallel Computing. New Paradigms for HPC in Industry and Academia","author":"B.S. Andersen","year":"2001","unstructured":"Andersen, B.S., Gustavson, F.G., Karaivanov, A., Marinova, M., Wasniewski, J., Yalamov, P.Y.: LAWRA: Linear algebra with recursive algorithms. In: S\u00f8revik, T., Manne, F., Moe, R., Gebremedhin, A.H. (eds.) PARA 2000. LNCS, vol.\u00a01947, pp. 38\u201351. Springer, Heidelberg (2001)"},{"key":"70_CR15","doi-asserted-by":"publisher","first-page":"214","DOI":"10.1145\/383738.383741","volume":"27","author":"B.S. Andersen","year":"2001","unstructured":"Andersen, B.S., Wasniewski, J., Gustavson, F.G.: A recursive formulation of Cholesky factorization of a matrix in packed storage. ACM Transactions on Mathematical Software (TOMS)\u00a027, 214\u2013244 (2001)","journal-title":"ACM Transactions on Mathematical Software (TOMS)"},{"key":"70_CR16","series-title":"Lecture Notes in Computational Science and Engineering","first-page":"46","volume-title":"Simulation and visualization on the grid: Parallelldatorcentrum, Kungl. Tekniska H\u00f6gskolan, proceedings 7th annual conference","author":"F.G. Gustavson","year":"1974","unstructured":"Gustavson, F.G.: New generalized data structures for matrices lead to a variety of high-performance algorithms. In: Engquist, B. (ed.) Simulation and visualization on the grid: Parallelldatorcentrum, Kungl. Tekniska H\u00f6gskolan, proceedings 7th annual conference. Lecture Notes in Computational Science and Engineering, vol.\u00a013, pp. 46\u201361. Springer, Heidelberg (1974)"},{"key":"70_CR17","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"crossref","first-page":"287","DOI":"10.1007\/3-540-48051-X_29","volume-title":"Applied Parallel Computing. Advanced Scientific Computing","author":"B.S. Andersen","year":"2002","unstructured":"Andersen, B.S., Gunnels, J.A., Gustavson, F., Wasniewski, J.: A recursive formulation of the inversion of symmetric positive defite matrices in packed storage data format. In: Fagerholm, J., Haataja, J., J\u00e4rvinen, J., Lyly, M., R\u00e5back, P., Savolainen, V. (eds.) PARA 2002. LNCS, vol.\u00a02367, pp. 287\u2013296. Springer, Heidelberg (2002)"},{"key":"70_CR18","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1145\/1067967.1067969","volume":"31","author":"B.S. Andersen","year":"2005","unstructured":"Andersen, B.S., Gunnels, J.A., Gustavson, F.G., Reid, J.K., Wa\u015bniewski, J.: A fully portable high performance minimal storage hybrid format Cholesky algorithm. ACM Transactions on Mathematical Software\u00a031, 201\u2013227 (2005)","journal-title":"ACM Transactions on Mathematical Software"},{"key":"70_CR19","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1147\/rd.471.0031","volume":"47","author":"F.G. Gustavson","year":"2003","unstructured":"Gustavson, F.G.: High-performance linear algebra algorithms using new generalized data structures for matrices. IBM J. Res. Dev.\u00a047, 31\u201355 (2003)","journal-title":"IBM J. Res. Dev."},{"key":"70_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1007\/11558958_2","volume-title":"Applied Parallel Computing","author":"F.G. Gustavson","year":"2006","unstructured":"Gustavson, F.G.: New generalized data structures for matrices lead to a variety of high performance dense linear algebra algorithms. In: Dongarra, J.J., Madsen, K., Wa\u015bniewski, J. (eds.) PARA 2004. LNCS, vol.\u00a03732, pp. 11\u201320. Springer, Heidelberg (2006)"},{"key":"70_CR21","unstructured":"Gustavson, F.G.: Algorithm Compiler Architecture Interaction Relative to Dense Linear Algebra. Technical Report RC23715 (W0509-039), IBM, T.J. Watson (2005)"},{"key":"70_CR22","doi-asserted-by":"publisher","first-page":"444","DOI":"10.1145\/305138.305231","volume-title":"Proceedings of the 13th international conference on Supercomputing","author":"S. Chatterjee","year":"1999","unstructured":"Chatterjee, S., Jain, V.V., Lebeck, A.R., Mundhra, S., Thottethodi, M.: Nonlinear array layouts for hierarchical memory systems. In: Proceedings of the 13th international conference on Supercomputing, pp. 444\u2013453. ACM Press, New York (1999)"},{"key":"70_CR23","doi-asserted-by":"publisher","first-page":"640","DOI":"10.1109\/TPDS.2003.1214317","volume":"14","author":"N. Park","year":"2003","unstructured":"Park, N., Hong, B., Prasanna, V.K.: Tiling, block data layout, and memory hierarchy performance. IEEE Trans.\u00a0Parallel and Distrib.\u00a0Systems\u00a014, 640\u2013654 (2003)","journal-title":"IEEE Trans.\u00a0Parallel and Distrib.\u00a0Systems"},{"key":"70_CR24","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"762","DOI":"10.1007\/11751649_84","volume-title":"Computational Science and Its Applications - ICCSA 2006","author":"J.R. Herrero","year":"2006","unstructured":"Herrero, J.R., Navarro, J.J.: Compiler-optimized kernels: An efficient alternative to hand-coded inner kernels. In: Gavrilova, M., Gervasi, O., Kumar, V., Tan, C.J.K., Taniar, D., Lagan\u00e0, A., Mun, Y., Choo, H. (eds.) ICCSA 2006. LNCS, vol.\u00a03984, pp. 762\u2013771. Springer, Heidelberg (2006)"},{"key":"70_CR25","doi-asserted-by":"crossref","unstructured":"Frens, J.D., Wise, D.S.: Auto-blocking matrix-multiplication or tracking BLAS3 performance from source code. In: Proc. 6th ACM SIGPLAN Symp. on Principles and Practice of Parallel Program, SIGPLAN Notices, pp. 206\u2013216 (1997)","DOI":"10.1145\/263767.263789"},{"key":"70_CR26","unstructured":"Wise, D.S., Frens, J.D.: Morton-order matrices deserve compilers\u2019 support. Technical Report TR 533, Computer Science Department, Indiana University (1999)"},{"key":"70_CR27","first-page":"222","volume-title":"Proc. of the 11th annual ACM symposium on Parallel algorithms and architectures","author":"S. Chatterjee","year":"1999","unstructured":"Chatterjee, S., Lebeck, A.R., Patnala, P.K., Thottethodi, M.: Recursive array layouts and fast parallel matrix multiplication. In: Proc. of the 11th annual ACM symposium on Parallel algorithms and architectures, pp. 222\u2013231. ACM Press, New York (1999)"},{"key":"70_CR28","doi-asserted-by":"crossref","unstructured":"Athanasaki, E., Koziris, N.: Fast indexing for blocked array layouts to improve multi-level cache locality. In: Interaction between Compilers and Computer Architectures, pp. 109\u2013119 (2004)","DOI":"10.1109\/INTERA.2004.1299515"},{"key":"70_CR29","first-page":"521","volume-title":"PARA 2006","author":"M. Bader","year":"2006","unstructured":"Bader, M., Mayer, C.: Cache oblivious matrix operations using Peano curves (These proceedings). In: PARA 2006, pp. 521\u2013530. Springer, Heidelberg (2006)"},{"key":"70_CR30","doi-asserted-by":"publisher","first-page":"805","DOI":"10.1002\/cpe.630","volume":"14","author":"V. Valsalam","year":"2002","unstructured":"Valsalam, V., Skjellum, A.: A framework for high-performance matrix multiplication based on hierarchical abstractions, algorithms and optimized low-level kernels. Concurrency and Computation: Practice and Experience\u00a014, 805\u2013839 (2002)","journal-title":"Concurrency and Computation: Practice and Experience"},{"key":"70_CR31","doi-asserted-by":"crossref","unstructured":"Athanasaki, E., Koziris, N., Tsanakas, P.: A tile size selection analysis for blocked array layouts. In: Interaction between Compilers and Computer Architectures, pp. 70\u201380 (2005)","DOI":"10.1109\/INTERACT.2005.1"},{"key":"70_CR32","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1016\/0045-7825(72)90005-9","volume":"1","author":"G. Fuchs","year":"1972","unstructured":"Fuchs, G., Roy, J., Schrem, E.: Hypermatrix solution of large sets of symmetric positive-definite linear equations. Comp. Meth. Appl. Mech. Eng.\u00a01, 197\u2013216 (1972)","journal-title":"Comp. Meth. Appl. Mech. Eng."},{"key":"70_CR33","unstructured":"Herrero, J.R., Navarro, J.J.: Automatic benchmarking and optimization of codes: an experience with numerical kernels. In: Int. Conf. on Software Engineering Research and Practice, pp. 701\u2013706. CSREA Press (2003)"},{"key":"70_CR34","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1016\/0020-0190(85)90049-3","volume":"20","author":"D.S. Wise","year":"1985","unstructured":"Wise, D.S.: Representing matrices as quadtrees for parallel processors. Information Processing Letters\u00a020, 195\u2013199 (1985)","journal-title":"Information Processing Letters"},{"key":"70_CR35","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"1058","DOI":"10.1007\/11752578_128","volume-title":"Parallel Processing and Applied Mathematics","author":"J.R. Herrero","year":"2006","unstructured":"Herrero, J.R., Navarro, J.J.: Adapting linear algebra codes to the memory hierarchy using a hypermatrix scheme. In: Wyrzykowski, R., Dongarra, J.J., Meyer, N., Wa\u015bniewski, J. (eds.) PPAM 2005. LNCS, vol.\u00a03911, pp. 1058\u20131065. Springer, Heidelberg (2006)"},{"key":"70_CR36","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1145\/181181.181561","volume-title":"Proceedings of the 8th International Conference on Supercomputing","author":"J.J. Navarro","year":"1994","unstructured":"Navarro, J.J., Juan, A., Lang, T.: MOB forms: A class of Multilevel Block Algorithms for dense linear algebra operations. In: Proceedings of the 8th International Conference on Supercomputing, pp. 354\u2013363. ACM Press, New York (1994)"},{"key":"70_CR37","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"124","DOI":"10.1007\/11752578_16","volume-title":"Parallel Processing and Applied Mathematics","author":"J.R. Herrero","year":"2006","unstructured":"Herrero, J.R., Navarro, J.J.: A study on load imbalance in parallel hypermatrix multiplication using OpenMP. In: Wyrzykowski, R., Dongarra, J.J., Meyer, N., Wa\u015bniewski, J. (eds.) PPAM 2005. LNCS, vol.\u00a03911, pp. 124\u2013131. Springer, Heidelberg (2006)"},{"key":"70_CR38","first-page":"211","volume-title":"Supercomputing 1998","author":"R.C. Whaley","year":"1998","unstructured":"Whaley, R.C., Dongarra, J.J.: Automatically tuned linear algebra software. In: Supercomputing 1998, pp. 211\u2013217. IEEE Computer Society Press, Los Alamitos (1998)"},{"key":"70_CR39","unstructured":"Goto, K., van de Geijn, R.: On reducing TLB misses in matrix multiplication. Technical Report CS-TR-02-55, Univ. of Texas at Austin (2002)"},{"key":"70_CR40","first-page":"919","volume-title":"PARA 2006","author":"J. Gunnels","year":"2006","unstructured":"Gunnels, J., Gustavson, F., Pingali, K., Yotov, K.: Is cache-oblivious DGEMM viable (These proceedings). In: PARA 2006, pp. 919\u2013928. Springer, Heidelberg (2006)"}],"container-title":["Lecture Notes in Computer Science","Applied Parallel Computing. State of the Art in Scientific Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-540-75755-9_70.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,4,27]],"date-time":"2021-04-27T10:31:24Z","timestamp":1619519484000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-540-75755-9_70"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[null]]},"ISBN":["9783540757542"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-3-540-75755-9_70","relation":{},"subject":[]}}