{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,5]],"date-time":"2024-09-05T05:30:11Z","timestamp":1725514211765},"publisher-location":"Berlin, Heidelberg","reference-count":26,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540681052"},{"type":"electronic","value":"9783540681113"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.1007\/978-3-540-68111-3_67","type":"book-chapter","created":{"date-parts":[[2008,5,28]],"date-time":"2008-05-28T16:27:12Z","timestamp":1211992032000},"page":"639-648","source":"Crossref","is-referenced-by-count":6,"title":["Parallel Tiled QR Factorization for Multicore Architectures"],"prefix":"10.1007","author":[{"given":"Alfredo","family":"Buttari","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Julien","family":"Langou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jakub","family":"Kurzak","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jack","family":"Dongarra","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"67_CR1","doi-asserted-by":"crossref","unstructured":"Pham, D., Asano, S., Bolliger, M., Day, M.N., Hofstee, H.P., Johns, C., Kahle, J., Kameyama, A., Keaty, J., Masubuchi, Y., Riley, M., Shippy, D., Stasiak, D., Suzuoki, M., Wang, M., Warnock, J., Weitzel, S., Wendel, D., Yamazaki, T., Yazawa, K.: The design and implementation of a first-generation CELL processor. In: IEEE International Solid-State Circuits Conference, pp. 184\u2013185 (2005)","DOI":"10.1109\/ICICDT.2005.1502588"},{"key":"67_CR2","unstructured":"Teraflops research chip, http:\/\/www.intel.com\/research\/platform\/terascale\/teraflops.htm"},{"key":"67_CR3","doi-asserted-by":"crossref","DOI":"10.1137\/1.9780898719604","volume-title":"LAPACK User\u2019s Guide","author":"E. Anderson","year":"1999","unstructured":"Anderson, E., Bai, Z., Bischof, C., Blackford, S., Demmel, J., Dongarra, J., Croz, J.D., Greenbaum, A., Hammarling, S., McKenney, A., Sorensen, D.: LAPACK User\u2019s Guide, 3rd edn. SIAM, Philadelphia (1999)","edition":"3"},{"key":"67_CR4","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/0010-4655(96)00017-3","volume":"97","author":"J. Choi","year":"1996","unstructured":"Choi, J., Demmel, J., Dhillon, I., Dongarra, J., Ostrouchov, S., Petitet, A., Stanley, K., Walker, D., Whaley, R.C.: ScaLAPACK: A portable linear algebra library for distributed memory computers - design issues and performance. Computer Physics Communications\u00a097, 1\u201315 (1996), (also as LAPACK Working Note #95)","journal-title":"Computer Physics Communications"},{"key":"67_CR5","unstructured":"Kurzak, J., Dongarra, J.: Implementing linear algebra routines on multi-core processors with pipelining and a look ahead. LAPACK Working Note 178 (September 2006), Also available as UT-CS-06-581"},{"key":"67_CR6","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-540-75755-9_1","volume-title":"Applied Parallel Computing. State of the Art in Scientific Computing","author":"A. Buttari","year":"2007","unstructured":"Buttari, A., Dongarra, J., Kurzak, J., Langou, J., Luszczek, P., Tomov, S.: The impact of multicore on math software. In: K\u00e5gstr\u00f6m, B., Elmroth, E., Dongarra, J., Wa\u015bniewski, J. (eds.) PARA 2006. LNCS, vol.\u00a04699, pp. 1\u201310. Springer, Heidelberg (2007)"},{"key":"67_CR7","doi-asserted-by":"publisher","first-page":"116","DOI":"10.1145\/1248377.1248397","volume-title":"SPAA 2007: Proceedings of the nineteenth annual ACM symposium on Parallel algorithms and architectures","author":"E. Chan","year":"2007","unstructured":"Chan, E., Quintana-Orti, E.S., Quintana-Orti, G., van de Geijn, R.: Supermatrix out-of-order scheduling of matrix operations for SMP and multi-core architectures. In: SPAA 2007: Proceedings of the nineteenth annual ACM symposium on Parallel algorithms and architectures, pp. 116\u2013125. ACM Press, New York (2007)"},{"issue":"1","key":"67_CR8","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1137\/S0036144503428693","volume":"46","author":"E. Elmroth","year":"2004","unstructured":"Elmroth, E., Gustavson, F., Jonsson, I., K\u00e5gstr\u00f6m, B.: Recursive blocked algorithms and hybrid data structures for dense matrix library software. SIAM Review\u00a046(1), 3\u201345 (2004)","journal-title":"SIAM Review"},{"key":"67_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"550","DOI":"10.1007\/978-3-540-75755-9_67","volume-title":"Applied Parallel Computing. State of the Art in Scientific Computing","author":"F. Gustavson","year":"2007","unstructured":"Gustavson, F., Karlsson, L., K\u00e5gstr\u00f6m, B.: Three algorithms for cholesky factorization on distributed memory using packed storage. In: K\u00e5gstr\u00f6m, B., Elmroth, E., Dongarra, J., Wa\u015bniewski, J. (eds.) PARA 2006. LNCS, vol.\u00a04699, pp. 550\u2013559. Springer, Heidelberg (2007)"},{"key":"67_CR10","unstructured":"Kurzak, J., Buttari, A., Dongarra, J.: Solving systems of linear equations on the CELL processor using Cholesky factorization. Technical Report UT-CS-07-596, Innovative Computing Laboratory, University of Tennessee Knoxville (April 2007)"},{"issue":"1","key":"67_CR11","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1145\/322358.322366","volume":"30","author":"R.E. Lord","year":"1983","unstructured":"Pham, D., Asano, S., Bolliger, M., Day, M.N., Hofstee, H.P., Johns, C., Kahle, J., Kameyama, A., Keaty, J., Masubuchi, Y., Riley, M., Shippy, D., Stasiak, D., Suzuoki, M., Wang, M., Warnock, J., Weitzel, S., Wendel, D., Yamazaki, T., Yazawa, K.: The design and implementation of a first-generation CELL processor. In: IEEE International Solid-State Circuits Conference, pp. 184\u2013185 (2005)","journal-title":"J. ACM"},{"key":"67_CR12","doi-asserted-by":"crossref","unstructured":"Dongarra, J.J., Hiromoto, R.E.: A collection of parallel linear equations routines for the Denelcor HEP 1(2), 133\u2013142 (December 1984)","DOI":"10.1016\/S0167-8191(84)90036-X"},{"key":"67_CR13","doi-asserted-by":"publisher","first-page":"225","DOI":"10.1145\/76263.76287","volume-title":"Supercomputing 1989: Proceedings of the 1989 ACM\/IEEE conference on Supercomputing","author":"R.C. Agarwal","year":"1989","unstructured":"Agarwal, R.C., Gustavson, F.G.: Vector and parallel algorithms for cholesky factorization on ibm 3090. In: Supercomputing 1989: Proceedings of the 1989 ACM\/IEEE conference on Supercomputing, pp. 225\u2013233. ACM Press, New York (1989)"},{"key":"67_CR14","unstructured":"Agarwal, R.C., Gustavson, F.G.: A parallel implementation of matrix multiplication and LU factorization on the IBM 3090. In: Proceedings of the IFIP WG 2.5 Working Group on Aspects of Computation on Asychronous Parallel Processors, Stanford CA, Augest 22-26,1988, North Holland, Amsterdam (1988)"},{"issue":"4","key":"67_CR15","doi-asserted-by":"publisher","first-page":"605","DOI":"10.1147\/rd.444.0605","volume":"44","author":"E. Elmroth","year":"2000","unstructured":"Elmroth, E., Gustavson, F.G.: Applying recursion to serial and parallel QR factorization leads to better performance. IBM Journal of Research and Development\u00a044(4), 605 (2000)","journal-title":"IBM Journal of Research and Development"},{"key":"67_CR16","volume-title":"Matrix Computations","author":"G. Golub","year":"1996","unstructured":"Golub, G., Van Loan, C.: Matrix Computations, 3rd edn. Johns Hopkins University Press, Baltimore (1996)","edition":"3"},{"key":"67_CR17","doi-asserted-by":"crossref","DOI":"10.1137\/1.9781611971408","volume-title":"Matrix Algorithms","author":"G.W. Stewart","year":"1998","unstructured":"Stewart, G.W.: Matrix Algorithms, 1st edn., vol.\u00a01. SIAM, Philadelphia (1998)","edition":"1"},{"key":"67_CR18","unstructured":"Yip, E.L.: FORTRAN Subroutines for Out-of-Core Solutions of Large Complex Linear Systems. Technical Report CR-159142, NASA (November 1979)"},{"key":"67_CR19","unstructured":"Quintana-Orti, E., van de Geijn, R.: Updating an LU factorization with pivoting, Technical Report TR-2006-42, The University of Texas at Austin, Department of Computer Sciences (2006), FLAME Working Note 21"},{"issue":"1","key":"67_CR20","doi-asserted-by":"publisher","first-page":"60","DOI":"10.1145\/1055531.1055534","volume":"31","author":"B.C. Gunter","year":"2005","unstructured":"Gunter, B.C., van de Geijn, R.A.: Parallel out-of-core computation and updating of the QR factorization. ACM Trans. Math. Softw.\u00a031(1), 60\u201378 (2005)","journal-title":"ACM Trans. Math. Softw."},{"issue":"8","key":"67_CR21","doi-asserted-by":"publisher","first-page":"1189","DOI":"10.1016\/0167-8191(95)00015-G","volume":"21","author":"M.W. Berry","year":"1995","unstructured":"Berry, M.W., Dongarra, J.J., Kim, Y.: A parallel algorithm for the reduction of a nonsymmetric matrix to block upper-hessenberg form. Parallel Comput.\u00a021(8), 1189\u20131211 (1995)","journal-title":"Parallel Comput."},{"key":"67_CR22","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"418","DOI":"10.1007\/3-540-48086-2_47","volume-title":"Parallel Processing and Applied Mathematics","author":"F.G. Gustavson","year":"2002","unstructured":"Gustavson, F.G.: New generalized data structures for matrices lead to a variety of high performance algorithms. In: Wyrzykowski, R., Dongarra, J., Paprzycki, M., Wa\u015bniewski, J. (eds.) PPAM 2001. LNCS, vol.\u00a02328, pp. 418\u2013436. Springer, Heidelberg (2002)"},{"issue":"1","key":"67_CR23","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1137\/0908009","volume":"8","author":"C. Bischof","year":"1987","unstructured":"Bischof, C., van Loan, C.: The WY representation for products of householder matrices. SIAM J. Sci. Stat. Comput.\u00a08(1), 2\u201313 (1987)","journal-title":"SIAM J. Sci. Stat. Comput."},{"issue":"1","key":"67_CR24","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1137\/0910005","volume":"10","author":"R. Schreiber","year":"1989","unstructured":"Schreiber, R., van Loan, C.: A storage-efficient WY representation for products of Householder transformations. SIAM J. Sci. Stat. Comput.\u00a010(1), 53\u201357 (1989)","journal-title":"SIAM J. Sci. Stat. Comput."},{"issue":"1","key":"67_CR25","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1137\/0908009","volume":"8","author":"C. Bischof","year":"1987","unstructured":"Bischof, C., van Loan, C.: The WY representation for products of householder matrices. SIAM J. Sci. Stat. Comput.\u00a08(1), 2\u201313 (1987)","journal-title":"SIAM J. Sci. Stat. Comput."},{"key":"67_CR26","unstructured":"Buttari, A., Langou, J., Kurzak, J., Dongarra, J.: Parallel Tiled QR Factorization for Multicore Architectures. Technical Report UT-CS-07-598, University of Tennessee (2007), LAPACK Working Note 190"}],"container-title":["Lecture Notes in Computer Science","Parallel Processing and Applied Mathematics"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-540-68111-3_67.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,18]],"date-time":"2023-05-18T14:46:55Z","timestamp":1684421215000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-540-68111-3_67"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[null]]},"ISBN":["9783540681052","9783540681113"],"references-count":26,"URL":"https:\/\/doi.org\/10.1007\/978-3-540-68111-3_67","relation":{},"subject":[]}}