{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,7]],"date-time":"2024-09-07T22:34:31Z","timestamp":1725748471480},"publisher-location":"Berlin, Heidelberg","reference-count":20,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642408199"},{"type":"electronic","value":"9783642408205"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-40820-5_4","type":"book-chapter","created":{"date-parts":[[2013,9,12]],"date-time":"2013-09-12T07:19:38Z","timestamp":1378970378000},"page":"39-48","source":"Crossref","is-referenced-by-count":0,"title":["A Fine-Grained Pipelined Implementation of LU Decomposition on SIMD Processors"],"prefix":"10.1007","author":[{"given":"Kai","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"ShuMing","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xi","family":"Ning","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"issue":"9","key":"4_CR1","doi-asserted-by":"publisher","first-page":"803","DOI":"10.1002\/cpe.728","volume":"15","author":"J.J. Dongarra","year":"2003","unstructured":"Dongarra, J.J., Luszczek, P., Petitet, A.: The LINPACK benchmark: past, present and future. Concurrency and Computation: Practice and Experience\u00a015(9), 803\u2013820 (2003)","journal-title":"Concurrency and Computation: Practice and Experience"},{"key":"4_CR2","unstructured":"LINPACK, \n                    \n                      http:\/\/www.netlib.org\/linpack"},{"key":"4_CR3","doi-asserted-by":"crossref","unstructured":"Michailidis, P.D., Margaritis, K.G.: Implementing parallel LU factorization with pipelining on a multicore using OpenMP. In: 2010 IEEE 13th International Conference on Computational Science and Engineering (CSE). IEEE (2010)","DOI":"10.1109\/CSE.2010.39"},{"issue":"2","key":"4_CR4","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1137\/1037042","volume":"37","author":"J.J.. Dongarra","year":"1995","unstructured":"Dongarra, J.J., Walker, D.W.: Software libraries for linear algebra computations on high performance computers. SIAM Review\u00a037(2), 151\u2013180 (1995)","journal-title":"SIAM Review"},{"key":"4_CR5","doi-asserted-by":"crossref","unstructured":"Whaley, R.C., Dongarra, J.J.: Automatically tuned linear algebra software. In: Proceedings of the 1998 ACM\/IEEE Conference on Supercomputing (CDROM). IEEE Computer Society (1998)","DOI":"10.1109\/SC.1998.10004"},{"key":"4_CR6","doi-asserted-by":"crossref","unstructured":"Anderson, M.J., Sheffield, D., Keutzer, K.: A predictive model for solving small linear algebra problems in gpu registers. In: 2012 IEEE 26th International Parallel & Distributed Processing Symposium (IPDPS). IEEE (2012)","DOI":"10.1109\/IPDPS.2012.11"},{"key":"4_CR7","doi-asserted-by":"crossref","unstructured":"Cupertino, L.F., et al.: LU Decomposition on GPUs: The Impact of Memory Access. In: 2010 22nd International Symposium on Computer Architecture and High Performance Computing Workshops (SBAC-PADW). IEEE (2010)","DOI":"10.1109\/SBAC-PADW.2010.10"},{"key":"4_CR8","unstructured":"Galoppo, N., et al.: LU-GPU: Efficient algorithms for solving dense linear systems on graphics hardware. In: Proceedings of the 2005 ACM\/IEEE Conference on Supercomputing. IEEE Computer Society (2005)"},{"key":"4_CR9","doi-asserted-by":"crossref","unstructured":"Lifflander, J., et al.: Dynamic Scheduling for Work Agglomeration on Heterogeneous Clusters. In: 2012 IEEE 26th International Parallel and Distributed Processing Symposium Workshops & PhD Forum (IPDPSW). IEEE (2012)","DOI":"10.1109\/IPDPSW.2012.297"},{"key":"4_CR10","doi-asserted-by":"crossref","unstructured":"Donfack, S., et al.: Hybrid static\/dynamic scheduling for already optimized dense matrix factorization. In: 2012 IEEE 26th International Parallel & Distributed Processing Symposium (IPDPS). IEEE (2012)","DOI":"10.1109\/IPDPS.2012.53"},{"key":"4_CR11","doi-asserted-by":"crossref","unstructured":"Lifflander, J., et al.: Mapping dense lu factorization on multicore supercomputer nodes. In: 2012 IEEE 26th International Parallel & Distributed Processing Symposium (IPDPS). IEEE (2012)","DOI":"10.1109\/IPDPS.2012.61"},{"key":"4_CR12","doi-asserted-by":"crossref","unstructured":"Venetis, I.E., Gao, G.R.: Mapping the LU decomposition on a many-core architecture: challenges and solutions. In: Proceedings of the 6th ACM Conference on Computing Frontiers. ACM (2009)","DOI":"10.1145\/1531743.1531756"},{"key":"4_CR13","doi-asserted-by":"crossref","unstructured":"Grigori, L., Demmel, J.W., Xiang, H.: Communication avoiding Gaussian elimination. In: Proceedings of the 2008 ACM\/IEEE Conference on Supercomputing. IEEE Press (2008)","DOI":"10.1109\/SC.2008.5214287"},{"issue":"1","key":"4_CR14","doi-asserted-by":"crossref","first-page":"60","DOI":"10.1109\/TC.2011.24","volume":"61","author":"M.K. Jaiswal","year":"2012","unstructured":"Jaiswal, M.K., Chandrachoodan, N.: Fpga-based high-performance and scalable block lu decomposition architecture. IEEE Transactions on Computers 61(1), 60\u201372 (2012)","journal-title":"IEEE Transactions on Computers"},{"key":"4_CR15","doi-asserted-by":"crossref","unstructured":"Zhuo, L., Prasanna, V.K.: High-performance and parameterized matrix factorization on FPGAs. In: International Conference on Field Programmable Logic and Applications, FPL 2006. IEEE (2006)","DOI":"10.1109\/FPL.2006.311238"},{"issue":"8","key":"4_CR16","doi-asserted-by":"crossref","first-page":"1057","DOI":"10.1109\/TC.2008.55","volume":"57","author":"L. Zhuo","year":"2008","unstructured":"Zhuo, L., Prasanna, V.K.: High-performance designs for linear algebra operations on reconfigurable hardware. IEEE Transactions on Computers 57(8), 1057\u20131071 (2008)","journal-title":"IEEE Transactions on Computers"},{"key":"4_CR17","doi-asserted-by":"crossref","unstructured":"Woh, M., Seo, S., Mahlke, S., Mudge, T., Chakrabarti, C., Flautner, K.: AnySP: Anytime Anywhere Anyway Signal Processing. In: Proceedings of the 31st Annual International Symposium on Computer Architecture (ISCA 2009), Austin, Texas, June 20-24 (2009)","DOI":"10.1145\/1555754.1555773"},{"key":"4_CR18","doi-asserted-by":"crossref","unstructured":"Flachs, B., Asano, S., Dhong, S.H., et al.: The Microarchitecture of the Synergistic Processor for a Cell Processor. IEEE Journal of Solid-State Circuits\u00a041(1) (January 2006)","DOI":"10.1109\/JSSC.2005.859332"},{"key":"4_CR19","unstructured":"Krashinsky, R., et al.: The vector-thread architecture. In: Proceedings of the 31st Annual International Symposium on Computer Architecture. IEEE (2004)"},{"issue":"2","key":"4_CR20","doi-asserted-by":"publisher","first-page":"214","DOI":"10.1007\/s11390-010-9318-0","volume":"25","author":"S.-M. Chen","year":"2010","unstructured":"Chen, S.-M., et al.: YHFT-QDSP: High-performance heterogeneous multi-core DSP. Journal of Computer Science and Technology\u00a025(2), 214\u2013224 (2010)","journal-title":"Journal of Computer Science and Technology"}],"container-title":["Lecture Notes in Computer Science","Network and Parallel Computing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-40820-5_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,17]],"date-time":"2019-05-17T02:35:00Z","timestamp":1558060500000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-40820-5_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642408199","9783642408205"],"references-count":20,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-40820-5_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2013]]}}}