{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T09:51:05Z","timestamp":1742982665226,"version":"3.40.3"},"publisher-location":"Cham","reference-count":15,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319173528"},{"type":"electronic","value":"9783319173535"}],"license":[{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015]]},"DOI":"10.1007\/978-3-319-17353-5_3","type":"book-chapter","created":{"date-parts":[[2015,4,17]],"date-time":"2015-04-17T07:16:14Z","timestamp":1429254974000},"page":"31-42","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Heterogenous Acceleration for Linear Algebra in Multi-coprocessor Environments"],"prefix":"10.1007","author":[{"given":"Azzam","family":"Haidar","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Piotr","family":"Luszczek","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stanimire","family":"Tomov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jack","family":"Dongarra","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,4,18]]},"reference":[{"issue":"2","key":"3_CR1","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1002\/cpe.1631","volume":"23","author":"C Augonnet","year":"2011","unstructured":"Augonnet, C., Thibault, S., Namyst, R., Wacrenier, P.-A.: StarPU: a unified platform for task scheduling on heterogeneous multicore architectures. Concur. Comput. Pract. Exp. 23(2), 187\u2013198 (2011)","journal-title":"Concur. Comput. Pract. Exp."},{"key":"3_CR2","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1145\/209937.209958","volume":"30","author":"RD Blumofe","year":"1995","unstructured":"Blumofe, R.D., Joerg, C.F., Kuszmaul, B.C., Leiserson, C.E., Randall, K.H., Zhou, Y.: Cilk: an efficient multithreaded runtime system. SIGPLAN Not. 30, 207\u2013216 (1995)","journal-title":"SIGPLAN Not."},{"key":"3_CR3","doi-asserted-by":"crossref","unstructured":"Haidar, A., Ltaief, H., Luszczek, P., Dongarra, J.: A comprehensive study of task coalescing for selecting parallelism granularity in a two-stage bidiagonal reduction. In: Proceedings of the IEEE International Parallel and Distributed Processing Symposium, Shanghai, China, 21\u201325 May 2012, pp. 25\u201335. IEEE Computer Society (2012)","DOI":"10.1109\/IPDPS.2012.13"},{"key":"3_CR4","unstructured":"Intel$$^{\\textregistered }$$ Xeon Phi\u2122 coprocessor system software developers guide. http:\/\/software.intel.com\/en-us\/articles\/"},{"key":"3_CR5","unstructured":"Math Kernel Library. http:\/\/software.intel.com\/intel-mkl\/"},{"key":"3_CR6","volume-title":"Intel$$^{\\textregistered }$$ Xeon Phi\u2122 Coprocessor High-Performance Programming","author":"J Jeffers","year":"2013","unstructured":"Jeffers, J., Reinders, J.: Intel$$^{\\textregistered }$$ Xeon Phi\u2122 Coprocessor High-Performance Programming. Morgan Kaufmann Publishers, San Francisco (2013)"},{"issue":"1","key":"3_CR7","doi-asserted-by":"crossref","first-page":"15","DOI":"10.1002\/cpe.1467","volume":"21","author":"J Kurzak","year":"2009","unstructured":"Kurzak, J., Ltaief, H., Dongarra, J.J., Badia, R.M.: Scheduling dense linear algebra operations on multicore processors. Concur. Comput. Pract. Exp. 21(1), 15\u201344 (2009)","journal-title":"Concur. Comput. Pract. Exp."},{"key":"3_CR8","unstructured":"Kurzak, J., Luszczek, P., YarKhan, A., Faverge, M., Langou, J., Bouwmeester, H., Dongarra, J.: Multithreading in the PLASMA Library. In Handbook of Multi and Many-Core Processing: Architecture, Algorithms, Programming, and Applications. Computer and Information Science Series. Chapman and Hall\/CRC, 26 April 2013"},{"key":"3_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"661","DOI":"10.1007\/978-3-642-31464-3_67","volume-title":"Parallel Processing and Applied Mathematics","author":"H Ltaief","year":"2012","unstructured":"Ltaief, H., Luszczek, P., Dongarra, J.: Enhancing parallelism of tile bidiagonal transformation on multicore architectures using tree reduction. In: Wyrzykowski, R., Dongarra, J., Karczewski, K., Wa\u015bniewski, J. (eds.) PPAM 2011, Part I. LNCS, vol. 7203, pp. 661\u2013670. Springer, Heidelberg (2012)"},{"key":"3_CR10","doi-asserted-by":"crossref","unstructured":"Luszczek, P., Ltaief, H., Dongarra, J.: Two-stage tridiagonal reduction for dense symmetric matrices using tile algorithms on multicore architectures. In: Proceedings of IPDPS 2011: IEEE International Parallel and Distributed Processing Symposium, Anchorage, Alaska, USA, 16\u201320 May 2011, pp. 944\u2013955. IEEE Computer Society (2011)","DOI":"10.1109\/IPDPS.2011.91"},{"key":"3_CR11","doi-asserted-by":"crossref","unstructured":"P\u00e9rez, J.M., Badia, R.M., Labarta, J.: A dependency-aware task-based programming environment for multi-core architectures. In: Proceedings of the 2008 IEEE International Conference on Cluster Computing, Tsukuba, Japan, 29 September\u20131 October 2008, pp. 142\u2013151. IEEE (2008)","DOI":"10.1109\/CLUSTR.2008.4663765"},{"issue":"6","key":"3_CR12","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1109\/2.214440","volume":"26","author":"MC Rinard","year":"1993","unstructured":"Rinard, M.C., Scales, D.J., Lam, M.S.: Jade: a high-level, machine-independent language for parallel programming. Computer 26(6), 28\u201338 (1993). doi:10.1109\/2.214440","journal-title":"Computer"},{"key":"3_CR13","first-page":"15","volume-title":"Parallel Processing and Artificial Intelligence","author":"LG Valiant","year":"1989","unstructured":"Valiant, L.G.: Bulk-synchronous parallel computers. In: Reeve, M. (ed.) Parallel Processing and Artificial Intelligence, pp. 15\u201322. Wiley, New York (1989)"},{"key":"3_CR14","doi-asserted-by":"publisher","unstructured":"Valiant, L. G.: A bridging model for parallel computation. Commun. ACM 33(8) (1990). doi:10.1145\/79173.79181","DOI":"10.1145\/79173.79181"},{"key":"3_CR15","unstructured":"YarKhan, A.: Dynamic task execution on shared and distributed memory architectures. Ph.D. thesis, University of Tennessee, December 2012"}],"container-title":["Lecture Notes in Computer Science","High Performance Computing for Computational Science -- VECPAR 2014"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-17353-5_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,9]],"date-time":"2023-08-09T14:55:31Z","timestamp":1691592931000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-17353-5_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015]]},"ISBN":["9783319173528","9783319173535"],"references-count":15,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-17353-5_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2015]]},"assertion":[{"value":"18 April 2015","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}