{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T12:41:23Z","timestamp":1781613683336,"version":"3.54.5"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319201184","type":"print"},{"value":"9783319201191","type":"electronic"}],"license":[{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015]]},"DOI":"10.1007\/978-3-319-20119-1_2","type":"book-chapter","created":{"date-parts":[[2015,6,19]],"date-time":"2015-06-19T10:36:48Z","timestamp":1434710208000},"page":"17-30","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":26,"title":["Matrix Multiplication on High-Density Multi-GPU Architectures: Theoretical and Experimental Investigations"],"prefix":"10.1007","author":[{"given":"Peng","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuxiang","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2015,6,20]]},"reference":[{"key":"2_CR1","first-page":"1","volume":"38","author":"S Robinson","year":"2005","unstructured":"Robinson, S.: Toward an optimal algorithm for matrix multiplication. SIAM News 38, 1\u20133 (2005)","journal-title":"SIAM News"},{"key":"2_CR2","volume-title":"The Theory of Matrices: with Applications","author":"P Lancaster","year":"1985","unstructured":"Lancaster, P., Tismenetsky, M.: The Theory of Matrices: with Applications. Academic Press, Waltham (1985)"},{"key":"2_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"280","DOI":"10.1007\/11841036_27","volume-title":"Algorithms \u2013 ESA 2006","author":"F Dorn","year":"2006","unstructured":"Dorn, F.: Dynamic programming and fast matrix multiplication. In: Azar, Y., Erlebach, T. (eds.) ESA 2006. LNCS, vol. 4168, pp. 280\u2013291. Springer, Heidelberg (2006)"},{"key":"2_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1007\/3-540-45545-0_15","volume-title":"A Family of High-Performance Matrix Multiplication Algorithms","author":"JA Gunnels","year":"2001","unstructured":"Gunnels, J.A., Henry, G.M., Van De Geijn, R.A.: A Family of high-performance matrix multiplication algorithms. In: Alexandrov, V.N., Dongarra, J.J., Juliano, B.A., Renner, R.S., Kenneth Tan, C.J. (eds.) ICCS 2001. LNCS, vol. 2073, pp. 51\u201360. Springer, Heidelberg (2001)"},{"key":"2_CR5","doi-asserted-by":"publisher","first-page":"138","DOI":"10.1016\/j.parco.2008.12.010","volume":"35","author":"J Kurzak","year":"2009","unstructured":"Kurzak, J., Alvaro, W., Dongarra, J.: Optimizing matrix multiplication for a short-vector SIMD architecture\u2013CELL processor. Parallel Comput. 35, 138\u2013150 (2009)","journal-title":"Parallel Comput."},{"key":"2_CR6","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1007\/BF02165411","volume":"13","author":"V Strassen","year":"1969","unstructured":"Strassen, V.: Gaussian elimination is not optimal. Numer. Math. 13, 354\u2013356 (1969)","journal-title":"Numer. Math."},{"key":"2_CR7","unstructured":"Coppersmith, D., Winograd, S.: Matrix multiplication via arithmetic progressions. In: Proceedings of the Nineteenth Annual ACM Symposium on Theory of Computing, pp. 1\u20136 (2004)"},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"Williams, V.V.: Multiplying matrices faster than Coppersmith-Winograd. In: Proceedings of the Forty-Fourth Annual ACM Symposium on Theory of Computing, pp. 887\u2013898 (2012)","DOI":"10.1145\/2213977.2214056"},{"key":"2_CR9","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1016\/0898-1221(95)00077-C","volume":"30","author":"CC Chou","year":"1995","unstructured":"Chou, C.C., Deng, Y.F., Li, G., Wang, Y.: Parallelizing strassens method for matrix multiplication on distributed-memory mimd architectures. Comput. Math. Appl. 30, 49\u201369 (1995)","journal-title":"Comput. Math. Appl."},{"key":"2_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1007\/978-3-540-77704-5_12","volume-title":"High-Performance Computing","author":"P D\u2019Alberto","year":"2008","unstructured":"D\u2019Alberto, P., Nicolau, A.: Using recursion to boost ATLAS\u2019s performance. In: Labarta, J., Joe, K., Sato, T. (eds.) ISHPC 2006 and ALPS 2006. LNCS, vol. 4759, pp. 142\u2013151. Springer, Heidelberg (2008)"},{"key":"2_CR11","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"305","DOI":"10.1007\/978-3-540-71351-7_24","volume-title":"High Performance Computing for Computational Science - VECPAR 2006","author":"S Ohshima","year":"2007","unstructured":"Ohshima, S., Kise, K., Katagiri, T., Yuba, T.: Parallel processing of matrix multiplication in a CPU and GPU heterogeneous environment. In: Dayd\u00e9, M., Palma, J.M.L.M., Coutinho, A.L.G.A., Pacitti, E., Lopes, J.C. (eds.) VECPAR 2006. LNCS, vol. 4395, pp. 305\u2013318. Springer, Heidelberg (2007)"},{"key":"2_CR12","doi-asserted-by":"publisher","first-page":"1017","DOI":"10.1016\/j.jpdc.2004.03.021","volume":"64","author":"D Irony","year":"2004","unstructured":"Irony, D., Toledo, S., Tiskin, A.: Communication lower bounds for distributed-memory matrix multiplication. J. Parallel Distrib. Comput. 64, 1017\u20131026 (2004)","journal-title":"J. Parallel Distrib. Comput."},{"key":"2_CR13","doi-asserted-by":"crossref","unstructured":"Fatahalian, K., Sugerman, J., Hanrahan, P.: Understanding the efficiency of GPU algorithms for matrix-matrix multiplication. In: Proceedings of the ACM SIGGRAPH\/EUROGRAPHICS Conference on Graphics Hardware, pp. 133\u2013137 (2004)","DOI":"10.1145\/1058129.1058148"},{"key":"2_CR14","doi-asserted-by":"publisher","first-page":"1033","DOI":"10.1109\/71.963416","volume":"12","author":"O Beaumont","year":"2001","unstructured":"Beaumont, O., Boudet, V., Rastello, F., Robert, Y.: Matrix multiplication on heterogeneous platforms. IEEE Trans. Parallel Distrib. Syst. 12, 1033\u20131051 (2001)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"2_CR15","doi-asserted-by":"crossref","unstructured":"Thottethodi, M., Chatterjee, S., Lebeck, A.R.: Tuning Strassen\u2019s matrix multiplication for memory efficiency. In: Proceedings of the 1998 ACM\/IEEE Conference on Supercomputing (CDROM), pp. 1\u201314 (1998)","DOI":"10.1109\/SC.1998.10045"},{"key":"2_CR16","doi-asserted-by":"crossref","unstructured":"Luo, Q., Drake, J.B.: A scalable parallel Strassen\u2019s matrix multiplication algorithm for distributed-memory computers. In: Proceedings of the 1995 ACM Symposium on Applied Computing, pp. 221\u2013226 (1995)","DOI":"10.1145\/315891.315965"},{"key":"2_CR17","doi-asserted-by":"publisher","first-page":"543","DOI":"10.1002\/cpe.4330060702","volume":"6","author":"J Choi","year":"1994","unstructured":"Choi, J., Walker, D.W., Dongarra, J.J.: PUMMA: parallel universal matrix multiplication algorithms on distributed memory concurrent computers. Concurrency: Pract. Experience 6, 543\u2013570 (1994)","journal-title":"Concurrency: Pract. Experience"},{"key":"2_CR18","doi-asserted-by":"publisher","first-page":"1727","DOI":"10.1090\/S0025-5718-2013-02770-6","volume":"83","author":"P Zhang","year":"2014","unstructured":"Zhang, P., Gao, Y., Fierson, J., Deng, Y.: Eigenanalysis-based task mapping on parallel computers with cellular networks. Math. Comput. 83, 1727\u20131756 (2014)","journal-title":"Math. Comput."},{"key":"2_CR19","doi-asserted-by":"publisher","first-page":"287","DOI":"10.1109\/TPDS.2010.89","volume":"22","author":"P Zhang","year":"2011","unstructured":"Zhang, P., Powell, R., Deng, Y.: Interlacing bypass rings to torus networks for more efficient networks. IEEE Trans. Parallel Distrib. Syst. 22, 287\u2013295 (2011)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"2_CR20","doi-asserted-by":"publisher","first-page":"984","DOI":"10.1109\/TPDS.2014.2315201","volume":"26","author":"P Zhang","year":"2015","unstructured":"Zhang, P., Deng, Y., Feng, R., Luo, X., Wu, J.: Evaluation of various networks configurated by adding bypass or torus links. IEEE Trans. Parallel Distrib. Syst. 26, 984\u2013996 (2015)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"2_CR21","doi-asserted-by":"crossref","unstructured":"Ballard, G., Demmel, J., Holtz, O., Lipshitz, B., Schwartz, O.: Communication-optimal parallel algorithm for strassen\u2019s matrix multiplication. In: Proceedings of the 24th ACM Symposium on Parallelism in Algorithms and Architectures, pp. 193\u2013204 (2012)","DOI":"10.1145\/2312005.2312044"},{"key":"2_CR22","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1145\/1356052.1356053","volume":"34","author":"K Goto","year":"2008","unstructured":"Goto, K., Geijn, R.A.: Anatomy of high-performance matrix multiplication. ACM Trans. Math. Softw. (TOMS) 34, 12 (2008)","journal-title":"ACM Trans. Math. Softw. (TOMS)"},{"key":"2_CR23","doi-asserted-by":"crossref","unstructured":"Barrachina, S., Castillo, M., Igual, F.D., Mayo, R., Quintana-Orti, E.S.: Evaluation and tuning of the level 3 CUBLAS for graphics processors. In: IEEE International Symposium on Parallel and Distributed Processing, IPDPS 2008, pp. 1\u20138 (2008)","DOI":"10.1109\/IPDPS.2008.4536485"},{"key":"2_CR24","unstructured":"Demmel, J.: LAPACK: a portable linear algebra library for supercomputers. In: IEEE Control Systems Society Workshop on Computer-Aided Control System Design, pp. 1\u20137 (1989)"},{"key":"2_CR25","unstructured":"CS-Storm specification. (2014). http:\/\/www.cray.com\/sites\/default\/files\/CrayCS-Storm.pdf"},{"key":"2_CR26","series-title":"Lecture Notes in Electrical Engineering","doi-asserted-by":"publisher","first-page":"127","DOI":"10.1007\/978-94-017-8798-7_16","volume-title":"Frontier and Innovation in Future Computing and Communications","author":"Y-C Fang","year":"2014","unstructured":"Fang, Y.-C., Gao, Y., Stap, C.: Future enterprise computing looking into 2020. In: Park, J.J., Zomaya, A., Jeong, H.-Y., Obaidat, M. (eds.) Frontier and Innovation in Future Computing and Communications. LNEE, vol. 301, pp. 127\u2013134. Springer, Heidelberg (2014)"},{"key":"2_CR27","volume-title":"The Algorithm Design Manual","author":"SS Skiena","year":"1998","unstructured":"Skiena, S.S.: The Algorithm Design Manual, vol. 1. Springer, Heidelberg (1998)"},{"key":"2_CR28","doi-asserted-by":"publisher","unstructured":"Zhang, P., Ling, L., Deng, Y.: A data-driven paradigm for mapping problems. Parallel Comput. (2015).  doi: 10.1016\/j.parco.2015.05.002 (In press)","DOI":"10.1016\/j.parco.2015.05.002"},{"key":"2_CR29","doi-asserted-by":"crossref","unstructured":"Huss-Lederman, S., Jacobson, E.M., Johnson, J.R., Tsao, A., Turnbull, T.: Implementation of Strassen\u2019s algorithm for matrix multiplication. In: Proceedings of the 1996 ACM\/IEEE Conference on Supercomputing, pp. 32\u201332 (1996)","DOI":"10.1145\/369028.369096"}],"container-title":["Lecture Notes in Computer Science","High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-20119-1_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,8]],"date-time":"2023-02-08T12:50:14Z","timestamp":1675860614000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-20119-1_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015]]},"ISBN":["9783319201184","9783319201191"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-20119-1_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015]]},"assertion":[{"value":"20 June 2015","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}