{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T08:14:42Z","timestamp":1759133682480},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2014,3,4]],"date-time":"2014-03-04T00:00:00Z","timestamp":1393891200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2015,11]]},"DOI":"10.1007\/s11227-014-1133-x","type":"journal-article","created":{"date-parts":[[2014,3,3]],"date-time":"2014-03-03T23:26:00Z","timestamp":1393889160000},"page":"3991-4014","source":"Crossref","is-referenced-by-count":13,"title":["Hierarchical approach to optimization of parallel matrix multiplication on large-scale platforms"],"prefix":"10.1007","volume":"71","author":[{"given":"Khalid","family":"Hasanov","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jean-No\u00ebl","family":"Quintin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexey","family":"Lastovetsky","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,3,4]]},"reference":[{"key":"1133_CR1","unstructured":"Top 500 supercomputer sites. http:\/\/www.top500.org\/"},{"key":"1133_CR2","doi-asserted-by":"crossref","unstructured":"van de Geijn RA, Jerrell W (1997) SUMMA: scalable universal matrix multiplication algorithm. Concurr Pract Exp 9(4):255\u2013274","DOI":"10.1002\/(SICI)1096-9128(199704)9:4<255::AID-CPE250>3.0.CO;2-2"},{"issue":"10","key":"1133_CR3","doi-asserted-by":"crossref","first-page":"1033","DOI":"10.1109\/71.963416","volume":"12","author":"O Beaumont","year":"2001","unstructured":"Beaumont O, Boudet V, Rastello F, Robert Y (2001) Matrix multiplication on heterogeneous platforms. IEEE Trans Parallel Distrib Syst 12(10):1033\u20131051","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"1133_CR4","doi-asserted-by":"crossref","unstructured":"Lastovetsky A, Dongarra J (2009) High performance heterogeneous computing. Wiley, New York","DOI":"10.1002\/9780470508206"},{"key":"1133_CR5","doi-asserted-by":"crossref","unstructured":"Gustavson FG (2012) Cache blocking for linear algebra algorithms. Parallel processing and applied mathematics. In: Lecture Notes in Computer Science, vol 7203. Springer, Berlin, pp 122\u2013132","DOI":"10.1007\/978-3-642-31464-3_13"},{"key":"1133_CR6","doi-asserted-by":"crossref","unstructured":"Frigo M, Leiserson CE, Prokop H, Ramachandran S (1999) Cache-oblivious algorithms. In: Proceedings of the 40th annual symposium on foundations of computer science, FOCS \u201999. IEEE Computer Society, Washington, DC, USA, p 285","DOI":"10.1109\/SFFCS.1999.814600"},{"key":"1133_CR7","doi-asserted-by":"crossref","unstructured":"Yotov K, Roeder T, Pingali K, Gunnels J, Gustavson F (2007) An experimental comparison of cache-oblivious and cache-conscious programs. In: Proceedings of the nineteenth annual ACM symposium on parallel algorithms and srchitectures., SPAA \u201907ACM, New York, NY, USA, pp 93\u2013104","DOI":"10.1145\/1248377.1248394"},{"issue":"11","key":"1133_CR8","doi-asserted-by":"crossref","first-page":"1105","DOI":"10.1109\/TPDS.2002.1058095","volume":"13","author":"S Chatterjee","year":"2002","unstructured":"Chatterjee S, Lebeck AR, Patnala PK, Mithuna T (2002) Recursive array layouts and fast matrix multiplication. IEEE Trans Parallel Distrib Syst 13(11):1105\u20131123","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"1133_CR9","unstructured":"Basic Linear Algebra Routines (BLAS). http:\/\/www.netlib.org\/blas\/"},{"key":"1133_CR10","unstructured":"Clint WR, Dongarra JJ (1998) Automatically tuned linear algebra software. Proceedings of the 1998 ACM\/IEEE conference on supercomputing. Supercomputing \u201998IEEE Computer Society, Washington, DC, USA, pp 1\u201327"},{"key":"1133_CR11","doi-asserted-by":"crossref","unstructured":"Goto K, van De Geijn RA (2008) Anatomy of high-performance matrix multiplication. ACM Trans Math Softw 34(3):1\u201325","DOI":"10.1145\/1356052.1356053"},{"key":"1133_CR12","unstructured":"Cannon LE (1969) A cellular computer to implement the Kalman filter algorithm. Ph.D. thesis, Bozeman, MT, USA"},{"issue":"1","key":"1133_CR13","doi-asserted-by":"crossref","first-page":"17","DOI":"10.1016\/0167-8191(87)90060-3","volume":"4","author":"GC Fox","year":"1987","unstructured":"Fox GC, Otto SW, Hey AJG (1987) Matrix algorithms on a hypercube I: matrix multiplication. Parallel Comput 4(1):17\u201331","journal-title":"Parallel Comput"},{"key":"1133_CR14","doi-asserted-by":"crossref","unstructured":"Jaeyoung C, Walker DW, Dongarra J (1994) PUMMA: parallel universal matrix multiplication algorithms on distributed memory concurrent computers. Concurr Pract Exp 6(7):543\u2013570","DOI":"10.1002\/cpe.4330060702"},{"issue":"7","key":"1133_CR15","doi-asserted-by":"crossref","first-page":"571","DOI":"10.1002\/cpe.4330060703","volume":"6","author":"S Huss-Lederman","year":"1994","unstructured":"Huss-Lederman S, Jacobson E, Tsao A, Zhang G (1994) Matrix multiplication on the Intel Touchstone Delta. Concurr Pract Exp 6(7):571\u2013594","journal-title":"Concurr Pract Exp"},{"issue":"5","key":"1133_CR16","doi-asserted-by":"crossref","first-page":"575","DOI":"10.1147\/rd.395.0575","volume":"39","author":"RC Agarwal","year":"1995","unstructured":"Agarwal RC, Balle SM, Gustavson FG, Joshi M, Palkar P (1995) A three-dimensional approach to parallel matrix multiplication. IBM J Res Dev 39(5):575\u2013582","journal-title":"IBM J Res Dev"},{"issue":"6","key":"1133_CR17","doi-asserted-by":"crossref","first-page":"673","DOI":"10.1147\/rd.386.0673","volume":"38","author":"RC Agarwal","year":"1994","unstructured":"Agarwal RC, Gustavson FG, Zubair M (1994) A High-performance matrix-multiplication algorithm on a distributed-memory parallel computer, using overlapped communication. IBM J Res Dev 38(6):673\u2013681","journal-title":"IBM J Res Dev"},{"key":"1133_CR18","doi-asserted-by":"crossref","DOI":"10.1137\/1.9780898719642","volume-title":"ScaLAPACK user\u2019s guide","author":"LS Blackford","year":"1997","unstructured":"Blackford LS, Choi J, Cleary A, D\u2019Azeuedo E, Demmel J, Dhillon I, Hammarling S, Henry G, Petitet A, Stanley K, Walker D, Whaley RC (1997) ScaLAPACK user\u2019s guide. Society for industrial and applied mathematics, Philadelphia"},{"key":"1133_CR19","doi-asserted-by":"crossref","unstructured":"Jaeyoung C (1997) A new parallel matrix multiplication algorithm on distributed-memory concurrent computers. In: High Performance Computing on the Information Superhighway, 1997. HPC, Asia \u201997, pp 224\u2013229","DOI":"10.1109\/HPC.1997.592151"},{"key":"1133_CR20","doi-asserted-by":"crossref","unstructured":"Krishnan M, Nieplocha J (2004) SRUMMA: a matrix multiplication algorithm suitable for clusters and scalable shared memory systems. In: Proceedings of parallel and distributed processing symposium","DOI":"10.1109\/IPDPS.2004.1303000"},{"key":"1133_CR21","doi-asserted-by":"crossref","unstructured":"Solomonik E, Demmel J (2011)Communication-optimal parallel 2.5D matrix multiplication and LU factorization algorithms. In: Euro-Par (2), Lecture Notes in Computer Science, vol 6853. Springer, Berlin, pp 90\u2013109","DOI":"10.1007\/978-3-642-23397-5_10"},{"key":"1133_CR22","unstructured":"U.S.Department of Energy: Exascale Programming Challenges. ASCR Exascale Programming Challenges Workshop (2011)"},{"key":"1133_CR23","unstructured":"Message passing interface forum. http:\/\/www.mpi-forum.org\/"},{"key":"1133_CR24","doi-asserted-by":"crossref","unstructured":"Barnett M, Gupta S, Payne DG, Shuler L, Robert A, van de Geijn, Watts J (1994) Interprocessor collective communication library (InterCom). In: Proceedings of the scalable high performance computing conference. IEEE Computer Society Press, New York, pp 357\u2013364","DOI":"10.1109\/SHPCC.1994.296665"},{"issue":"6","key":"1133_CR25","doi-asserted-by":"crossref","first-page":"809","DOI":"10.1016\/j.jpdc.2007.11.003","volume":"68","author":"P Patarasuk","year":"2008","unstructured":"Patarasuk P, Yuan X, Faraj A (2008) Techniques for pipelined broadcast on ethernet switched clusters. J Parallel Distrib Comput 68(6):809\u2013824","journal-title":"J Parallel Distrib Comput"},{"issue":"1","key":"1133_CR26","doi-asserted-by":"crossref","first-page":"49","DOI":"10.1177\/1094342005051521","volume":"19","author":"R Thakur","year":"2005","unstructured":"Thakur R, Rabenseifner R, Gropp W (2005) Optimization of collective communication operations in MPICH. Int J High Perform Comput Appl 19(1):49\u201366","journal-title":"Int J High Perform Comput Appl"},{"key":"1133_CR27","doi-asserted-by":"crossref","unstructured":"Scott DS (1991) Efficient all-to-all communication patterns in hypercube and mesh topologies. In: Proceedings of the sixth conference distributed memory computing, pp 398\u2013403","DOI":"10.1109\/DMCC.1991.633174"},{"key":"1133_CR28","doi-asserted-by":"crossref","unstructured":"Graham RL, Venkata MG, Ladd J, Shamis P, Rabinovitz I, Filipov V, Shainer G (2011) Cheetah: a framework for scalable hierarchical collective operations. CCGRID, pp 73\u201383","DOI":"10.1109\/CCGrid.2011.42"},{"key":"1133_CR29","doi-asserted-by":"crossref","unstructured":"Alm\u00e1si G, Heidelberger P, Archer CJ, Martorell X, Erway CC, Moreira JE, Steinmacher-Burow B, Zheng Y (2005) Optimization of MPI collective communication on BlueGene\/L systems. In: Proceedings of the 19th annual international conference on supercomputing., ICS \u201905ACM, New York, NY, USA, pp 253\u2013262","DOI":"10.1145\/1088149.1088183"},{"key":"1133_CR30","doi-asserted-by":"crossref","unstructured":"Kumar S, Dozsa G, Almasi G, Heidelberger P, Chen D, Giampapa ME, Blocksome M, Faraj A, Parker J, Ratterman J, Smith B, Archer CJ (2008) The deep computing messaging framework: generalized scalable message passing on the Blue Gene\/P supercomputer. In: Proceedings of the 22nd annual international conference on supercomputing., ICS \u201908ACM, New York, NY, USA, pp 94\u2013103","DOI":"10.1145\/1375527.1375544"},{"key":"1133_CR31","doi-asserted-by":"crossref","unstructured":"Hoefler T, Siebert C, Rehm W (2007) A practically constant-time MPI broadcast algorithm for large-scale InfiniBand clusters with multicast. In: IPDPS, IEEE, New York, pp 1\u20138","DOI":"10.1109\/IPDPS.2007.370475"},{"issue":"3","key":"1133_CR32","doi-asserted-by":"crossref","first-page":"167","DOI":"10.1023\/B:IJPP.0000029272.69895.c1","volume":"32","author":"J Liu","year":"2004","unstructured":"Liu J, Wu J, Panda DK (2004) High performance RDMA-based MPI implementation over InfiniBand. Int J Parallel Progr 32(3):167\u2013198","journal-title":"Int J Parallel Progr"},{"issue":"3","key":"1133_CR33","doi-asserted-by":"crossref","first-page":"389","DOI":"10.1016\/S0167-8191(06)80021-9","volume":"20","author":"RW Hockney","year":"1994","unstructured":"Hockney RW (1994) The communication challenge for MPP: Intel Paragon and Meiko CS-2. Parallel Comput 20(3):389\u2013398","journal-title":"Parallel Comput"},{"key":"1133_CR34","unstructured":"Pjes\u0306ivac-Grbovi\u0107 J (2007) Towards Automatic and Adaptive Optimizations of MPI Collective Operations. Ph.D. thesis, University of Tennessee, Knoxville"},{"key":"1133_CR35","unstructured":"MPICH-A Portable Implementation of MPI. http:\/\/www.mpich.org\/"},{"key":"1133_CR36","doi-asserted-by":"crossref","unstructured":"Gabriel E, Fagg G, Bosilca G, Angskun T, Dongarra J, Squyres J, Sahay V, Kambadur P, Barrett B, Lumsdaine A, Castain R, Daniel D, Graham R, Woodall T (2004) Open MPI: goals, concept, and design of a next generation MPI implementation. In: Proceedings, 11th European PVM\/MPI Users\u2019 Group Meeting, pp 97\u2013104","DOI":"10.1007\/978-3-540-30218-6_19"},{"key":"1133_CR37","doi-asserted-by":"crossref","unstructured":"Quintin J., Hasanov K, Lastovetsky A (2013) Hierarchical parallel matrix multiplication on large-scale distributed memory platforms. In: 42nd International conference on parallel processing (ICPP 2013). IEEE, New York, pp 754\u2013762","DOI":"10.1109\/ICPP.2013.89"},{"key":"1133_CR38","unstructured":"Kondo M (2012) Report on Exascale Architecture. In: IESP Meeting, Japan"},{"key":"1133_CR39","unstructured":"Grid5000. http:\/\/www.grid5000.fr"},{"issue":"3\u20134","key":"1133_CR40","doi-asserted-by":"crossref","first-page":"247","DOI":"10.1007\/s00450-011-0168-y","volume":"26","author":"P Balaji","year":"2011","unstructured":"Balaji P, Gupta R, Vishnu A, Beckman P (2011) Mapping communication layouts to network hardware characteristics on massive-scale blue gene systems. Comput Sci R D 26(3\u20134):247\u2013256","journal-title":"Comput Sci R D"},{"key":"1133_CR41","unstructured":"Blackford LS, Whaley RC (1998) ScaLAPACK Evaluation and Performance at the DoD MSRCs. Tech. Rep. LAPACK Working Note No. 136, Technical Report UT CS-98-388, University of Tennessee, Knoxville, TN (1998)"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1133-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-014-1133-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1133-x","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,8]],"date-time":"2019-08-08T03:44:01Z","timestamp":1565235841000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-014-1133-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,3,4]]},"references-count":41,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2015,11]]}},"alternative-id":["1133"],"URL":"https:\/\/doi.org\/10.1007\/s11227-014-1133-x","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,3,4]]}}}