{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T12:16:05Z","timestamp":1763468165558},"reference-count":58,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2014,1,21]],"date-time":"2014-01-21T00:00:00Z","timestamp":1390262400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2014,6]]},"DOI":"10.1007\/s11227-014-1098-9","type":"journal-article","created":{"date-parts":[[2014,1,20]],"date-time":"2014-01-20T09:55:16Z","timestamp":1390211716000},"page":"1418-1440","source":"Crossref","is-referenced-by-count":21,"title":["A Matrix\u2013Matrix Multiplication methodology for single\/multi-core architectures using SIMD"],"prefix":"10.1007","volume":"68","author":[{"given":"Vasilios","family":"Kelefouras","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Angeliki","family":"Kritikakou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Costas","family":"Goutis","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,1,21]]},"reference":[{"key":"1098_CR1","doi-asserted-by":"crossref","unstructured":"Agakov F, Bonilla E, Cavazos J, Franke B, Fursin G, O\u2019Boyle MFP, Thomson J, Toussaint M, Williams CKI (2006) Using machine learning to focus iterative optimization. In: Proceedings of the international symposium on code generation and optimization, CGO \u201906. IEEE Computer Society, Washington, DC, USA, pp 295\u2013305. doi: 10.1109\/CGO.2006.37","DOI":"10.1109\/CGO.2006.37"},{"key":"1098_CR2","doi-asserted-by":"crossref","first-page":"345","DOI":"10.1145\/197405.197406","volume":"26","author":"DF Bacon","year":"1994","unstructured":"Bacon DF, Graham SL, Oliver SJ (1994) Compiler transformations for high-performance computing. ACM Comput Surv 26:345\u2013420","journal-title":"ACM Comput Surv"},{"key":"1098_CR3","doi-asserted-by":"crossref","unstructured":"Bilmes J, Asanovi\u0107 K, Chin C, Demmel J (1997) Optimizing matrix multiply using PHiPAC: a portable, high-performance, ANSI C coding methodology. In: Proceedings of the international conference on supercomputing. ACM SIGARC, Vienna, Austria","DOI":"10.1145\/263580.263662"},{"key":"1098_CR4","doi-asserted-by":"crossref","first-page":"386","DOI":"10.1137\/0613026","volume":"13","author":"P Bj\u00f8rstad","year":"1992","unstructured":"Bj\u00f8rstad P, Manne F, S\u00f8revik T, Vajtersic M (1992) Efficient matrix multiplication on simd computers. SIAM J Matrix Anal Appl 13:386\u2013401","journal-title":"SIAM J Matrix Anal Appl"},{"key":"1098_CR5","doi-asserted-by":"crossref","DOI":"10.1137\/1.9780898719642","volume-title":"ScaLAPACK user\u2019s guide","author":"LS Blackford","year":"1997","unstructured":"Blackford LS, Choi J, Cleary A, D\u2019Azeuedo E, Demmel J, Dhillon I, Hammarling S, Henry G, Petitet A, Stanley K, Walker D, Whaley RC (1997) ScaLAPACK user\u2019s guide. Society for industrial and applied mathematics, Philadelphia, PA"},{"issue":"8","key":"1098_CR6","doi-asserted-by":"crossref","first-page":"207","DOI":"10.1145\/209937.209958","volume":"30","author":"RD Blumofe","year":"1995","unstructured":"Blumofe RD, Joerg CF, Kuszmaul BC, Leiserson CE, Randall KH, Zhou Y (1995) Cilk: an efficient multithreaded runtime system. SIGPLAN Not 30(8):207\u2013216. doi: 10.1145\/209937.209958","journal-title":"SIGPLAN Not"},{"key":"1098_CR7","doi-asserted-by":"crossref","unstructured":"Chatterjee S, Lebeck AR, Patnala PK, Thottethodi M (1999) Recursive array layouts and fast parallel matrix multiplication. In: Proceedings of 11th annual ACM symposium on parallel algorithms and architectures, pp 222\u2013231","DOI":"10.1145\/305619.305645"},{"key":"1098_CR8","doi-asserted-by":"crossref","unstructured":"Choi J (1998) A new parallel matrix multiplication algorithm on distributed-memory concurrent computers. Concurr Pract Exp 10(8):655\u2013670","DOI":"10.1002\/(SICI)1096-9128(199807)10:8<655::AID-CPE369>3.0.CO;2-O"},{"key":"1098_CR9","unstructured":"Cooper KD, Subramanian D, Torczon L (2001) Adaptive optimizing compilers for the 21st century. J Supercomput 23:2002"},{"key":"1098_CR10","unstructured":"Desprez F, Suter F (2002) Impact of mixed-parallelismpon parallel implementations of Strassen and Winograd matrix multiplication algorithms. Rapport de recherche RR-4482, INRIA. http:\/\/hal.inria.fr\/inria-00072106"},{"issue":"8","key":"1098_CR11","doi-asserted-by":"crossref","first-page":"771","DOI":"10.1002\/cpe.791","volume":"16","author":"F Desprez","year":"2004","unstructured":"Desprez F, Suter F (2004) Impact of mixed-parallelism on parallel implementations of the strassen and winograd matrix multiplication algorithms: Research articles. Concurr Comput Pract Exp 16(8):771\u2013797. doi: 10.1002\/cpe.v16:8","journal-title":"Concurr Comput Pract Exp"},{"key":"1098_CR12","doi-asserted-by":"crossref","DOI":"10.21236\/ADA479065","volume-title":"The fastest fourier transform in the west","author":"M Frigo","year":"1997","unstructured":"Frigo M, Johnson SG (1997) The fastest fourier transform in the west. Tech. rep, Cambridge, MA"},{"key":"1098_CR13","doi-asserted-by":"crossref","unstructured":"Garcia E, Venetis IE, Khan R, Gao GR (2010) Optimized dense matrix multiplication on a many-core architecture. In: Proceedings of the 16th international Euro-Par conference on parallel processing: Part II, Euro-Par\u201910. Springer-Verlag, Berlin, Heidelberg, pp 316\u2013327. http:\/\/dl.acm.org\/citation.cfm?id=1885276.1885308","DOI":"10.1007\/978-3-642-15291-7_29"},{"key":"1098_CR14","volume-title":"Summa: scalable universal matrix multiplication algorithm","author":"RAVD Geijn","year":"1997","unstructured":"Geijn RAVD, Watts J (1997) Summa: scalable universal matrix multiplication algorithm. Tech. rep., Cambridge, MA"},{"key":"1098_CR15","volume-title":"On reducing tlb misses in matrix multiplication","author":"K Goto","year":"2002","unstructured":"Goto K, van de Geijn R (2002) On reducing tlb misses in matrix multiplication. Tech. rep., Cambridge, MA"},{"key":"1098_CR16","doi-asserted-by":"crossref","unstructured":"Goto K, van de Geijn RA (2008) Anatomy of high-performance matrix multiplication. ACM Trans Math Softw 34(3):12:1\u201312:25. doi: 10.1145\/1356052.1356053","DOI":"10.1145\/1356052.1356053"},{"key":"1098_CR17","unstructured":"Granston E, Holler A (2001) Automatic recommendation of compiler options. In: Proceedings of the workshop on feedback-directed and dynamic optimization FDDO"},{"key":"1098_CR18","unstructured":"Guennebaud G, Jacob B, et al (2010) Eigen v3. http:\/\/eigen.tuxfamily.org"},{"key":"1098_CR19","unstructured":"Hall JD, Carr NA, Hart JC (2003) Cache and bandwidth aware matrix multiplication on the gpu. Tech. rep., Cambridge, MA"},{"issue":"4","key":"1098_CR20","doi-asserted-by":"crossref","first-page":"48","DOI":"10.1002\/scj.10551","volume":"36","author":"M Hattori","year":"2005","unstructured":"Hattori M, Ito N, Chen W, Wada K (2005) Parallel matrix-multiplication algorithm for distributed parallel computers. Syst Comput Jpn 36(4):48\u201359. doi: 10.1002\/scj.v36:4","journal-title":"Syst Comput Jpn"},{"key":"1098_CR21","doi-asserted-by":"crossref","unstructured":"Hunold S, Rauber T (2005) Automatic tuning of pdgemm towards optimal performance. In: Proceedings of the 11th international Euro-Par conference on parallel processing, Euro-Par\u201905. Springer-Verlag, Berlin, pp 837\u2013846. doi: 10.1007\/11549468_91","DOI":"10.1007\/11549468_91"},{"key":"1098_CR22","doi-asserted-by":"crossref","unstructured":"Hunold S, Rauber T, R\u00fcnger G (2004) Multilevel hierarchical matrix multiplication on clusters. In: Proceedings of the 18th annual international conference on supercomputing, ICS\u201904. ACM, New York, NY, pp. 136\u2013145. doi: 10.1145\/1006209.1006230","DOI":"10.1145\/1006209.1006230"},{"key":"1098_CR23","unstructured":"Intel (2012) Intel mkl. Available at http:\/\/software.intel.com\/en-us\/intel-mkl"},{"key":"1098_CR24","unstructured":"Jiang C, Snir M (2005) Automatic tuning matrix multiplication performance on graphics hardware. In: In the proceedings of the 14th international conference on parallel architecture and compilation techniques (PACT), pp 185\u2013196"},{"key":"1098_CR25","doi-asserted-by":"crossref","unstructured":"Kisuki T, Knijnenburg PMW, O\u2019Boyle MFP, Bodin F, Wijshoff HAG (1999) A feasibility study in iterative compilation. In: Proceedings of the 2nd international symposium on high performance computing, ISHPC\u201999. Springer-Verlag, London, pp. 121\u2013132. http:\/\/dl.acm.org\/citation.cfm?id=646347.690219","DOI":"10.1007\/BFb0094916"},{"key":"1098_CR26","doi-asserted-by":"crossref","unstructured":"KKrishnan M, Nieplocha J (2004) Srumma: a matrix multiplication algorithm suitable for clusters and scalable shared memory systems. Parallel and distributed processing symposium, international 1, 70b. doi: 10.1109\/IPDPS.2004.1303000","DOI":"10.1109\/IPDPS.2004.1303000"},{"key":"1098_CR27","unstructured":"Krishnan, M., Nieplocha, J.: Memory efficient parallel matrix multiplication operation for irregular problems. In: Proceedings of the 3rd conference on Computing frontiers, CF \u201906, pp. 229\u2013240. ACM, New York, NY, USA (2006). DOI 10.1145\/1128022.1128054. URL http:\/\/doi.acm.org\/10.1145\/1128022.1128054"},{"key":"1098_CR28","volume-title":"Gotoblas\u2014anatomy of a fast matrix multiplication","author":"A Krivutsenko","year":"2008","unstructured":"Krivutsenko A (2008) Gotoblas\u2014anatomy of a fast matrix multiplication. Tech. rep., Cambridge, MA"},{"issue":"5","key":"1098_CR29","doi-asserted-by":"crossref","first-page":"832","DOI":"10.1109\/JPROC.2008.917732","volume":"96","author":"M Kulkarni","year":"2008","unstructured":"Kulkarni M, Pingali K (2008) An experimental study of self-optimizing dense linear algebra software. Proc IEEE 96(5):832\u2013848","journal-title":"Proc IEEE"},{"issue":"6","key":"1098_CR30","doi-asserted-by":"crossref","first-page":"171","DOI":"10.1145\/996893.996863","volume":"39","author":"P Kulkarni","year":"2004","unstructured":"Kulkarni P, Hines S, Hiser J, Whalley D, Davidson J, Jones D (2004) Fast searches for effective optimization phase sequences. SIGPLAN Not 39(6):171\u2013182. doi: 10.1145\/996893.996863","journal-title":"SIGPLAN Not"},{"key":"1098_CR31","doi-asserted-by":"crossref","unstructured":"Kulkarni PA, Whalley DB, Tyson GS, Davidson JW (2009) Practical exhaustive optimization phase order exploration and evaluation. ACM Trans Archit Code Optim 6(1):1:1\u20131:36","DOI":"10.1145\/1509864.1509865"},{"issue":"3","key":"1098_CR32","doi-asserted-by":"crossref","first-page":"138","DOI":"10.1016\/j.parco.2008.12.010","volume":"35","author":"J Kurzak","year":"2009","unstructured":"Kurzak J, Alvaro W, Dongarra J (2009) Optimizing matrix multiplication for a short-vector simd architecture\u2014cell processor. Parallel Comput 35(3):138\u2013150. doi: 10.1016\/j.parco.2008.12.010","journal-title":"Parallel Comput"},{"key":"1098_CR33","doi-asserted-by":"crossref","unstructured":"Michaud P (2011) Replacement policies for shared caches on symmetric multicores: a programmer-centric point of view. In: Proceedings of the 6th international conference on high performance and embedded architectures and compilers, HiPEAC\u201911. ACM, New York, NY, pp. 187\u2013196. doi: 10.1145\/1944862.1944890","DOI":"10.1145\/1944862.1944890"},{"key":"1098_CR34","doi-asserted-by":"crossref","unstructured":"Milder PA, Franchetti F, Hoe JC, P\u00fcschel M (2012) Computer generation of hardware for linear digital signal processing transforms. ACM Trans Des Autom Electron Syst 17(2). http:\/\/dblp.uni-trier.de\/db\/journals\/todaes\/todaes17.html#MilderFHP12","DOI":"10.1145\/2159542.2159547"},{"key":"1098_CR35","doi-asserted-by":"crossref","unstructured":"Monsifrot A, Bodin F, Quiniou R (2002) A machine learning approach to automatic production of compiler heuristics. In: Proceedings of the 10th international conference on artificial intelligence: methodology, systems, and applications, AIMSA\u201902. Springer-Verlag, London, pp 41\u201350. http:\/\/dl.acm.org\/citation.cfm?id=646053.677574","DOI":"10.1007\/3-540-46148-5_5"},{"key":"1098_CR36","doi-asserted-by":"crossref","first-page":"2001","DOI":"10.1109\/69.908985","volume":"13","author":"B Moon","year":"2001","unstructured":"Moon B, Jagadish HV, Faloutsos C, Saltz JH (2001) Analysis of the clustering properties of the hilbert space-filling curve. IEEE Trans Knowl Data Eng 13:2001","journal-title":"IEEE Trans Knowl Data Eng"},{"issue":"6","key":"1098_CR37","doi-asserted-by":"crossref","first-page":"89","DOI":"10.1145\/1273442.1250746","volume":"42","author":"N Nethercote","year":"20007","unstructured":"Nethercote N, Seward J (20007) Valgrind: a framework for heavyweight dynamic binary instrumentation. SIGPLAN Not 42(6):89\u2013100. doi: 10.1145\/1273442.1250746","journal-title":"SIGPLAN Not"},{"key":"1098_CR38","doi-asserted-by":"crossref","unstructured":"Nikolopoulos DS (2003) Code and data transformations for improving shared cache performance on smt processors. In: ISHPC, pp 54\u201369","DOI":"10.1007\/978-3-540-39707-6_5"},{"key":"1098_CR39","unstructured":"Openblas (2012) An optimized blas library. URL available at http:\/\/xianyi.github.com\/OpenBLAS\/"},{"key":"1098_CR40","doi-asserted-by":"crossref","unstructured":"Park E, Kulkarni S, Cavazos J (2011) An evaluation of different modeling techniques for iterative compilation. In: Proceedings of the 14th international conference on compilers, architectures and synthesis for embedded systems, CASES\u201911. ACM, New York, NY, pp. 65\u201374. doi: 10.1145\/2038698.2038711","DOI":"10.1145\/2038698.2038711"},{"key":"1098_CR41","unstructured":"Pinter SS (1996) Register allocation with instruction scheduling: a new approach. J Prog Lang 4(1):21\u201338"},{"key":"1098_CR42","doi-asserted-by":"crossref","unstructured":"R\u00fcnger G., Schwind M (2010 Fast recursive matrix multiplication for multi-core architectures. Procedia Comput Sci 1(1):67\u201376. International conference on computational science 2010 (ICCS 2010)","DOI":"10.1016\/j.procs.2010.04.009"},{"key":"1098_CR43","unstructured":"See homepage for details: Atlas homepage (2012). Http:\/\/math-atlas.sourceforge.net\/"},{"issue":"3","key":"1098_CR44","first-page":"14:1","volume":"10","author":"G Shobaki","year":"2008","unstructured":"Shobaki G, Shawabkeh M, Rmaileh NEA (2008) Preallocation instruction scheduling with register pressure minimization using a combinatorial optimization approach. ACM Trans Archit Code Optim 10(3):14:1\u201314:31. doi: 10.1145\/2512432","journal-title":"ACM Trans Archit Code Optim"},{"issue":"5","key":"1098_CR45","doi-asserted-by":"crossref","first-page":"77","DOI":"10.1145\/780822.781141","volume":"38","author":"M Stephenson","year":"2003","unstructured":"Stephenson M, Amarasinghe S, Martin M, O\u2019Reilly UM (2003) Meta optimization: improving compiler heuristics with machine learning. SIGPLAN Not 38(5):77\u201390. doi: 10.1145\/780822.781141","journal-title":"SIGPLAN Not"},{"issue":"3","key":"1098_CR46","doi-asserted-by":"crossref","first-page":"354","DOI":"10.1007\/BF02165411","volume":"14","author":"V Strassen","year":"1969","unstructured":"Strassen V (1969) Gaussian elimination is not optimal. Numerische Mathematik 14(3):354\u2013356","journal-title":"Numerische Mathematik"},{"issue":"4","key":"1098_CR47","doi-asserted-by":"crossref","first-page":"46:1","DOI":"10.1145\/2400682.2400705","volume":"9","author":"M Tartara","year":"2013","unstructured":"Tartara M, Crespi Reghizzi S (2013) Continuous learning of compiler heuristics. ACM Trans Archit Code Optim 9(4):46:1\u201346:25. doi: 10.1145\/2400682.2400705","journal-title":"ACM Trans Archit Code Optim"},{"key":"1098_CR48","doi-asserted-by":"crossref","unstructured":"Thottethodi M, Chatterjee S, Lebeck AR (1998) Tuning strassen\u2019s matrix multiplication for memory efficiency. In: In proceedings of SC98 (CD-ROM)","DOI":"10.1109\/SC.1998.10045"},{"key":"1098_CR49","doi-asserted-by":"crossref","unstructured":"Triantafyllis S, Vachharajani M, Vachharajani N, August DI (2003) Compiler optimization-space exploration. In: Proceedings of the international symposium on Code generation and optimization: feedback-directed and runtime optimization, CGO \u201903. IEEE Computer Society, Washington, DC, USA, pp 204\u2013215. http:\/\/dl.acm.org\/citation.cfm?id=776261.776284","DOI":"10.1109\/CGO.2003.1191546"},{"key":"1098_CR50","unstructured":"Tsilikas G, Fleury M (2004) Matrix multiplication performance on commodity shared-memory multiprocessors. In: Proceedings of the international conference on parallel computing in electrical engineering, PARELEC \u201904. IEEE Computer Society, Washington, DC, USA, pp 13\u201318. doi: 10.1109\/PARELEC.2004.43"},{"key":"1098_CR51","doi-asserted-by":"crossref","unstructured":"Whaley RC, Dongarra J (1997) Automatically tuned linear algebra software. Tech. Rep. UT-CS-97-366, University of Tennessee","DOI":"10.1109\/SC.1998.10004"},{"key":"1098_CR52","unstructured":"Whaley RC, Dongarra J J (1998) Automatically tuned linear algebra software. In: Proceedings of the 1998 ACM\/IEEE conference on supercomputing, Supercomputing \u201998. IEEE Computer Society, San Jose, CA, pp 1\u201327"},{"key":"1098_CR53","doi-asserted-by":"crossref","unstructured":"Whaley RC, Dongarra J (1999) Automatically tuned linear algebra software. In: Ninth SIAM conference on parallel processing for scientific computing. CD-ROM proceedings","DOI":"10.1109\/SC.1998.10004"},{"issue":"2","key":"1098_CR54","doi-asserted-by":"crossref","first-page":"101","DOI":"10.1002\/spe.626","volume":"35","author":"RC Whaley","year":"2005","unstructured":"Whaley RC, Petitet A (2005) Minimizing development and maintenance costs in supporting persistently optimized BLAS. Softw Pract Exp 35(2):101\u2013121","journal-title":"Softw Pract Exp"},{"issue":"1\u20132","key":"1098_CR55","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1016\/S0167-8191(00)00087-9","volume":"27","author":"RC Whaley","year":"2001","unstructured":"Whaley RC, Petitet A, Dongarra JJ (2001) Automated empirical optimization of software and the ATLAS project. Parallel Comput 27(1\u20132):3\u201335","journal-title":"Parallel Comput"},{"key":"1098_CR56","doi-asserted-by":"crossref","unstructured":"Yotov K, Li X, Ren G, Garzaran M, Padua D, Pingali K, Stodghill P (2005) Is search really necessary to generate high-performance blas? Proceedings of the IEEE 93(2)","DOI":"10.1109\/JPROC.2004.840444"},{"key":"1098_CR57","doi-asserted-by":"crossref","unstructured":"Yuan N, Zhou Y, Tan G, Zhang J., Fan D (2009) High performance matrix multiplication on many cores. In: Proceedings of the 15th international Euro-Par conference on parallel processing, Euro-Par \u201909. Springer-Verlag, Berlin, pp. 948\u2013959. doi: 10.1007\/978-3-642-03869-3_87","DOI":"10.1007\/978-3-642-03869-3_87"},{"key":"1098_CR58","doi-asserted-by":"crossref","unstructured":"Zhuravlev S, Saez JC, Fedorova A, Prieto M (2012) Survey of scheduling techniques for addressing shared resources in multicore processors. ACM Comput Surv 45(1):1\u201328","DOI":"10.1145\/2379776.2379780"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1098-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-014-1098-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1098-9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,6]],"date-time":"2019-08-06T20:39:07Z","timestamp":1565123947000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-014-1098-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,1,21]]},"references-count":58,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2014,6]]}},"alternative-id":["1098"],"URL":"https:\/\/doi.org\/10.1007\/s11227-014-1098-9","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,1,21]]}}}