{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:19:40Z","timestamp":1740122380659,"version":"3.37.3"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2021,4,12]],"date-time":"2021-04-12T00:00:00Z","timestamp":1618185600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2021,4,12]],"date-time":"2021-04-12T00:00:00Z","timestamp":1618185600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100004085","name":"Ministry of Education, Science and Technology","doi-asserted-by":"crossref","award":["NRF-2015M3C4A7065662"],"award-info":[{"award-number":["NRF-2015M3C4A7065662"]}],"id":[{"id":"10.13039\/501100004085","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cluster Comput"],"published-print":{"date-parts":[[2023,10]]},"DOI":"10.1007\/s10586-021-03274-8","type":"journal-article","created":{"date-parts":[[2021,4,12]],"date-time":"2021-04-12T06:02:30Z","timestamp":1618207350000},"page":"2539-2549","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Improving blocked matrix-matrix multiplication routine by utilizing AVX-512 instructions on intel knights landing and xeon scalable processors"],"prefix":"10.1007","volume":"26","author":[{"given":"Yoosang","family":"Park","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Raehyun","family":"Kim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Thi My Tuyen","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7321-9682","authenticated-orcid":false,"given":"Jaeyoung","family":"Choi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,4,12]]},"reference":[{"key":"3274_CR1","unstructured":"High-Performance Computing (HPC). https:\/\/www.nics.tennessee.edu\/computing-resources\/what-is-hpc. Accessed 22 Nov 2020"},{"key":"3274_CR2","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1177\/1094342015597083","volume":"31","author":"A Geist","year":"2017","unstructured":"Geist, A., Reed, D.A.: A survey of high-performance computing scaling challenges. Int. J. High Perform. Comput. Appl. 31, 104\u2013113 (2017)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"3274_CR3","doi-asserted-by":"publisher","first-page":"1422","DOI":"10.1007\/s11227-018-2238-4","volume":"74","author":"P Thoman","year":"2018","unstructured":"Thoman, P., Dichev, K., Heller, T., Iakymchuk, R., Aguilar, X., Hasanov, K., Gschwandtner, P., Lemarinier, P., Markidis, S., Jordan, H., Fahringer, T., Katrinis, K., Laure, E., Nikolopoulos, D.S.: A taxonomy of task-based parallel programming technologies for high-performance computing. J. SuperComput. 74, 1422\u20131434 (2018)","journal-title":"J. SuperComput."},{"key":"3274_CR4","unstructured":"Basic Linear Algebra Subprograms (BLAS). http:\/\/www.netlib.org\/blas. Accessed 22 Nov 2020."},{"key":"3274_CR5","unstructured":"Parallel Basic Linear Algebra Subprograms (PBLAS). http:\/\/www.netlib.org\/scalapack\/pblas_qref.html. Accessed 22 Nov 2020."},{"key":"3274_CR6","doi-asserted-by":"crossref","unstructured":"Filippone, S.: Parallel libraries on distributed memory architectures: The IBM Parallel ESSL. In: Wa\u015bniewski, J., Dongarra, J., Madsen, K., Olesen, D. (eds.) Applied Parallel Computing Industrial Computation and Optimization, pp.247-255. Springer (1996)","DOI":"10.1007\/3-540-62095-8_26"},{"key":"3274_CR7","unstructured":"ScaLAPACK. http:\/\/www.netlib.org\/scalapack. Accessed 22 Nov 2020."},{"key":"3274_CR8","unstructured":"Intel Math Kernel Library (MKL). https:\/\/software.intel.com\/content\/www\/us\/en\/develop\/tools\/math-kernel-library.html. Accessed 22 Nov 2020."},{"key":"3274_CR9","doi-asserted-by":"publisher","unstructured":"Yan, D., Wang, W., Chu, X.: Optimizing batched winograd convolution on GPUs. Proceedings of the 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (PPoPP \u201920). https:\/\/doi.org\/10.1145\/3332466.3374520","DOI":"10.1145\/3332466.3374520"},{"key":"3274_CR10","doi-asserted-by":"publisher","first-page":"359","DOI":"10.1007\/s10586-019-02927-z","volume":"23","author":"S Catal\u00e1n","year":"2020","unstructured":"Catal\u00e1n, S., Castell\u00f3, A., Igual, F.D., Rodr\u00edguez-S\u00e1nchez, R., Quintana-Ort\u00ed, E.S.: Programming parallel dense matrix factorizations with look-ahead and OpenMP. Cluster Comput. 23, 359\u2013375 (2020)","journal-title":"Cluster Comput."},{"key":"3274_CR11","doi-asserted-by":"publisher","DOI":"10.1145\/3378671","author":"G Frison","year":"2020","unstructured":"Frison, G., Sartor, T., Zanelli, A., Diehl, M.: The BLAS API of BLASFEO: optimizing performance for small matrices. ACM Transact. Mathem. Software (2020). https:\/\/doi.org\/10.1145\/3378671","journal-title":"ACM Transact. Mathem. Software"},{"key":"3274_CR12","doi-asserted-by":"publisher","DOI":"10.1145\/3434402","author":"PS Labini","year":"2021","unstructured":"Labini, P.S., Cianfriglia, M., Perri, D., Gervasi, O., Fursin, G., Lokhmotov, A., Nugteren, C., Carpentieri, B., Zollo, F., Vella, F.: On the anatomy of predictive models for accelerating GPU convolution kernels and beyond. ACM Transact. Architect. Code Optimiz. (2021). https:\/\/doi.org\/10.1145\/3434402","journal-title":"ACM Transact. Architect. Code Optimiz."},{"key":"3274_CR13","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/5712.001.0001","volume-title":"PVM: Parallel virtual machine: a users' guide and tutorial for networked parallel computing","author":"A Geist","year":"1994","unstructured":"Geist, A., Beguelin, A., Dongarra, J., Jiang, W., Manchek, R., Sunderam, V.: PVM: Parallel virtual machine: a users\u2019 guide and tutorial for networked parallel computing. MIT Press, Cambridge, MA (1994)"},{"key":"3274_CR14","doi-asserted-by":"publisher","unstructured":"Kotsifakou, M., Srivastava, P., Sinclair, M.D., Komuravelli, R., Adve, V., Adve, S.: HPVM: Heterogeneous parallel virtual machine. proceedings of the 23rd ACM SIGPLAN symposium on principles and practice of parallel programming (2018). https:\/\/doi.org\/10.1145\/3178487.3178493","DOI":"10.1145\/3178487.3178493"},{"key":"3274_CR15","doi-asserted-by":"crossref","unstructured":"Hempel, R.: The MPI standard for message passing. In: Gentzsch, W. Harms, U. (eds.) High-Performance Computing and Networking, pp. 247-252. Springer, (1994)","DOI":"10.1007\/3-540-57981-8_126"},{"key":"3274_CR16","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2016.38","author":"J Zhang","year":"2016","unstructured":"Zhang, J., Lu, X., Panda, D.K.: High performance mpi library for container-based HPC cloud on InfiniBand clusters. Int. Conf. Parallel Process. (2016). https:\/\/doi.org\/10.1109\/ICPP.2016.38","journal-title":"Int. Conf. Parallel Process."},{"key":"3274_CR17","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1109\/99.660313","volume":"5","author":"L Dagum","year":"1998","unstructured":"Dagum, L., Menon, R.: OpenMP: an industry standard API for shared-memory programming. IEEE Comput. Sci. Eng. 5, 46\u201355 (1998)","journal-title":"IEEE Comput. Sci. Eng."},{"key":"3274_CR18","doi-asserted-by":"publisher","first-page":"404","DOI":"10.1109\/TPDS.2008.105","volume":"20","author":"E Ayguade","year":"2008","unstructured":"Ayguade, E., Copty, N., Duran, A., Hoeflinger, J., Lin, Y., Massaioli, F., Teruel, X., Unnikrishnan, P., Zhang, H.: The design of OpenMP tasks. IEEE Transact. Parallel Distr. Syst. 20, 404\u2013418 (2008)","journal-title":"IEEE Transact. Parallel Distr. Syst."},{"key":"3274_CR19","doi-asserted-by":"publisher","first-page":"e5728","DOI":"10.1002\/cpe.5728","volume":"32","author":"M Diener","year":"2020","unstructured":"Diener, M., Kale, L.V., Bodony, D.J.: Heterogeneous computing with OpenMP and Hydra. Concurr. Comput.: Practice Exp. 32, e5728 (2020)","journal-title":"Concurr. Comput.: Practice Exp."},{"key":"3274_CR20","doi-asserted-by":"publisher","unstructured":"Sampath, S., Sagar, B.B., Nanjesh, B.R.: Performance evaluation and comparison of MPI and PVM using a cluster based parallel computing architecture. International Conference on Circuits, Power and Computing Technologies (2013). https:\/\/doi.org\/10.1109\/ICCPCT.2013.6529020","DOI":"10.1109\/ICCPCT.2013.6529020"},{"key":"3274_CR21","doi-asserted-by":"crossref","unstructured":"Lusk, E., Chan, A.: Early Experiments with the OpenMP\/MPI Hybrid Programming model. In: Eigenmann, R., de Supinski, B.R. (eds.) OpenMP in a New Era of Parallelism, pp. 36-47. Springer (2008)","DOI":"10.1007\/978-3-540-79561-2_4"},{"key":"3274_CR22","volume-title":"Intel Xeon Phi Processor High Performance Programming: Knights","author":"J Jeffers","year":"2016","unstructured":"Jeffers, J., Reinders, J., Sodani, A.: Intel Xeon Phi Processor High Performance Programming: Knights, Landing Morgan Kaufmann, Burlington, Massachusetts (2016)","edition":"Landing"},{"key":"3274_CR23","unstructured":"Zhao, Z., Marsman, M., Wende, F., Kim, J.: Performance of hybrid MPI\/OpenMP VASP on Cray XC40 based on Intel Knights landing many integrated core architecture. Cray User Group Proceedings (2017)"},{"key":"3274_CR24","unstructured":"Basic Linear Algebra Communication Subprograms (BLACS). http:\/\/www.netlib.org\/blacs. Accessed 22 Nov 2020"},{"key":"3274_CR25","doi-asserted-by":"publisher","unstructured":"Walker, D., Sawyer, W., Deshpande, V.: An MPI implementation of the BLACS. Proceedings of 3rd International Conference on High Performance Computing (1996). https:\/\/doi.org\/10.1109\/HIPC.1996.565864","DOI":"10.1109\/HIPC.1996.565864"},{"key":"3274_CR26","doi-asserted-by":"publisher","first-page":"791","DOI":"10.1007\/s00607-016-0537-2","volume":"99","author":"C Chen","year":"2017","unstructured":"Chen, C., Fang, J., Tang, T., Yang, C.: LU factorization on heterogeneous systems: an energy-efficient appraoch towards high performance. Computing 99, 791\u2013811 (2017)","journal-title":"Computing"},{"key":"3274_CR27","doi-asserted-by":"publisher","unstructured":"Nagasaka, Y., Matsuoka, S., Azad, A., Buluc, A.: High-Performance Sparse Matrix-Matrix Products on Intel KNL and Multicore Architectures. Proceedings of the 47th International Conference on Parallel Processing Companion (2018). https:\/\/doi.org\/10.1145\/3229710.3229720","DOI":"10.1145\/3229710.3229720"},{"key":"3274_CR28","doi-asserted-by":"publisher","first-page":"1785","DOI":"10.1007\/s10586-018-2810-y","volume":"21","author":"R Lim","year":"2018","unstructured":"Lim, R., Lee, Y., Kim, R., Choi, J.: An implementation of matrix-matrix multiplication on the Intel KNL processor with AVX-512. Cluster Comput. 21, 1785\u20131795 (2018)","journal-title":"Cluster Comput."},{"key":"3274_CR29","doi-asserted-by":"publisher","first-page":"7895","DOI":"10.1007\/s11227-018-2702-1","volume":"75","author":"R Lim","year":"2019","unstructured":"Lim, R., Lee, Y., Kim, R., Choi, J., Lee, M.: Auto-tuning GEMM kernels on the Intel KNL and Intel Skylake-SP processors. J. Supercomput. 75, 7895\u20137908 (2019)","journal-title":"J. Supercomput."},{"key":"3274_CR30","unstructured":"Zhang, X., Wang, Q., Werber, S.: Openblas. http:\/\/www.openblas.net. Accessed 22 Nov. 2020"},{"key":"3274_CR31","first-page":"321","volume":"27","author":"RC Whaley","year":"2001","unstructured":"Whaley, R.C., Petitet, A., Dongarra, J.: Automated empirical optimizations of software and the atlas project. Parallel Comput. 27, 321\u2013354 (2001)","journal-title":"Parallel Comput."},{"key":"3274_CR32","doi-asserted-by":"publisher","DOI":"10.1145\/2764454","author":"FG van Zee","year":"2015","unstructured":"van Zee, F.G., van de Geijn, R.A.: BLIS: a framework for rapidly instantiating BLAS functionality. ACM Transact. Mathem. Softw. (2015). https:\/\/doi.org\/10.1145\/2764454","journal-title":"ACM Transact. Mathem. Softw."},{"key":"3274_CR33","doi-asserted-by":"publisher","first-page":"255","DOI":"10.1002\/(SICI)1096-9128(199704)9:4<255::AID-CPE250>3.0.CO;2-2","volume":"94","author":"RA van de Geijn","year":"1997","unstructured":"van de Geijn, R.A., Watts, J.: SUMMA: scalable universal matrix multiplication algorithm. Concurr. Practice Exp. 94, 255\u2013274 (1997)","journal-title":"Concurr. Practice Exp."},{"key":"3274_CR34","doi-asserted-by":"publisher","unstructured":"Choi, J.: A fast scalable universal matrix multiplication algorithm on distributed-memory concurrent computers. Proceedings of 11th International Parallel Processing Symposium (1997). https:\/\/doi.org\/10.1109\/IPPS.1997.580916","DOI":"10.1109\/IPPS.1997.580916"},{"key":"3274_CR35","doi-asserted-by":"publisher","first-page":"655","DOI":"10.1002\/(SICI)1096-9128(199807)10:8<655::AID-CPE369>3.0.CO;2-O","volume":"10","author":"J Choi","year":"1998","unstructured":"Choi, J.: A new parallel matrix multiplication algorithm on distributed-memory concurrent computers. Concurr. Practice Exp. 10, 655\u2013670 (1998)","journal-title":"Concurr. Practice Exp."},{"key":"3274_CR36","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1356052.1356053","volume":"34","author":"K Goto","year":"2008","unstructured":"Goto, K., van de Geijn, R.A.: Anatomy of high-performance matrix multiplication. ACM Transact. Mathem. Softw. 34, 1\u201325 (2008). https:\/\/doi.org\/10.1145\/1356052.1356053","journal-title":"ACM Transact. Mathem. Softw."},{"key":"3274_CR37","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45545-0_15","author":"JA Gunnels","year":"2001","unstructured":"Gunnels, J.A., Henry, G.M., van de Geijn, R.A.: A family of high-performance matrix multiplication algorithms. Int. Conf. Comput. Sci. (2001). https:\/\/doi.org\/10.1007\/3-540-45545-0_15","journal-title":"Int. Conf. Comput. Sci."},{"key":"3274_CR38","doi-asserted-by":"publisher","DOI":"10.1145\/3293320.3293334","author":"R Kim","year":"2019","unstructured":"Kim, R., Choi, J., Lee, M.: Optimizing parallel GEMM routines using auto-tuning with Intel AVX-512. Int. Conf. High Perf. Comput. Asia-Pac. Region (2019). https:\/\/doi.org\/10.1145\/3293320.3293334","journal-title":"Int. Conf. High Perf. Comput. Asia-Pac. Region"},{"key":"3274_CR39","doi-asserted-by":"publisher","DOI":"10.1145\/3176364.3176374","author":"R Lim","year":"2018","unstructured":"Lim, R., Lee, Y., Kim, R., Choi, J.: OpenMP-based parallel implementation of matrix-matrix multiplication on the intel knights landing. Workshops of HPC Asia (2018). https:\/\/doi.org\/10.1145\/3176364.3176374","journal-title":"Workshops of HPC Asia"},{"key":"3274_CR40","unstructured":"Recommended value of block size for Intel processor. https:\/\/software.intel.com\/content\/www\/us\/en\/develop\/documentation\/mkl-linux-developer-guide\/top\/intel-math-kernel-library-benchmarks\/intel-distribution-for-linpack-benchmark\/configuring-parameters.html. Accessed 22 Nov 2020."}],"container-title":["Cluster Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-021-03274-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10586-021-03274-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-021-03274-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,26]],"date-time":"2023-08-26T20:20:54Z","timestamp":1693081254000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10586-021-03274-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,4,12]]},"references-count":40,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2023,10]]}},"alternative-id":["3274"],"URL":"https:\/\/doi.org\/10.1007\/s10586-021-03274-8","relation":{},"ISSN":["1386-7857","1573-7543"],"issn-type":[{"type":"print","value":"1386-7857"},{"type":"electronic","value":"1573-7543"}],"subject":[],"published":{"date-parts":[[2021,4,12]]},"assertion":[{"value":"25 November 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 March 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 March 2021","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 April 2021","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}