{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T06:24:59Z","timestamp":1757312699330,"version":"3.37.3"},"reference-count":25,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2022,10,26]],"date-time":"2022-10-26T00:00:00Z","timestamp":1666742400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,10,26]],"date-time":"2022-10-26T00:00:00Z","timestamp":1666742400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key R &D Program of China","doi-asserted-by":"crossref","award":["2018YFB0204404"],"award-info":[{"award-number":["2018YFB0204404"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["CCF Trans. HPC"],"published-print":{"date-parts":[[2023,3]]},"DOI":"10.1007\/s42514-022-00128-6","type":"journal-article","created":{"date-parts":[[2022,10,26]],"date-time":"2022-10-26T11:03:34Z","timestamp":1666782214000},"page":"84-96","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Evolving the HPL benchmark towards multi-GPGPU clusters"],"prefix":"10.1007","volume":"5","author":[{"given":"Qiao","family":"Sun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenjing","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiachang","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huiyuan","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,10,26]]},"reference":[{"key":"128_CR1","doi-asserted-by":"publisher","unstructured":"Awan, A.A., Hamidouche, K., Venkatesh, A., Panda, D.K.: Efficient large message broadcast using nccl and cuda-aware mpi for deep learning. In: Proceedings of the 23rd European MPI Users\u2019 Group Meeting, pp 15\u201322 (2016). https:\/\/doi.org\/10.1145\/2966884.2966912","DOI":"10.1145\/2966884.2966912"},{"issue":"3\u20134","key":"128_CR2","doi-asserted-by":"publisher","first-page":"153","DOI":"10.1007\/s00450-011-0161-5","volume":"26","author":"M Bach","year":"2011","unstructured":"Bach, M., Kretz, M., Lindenstruth, V., Rohr, D.: Optimized hpl for amd gpu and multi-core cpu usage. Comput. Sci. 26(3\u20134), 153\u2013164 (2011). https:\/\/doi.org\/10.1007\/s00450-011-0161-5","journal-title":"Comput. Sci."},{"key":"128_CR3","unstructured":"BLAS: Basic linear algebra subprograms. http:\/\/www.netlib.org\/blas\/ (2021)"},{"key":"128_CR4","unstructured":"cuBLAS: the CUDA basic linear algebra subroutine library. https:\/\/developer.nvidia.com\/cublas (2010)"},{"key":"128_CR5","unstructured":"CUDA: Compute unified device architecture. https:\/\/developer.nvidia.com\/cuda-downloads (2022)"},{"key":"128_CR6","unstructured":"Dongarra, J., Luszczek, P.: TOP500. https:\/\/www.top500.org\/ (2011)"},{"key":"128_CR7","doi-asserted-by":"publisher","first-page":"803","DOI":"10.1002\/cpe.728","volume":"15","author":"J Dongarra","year":"2003","unstructured":"Dongarra, J., Luszczek, P., Petitet, A.: The linpack benchmark: past, present and future. Concurr. Comput. Pract. Exp. 15, 803\u2013820 (2003). https:\/\/doi.org\/10.1002\/cpe.728","journal-title":"Concurr. Comput. Pract. Exp."},{"key":"128_CR8","doi-asserted-by":"publisher","unstructured":"Endo, T., Matsuoka, S., Nukada, A., Maruyama, N.: Linpack evaluation on a supercomputer with heterogeneous accelerators. In: 2010 IEEE International Symposium on Parallel and Distributed Processing (IPDPS), pp. 1\u20138 (2010). https:\/\/doi.org\/10.1109\/IPDPS.2010.5470353","DOI":"10.1109\/IPDPS.2010.5470353"},{"key":"128_CR9","doi-asserted-by":"publisher","unstructured":"Fatica, M.: Accelerating linpack with cuda on heterogenous clusters, pp 46\u201351 (2009). https:\/\/doi.org\/10.1145\/1513895.1513901","DOI":"10.1145\/1513895.1513901"},{"key":"128_CR10","doi-asserted-by":"publisher","unstructured":"Heinecke, A.: et\u00a0al. Design and implementation of the linpack benchmark for single and multi-node systems based on intel \u00aexeon phi coprocessor. In: 2013 IEEE 27th International Symposium on Parallel and Distributed Processing, pp. 126\u2013137 (2013). https:\/\/doi.org\/10.1109\/IPDPS.2013.113","DOI":"10.1109\/IPDPS.2013.113"},{"key":"128_CR11","doi-asserted-by":"crossref","unstructured":"Jia, Y., Luszczek, P., Dongarra, J.: Multi-gpu implementation of lu factorization. Proc. Comput. Sci. 9(106\u2013115), (2012). https:\/\/doi.org\/10.1016\/j.procs.2012.04.012, Proceedings of the International Conference on Computational Science, ICCS","DOI":"10.1016\/j.procs.2012.04.012"},{"issue":"7","key":"128_CR12","doi-asserted-by":"publisher","first-page":"1814","DOI":"10.1109\/TPDS.2014.2321742","volume":"26","author":"G Jo","year":"2015","unstructured":"Jo, G., Nah, J., Lee, J., Kim, J., Lee, J.: Accelerating linpack with mpi-opencl on clusters of multi-gpu nodes. IEEE Trans. Parallel Distrib. Syst. 26(7), 1814\u20131825 (2015). https:\/\/doi.org\/10.1109\/TPDS.2014.2321742","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"issue":"8","key":"128_CR13","doi-asserted-by":"publisher","first-page":"1613","DOI":"10.1109\/TPDS.2012.242","volume":"24","author":"J Kurzak","year":"2013","unstructured":"Kurzak, J., Luszczek, P., Faverge, M., Dongarra, J.: Lu factorization with partial pivoting for a multicore system with accelerators. IEEE Trans. Parallel Distrib. Syst. 24(8), 1613\u20131621 (2013). https:\/\/doi.org\/10.1109\/TPDS.2012.242","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"128_CR14","unstructured":"Meet two of the most powerful supercomputers on the planet. https:\/\/www.ibm.com\/thought-leadership\/summit-supercomputer\/ (2021)"},{"key":"128_CR15","unstructured":"NUMA support for linux. https:\/\/github.com\/numactl (2022)"},{"key":"128_CR16","unstructured":"Open MPI: Version 4.0. https:\/\/www.open-mpi.org\/software\/ompi\/v4.0\/ (2022)"},{"key":"128_CR17","unstructured":"OpenCL: Open computing language. https:\/\/opencl.org\/ (2022)"},{"key":"128_CR18","unstructured":"rocBLAS: Next generation BLAS implementation for ROCm platform. https:\/\/github.com\/ROCmSoftwarePlatform\/rocBLAS (2022)"},{"key":"128_CR19","unstructured":"ROCm developer tools and programing languages. https:\/\/github.com\/ROCm-Developer-Tools (2022)"},{"key":"128_CR20","doi-asserted-by":"publisher","unstructured":"Shui, C., et al.: Revisiting linpack algorithm on large-scale cpu-gpu heterogeneous systems, pp. 411\u2013412 (2020). https:\/\/doi.org\/10.1145\/3332466.3374530","DOI":"10.1145\/3332466.3374530"},{"key":"128_CR21","unstructured":"Supercomputer Fugaku: Fujitsu Global. https:\/\/www.fujitsu.com\/global\/about\/innovation\/fugaku\/ (2021)"},{"issue":"4","key":"128_CR22","doi-asserted-by":"publisher","first-page":"1065","DOI":"10.1137\/S0895479896297744","volume":"18","author":"S Toledo","year":"1997","unstructured":"Toledo, S.: Locality of reference in LU decomposition with partial pivoting. Siam J. Matrix Anal. Appl. 18(4), 1065\u20131081 (1997). https:\/\/doi.org\/10.1137\/S0895479896297744","journal-title":"Siam J. Matrix Anal. Appl."},{"issue":"2","key":"128_CR23","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1109\/MCSE.2017.25","volume":"19","author":"E Track","year":"2017","unstructured":"Track, E., Forbes, N., Strawn, G.: The end of Moore\u2019s law. Comput. Sci. Eng. 19(2), 4\u20136 (2017). https:\/\/doi.org\/10.1109\/MCSE.2017.25","journal-title":"Comput. Sci. Eng."},{"issue":"3","key":"128_CR24","doi-asserted-by":"publisher","first-page":"14:1","DOI":"10.1145\/2764454","volume":"41","author":"FGV Zee","year":"2015","unstructured":"Zee, F.G.V., van de Geijn, R.A.: BLIS: a framework for rapidly instantiating BLAS functionality. ACM Trans. Math. Softw. 41(3), 14:1-14:33 (2015). https:\/\/doi.org\/10.1145\/2764454","journal-title":"ACM Trans. Math. Softw."},{"key":"128_CR25","doi-asserted-by":"publisher","DOI":"10.1360\/crad20060328","author":"W Zhang","year":"2006","unstructured":"Zhang, W.: Emulation and forecast of hpl test performance. J. Comput. Res. Dev. (2006). https:\/\/doi.org\/10.1360\/crad20060328","journal-title":"J. Comput. Res. Dev."}],"container-title":["CCF Transactions on High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-022-00128-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42514-022-00128-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-022-00128-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,3,26]],"date-time":"2023-03-26T21:40:11Z","timestamp":1679866811000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42514-022-00128-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,26]]},"references-count":25,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2023,3]]}},"alternative-id":["128"],"URL":"https:\/\/doi.org\/10.1007\/s42514-022-00128-6","relation":{},"ISSN":["2524-4922","2524-4930"],"issn-type":[{"type":"print","value":"2524-4922"},{"type":"electronic","value":"2524-4930"}],"subject":[],"published":{"date-parts":[[2022,10,26]]},"assertion":[{"value":"30 October 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 September 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 October 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}