{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:38:41Z","timestamp":1740123521796,"version":"3.37.3"},"reference-count":29,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2021,5,3]],"date-time":"2021-05-03T00:00:00Z","timestamp":1620000000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,5,3]],"date-time":"2021-05-03T00:00:00Z","timestamp":1620000000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["109000$"],"award-info":[{"award-number":["109000$"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Guangzhou Produce & Research Fund","award":["298800$"],"award-info":[{"award-number":["298800$"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2021,12]]},"DOI":"10.1007\/s11227-021-03829-x","type":"journal-article","created":{"date-parts":[[2021,5,3]],"date-time":"2021-05-03T10:02:58Z","timestamp":1620036178000},"page":"13739-13756","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Design of a simulation model for high performance LINPACK in hybrid CPU-GPU systems"],"prefix":"10.1007","volume":"77","author":[{"given":"Yichang","family":"Hu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lu","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,5,3]]},"reference":[{"issue":"2","key":"3829_CR1","doi-asserted-by":"publisher","first-page":"57","DOI":"10.4018\/jdst.2010040104","volume":"1","author":"H Adalsteinsson","year":"2010","unstructured":"Adalsteinsson H, Cranford S, Evensky DA, Kenny JP, Mayo J, Pinar A, Janssen CL (2010) A simulator for large-scale parallel computer architectures. Int J Distrib Syst Technol 1(2):57\u201373. https:\/\/doi.org\/10.4018\/jdst.2010040104","journal-title":"Int J Distrib Syst Technol"},{"key":"3829_CR2","unstructured":"AMD (2017) Hpl-rocm. https:\/\/github.com\/rocmarchive\/HPL-ROCm"},{"key":"3829_CR3","doi-asserted-by":"publisher","unstructured":"Ben-Nun T, Sutton M, Pai S, Pingali K (2017) Groute: An Asynchronous Multi-GPU Programming Model for Irregular Computations. In: Proceedings of the 22nd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, Association for Computing Machinery, New York, NY, USA, PPoPP \u201917, pp 235\u2013248, https:\/\/doi.org\/10.1145\/3018743.3018756,","DOI":"10.1145\/3018743.3018756"},{"issue":"8","key":"3829_CR4","doi-asserted-by":"publisher","first-page":"791","DOI":"10.1007\/s00607-016-0537-2","volume":"99","author":"C Chen","year":"2017","unstructured":"Chen C, Fang J, Tang T, Yang C (2017) LU factorization on heterogeneous systems: an energy-efficient approach towards high performance. Computing 99(8):791\u2013811. https:\/\/doi.org\/10.1007\/s00607-016-0537-2","journal-title":"Computing"},{"key":"3829_CR5","unstructured":"Cornebize T, Heinrich FC, Legrand A, Vienne J (2017) Emulating High Performance Linpack on a Commodity Server at the Scale of a Supercomputer, https:\/\/hal.inria.fr\/hal-01654804, working paper or preprint"},{"key":"3829_CR6","doi-asserted-by":"publisher","unstructured":"Cornebize T, Legrand A, Heinrich FC (2019) Fast and Faithful Performance Prediction of MPI Applications: the HPL Case Study. In: 2019 IEEE International Conference on Cluster Computing (CLUSTER), pp 1\u201311, https:\/\/doi.org\/10.1109\/CLUSTER.2019.8891011","DOI":"10.1109\/CLUSTER.2019.8891011"},{"key":"3829_CR7","doi-asserted-by":"publisher","unstructured":"Davies T, Karlsson C, Liu H, Ding C, Chen Z (2011) High Performance Lipack Benchmark: A Fault Tolerant Implementation Without Checkpointing. In: Proceedings of the International Conference on Supercomputing, Association for Computing Machinery, New York, NY, USA, ICS \u201911, p 162\u2013171, https:\/\/doi.org\/10.1145\/1995896.1995923","DOI":"10.1145\/1995896.1995923"},{"issue":"8","key":"3829_CR8","doi-asserted-by":"publisher","first-page":"2387","DOI":"10.1109\/TPDS.2017.2669305","volume":"28","author":"A Degomme","year":"2017","unstructured":"Degomme A, Legrand A, Markomanolis GS, Quinson M, Stillwell M, Suter F (2017) Simulating MPI applications: the SMPI approach. IEEE Trans Parall Distrib Syst 28(8):2387\u20132400. https:\/\/doi.org\/10.1109\/TPDS.2017.2669305","journal-title":"IEEE Trans Parall Distrib Syst"},{"key":"3829_CR9","unstructured":"Dittmer S, Kluth T, Henriksen MTR, Maass P (2020) Deep image prior for 3d magnetic particle imaging: a quantitative comparison of regularization techniques on open mpi dataset. arXiv:2007.01593"},{"issue":"4","key":"3829_CR10","doi-asserted-by":"publisher","first-page":"42102","DOI":"10.1007\/s11432-017-9221-0","volume":"61","author":"X Gan","year":"2018","unstructured":"Gan X, Hu Y, Liu J, Chi L, Xu H, Gong C, Li S, Yan Y (2018) Customizing the HPL for China accelerator. Sci China Inf Sci 61(4):42102. https:\/\/doi.org\/10.1007\/s11432-017-9221-0","journal-title":"Sci China Inf Sci"},{"key":"3829_CR11","unstructured":"Haitao Zhao Leisheng Li, Wenhao Yang, Hui Zhao, Huiyuan Li JS (2020) Research on HPL parallelcComputing model for a class of complex heterogeneous supercomputer system. http:\/\/www.jfdc.cnic.cn"},{"issue":"11","key":"3829_CR12","doi-asserted-by":"publisher","first-page":"2034","DOI":"10.3390\/app8112034","volume":"8","author":"M Hemmatpour","year":"2018","unstructured":"Hemmatpour M, Montrucchio B, Rebaudengo M (2018) Communicating efficiently on cluster-based remote direct memory access (RDMA) over infiniband protocol. Appl Sci 8(11):2034","journal-title":"Appl Sci"},{"key":"3829_CR13","doi-asserted-by":"publisher","unstructured":"Hjelm N, Pritchard H, Guti\u00e9rrez SK, Holmes DJ, Castain R, Skjellum A (2019) MPI Sessions: Evaluation of an Implementation in Open MPI. In: 2019 IEEE International Conference on Cluster Computing (CLUSTER), pp 1\u201311, https:\/\/doi.org\/10.1109\/CLUSTER.2019.8891002","DOI":"10.1109\/CLUSTER.2019.8891002"},{"key":"3829_CR14","doi-asserted-by":"publisher","unstructured":"Huang J, Lu L (2019) Performance Optimization of High-Performance Linpack Based on GPU-Centric Model on Heterogeneous Systems. In: 2019 IEEE International Conference on Parallel Distributed Processing with Applications, Big Data Cloud Computing, Sustainable Computing Communications, Social Computing Networking (ISPA\/BDCloud\/SocialCom\/SustainCom), pp 1371\u20131377, https:\/\/doi.org\/10.1109\/ISPA-BDCloud-SustainCom-SocialCom48970.2019.00197","DOI":"10.1109\/ISPA-BDCloud-SustainCom-SocialCom48970.2019.00197"},{"issue":"7","key":"3829_CR15","doi-asserted-by":"publisher","first-page":"1814","DOI":"10.1109\/TPDS.2014.2321742","volume":"26","author":"G Jo","year":"2015","unstructured":"Jo G, Nah J, Lee J, Kim J, Lee J (2015) Accelerating LINPACK with MPI-OpenCL oncClusters of multi-GPU nodes. IEEE Trans Parallel Distrib Syst 26(7):1814\u20131825","journal-title":"IEEE Trans Parallel Distrib Syst"},{"issue":"1","key":"3829_CR16","doi-asserted-by":"publisher","first-page":"e4298","DOI":"10.1002\/cpe.4298","volume":"30","author":"J Kwack","year":"2018","unstructured":"Kwack J, Bauer GH (2018) HPCG and HPGMG benchmark tests on multiple program, multiple data (MPMD) mode on Blue Waters\u2013A Cray XE6\/XK7 hybrid system. Concurr Comput: Pract Exp 30(1):e4298. https:\/\/doi.org\/10.1002\/cpe.4298","journal-title":"Concurr Comput: Pract Exp"},{"key":"3829_CR17","doi-asserted-by":"publisher","DOI":"10.1007\/s11227-020-03319-6","author":"F Lin","year":"2020","unstructured":"Lin F, Liu Y, Guo Y, Qian D (2020) ELS: Emulation system for debugging and tuning large-scale parallel programs on small clusters. J Supercomput. https:\/\/doi.org\/10.1007\/s11227-020-03319-6","journal-title":"J Supercomput"},{"issue":"8","key":"3829_CR18","doi-asserted-by":"publisher","first-page":"2810","DOI":"10.1109\/JSTARS.2019.2920077","volume":"12","author":"J Liu","year":"2019","unstructured":"Liu J, Xue Y, Ren K, Song J, Windmill C, Merritt P (2019) High-performance time-series quantitative retrieval from satellite images on a GPU cluster. IEEE J Sel Topics App Earth Observ Remote Sens 12(8):2810\u20132821. https:\/\/doi.org\/10.1109\/JSTARS.2019.2920077","journal-title":"IEEE J Sel Topics App Earth Observ Remote Sens"},{"issue":"1","key":"3829_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s13673-017-0124-3","volume":"8","author":"JP Martin","year":"2018","unstructured":"Martin JP, Kandasamy A, Chandrasekaran K (2018) Exploring the support for high performance applications in the container runtime environment. Human-centric Comput Inf Sci 8(1):1. https:\/\/doi.org\/10.1186\/s13673-017-0124-3","journal-title":"Human-centric Comput Inf Sci"},{"key":"3829_CR20","doi-asserted-by":"publisher","unstructured":"McCalpin JD (2018) HPL and DGEMM Performance Variability on the Xeon Platinum 8160 Processor. In: SC18: International Conference for High Performance Computing, Networking, Storage and Analysis, pp 225\u2013237, https:\/\/doi.org\/10.1109\/SC.2018.00021","DOI":"10.1109\/SC.2018.00021"},{"key":"3829_CR21","doi-asserted-by":"publisher","unstructured":"Mohammadi M, Bazhirov T (2018) Comparative Benchmarking of Cloud Computing Vendors with High Performance Linpack. In: Proceedings of the 2nd International Conference on High Performance Compilation, Computing and Communications, Association for Computing Machinery, New York, NY, USA, HP3C, pp 1\u20135, https:\/\/doi.org\/10.1145\/3195612.3195613","DOI":"10.1145\/3195612.3195613"},{"issue":"1","key":"3829_CR22","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1109\/TPDS.2016.2543725","volume":"28","author":"M Mubarak","year":"2017","unstructured":"Mubarak M, Carothers CD, Ross RB, Carns P (2017) Enabling parallel simulation of large-scale HPC network systems. IEEE Trans Parallel Distrib Syst 28(1):87\u2013100. https:\/\/doi.org\/10.1109\/TPDS.2016.2543725","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"3829_CR23","doi-asserted-by":"crossref","unstructured":"Rohr D, De Cuveland J, Lindenstruth V (2016) A Model for Weak Scaling to Many GPUs at the Basis of the Linpack Benchmark. In: 2016 IEEE International Conference on Cluster Computing (CLUSTER), pp 192\u2013202","DOI":"10.1109\/CLUSTER.2016.15"},{"key":"3829_CR24","unstructured":"V\u00e9gh J (2018) Limitations of performance of exascale applications and supercomputers they are running on. arXiv:1808.05338"},{"key":"3829_CR25","doi-asserted-by":"publisher","unstructured":"Yang C, Chen C, Tang T, Chen X, Fang J, Xue J (2016) An Energy-Efficient Implementation of LU Factorization on Heterogeneous Systems. In: 2016 IEEE 22nd International Conference on Parallel and Distributed Systems (ICPADS), pp 971\u2013979, https:\/\/doi.org\/10.1109\/ICPADS.2016.0130","DOI":"10.1109\/ICPADS.2016.0130"},{"key":"3829_CR26","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1016\/j.jpdc.2016.12.023","volume":"104","author":"W Yang","year":"2017","unstructured":"Yang W, Li K, Li K (2017) A hybrid computing method of spmv on cpu-gpu heterogeneous computing systems. J Parallel Distrib Comput 104:49\u201360","journal-title":"J Parallel Distrib Comput"},{"issue":"6","key":"3829_CR27","first-page":"1398","volume":"14","author":"C Yong","year":"2018","unstructured":"Yong C, Lee GW, Huh EN (2018) Proposal of container-based HPC structures and performance analysis. J Inf Process Syst 14(6):1398\u20131404","journal-title":"J Inf Process Syst"},{"key":"3829_CR28","unstructured":"Zhang Wenli and Fan Jianping CM (2004) Emulation and Forecast of HPL Test Performance. http:\/\/crad.ict.ac.cn"},{"key":"3829_CR29","doi-asserted-by":"publisher","unstructured":"Zheng G, Kakulapati G, Kale LV (2004) BigSim: A Parallel Simulator for Performance Prediction of Extremely Large Parallel Machines. In: 18th International Parallel and Distributed Processing Symposium, 2004. Proceedings., p\u00a078, https:\/\/doi.org\/10.1109\/IPDPS.2004.1303013","DOI":"10.1109\/IPDPS.2004.1303013"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-021-03829-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-021-03829-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-021-03829-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,11,16]],"date-time":"2021-11-16T11:20:46Z","timestamp":1637061646000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-021-03829-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,5,3]]},"references-count":29,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2021,12]]}},"alternative-id":["3829"],"URL":"https:\/\/doi.org\/10.1007\/s11227-021-03829-x","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"type":"print","value":"0920-8542"},{"type":"electronic","value":"1573-0484"}],"subject":[],"published":{"date-parts":[[2021,5,3]]},"assertion":[{"value":"18 April 2021","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 May 2021","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}