{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T18:53:37Z","timestamp":1771700017033,"version":"3.50.1"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2020,5,23]],"date-time":"2020-05-23T00:00:00Z","timestamp":1590192000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,5,23]],"date-time":"2020-05-23T00:00:00Z","timestamp":1590192000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"name":"National Key R&D Program of China","award":["2016YFB0200100"],"award-info":[{"award-number":["2016YFB0200100"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61732002"],"award-info":[{"award-number":["61732002"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2021,2]]},"DOI":"10.1007\/s11227-020-03319-6","type":"journal-article","created":{"date-parts":[[2020,5,23]],"date-time":"2020-05-23T13:02:39Z","timestamp":1590238959000},"page":"1635-1666","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["ELS: Emulation system for debugging and tuning large-scale parallel programs on small clusters"],"prefix":"10.1007","volume":"77","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1411-0115","authenticated-orcid":false,"given":"Fang","family":"Lin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yayu","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Depei","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,5,23]]},"reference":[{"key":"3319_CR1","unstructured":"CUDA-GDB homepage [online]. https:\/\/developer.nvidia.com\/cuda-gdb. Accessed 1 Aug 2019"},{"key":"3319_CR2","unstructured":"Distributed Debugging Tool (DDT) homepage [online]. https:\/\/developer.arm.com\/products\/software-development-tools\/hpc\/arm-forge. Accessed 11 July 2019"},{"key":"3319_CR3","unstructured":"Dyninst homepage [online]. https:\/\/dyninst.org\/. Accessed 18 Aug 2019"},{"key":"3319_CR4","unstructured":"GDB homepage [online]. http:\/\/www.gnu.org\/software\/gdb\/. Accessed 9 June 2019"},{"key":"3319_CR5","unstructured":"HPCTOOLKIT homepage [online]. http:\/\/hpctoolkit.org\/index.html. Accessed 20 Aug 2019"},{"key":"3319_CR6","unstructured":"MPI Documents [online]. https:\/\/www.mpi-forum.org\/docs\/. Accessed 14 Nov 2018"},{"key":"3319_CR7","unstructured":"MVAPICH homepage [online]. http:\/\/mvapich.cse.ohio-state.edu\/. Accessed 14 Nov 2018"},{"key":"3319_CR8","unstructured":"TAU homepage [online]. https:\/\/www.cs.uoregon.edu\/research\/tau\/home.php. Accessed 20 Aug 2019"},{"key":"3319_CR9","unstructured":"THE NAS PARALLEL BENCHMARKS [online]. https:\/\/www.nas.nasa.gov\/publications\/npb.html. Accessed 25 Dec 2018"},{"key":"3319_CR10","unstructured":"TotalView for HPC homepage [Online]. https:\/\/www.roguewave.com\/products-services\/totalview. Accessed 11 July 2019"},{"issue":"6","key":"3319_CR11","doi-asserted-by":"crossref","first-page":"685","DOI":"10.1002\/cpe.1553","volume":"22","author":"L Adhianto","year":"2010","unstructured":"Adhianto L, Banerjee S, Fagan M, Krentel M, Marin G, Mellor-Crummey J, Tallent NR (2010) Hpctoolkit: tools for performance analysis of optimized parallel programs. Concurr Comput Pract Exp 22(6):685\u2013701","journal-title":"Concurr Comput Pract Exp"},{"key":"3319_CR12","doi-asserted-by":"publisher","first-page":"230","DOI":"10.1016\/j.jpdc.2017.06.008","volume":"109","author":"A Bahmani","year":"2017","unstructured":"Bahmani A, Mueller F (2017) Scalable communication event tracing via clustering. J Parallel Distrib Comput 109:230\u2013244","journal-title":"J Parallel Distrib Comput"},{"key":"3319_CR13","doi-asserted-by":"crossref","unstructured":"Bouteiller A, Bosilca G, Dongarra J (2007) Retrospect: deterministic replay of MPI applications for interactive distributed debugging. In: European Parallel Virtual Machine\/Message Passing Interface Users\u2019 Group Meeting, Springer. pp 297\u2013306","DOI":"10.1007\/978-3-540-75416-9_41"},{"key":"3319_CR14","doi-asserted-by":"crossref","unstructured":"Clemencon C, Fritscher J, Meehan MJ, R\u00fchl R (1995) An implementation of race detection and deterministic replay with MPI. In: European Conference on Parallel Processing, Springer. pp 155\u2013166","DOI":"10.1007\/BFb0020462"},{"key":"3319_CR15","doi-asserted-by":"crossref","unstructured":"Danalis A, Marin G, McCurdy C, Meredith JS, Roth PC, Spafford K, Tipparaju V, Vetter JS (2010) The scalable heterogeneous computing (SHOC) benchmark suite. In: Proceedings of the 3rd Workshop on General-Purpose Computation on Graphics Processing Units. pp 63\u201374","DOI":"10.1145\/1735688.1735702"},{"key":"3319_CR16","doi-asserted-by":"crossref","unstructured":"DeFreez D, Bhowmick A, Laguna I, Rubio-Gonz\u00e1lez C (2020) Detecting and reproducing error-code propagation bugs in mpi implementations. In: Proceedings of the 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. pp 187\u2013201","DOI":"10.1145\/3332466.3374515"},{"key":"3319_CR17","doi-asserted-by":"crossref","unstructured":"DeSouza J, Kuhn B, De Supinski BR, Samofalov V, Zheltov S, Bratanov S (2005) Automated, scalable debugging of mpi programs with intel\u00ae message checker. In: Proceedings of the Second International Workshop on Software Engineering for High Performance Computing System Applications. pp 78\u201382","DOI":"10.1145\/1145319.1145342"},{"key":"3319_CR18","doi-asserted-by":"crossref","unstructured":"de\u00a0Kergommeaux JC, Ronsse M, De\u00a0Bosschere K (1999) Mpl: Efficient record\/replay of nondeterministic features of message passing libraries. In: European Parallel Virtual Machine\/message Passing Interface Users\u2019 Group Meeting, Springer. pp 141\u2013148","DOI":"10.1007\/3-540-48158-3_18"},{"key":"3319_CR19","doi-asserted-by":"crossref","unstructured":"Elis B, Yang D, Schulz M (2019) Qmpi: a next generation MPI profiling interface for modern HPC platforms. In: Proceedings of the 26th European MPI Users\u2019 Group Meeting. pp 1\u201310","DOI":"10.1145\/3343211.3343215"},{"issue":"1","key":"3319_CR20","doi-asserted-by":"publisher","first-page":"361","DOI":"10.1007\/s11227-010-0440-0","volume":"59","author":"R Filgueira","year":"2012","unstructured":"Filgueira R, Carretero J, Singh DE, Calderon A, N\u00fa\u00f1ez A (2012) Dynamic-compi: dynamic optimization techniques for mpi parallel applications. J Supercomput 59(1):361\u2013391","journal-title":"J Supercomput"},{"issue":"6","key":"3319_CR21","doi-asserted-by":"crossref","first-page":"702","DOI":"10.1002\/cpe.1556","volume":"22","author":"M Geimer","year":"2010","unstructured":"Geimer M, Wolf F, Wylie BJ, \u00c1brah\u00e1m E, Becker D, Mohr B (2010) The scalasca performance toolset architecture. Concurr Comput Pract Exp 22(6):702\u2013719","journal-title":"Concurr Comput Pract Exp"},{"key":"3319_CR22","doi-asserted-by":"crossref","unstructured":"Gioachin F, Zheng G, Kal\u00e9 LV (2010) Debugging large scale applications in a virtualized environment. In: International Workshop on Languages and Compilers for Parallel Computing, Springer. pp 199\u2013214","DOI":"10.1007\/978-3-642-19595-2_14"},{"key":"3319_CR23","doi-asserted-by":"crossref","unstructured":"Guo X, Lin Y, Xu X, Zhang X (2011) Ps-sim: An execution-driven performance simulation technology based on process-switch. In: International Conference on Computer Science, Environment, Ecoinformatics, and Education, Springer. pp 15\u201322","DOI":"10.1007\/978-3-642-23324-1_4"},{"issue":"1","key":"3319_CR24","first-page":"19","volume":"28","author":"W Haque","year":"2006","unstructured":"Haque W (2006) Concurrent deadlock detection in parallel programs. Int J Comput Appl 28(1):19\u201325","journal-title":"Int J Comput Appl"},{"issue":"10","key":"3319_CR25","doi-asserted-by":"publisher","first-page":"4390","DOI":"10.1007\/s11227-017-2023-9","volume":"73","author":"S H\u00f6finger","year":"2017","unstructured":"H\u00f6finger S, Haunschmid E (2017) Modelling parallel overhead from simple run-time records. J Supercomput 73(10):4390\u20134406","journal-title":"J Supercomput"},{"key":"3319_CR26","doi-asserted-by":"crossref","unstructured":"Kale LV, Krishnan S (1993) Charm++ a portable concurrent object oriented system based on c++. In: Proceedings of the Eighth Annual Conference on Object-oriented Programming Systems, Languages, and Applications. pp 91\u2013108","DOI":"10.1145\/165854.165874"},{"key":"3319_CR27","doi-asserted-by":"crossref","unstructured":"Krammer B, Bidmon K, M\u00fcller MS, Resch MM (2003) Marmot: An MPI analysis and checking tool. In: ParCo, vol 13, pp 493\u2013500. Citeseer","DOI":"10.1016\/S0927-5452(04)80063-7"},{"key":"3319_CR28","doi-asserted-by":"crossref","unstructured":"Kranzlm\u00fcller D, Schaubschl\u00e4ger C, Volkert J (2001) An integrated record&replay mechanism for nondeterministic message passing programs. In: European Parallel Virtual Machine\/message Passing Interface Users\u2019 Group Meeting, Springer. pp 192\u2013200","DOI":"10.1007\/3-540-45417-9_28"},{"key":"3319_CR29","doi-asserted-by":"crossref","unstructured":"Kranzlm\u00fcller D, Volkert J (1999) Nope: A nondeterministic program evaluator. In: International Conference of the Austrian Center for Parallel Computation, Springer. pp 490\u2013499","DOI":"10.1007\/3-540-49164-3_47"},{"key":"3319_CR30","doi-asserted-by":"crossref","unstructured":"Li H, Chen Z, Gupta R (2017) Parastack: Efficient hang detection for MPI programs at large scale. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis. pp 1\u201312","DOI":"10.1145\/3126908.3126938"},{"issue":"4","key":"3319_CR31","doi-asserted-by":"publisher","first-page":"359","DOI":"10.1145\/3296979.3192390","volume":"53","author":"B Liu","year":"2018","unstructured":"Liu B, Huang J (2018) D4: fast concurrency debugging with parallel differential analysis. ACM SIGPLAN Not 53(4):359\u2013373","journal-title":"ACM SIGPLAN Not"},{"key":"3319_CR32","doi-asserted-by":"crossref","unstructured":"Liu X, Mellor-Crummey J (2013) A data-centric profiler for parallel programs. In: SC\u201913: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis, IEEE. pp 1\u201312","DOI":"10.1145\/2503210.2503297"},{"issue":"4","key":"3319_CR33","first-page":"738","volume":"36","author":"Y Liu","year":"2013","unstructured":"Liu Y, Zhi YZ, Zhang X, Li H, Jiao L, Zhang P, Su YM, Ni ZH, Qian DP (2013) Simhpc: An execution-driven simulator for high-performance computers. Jisuanji Xuebao(Chinese Journal of Computers) 36(4):738\u2013746","journal-title":"Jisuanji Xuebao(Chinese Journal of Computers)"},{"issue":"2","key":"3319_CR34","doi-asserted-by":"publisher","first-page":"93","DOI":"10.1002\/cpe.705","volume":"15","author":"G Luecke","year":"2003","unstructured":"Luecke G, Chen H, Coyle J, Hoekstra J, Kraeva M, Zou Y (2003) MPI-check: a tool for checking Fortran 90 MPI programs. Concurr Comput Pract Exp 15(2):93\u2013100","journal-title":"Concurr Comput Pract Exp"},{"issue":"2\u20134","key":"3319_CR35","doi-asserted-by":"publisher","first-page":"117","DOI":"10.1002\/cpe.931","volume":"17","author":"A Malony","year":"2005","unstructured":"Malony A, Shende S, Trebon N, Ray J, Armstrong R, Rasmussen C, Sottile M (2005) Performance technology for parallel and distributed component software. Concurr Comput Pract Exp 17(2\u20134):117\u2013141","journal-title":"Concurr Comput Pract Exp"},{"key":"3319_CR36","unstructured":"Maruyama M, Tsumura T, Nakashima H (2005) Parallel program debugging based on data-replay. In: IASTED PDCS. pp 151\u2013156"},{"issue":"1","key":"3319_CR37","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1023\/A:1015789220266","volume":"23","author":"J Mellor-Crummey","year":"2002","unstructured":"Mellor-Crummey J, Fowler RJ, Marin G, Tallent N (2002) HPCView: A tool for top-down analysis of node performance. J Supercomput 23(1):81\u2013104","journal-title":"J Supercomput"},{"key":"3319_CR38","doi-asserted-by":"crossref","unstructured":"Mueller F, Wu X, Schulz M, De\u00a0Supinski BR, Gamblin T (2010) Scalatrace: tracing, analysis and modeling of HPC codes at scale. In: International Workshop on Applied Parallel Computing, Springer. pp 410\u2013418","DOI":"10.1007\/978-3-642-28145-7_40"},{"issue":"8","key":"3319_CR39","doi-asserted-by":"publisher","first-page":"696","DOI":"10.1016\/j.jpdc.2008.09.001","volume":"69","author":"M Noeth","year":"2009","unstructured":"Noeth M, Ratn P, Mueller F, Schulz M, De Supinski BR (2009) Scalatrace: scalable compression and replay of communication traces for high-performance computing. J Parallel Distrib Comput 69(8):696\u2013710","journal-title":"J Parallel Distrib Comput"},{"key":"3319_CR40","doi-asserted-by":"crossref","unstructured":"Pham A, J\u00e9ron T, Quinson M (2017) Verifying MPI applications with simgridmc. In: Proceedings of the First International Workshop on Software Correctness for HPC Applications. pp 28\u201333","DOI":"10.1145\/3145344.3145345"},{"key":"3319_CR41","doi-asserted-by":"crossref","unstructured":"Prakash S, Bagrodia RL (1998) MPI-SIM: Using parallel simulation to evaluate MPI programs. In: 1998 Winter Simulation Conference. Proceedings (Cat. No. 98CH36274), vol 1, IEEE. pp 467\u2013474","DOI":"10.1109\/WSC.1998.745023"},{"key":"3319_CR42","doi-asserted-by":"crossref","unstructured":"Siegel SF (2007) Model checking nonblocking MPI programs. In: International Workshop on Verification, Model Checking, and Abstract Interpretation. Springer, pp 44\u201358","DOI":"10.1007\/978-3-540-69738-1_3"},{"key":"3319_CR43","doi-asserted-by":"crossref","unstructured":"Siegel SF (2007) Verifying parallel programs with MPI-Spin. In: European Parallel Virtual Machine\/message Passing Interface Users\u2019 Group Meeting. Springer, pp 13\u201314","DOI":"10.1007\/978-3-540-75416-9_8"},{"key":"3319_CR44","doi-asserted-by":"crossref","unstructured":"Spear W, Malony A, Morris A, Shende S (2006) Integrating TAU with eclipse: a performance analysis system in an integrated development environment. In: International Conference on High Performance Computing and Communications. Springer, pp 230\u2013239","DOI":"10.1007\/11847366_24"},{"key":"3319_CR45","doi-asserted-by":"crossref","unstructured":"Su P, Jiao S, Chabbi M, Liu X (2019) Pinpointing performance inefficiencies via lightweight variance profiling. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis. pp 1\u201319","DOI":"10.1145\/3295500.3356167"},{"key":"3319_CR46","doi-asserted-by":"crossref","unstructured":"Taheri S, Briggs I, Burtscher M, Gopalakrishnan G (2019) Difftrace: Efficient whole-program trace analysis and diffing for debugging. In: 2019 IEEE International Conference on Cluster Computing (CLUSTER). IEEE. pp 1\u201312","DOI":"10.1109\/CLUSTER.2019.8891027"},{"key":"3319_CR47","doi-asserted-by":"crossref","unstructured":"Taheri S, Devale S, Gopalakrishnan G, Burtscher M (2017) Parlot: Efficient whole-program call tracing for hpc applications. In: Programming and Performance Visualization Tools. Springer, pp 162\u2013184","DOI":"10.1007\/978-3-030-17872-7_10"},{"key":"3319_CR48","doi-asserted-by":"crossref","unstructured":"Vakkalanka SS, Sharma S, Gopalakrishnan G, Kirby RM (2008) ISP: A tool for model checking MPI programs. In: Proceedings of the 13th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. pp 285\u2013286","DOI":"10.1145\/1345206.1345258"},{"key":"3319_CR49","doi-asserted-by":"crossref","unstructured":"Vetter JS, De\u00a0Supinski BR (2000) Dynamic software testing of MPI applications with umpire. In: SC\u201900: Proceedings of the 2000 ACM\/IEEE Conference on Supercomputing. IEEE, pp 51\u201361","DOI":"10.1109\/SC.2000.10055"},{"issue":"4","key":"3319_CR50","doi-asserted-by":"publisher","first-page":"261","DOI":"10.1145\/1594835.1504214","volume":"44","author":"A Vo","year":"2009","unstructured":"Vo A, Vakkalanka S, DeLisi M, Gopalakrishnan G, Kirby RM, Thakur R (2009) Formal verification of practical MPI programs. ACM Sigplan Not 44(4):261\u2013270","journal-title":"ACM Sigplan Not"},{"key":"3319_CR51","doi-asserted-by":"crossref","unstructured":"Xue R, Liu X, Wu M, Guo Z, Chen W, Zheng W, Zhang Z, Voelker G (2009) MPIWiz: Subgroup reproducible replay of MPI applications. In: Proceedings of the 14th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. pp 251\u2013260","DOI":"10.1145\/1594835.1504213"},{"key":"3319_CR52","doi-asserted-by":"crossref","unstructured":"Ye F, Zhao J, Sarkar V (2018) Detecting MPI usage anomalies via partial program symbolic execution. In: SC18: International Conference for High Performance Computing, Networking, Storage and Analysis. IEEE, pp 794\u2013806","DOI":"10.1109\/SC.2018.00066"},{"issue":"5","key":"3319_CR53","doi-asserted-by":"publisher","first-page":"305","DOI":"10.1145\/1837853.1693493","volume":"45","author":"J Zhai","year":"2010","unstructured":"Zhai J, Chen W, Zheng W (2010) Phantom: predicting performance of parallel applications on large-scale parallel machines using a single node. ACM Sigplan Not 45(5):305\u2013314","journal-title":"ACM Sigplan Not"},{"key":"3319_CR54","unstructured":"Zheng G, Kakulapati G, Kal\u00e9 LV (2004) Bigsim: A parallel simulator for performance prediction of extremely large parallel machines. In: 18th International Parallel and Distributed Processing Symposium, 2004. Proceedings. IEEE, p 78"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-020-03319-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-020-03319-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-020-03319-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,10,1]],"date-time":"2023-10-01T11:52:53Z","timestamp":1696161173000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-020-03319-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,5,23]]},"references-count":54,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2021,2]]}},"alternative-id":["3319"],"URL":"https:\/\/doi.org\/10.1007\/s11227-020-03319-6","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,5,23]]},"assertion":[{"value":"23 May 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}