{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T23:42:26Z","timestamp":1783554146763,"version":"3.55.0"},"reference-count":115,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2022,12,12]],"date-time":"2022-12-12T00:00:00Z","timestamp":1670803200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,12,12]],"date-time":"2022-12-12T00:00:00Z","timestamp":1670803200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Front. Comput. Sci."],"published-print":{"date-parts":[[2023,8]]},"DOI":"10.1007\/s11704-022-2096-3","type":"journal-article","created":{"date-parts":[[2022,12,12]],"date-time":"2022-12-12T09:03:21Z","timestamp":1670835801000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Software approaches for resilience of high performance computing systems: a survey"],"prefix":"10.1007","volume":"17","author":[{"given":"Jie","family":"Jia","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guozhen","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yulin","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Depei","family":"Qian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,12,12]]},"reference":[{"key":"2096_CR1","unstructured":"Dongarra J. Report on the fujitsu fugaku system. University of Tennessee-Knoxville Innovative Computing Laboratory, Tech. Rep. ICLUT-20-06, 2020"},{"key":"2096_CR2","doi-asserted-by":"crossref","unstructured":"Di Martino C, Kramer W, Kalbarczyk Z, Iyer R. Measuring and understanding extreme-scale application resilience: a field study of 5, 000, 000 HPC application runs. In: Proceedings of the 45th Annual IEEE\/IFIP International Conference on Dependable Systems and Networks. 2015, 25\u201336","DOI":"10.1109\/DSN.2015.50"},{"key":"2096_CR3","doi-asserted-by":"crossref","unstructured":"Hursey J, Squyres J M, Mattox T I, Lumsdaine A. The design and implementation of checkpoint\/restart process fault Tolerance for open MPI. In: Proceedings of 2007 IEEE International Parallel and Distributed Processing Symposium. 2007, 1\u20138","DOI":"10.1109\/IPDPS.2007.370605"},{"issue":"4","key":"2096_CR4","doi-asserted-by":"publisher","first-page":"374","DOI":"10.1177\/1094342009347767","volume":"23","author":"F Cappello","year":"2009","unstructured":"Cappello F, Geist A, Gropp B, Kale L, Kramer B, Snir M. Toward exascale resilience. The International Journal of High Performance Computing Applications, 2009, 23(4): 374\u2013388","journal-title":"The International Journal of High Performance Computing Applications"},{"issue":"3","key":"2096_CR5","doi-asserted-by":"publisher","first-page":"1302","DOI":"10.1007\/s11227-013-0884-0","volume":"65","author":"I P Egwutuoha","year":"2013","unstructured":"Egwutuoha I P, Levy D, Selic B, Chen S. A survey of fault tolerance mechanisms and checkpoint\/restart implementations for high performance computing systems. The Journal of Supercomputing, 2013, 65(3): 1302\u20131326","journal-title":"The Journal of Supercomputing"},{"key":"2096_CR6","first-page":"181","volume":"15","author":"K Bergman","year":"2008","unstructured":"Bergman K, Borkar S, Campbell D, Carlson W, Dally W, Denneau M, Franzon P, Harrod W, Hill K, Hiller J, et al. Exascale computing study: Technology challenges in achieving exascale systems. Defense Advanced Research Projects Agency Information Processing Techniques Office (DARPA IPTO), Tech. Rep, 2008, 15: 181","journal-title":"Defense Advanced Research Projects Agency Information Processing Techniques Office (DARPA IPTO), Tech. Rep"},{"key":"2096_CR7","doi-asserted-by":"crossref","unstructured":"Gupta S, Patel T, Engelmann C, Tiwari D. Failures in large scale systems: Long-term measurement, analysis, and implications. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis. 2017, 44","DOI":"10.1145\/3126908.3126937"},{"key":"2096_CR8","doi-asserted-by":"crossref","unstructured":"Radojkovic P, Marazakis M, Carpenter P, Jeyapaul R, Gizopoulos D, Schulz M, Armejach A, Ayguade E A, Bodin F, Canal R, et al. Towards resilient EU HPC systems: A blueprint. PhD thesis, European HPC resilience initiative, 2020","DOI":"10.1145\/3310273.3323434"},{"issue":"1","key":"2096_CR9","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1109\/TDSC.2004.2","volume":"1","author":"A Avizienis","year":"2004","unstructured":"Avizienis A, Laprie J C, Randell B, Landwehr C. Basic concepts and taxonomy of dependable and secure computing. IEEE Transactions on Dependable and Secure Computing, 2004, 1(1): 11\u201333","journal-title":"IEEE Transactions on Dependable and Secure Computing"},{"key":"2096_CR10","volume-title":"Architecture Design for Soft Errors","author":"S Mukherjee","year":"2008","unstructured":"Mukherjee S. Architecture Design for Soft Errors. San Francisco: Morgan Kaufmann, 2008"},{"key":"2096_CR11","unstructured":"Tan L, DeBardeleben N. Failure analysis and quantification for contemporary and future supercomputers. 2019, arXiv preprint arXiv: 1911.02118"},{"key":"2096_CR12","unstructured":"Shoji F, Matsui S, Okamoto M, Sueyasu F, Tsukamoto T, Uno A, Yamamoto K. Long term failure analysis of 10 peta-scale supercomputer. In: Proceedings of HPC in Asia Session at ISC 2015. 2015"},{"key":"2096_CR13","unstructured":"Das A, Mueller F, Siegel C, Vishnu A. Desh: deep learning for system health prediction of lead times to failure in HPC. In: Proceedings of the 27th International Symposium on High-Performance Parallel and Distributed Computing. 2018, 40\u201351"},{"key":"2096_CR14","doi-asserted-by":"crossref","unstructured":"Di Martino C, Kalbarczyk Z, Iyer R K, Baccanico F, Fullop J, Kramer W. Lessons learned from the analysis of system failures at petascale: the case of blue waters. In: Proceedings of the 44th Annual IEEE\/IFIP International Conference on Dependable Systems and Networks. 2014, 610\u2013621","DOI":"10.1109\/DSN.2014.62"},{"key":"2096_CR15","doi-asserted-by":"crossref","unstructured":"El-Sayed N, Schroeder B. Reading between the lines of failure logs: understanding how HPC systems fail. In: Proceedings of the 43rd Annual IEEE\/IFIP International Conference on Dependable Systems and Networks. 2013, 1\u201312","DOI":"10.1109\/DSN.2013.6575356"},{"key":"2096_CR16","doi-asserted-by":"crossref","unstructured":"Bode B, Butler M, Dunning T, Hoeer T, Kramer W, Gropp W, WenMei H. The blue waters super-system for super-science. In: Contemporary High Performance Computing: From Petascale toward Exascale, 339\u2013366. Chapman and Hall\/CRC, 2013","DOI":"10.1201\/9781351104005-13"},{"key":"2096_CR17","doi-asserted-by":"crossref","unstructured":"Bland B. Titan \u2014 Early experience with the titan system at oak ridge national laboratory. In: Proceedings of 2012 SC Companion: High Performance Computing, Networking Storage and Analysis. 2012, 2189\u20132211","DOI":"10.1109\/SC.Companion.2012.356"},{"key":"2096_CR18","doi-asserted-by":"crossref","unstructured":"Bautista-Gomez L, Gainaru A, Perarnau S, Tiwari D, Gupta S, Engelmann C, Cappello F, Snir M. Reducing waste in extreme scale systems through introspective analysis. In: Proceedings of 2016 IEEE International Parallel and Distributed Processing Symposium. 2016, 212\u2013221","DOI":"10.1109\/IPDPS.2016.100"},{"key":"2096_CR19","doi-asserted-by":"crossref","unstructured":"Tiwari D, Gupta S, Vazhkudai S S. Lazy checkpointing: exploiting temporal locality in failures to mitigate checkpointing overheads on extreme-scale systems. In: Proceedings of the 44th Annual IEEE\/IFIP International Conference on Dependable Systems and Networks. 2014, 25\u201336","DOI":"10.1109\/DSN.2014.101"},{"key":"2096_CR20","doi-asserted-by":"crossref","unstructured":"Tiwari D, Gupta S, Gallarno G, Rogers J, Maxwell D. Reliability lessons learned from GPU experience with the Titan supercomputer at Oak Ridge leadership computing facility. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis. 2015, 1\u201312","DOI":"10.1145\/2807591.2807666"},{"key":"2096_CR21","first-page":"494","volume":"46","author":"P H Hargrove","year":"2006","unstructured":"Hargrove P H, Duell J C. Berkeley lab checkpoint\/restart (BLCR) for Linux clusters. Journal of Physics: Conference Series, 2006, 46: 494\u2013499","journal-title":"Journal of Physics: Conference Series"},{"key":"2096_CR22","doi-asserted-by":"crossref","unstructured":"Ansel J, Arya K, Cooperman G. DMTCP: Transparent checkpointing for cluster computations and the desktop. In: Proceedings of 2009 IEEE International Symposium on Parallel &amp; Distributed Processing. 2009, 1\u201312","DOI":"10.1109\/IPDPS.2009.5161063"},{"key":"2096_CR23","doi-asserted-by":"crossref","unstructured":"Bautista-Gomez L, Tsuboi S, Komatitsch D, Cappello F, Maruyama N, Matsuoka S. FTI: High performance fault tolerance interface for hybrid systems. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis. 2011, 1\u201312","DOI":"10.1145\/2063384.2063427"},{"key":"2096_CR24","unstructured":"Zhong H, Nieh J. Crak: Linux checkpoint\/restart as a kernel module. Technical Report, Citeseer, 2001"},{"issue":"S1","key":"2096_CR25","doi-asserted-by":"publisher","first-page":"361","DOI":"10.1145\/844128.844162","volume":"36","author":"S Osman","year":"2002","unstructured":"Osman S, Subhraveti D, Su G, Nieh J. The design and implementation of zap: a system for migrating computing environments. ACM SIGOPS Operating Systems Review, 2002, 36(S1): 361\u2013376","journal-title":"ACM SIGOPS Operating Systems Review"},{"issue":"4","key":"2096_CR26","doi-asserted-by":"publisher","first-page":"479","DOI":"10.1177\/1094342005056139","volume":"19","author":"S Sankaran","year":"2005","unstructured":"Sankaran S, Squyres J M, Barrett B, Sahay V, Lumsdaine A, Duell J, Hargrove P, Roman E. The LAM\/MPI checkpoint\/restart framework: system-initiated checkpointing. The International Journal of High Performance Computing Applications, 2005, 19(4): 479\u2013493","journal-title":"The International Journal of High Performance Computing Applications"},{"key":"2096_CR27","doi-asserted-by":"crossref","unstructured":"Wang C, Mueller F, Engelmann C, Scott S L. Hybrid checkpointing for MPI jobs in HPC environments. In: Proceedings of the 16th International Conference on Parallel and Distributed Systems. 2010, 524\u2013533","DOI":"10.1109\/ICPADS.2010.48"},{"key":"2096_CR28","doi-asserted-by":"crossref","unstructured":"Sancho J C, Petrini F, Johnson G, Frachtenberg E. On the feasibility of incremental checkpointing for scientific computing. In: Proceedings of the 18th International Parallel and Distributed Processing Symposium. 2004, 58","DOI":"10.1109\/IPDPS.2004.1302982"},{"key":"2096_CR29","doi-asserted-by":"crossref","unstructured":"Agarwal S, Garg R, Gupta M S, Moreira J E. Adaptive incremental checkpointing for massively parallel systems. In: Proceedings of the 18th Annual International Conference on Supercomputing. 2004, 277\u2013286","DOI":"10.1145\/1006209.1006248"},{"key":"2096_CR30","doi-asserted-by":"crossref","unstructured":"Bosilca G, Bouteiller A, Cappello F, Djilali S, Fedak G, Germain C, Herault T, Lemarinier P, Lodygensky O, Magniette F, Neri V, Selikhov A. MPICh-V: toward a scalable fault tolerant MPI for volatile nodes. In: Proceedings of 2002 ACM\/IEEE Conference on Supercomputing. 2002, 29","DOI":"10.1109\/SC.2002.10048"},{"key":"2096_CR31","doi-asserted-by":"crossref","unstructured":"Bronevetsky G, Marques D, Pingali K, Stodghill P. Automated application-level checkpointing of MPI programs. In: Proceedings of the 9th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. 2003, 84\u201394","DOI":"10.1145\/781498.781513"},{"issue":"4","key":"2096_CR32","doi-asserted-by":"publisher","first-page":"285","DOI":"10.1023\/A:1024504726988","volume":"31","author":"R L Graham","year":"2003","unstructured":"Graham R L, Choi S E, Daniel D J, Desai N N, Minnich R G, Rasmussen C E, Risinger L D, Sukalski M W. A network-failure-tolerant message-passing system for terascale clusters. International Journal of Parallel Programming, 2003, 31(4): 285\u2013303","journal-title":"International Journal of Parallel Programming"},{"key":"2096_CR33","unstructured":"Woo N, Choi S, Jung h, Moon J, Yeom H Y, Park T, Park H. MPICHGF: providing fault tolerance on grid environments. In: Proceedings of the 3rd IEEE\/\/ACM International Symposium on Cluster Computing and the Grid (CCGrid2003), the Poster and Research Demo Session. 2003"},{"key":"2096_CR34","unstructured":"Zheng G, Shi L, Kale L V. FTC-Charm++: an in-memory checkpoint-based fault tolerant runtime for Charm++ and MPI. In: Proceedings of 2004 IEEE International Conference on Cluster Computing. 2004, 93\u2013103"},{"issue":"3","key":"2096_CR35","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1145\/1075395.1075402","volume":"39","author":"Y Zhang","year":"2005","unstructured":"Zhang Y, Wong D, Zheng W. User-level checkpoint and recovery for LAM\/MPI. ACM SIGOPS Operating Systems Review, 2005, 39(3): 72\u201381","journal-title":"ACM SIGOPS Operating Systems Review"},{"issue":"1","key":"2096_CR36","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1016\/j.future.2007.02.002","volume":"24","author":"D Buntinas","year":"2008","unstructured":"Buntinas D, Coti C, Herault T, Lemarinier P, Pilard L, Rezmerita A, Rodriguez E, Cappello F. Blocking vs. non-blocking coordinated checkpointing for large-scale fault tolerant MPI Protocols. Future Generation Computer Systems, 2008, 24(1): 73\u201384","journal-title":"Future Generation Computer Systems"},{"key":"2096_CR37","doi-asserted-by":"crossref","unstructured":"Ruscio J F, Heffner M A, Varadarajan S. DejaVu: transparent user-level checkpointing, migration, and recovery for distributed systems. In: Proceedings of 2007 IEEE International Parallel and Distributed Processing Symposium. 2007, 1\u201310","DOI":"10.1109\/IPDPS.2007.370309"},{"key":"2096_CR38","doi-asserted-by":"crossref","unstructured":"Cao J, Arya K, Garg R, Matott S, Panda D K, Subramoni H, Vienne J, Cooperman G. System-level scalable checkpoint-restart for petascale computing. In: Proceedings of the 22nd International Conference on Parallel and Distributed Systems. 2016, 932\u2013941","DOI":"10.1109\/ICPADS.2016.0125"},{"key":"2096_CR39","doi-asserted-by":"crossref","unstructured":"Garg R, Price G, Cooperman G. MANA for MPI: MPI-agnostic network-agnostic transparent checkpointing. In: Proceedings of the 28th International Symposium on High-Performance Parallel and Distributed Computing. 2019, 49\u201360","DOI":"10.1145\/3307681.3325962"},{"issue":"3","key":"2096_CR40","doi-asserted-by":"publisher","first-page":"305","DOI":"10.1177\/1094342015623623","volume":"30","author":"I Laguna","year":"2016","unstructured":"Laguna I, Richards D F, Gamblin T, Schulz M, De Supinski B R, Mohror K, Pritchard H. Evaluating and extending user-level fault tolerance in MPI applications. The International Journal of High Performance Computing Applications, 2016, 30(3): 305\u2013319","journal-title":"The International Journal of High Performance Computing Applications"},{"issue":"3","key":"2096_CR41","doi-asserted-by":"publisher","first-page":"e4863","DOI":"10.1002\/cpe.4863","volume":"32","author":"S Chakraborty","year":"2020","unstructured":"Chakraborty S, Laguna I, Emani M, Mohror K, Panda D K, Schulz M, Subramoni H. EREINIT: scalable and efficient fault-tolerance for bulk-synchronous MPI applications. Concurrency and Computation: Practice and Experience, 2020, 32(3): e4863","journal-title":"Concurrency and Computation: Practice and Experience"},{"key":"2096_CR42","doi-asserted-by":"crossref","unstructured":"Georgakoudis G, Guo L, Laguna I. Reinit++: evaluating the performance of global-restart recovery methods for MPI fault tolerance. In: Proceedings of the 35th International Conference on High Performance Computing. 2020, 536\u2013554","DOI":"10.1007\/978-3-030-50743-5_27"},{"key":"2096_CR43","doi-asserted-by":"crossref","unstructured":"Bronevetsky G, Marques D J, Pingali K K, Rugina R, McKee S A. Compiler-enhanced incremental checkpointing for OpenMP applications. In: Proceedings of the 13th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. 2008, 275\u2013276","DOI":"10.1145\/1345206.1345253"},{"issue":"3","key":"2096_CR44","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1007\/s11227-010-0383-5","volume":"57","author":"R Arora","year":"2011","unstructured":"Arora R, Bangalore P, Mernik M. A technique for non-invasive application-level checkpointing. The Journal of Supercomputing, 2011, 57(3): 227\u2013255","journal-title":"The Journal of Supercomputing"},{"key":"2096_CR45","unstructured":"Ba T N, Arora R. A tool for semi-automatic application-level checkpointing. In: Technical Posters at the International Conference for High Performance Computing, Networking, Storage and Analysis. 2016, 16\u201320"},{"key":"2096_CR46","unstructured":"Quinlan D, Liao C. The ROSE source-to-source compiler infrastructure. In: Proceedings of the Cetus Users and Compiler Infrastructure Workshop. 2011, 1\u20133"},{"issue":"3","key":"2096_CR47","doi-asserted-by":"publisher","first-page":"501","DOI":"10.1109\/TPDS.2018.2866794","volume":"30","author":"F Shahzad","year":"2019","unstructured":"Shahzad F, Thies J, Kreutzer M, Zeiser T, Hager G, Wellein G. CRAFT: a library for easier application-level checkpoint\/restart and automatic fault tolerance. IEEE Transactions on Parallel and Distributed Systems, 2019, 30(3): 501\u2013514","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"2096_CR48","doi-asserted-by":"crossref","unstructured":"Takizawa H, Sato K, Komatsu K, Kobayashi H. CheCUDA: a checkpoint\/restart tool for CUDA applications. In: Proceedings of 2009 International Conference on Parallel and Distributed Computing, Applications and Technologies. 2009, 408\u2013413","DOI":"10.1109\/PDCAT.2009.78"},{"key":"2096_CR49","unstructured":"Garg R. Extending the domain of transparent checkpoint-restart for large-scale HPC. Northeastern University, Dissertation, 2019"},{"key":"2096_CR50","doi-asserted-by":"crossref","unstructured":"Garg R, Mohan A, Sullivan M, Cooperman G. CRUM: checkpoint-restart support for CUDA\u2019s unified memory. In: Proceedings of 2018 IEEE International Conference on Cluster Computing. 2018, 302\u2013313","DOI":"10.1109\/CLUSTER.2018.00047"},{"key":"2096_CR51","doi-asserted-by":"crossref","unstructured":"Jain T, Cooperman G. CRAC: Checkpoint-restart architecture for CUDA with streams and UVM. In: Proceedings of International Conference for High Performance Computing, Networking, Storage and Analysis. 2020, 1\u201315","DOI":"10.1109\/SC41405.2020.00081"},{"key":"2096_CR52","doi-asserted-by":"crossref","unstructured":"Lee K, Sullivan M B, Hari S K S, Tsai T, Keckler S W, Erez M. GPU snapshot: checkpoint offloading for GPU-dense systems. In: Proceedings of the ACM International Conference on Supercomputing. 2019, 171\u2013183","DOI":"10.1145\/3330345.3330361"},{"key":"2096_CR53","doi-asserted-by":"crossref","unstructured":"Kannan S, Farooqui N, Gavrilovska A, Schwan K. HeteroCheckpoint: efficient checkpointing for accelerator-based systems. In: Proceedings of the 44th Annual IEEE\/IFIP International Conference on Dependable Systems and Networks. 2014, 738\u2013743","DOI":"10.1109\/DSN.2014.76"},{"key":"2096_CR54","doi-asserted-by":"crossref","unstructured":"Vaidya N H. A case for two-level distributed recovery schemes. In: Proceedings of the 1995 ACM SIGMETRICS Joint International Conference on Measurement and Modeling of Computer Systems. 1995, 64\u201373","DOI":"10.1145\/223587.223596"},{"issue":"1\u20132","key":"2096_CR55","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1023\/A:1008181429693","volume":"16","author":"J Haines","year":"2000","unstructured":"Haines J, Lakamraju V, Koren I, Krishna C M. Application-level fault tolerance as a complement to system-level fault tolerance. The Journal of Supercomputing, 2000, 16(1\u20132): 53\u201368","journal-title":"The Journal of Supercomputing"},{"issue":"1","key":"2096_CR56","doi-asserted-by":"publisher","first-page":"244","DOI":"10.1109\/TPDS.2016.2546248","volume":"28","author":"S Di","year":"2017","unstructured":"Di S, Robert Y, Vivien F, Cappello F. Toward an optimal online checkpoint solution under a two-level HPC checkpoint model. IEEE Transactions on Parallel and Distributed Systems, 2017, 28(1): 244\u2013259","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"issue":"7","key":"2096_CR57","doi-asserted-by":"publisher","first-page":"1212","DOI":"10.1109\/TC.2016.2643660","volume":"66","author":"A Benoit","year":"2017","unstructured":"Benoit A, Cavelan A, Le F\u00e8vre V, Robert Y, Sun H. Towards optimal multi-level checkpointing. IEEE Transactions on Computers, 2017, 66(7): 1212\u20131226","journal-title":"IEEE Transactions on Computers"},{"key":"2096_CR58","doi-asserted-by":"crossref","unstructured":"Ferreira K, Stearley J, Laros J H, Oldfield R, Pedretti K, Brightwell R, Riesen R, Bridges P G, Arnold D. Evaluating the viability of process replication reliability for exascale systems. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis. 2011, 1\u201312","DOI":"10.1145\/2063384.2063443"},{"key":"2096_CR59","doi-asserted-by":"crossref","unstructured":"Wu P, Ding C, Chen L, Gao F, Davies T, Karlsson C, Chen Z. Fault tolerant matrix-matrix multiplication: Correcting soft errors on-line. In: Proceedings of the 2nd Workshop on Scalable Algorithms for Large-Scale Systems. 2011, 25\u201328","DOI":"10.1145\/2133173.2133185"},{"key":"2096_CR60","doi-asserted-by":"crossref","unstructured":"Fiala D, Mueller F, Engelmann C, Riesen R, Ferreira K, Brightwell R. Detection and correction of silent data corruption for large-scale high-performance computing. In: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis. 2012, 1\u201312","DOI":"10.1109\/SC.2012.49"},{"key":"2096_CR61","doi-asserted-by":"crossref","unstructured":"Wang Z, Yang X, Zhou Y. MMPI: a scalable fault tolerance mechanism for MPI large scale parallel computing. In: Proceedings of the 10th IEEE International Conference on Computer and Information Technology. 2010, 1251\u20131256","DOI":"10.1109\/CIT.2010.226"},{"key":"2096_CR62","doi-asserted-by":"crossref","unstructured":"Hussain Z, Znati T, Melhem R. Partial redundancy in HPC systems with non-uniform node reliabilities. In: Proceedings of International Conference for High Performance Computing, Networking, Storage and Analysis. 2018, 566\u2013576","DOI":"10.1109\/SC.2018.00047"},{"key":"2096_CR63","doi-asserted-by":"crossref","unstructured":"Elliott J, Kharbas K, Fiala D, Mueller F, Ferreira K, Engelmann C. Combining partial redundancy and checkpointing for HPC. In: Proceedings of the 32nd International Conference on Distributed Computing Systems. 2012, 615\u2013626","DOI":"10.1109\/ICDCS.2012.56"},{"issue":"8","key":"2096_CR64","doi-asserted-by":"publisher","first-page":"2213","DOI":"10.1109\/TC.2014.2360536","volume":"64","author":"C George","year":"2015","unstructured":"George C, Vadhiyar S. Fault tolerance on large scale systems using adaptive process replication. IEEE Transactions on Computers, 2015, 64(8): 2213\u20132225","journal-title":"IEEE Transactions on Computers"},{"key":"2096_CR65","doi-asserted-by":"crossref","unstructured":"Quinn H, Graham P. Terrestrial-based radiation upsets: a cautionary tale. In: Proceedings of the 13th Annual IEEE Symposium on Field-Programmable Custom Computing Machines. 2005, 193\u2013202","DOI":"10.1109\/FCCM.2005.61"},{"issue":"2","key":"2096_CR66","doi-asserted-by":"publisher","first-page":"100","DOI":"10.1145\/1897816.1897844","volume":"54","author":"B Schroeder","year":"2011","unstructured":"Schroeder B, Pinheiro E, Weber W D. DRAM errors in the wild: a large-scale field study. Communications of the ACM, 2011, 54(2): 100\u2013107","journal-title":"Communications of the ACM"},{"key":"2096_CR67","doi-asserted-by":"crossref","unstructured":"Sedaghat Y, Miremadi S G, Fazeli M. A software-based error detection technique using encoded signatures. In: Proceedings of the 21st IEEE International Symposium on Defect and Fault Tolerance in VLSI Systems. 2006, 389\u2013400","DOI":"10.1109\/DFT.2006.11"},{"key":"2096_CR68","doi-asserted-by":"crossref","unstructured":"Miremadi G, Harlsson J, Gunneflo U, Torin J. Two software techniques for on-line error detection. In: Proceedings of the 22nd International Symposium on Fault-Tolerant Computing. 1992, 328\u2013335","DOI":"10.1109\/FTCS.1992.243568"},{"issue":"9","key":"2096_CR69","doi-asserted-by":"publisher","first-page":"1233","DOI":"10.1109\/TC.2011.101","volume":"60","author":"R Vemu","year":"2011","unstructured":"Vemu R, Abraham J. CEDA: control-flow error detection using assertions. IEEE Transactions on Computers, 2011, 60(9): 1233\u20131245","journal-title":"IEEE Transactions on Computers"},{"key":"2096_CR70","doi-asserted-by":"crossref","unstructured":"Zarandi H R, Maghsoudloo M, Khoshavi N. Two efficient software techniques to detect and correct control-flow errors. In: Proceedings of the 16th Pacific Rim International Symposium on Dependable Computing. 2010, 141\u2013148","DOI":"10.1109\/PRDC.2010.10"},{"issue":"8","key":"2096_CR71","doi-asserted-by":"publisher","first-page":"381","DOI":"10.1145\/2692916.2555279","volume":"49","author":"L B Gomez","year":"2014","unstructured":"Gomez L B, Cappello F. Detecting silent data corruption through data dynamic monitoring for scientific applications. ACM SIGPLAN Notices, 2014, 49(8): 381\u2013382","journal-title":"ACM SIGPLAN Notices"},{"key":"2096_CR72","doi-asserted-by":"crossref","unstructured":"Berrocal E, Bautista-Gomez L, Di S, Lan Z, Cappello F. Lightweight silent data corruption detection based on runtime data analysis for HPC applications. In: Proceedings of the 24th International Symposium on High-Performance Parallel and Distributed Computing. 2015, 275\u2013278","DOI":"10.1145\/2749246.2749253"},{"key":"2096_CR73","doi-asserted-by":"crossref","unstructured":"LeBlanc T, Anand R, Gabriel E, Subhlok J. VolpexMPI: an MPI library for execution of parallel applications on volatile nodes. In: Proceedings of the 16th European Parallel Virtual Machine \/ Message Passing Interface Users\u2019 Group Meeting. 2009, 124\u2013133","DOI":"10.1007\/978-3-642-03770-2_19"},{"key":"2096_CR74","doi-asserted-by":"crossref","unstructured":"Engelmann C, Boehm S. Redundant execution of HPC applications with MR-MPI. In: Proceedings of the 10th IASTED International Conference on Parallel and Distributed Computing and Networks. 2011, 31\u201338","DOI":"10.2316\/P.2011.719-031"},{"issue":"12","key":"2096_CR75","doi-asserted-by":"publisher","first-page":"3642","DOI":"10.1109\/TPDS.2017.2735971","volume":"28","author":"E Berrocal","year":"2017","unstructured":"Berrocal E, Bautista-Gomez L, Di S, Lan Z, Cappello F. Toward general software level silent data corruption detection for parallel applications. IEEE Transactions on Parallel and Distributed Systems, 2017, 28(12): 3642\u20133655","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"2096_CR76","doi-asserted-by":"crossref","unstructured":"Fiala D, Ferreira K B, Mueller F, Engelmann C. A tunable, software-based DRAM error detection and correction library for HPC. In: Proceedings of European Conference on Parallel Processing. 2012, 251\u2013261","DOI":"10.1007\/978-3-642-29740-3_29"},{"key":"2096_CR77","doi-asserted-by":"crossref","unstructured":"Fiala D, Mueller F, Ferreira K B. FlipSphere: a software-based DRAM error detection and correction library for HPC. In: Proceedings of the 20th International Symposium on Distributed Simulation and Real Time Applications. 2016, 19\u201328","DOI":"10.1109\/DS-RT.2016.27"},{"key":"2096_CR78","doi-asserted-by":"crossref","unstructured":"Fiala D, Mueller F, Ferreira K, Engelmann C. Mini-Ckpts: surviving OS failures in persistent memory. In: Proceedings of 2016 International Conference on Supercomputing. 2016, 7","DOI":"10.1145\/2925426.2926295"},{"issue":"6","key":"2096_CR79","doi-asserted-by":"publisher","first-page":"518","DOI":"10.1109\/TC.1984.1676475","volume":"C-33","author":"K H Huang","year":"1984","unstructured":"Huang K H, Abraham J A. Algorithm-based fault tolerance for matrix operations. IEEE Transactions on Computers, 1984, C-33(6): 518\u2013528","journal-title":"IEEE Transactions on Computers"},{"issue":"11","key":"2096_CR80","doi-asserted-by":"publisher","first-page":"1434","DOI":"10.1109\/12.8712","volume":"37","author":"F T Luk","year":"1988","unstructured":"Luk F T, Park H. Fault-tolerant matrix triangularizations on systolic arrays. IEEE Transactions on Computers, 1988, 37(11): 1434\u20131438","journal-title":"IEEE Transactions on Computers"},{"issue":"2","key":"2096_CR81","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1016\/0743-7315(88)90027-5","volume":"5","author":"F T Luk","year":"1988","unstructured":"Luk F T, Park H. An analysis of algorithm-based fault tolerance techniques. Journal of Parallel and Distributed Computing, 1988, 5(2): 172\u2013184","journal-title":"Journal of Parallel and Distributed Computing"},{"issue":"2","key":"2096_CR82","doi-asserted-by":"publisher","first-page":"10","DOI":"10.1145\/2686892","volume":"1","author":"A Bouteiller","year":"2015","unstructured":"Bouteiller A, Herault T, Bosilca G, Du P, Dongarra J. Algorithm-based fault tolerance for dense matrix factorizations, multiple failures and accuracy. ACM Transactions on Parallel Computing, 2015, 1(2): 10","journal-title":"ACM Transactions on Parallel Computing"},{"issue":"8","key":"2096_CR83","doi-asserted-by":"publisher","first-page":"167","DOI":"10.1145\/2517327.2442533","volume":"48","author":"Z Chen","year":"2013","unstructured":"Chen Z. Online-ABFT: an online algorithm based fault tolerance scheme for soft error detection in iterative methods. ACM SIGPLAN Notices, 2013, 48(8): 167\u2013176","journal-title":"ACM SIGPLAN Notices"},{"key":"2096_CR84","doi-asserted-by":"crossref","unstructured":"Tao D, Song S L, Krishnamoorthy S, Wu P, Liang X, Zhang E Z, Kerbyson D, Chen Z. New-sum: a novel online ABFT scheme for general iterative methods. In: Proceedings of the 25th ACM International Symposium on High-Performance Parallel and Distributed Computing. 2016, 43\u201355","DOI":"10.1145\/2907294.2907306"},{"key":"2096_CR85","doi-asserted-by":"crossref","unstructured":"Sch\u00f6ll A, Braun C, Kochte M A, Wunderlich H J. Efficient algorithm-based fault tolerance for sparse matrix operations. In: Proceedings of the 46th Annual IEEE\/IFIP International Conference on Dependable Systems and Networks. 2016, 251\u2013262","DOI":"10.1109\/DSN.2016.31"},{"key":"2096_CR86","doi-asserted-by":"crossref","unstructured":"Shantharam M, Srinivasmurthy S, Raghavan P. Fault tolerant preconditioned conjugate gradient for sparse linear system solution. In: Proceedings of the 26th ACM International Conference on Supercomputing. 2012, 69\u201378","DOI":"10.1145\/2304576.2304588"},{"key":"2096_CR87","doi-asserted-by":"crossref","unstructured":"Zhu Y, Liu Y, Li M, Qian D. Block-checksum-based fault tolerance for matrix multiplication on large-scale parallel systems. In: Proceedings of the 20th International Conference on High Performance Computing and Communications; IEEE 16th International Conference on Smart City; IEEE 4th International Conference on Data Science and Systems. 2018, 172\u2013179","DOI":"10.1109\/HPCC\/SmartCity\/DSS.2018.00054"},{"key":"2096_CR88","doi-asserted-by":"publisher","first-page":"42674","DOI":"10.1109\/ACCESS.2020.2975832","volume":"8","author":"Y Zhu","year":"2020","unstructured":"Zhu Y, Liu Y, Zhang G. FT-PBLAS: PBLAS-based fault-tolerant linear algebra computation on high-performance computing systems. IEEE Access, 2020, 8: 42674\u201342688","journal-title":"IEEE Access"},{"issue":"12","key":"2096_CR89","doi-asserted-by":"publisher","first-page":"1628","DOI":"10.1109\/TPDS.2008.58","volume":"19","author":"Z Chen","year":"2008","unstructured":"Chen Z, Dongarra J. Algorithm-based fault tolerance for fail-stop failures. IEEE Transactions on Parallel and Distributed Systems, 2008, 19(12): 1628\u20131641","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"2096_CR90","doi-asserted-by":"crossref","unstructured":"Roche T, Cunche M, Roch J L. Algorithm-based fault tolerance applied to P2P computing networks. In: Proceedings of the 1st International Conference on Advances in P2P Systems. 2009, 144\u2013149","DOI":"10.1109\/AP2PS.2009.30"},{"issue":"5","key":"2096_CR91","doi-asserted-by":"publisher","first-page":"1323","DOI":"10.1109\/TPDS.2014.2320502","volume":"26","author":"D Hakkarinen","year":"2015","unstructured":"Hakkarinen D, Wu P, Chen Z. Fail-stop failure algorithm-based fault tolerance for Cholesky decomposition. IEEE Transactions on Parallel and Distributed Systems, 2015, 26(5): 1323\u20131335","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"2096_CR92","doi-asserted-by":"crossref","unstructured":"Davies T, Karlsson C, Liu H, Ding C, Chen Z. High performance linpack benchmark: a fault tolerant implementation without checkpointing. In: Proceedings of the International Conference on Supercomputing. 2011, 162\u2013171","DOI":"10.1145\/1995896.1995923"},{"key":"2096_CR93","doi-asserted-by":"crossref","unstructured":"Chen J, Li S, Chen Z. GPU-ABFT: optimizing algorithm-based fault tolerance for heterogeneous systems with GPUs. In: Proceedings of 2016 IEEE International Conference on Networking, Architecture and Storage. 2016, 1\u20132","DOI":"10.1109\/NAS.2016.7549404"},{"key":"2096_CR94","doi-asserted-by":"crossref","unstructured":"Chen J, Li H, Li S, Liang X, Wu P, Tao D, Ouyang K, Liu Y, Zhao K, Guan Q, Chen Z. Fault tolerant one-sided matrix decompositions on heterogeneous systems with GPUs. In: Proceedings of International Conference for High Performance Computing, Networking, Storage and Analysis. 2018, 854\u2013865","DOI":"10.1109\/SC.2018.00071"},{"key":"2096_CR95","doi-asserted-by":"crossref","unstructured":"Braun C, Halder S, Wunderlich H J. A-ABFT: autonomous algorithm-based fault tolerance for matrix multiplications on graphics processing units. In: Proceedings of the 44th Annual IEEE\/IFIP International Conference on Dependable Systems and Networks. 2014, 443\u2013454","DOI":"10.1109\/DSN.2014.48"},{"issue":"3","key":"2096_CR96","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1023\/A:1011494323443","volume":"4","author":"S Ranganathan","year":"2001","unstructured":"Ranganathan S, George A D, Todd R W, Chidester M C. Gossip-style failure detection and distributed consensus for scalable heterogeneous clusters. Cluster Computing, 2001, 4(3): 197\u2013209","journal-title":"Cluster Computing"},{"key":"2096_CR97","doi-asserted-by":"crossref","unstructured":"Gabel M, Schuster A, Bachrach R G, Bj\u00f8rner N. Latent fault detection in large scale services. In: Proceedings of the IEEE\/IFIP International Conference on Dependable Systems and Networks. 2012, 1\u201312","DOI":"10.1109\/DSN.2012.6263932"},{"key":"2096_CR98","doi-asserted-by":"crossref","unstructured":"Wu L, Luo H, Zhan J, Meng D. A runtime fault detection method for HPC cluster. In: Proceedings of the 12th International Conference on Parallel and Distributed Computing, Applications and Technologies. 2011, 68\u201372","DOI":"10.1109\/PDCAT.2011.9"},{"key":"2096_CR99","doi-asserted-by":"crossref","unstructured":"Ghiasvand S, Ciorba F M. Anomaly detection in high performance computers: a vicinity perspective. In: Proceedings of the 18th International Symposium on Parallel and Distributed Computing. 2019, 112\u2013120","DOI":"10.1109\/ISPDC.2019.00024"},{"issue":"4","key":"2096_CR100","doi-asserted-by":"publisher","first-page":"363","DOI":"10.1080\/17445760.2013.803686","volume":"29","author":"I P Egwutuoha","year":"2014","unstructured":"Egwutuoha I P, Chen S, Levy D, Selic B, Calvo R. Cost-oriented proactive fault tolerance approach to high performance computing (HPC) in the cloud. International Journal of Parallel, Emergent and Distributed Systems, 2014, 29(4): 363\u2013378","journal-title":"International Journal of Parallel, Emergent and Distributed Systems"},{"key":"2096_CR101","doi-asserted-by":"crossref","unstructured":"Borghesi A, Libri A, Benini L, Bartolini A. Online anomaly detection in HPC systems. In: Proceedings of 2019 IEEE International Conference on Artificial Intelligence Circuits and Systems. 2019, 229\u2013233","DOI":"10.1109\/AICAS.2019.8771527"},{"issue":"4","key":"2096_CR102","doi-asserted-by":"publisher","first-page":"739","DOI":"10.1109\/TPDS.2021.3082802","volume":"33","author":"A Borghesi","year":"2022","unstructured":"Borghesi A, Molan M, Milano M, Bartolini A. Anomaly detection and anticipation in high performance computing systems. IEEE Transactions on Parallel and Distributed Systems, 2022, 33(4): 739\u2013750","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"2096_CR103","doi-asserted-by":"crossref","unstructured":"Dani M C, Doreau H, Alt S. K-means application for anomaly detection and log classification in HPC. In: Proceedings of the 30th International Conference on Industrial, Engineering and Other Applications of Applied Intelligent Systems. 2017, 201\u2013210","DOI":"10.1007\/978-3-319-60045-1_23"},{"key":"2096_CR104","doi-asserted-by":"crossref","unstructured":"Zhu B, Wang G, Liu X, Hu D, Lin S, Ma J. Proactive drive failure prediction for large scale storage systems. In: Proceedings of the 29th Symposium on Mass Storage Systems and Technologies. 2013, 1\u20135","DOI":"10.1109\/MSST.2013.6558427"},{"key":"2096_CR105","unstructured":"Fulp E W, Fink G A, Haack J N. Predicting computer system failures using support vector machines. In: Proceedings of the 1st USENIX Conference on Analysis of System Logs. 2008, 5"},{"key":"2096_CR106","doi-asserted-by":"crossref","unstructured":"Ganguly S, Consul A, Khan A, Bussone B, Richards J, Miguel A. A practical approach to hard disk failure prediction in cloud platforms: big data model for failure management in datacenters. In: Proceedings of the 2nd International Conference on Big Data Computing Service and Applications. 2016, 105\u2013116","DOI":"10.1109\/BigDataService.2016.10"},{"key":"2096_CR107","doi-asserted-by":"publisher","first-page":"493","DOI":"10.1016\/S0927-5452(04)80063-7","volume":"13","author":"B Krammer","year":"2004","unstructured":"Krammer B, Bidmon K, M\u00fcller M S, Resch M M. MARMOT: an MPI analysis and checking tool. Advances in Parallel Computing, 2004, 13: 493\u2013500","journal-title":"Advances in Parallel Computing"},{"key":"2096_CR108","doi-asserted-by":"crossref","unstructured":"Vetter J S, De Supinski B R. Dynamic software testing of MPI applications with Umpire. In: Proceedings of 2000 ACM\/IEEE Conference on Supercomputing. 2000, 51","DOI":"10.1109\/SC.2000.10055"},{"key":"2096_CR109","doi-asserted-by":"crossref","unstructured":"Gao J, Yu K, Qing P. A scalable runtime fault detection mechanism for high performance computing. In: Proceedings of the 2nd Information Technology, Networking, Electronic and Automation Control Conference. 2017, 490\u2013495","DOI":"10.1109\/ITNEC.2017.8284780"},{"key":"2096_CR110","doi-asserted-by":"crossref","unstructured":"Kharbas K, Kim D, Hoefler T, Mueller F. Assessing HPC failure detectors for MPI jobs. In: Proceedings of the 20th Euromicro International Conference on Parallel, Distributed and Network-Based Processing. 2012, 81\u201388","DOI":"10.1109\/PDP.2012.11"},{"key":"2096_CR111","unstructured":"Liang Y, Zhang Y, Sivasubramaniam A, Jette M, Sahoo R. BlueGene\/L failure analysis and prediction models. In: Proceedings of the International Conference on Dependable Systems and Networks. 2006, 425\u2013434"},{"key":"2096_CR112","doi-asserted-by":"crossref","unstructured":"Gainaru A, Cappello F, Snir M, Kramer W. Fault prediction under the microscope: a closer look into HPC systems. In: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis. 2012, 1\u201311","DOI":"10.1109\/SC.2012.57"},{"key":"2096_CR113","doi-asserted-by":"crossref","unstructured":"Gainaru A, Cappello F, Kramer W. Taming of the shrew: modeling the normal and faulty behaviour of large-scale HPC systems. In: Proceedings of the 26th International Parallel and Distributed Processing Symposium. 2012, 1168\u20131179","DOI":"10.1109\/IPDPS.2012.107"},{"key":"2096_CR114","doi-asserted-by":"crossref","unstructured":"Pelaez A, Quiroz A, Browne J C, Chuah E, Parashar M. Online failure prediction for HPC resources using decentralized clustering. In: Proceedings of the 21st International Conference on High Performance Computing. 2014, 1\u20139","DOI":"10.1109\/HiPC.2014.7116903"},{"key":"2096_CR115","doi-asserted-by":"crossref","unstructured":"Gunawi H S, Suminto R O, Sears R, Golliher C, Sundararaman S, et al. Fail-slow at scale: evidence of hardware performance faults in large production systems. In: Proceedings of the 16th USENIX Conference on File and Storage Technologies. 2018, 1\u201314","DOI":"10.1145\/3242086"}],"container-title":["Frontiers of Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-022-2096-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11704-022-2096-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-022-2096-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,18]],"date-time":"2024-09-18T20:25:45Z","timestamp":1726691145000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11704-022-2096-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,12]]},"references-count":115,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2023,8]]}},"alternative-id":["2096"],"URL":"https:\/\/doi.org\/10.1007\/s11704-022-2096-3","relation":{},"ISSN":["2095-2228","2095-2236"],"issn-type":[{"value":"2095-2228","type":"print"},{"value":"2095-2236","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12,12]]},"assertion":[{"value":"16 February 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 September 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 December 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"174105"}}