{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,7]],"date-time":"2024-09-07T17:10:29Z","timestamp":1725729029276},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2021,9,1]],"date-time":"2021-09-01T00:00:00Z","timestamp":1630454400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2021,9,1]],"date-time":"2021-09-01T00:00:00Z","timestamp":1630454400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Front. Comput. Sci."],"published-print":{"date-parts":[[2021,12]]},"DOI":"10.1007\/s11704-020-0190-y","type":"journal-article","created":{"date-parts":[[2021,9,1]],"date-time":"2021-09-01T16:03:28Z","timestamp":1630512208000},"update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["User-level failure detection and auto-recovery of parallel programs in HPC systems"],"prefix":"10.1007","volume":"15","author":[{"given":"Guozhen","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hailong","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Depei","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,9,1]]},"reference":[{"key":"190_CR1","doi-asserted-by":"publisher","first-page":"1302","DOI":"10.1007\/s11227-013-0884-0","volume":"65","author":"I P Egwutuoha","year":"2013","unstructured":"Egwutuoha I P, Levy D, Selic B, Chen S. A survey of fault tolerance mechanisms and checkpoint\/restart implementations for high performance computing systems. The Journal of Supercomputing, 2013, 65: 1302\u20131326","journal-title":"The Journal of Supercomputing"},{"key":"190_CR2","unstructured":"Lu C D. Failure data analysis of hpc systems. 2013, arXiv preprint arXiv:1302.4779"},{"key":"190_CR3","doi-asserted-by":"publisher","first-page":"374","DOI":"10.1177\/1094342009347767","volume":"23","author":"F Cappello","year":"2009","unstructured":"Cappello F, Geist A, Gropp W D, Kale L V, Kramer W T, Snir M. Toward exascale resilience. International Journal of High Performance Computing Applications, 2009, 23: 374\u2013385","journal-title":"International Journal of High Performance Computing Applications"},{"key":"190_CR4","doi-asserted-by":"crossref","unstructured":"Bertier M, Marin O, Sens P. Performance analysis of a hierarchical failure detector. In: Proceedings of the 2003 International Conference on Dependable Systems and Networks. 2003, 635\u2013644","DOI":"10.1109\/DSN.2003.1209973"},{"key":"190_CR5","doi-asserted-by":"publisher","first-page":"911","DOI":"10.1002\/cpe.701","volume":"14","author":"G R Luecke","year":"2002","unstructured":"Luecke G R, Zou Y, Coyle J, Hoekstra J, Kraeva M. Deadlock detection in MPI programs. Concurrency and Computation: Practice and Experience, 2002, 14: 911\u2013932","journal-title":"Concurrency and Computation: Practice and Experience"},{"key":"190_CR6","doi-asserted-by":"crossref","unstructured":"Gao Q, Qin F, Panda D K. DMTracker: finding bugs in large-scale parallel programs by detecting anomaly in data movements. In: Proceedings of the 2007 ACM\/IEEE Conference on Supercomputing. 2007, 1\u201312","DOI":"10.1145\/1362622.1362643"},{"key":"190_CR7","doi-asserted-by":"crossref","unstructured":"Arnold D C, Ahn D H, De Supinski B R, Lee G L, Miller B P, Schulz M. Stack trace analysis for large-scale debugging. In: Proceddings of the 2007 IEEE International Parallel and Distributed Processing Symposium. 2007, 1\u201310","DOI":"10.1109\/IPDPS.2007.370254"},{"key":"190_CR8","doi-asserted-by":"crossref","unstructured":"Laguna I, Gamblin T, De Supinski B R, Bagchi S, Bronevetsky G, Anh D H, Schulz M, Rountree B. Large scale debugging of parallel tasks with AutomaDeD. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis. 2011, 1\u201310","DOI":"10.1145\/2063384.2063451"},{"key":"190_CR9","doi-asserted-by":"crossref","unstructured":"Wu X, Mueller F. Elastic and scalable tracing and accurate replay of non-deterministic events. In: Proceedings of the 27th International ACM Conference on International Conference on Supercomputing. 2013, 59\u201368","DOI":"10.1145\/2464996.2465001"},{"key":"190_CR10","doi-asserted-by":"crossref","unstructured":"Gupta R, Beckman P, Park B H, Lusk E, Hargrove P, Geist A, Panda D, Lumsdaine A, Dongarra J. CIFTS: a coordinated infrastructure for fault-tolerant systems. In: Proceedings of the 2009 International Conference on Parallel Processing. 2009, 237\u2013245","DOI":"10.1109\/ICPP.2009.20"},{"key":"190_CR11","doi-asserted-by":"publisher","first-page":"71892","DOI":"10.1109\/ACCESS.2018.2882394","volume":"6","author":"G Z Zhang","year":"2018","unstructured":"Zhang G Z, Liu Y, Yang H L, Qian D P. A lightweight and flexible tool for distinguishing between hardware malfunctions and program bugs in debugging large-scale programs. IEEE Access, 2018, 6: 71892\u201371905","journal-title":"IEEE Access"},{"key":"190_CR12","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1177\/1094342017711505","volume":"32","author":"G Bosilca","year":"2018","unstructured":"Bosilca G, Bouteiller A, Guermouche A, Herault T, Robert Y, Sens P, Dongarra J. A failure detector for HPC platforms. The International Journal of High Performance Computing Applications, 2018, 32: 139\u2013158","journal-title":"The International Journal of High Performance Computing Applications"},{"key":"190_CR13","doi-asserted-by":"crossref","unstructured":"Berrocal E, Bautista-Gomez L, Di S, Lan Z L, Cappello F. Lightweight silent data corruption detection based on runtime data analysis for HPC applications. In: Proceedings of the 24th International Symposium on High-Performance Parallel and Distributed Computing. 2015, 275\u2013278","DOI":"10.1145\/2749246.2749253"},{"key":"190_CR14","doi-asserted-by":"crossref","unstructured":"Berrocal E, Bautista-Gomez L, Di S, Cappello F. Exploring partial replication to improve lightweight silent data corruption detection for HPC applications. In: Proceedings of Europe Conference on Parallel Processing. 2016, 419\u2013430","DOI":"10.1007\/978-3-319-43659-3_31"},{"issue":"12","key":"190_CR15","doi-asserted-by":"publisher","first-page":"3642","DOI":"10.1109\/TPDS.2017.2735971","volume":"28","author":"E Berrocal","year":"2017","unstructured":"Berrocal E, Bautista-Gomez L, Di S, Lan Z L, Cappello F. Toward general software level silent data corruption detection for parallel applications. IEEE Transactions on Parallel and Distributed Systtems, 2017, 28(12): 3642\u20133655","journal-title":"IEEE Transactions on Parallel and Distributed Systtems"},{"key":"190_CR16","doi-asserted-by":"crossref","unstructured":"Guo L Z, Li D, Laguna I, Schulz M. FlipTracker: understanding natural error resilience in HPC applications. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis. 2018, 94\u2013107","DOI":"10.1109\/SC.2018.00011"},{"key":"190_CR17","doi-asserted-by":"crossref","unstructured":"Gomez L B, Cappello F. Detecting silent data corruption through data dynamic monitoring for scientific applications. In: Proceedings of the 19th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. 2014, 381\u2013382","DOI":"10.1145\/2692916.2555279"},{"key":"190_CR18","first-page":"277","volume":"19","author":"O Subasi","year":"2018","unstructured":"Subasi O, Di S, Gomez L B, Balaprakash P, Unsal O, Labarta J, Cristal A, Krishnamoorthy S, Cappello F. Exploring the capabilities of support vector machines in detecting silent data corruptions. Sustainable Computing: Informatics and Systems, 2018, 19: 277\u2013290","journal-title":"Sustainable Computing: Informatics and Systems"},{"key":"190_CR19","doi-asserted-by":"crossref","unstructured":"Liu J, Agrawal G. Soft error detection for iterative applications using offline training. In: Proceedings of the 23rd IEEE International Conference on High Performance Computing. 2016, 2\u201311","DOI":"10.1109\/HiPC.2016.011"},{"key":"190_CR20","doi-asserted-by":"crossref","unstructured":"Hassani A, Skjellum A, Brightwell R. Design and evaluation of FA-MPI, a transactional resilience scheme for non-blocking MPI. In: Proceedings of the 44th Annual IEEE\/IFIP International Conference on Dependable Systems and Networks. 2014, 750\u2013755","DOI":"10.1109\/DSN.2014.78"},{"issue":"20","key":"190_CR21","doi-asserted-by":"crossref","first-page":"460","DOI":"10.1109\/TPDS.2008.128","volume":"4","author":"Y Li","year":"2009","unstructured":"Li Y, Lan Z, Gujrati P, Sun X H. Fault-aware runtime strategies for highperformance computing. IEEE Transactions on Parallel and Distributed Systems, 2009, 4(20): 460\u2013473","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"issue":"1","key":"190_CR22","first-page":"494","volume":"46","author":"P H Hargrove","year":"2006","unstructured":"Hargrove P H, Duell J C. Berkeley lab checkpoint\/restart (BLCR) for linux clusters. Journal of Physics: Conference Series, 2006, 46(1): 494\u2013499","journal-title":"Journal of Physics: Conference Series"},{"key":"190_CR23","unstructured":"Gomez L B, Tsuboi S, Komatitsch D, Cappello F, Maruyama N, Matsuoka S. FTI: high performance fault tolerance interface for hybrid systems. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis. 2011, 1\u201312"},{"issue":"1","key":"190_CR24","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1016\/j.future.2007.02.002","volume":"24","author":"D Buntinas","year":"2008","unstructured":"Buntinas D, Coti C, Herault T, Lemarinier P, Pilard L. Blocking vs. non-blocking coordinated checkpointing for large-scale fault tolerant MPI protocols. Future Generation Computer Systems, 2008, 24(1): 73\u201384","journal-title":"Future Generation Computer Systems"},{"key":"190_CR25","doi-asserted-by":"crossref","unstructured":"Di S, Bouguerra M S, Gomez L B, Cappello F. Optimization of multilevel checkpoint model for large scale HPC applications. In: Proceedings of the 28th IEEE International Parallel and Distributed Processing Symposium. 2014, 1181\u20131190","DOI":"10.1109\/IPDPS.2014.122"},{"key":"190_CR26","doi-asserted-by":"crossref","unstructured":"Ho J C Y, Wang C L, Lau F C M. Scalable group-based checkpoint\/restart for large-scale message-passing systems. In: Proceedings of 2008 IEEE International Symposium on Parallel and Distributed Processing. 2008, 1\u201312","DOI":"10.1109\/IPDPS.2008.4536302"},{"key":"190_CR27","doi-asserted-by":"crossref","unstructured":"Agarwal S, Garg R, Gupta M S, Moreira J E. Adaptive incremental checkpointing for massively parallel systems. In: Proceedings of the 18th Annual International Conference on Supercomputing. 2004, 277\u2013286","DOI":"10.1145\/1006209.1006248"},{"key":"190_CR28","doi-asserted-by":"crossref","unstructured":"Nicolae B, Cappello F. AI-Ckpt: leveraging memory access patterns for adaptive asynchronous incremental checkpointing. In: Proceedings of the 22nd International Symposium on High-Performance Parallel and Distributed Computing. 2013, 155\u2013166","DOI":"10.1145\/2493123.2462918"},{"key":"190_CR29","doi-asserted-by":"publisher","first-page":"66","DOI":"10.1016\/j.future.2013.04.017","volume":"30","author":"K B Ferreira","year":"2014","unstructured":"Ferreira K B, Riesen R, Bridges P, Arnold D, Brightwell R. Accelerating incremental checkpointing for extreme-scale computing. Future Generation Computer Systems, 2014, 30: 66\u201377","journal-title":"Future Generation Computer Systems"},{"key":"190_CR30","doi-asserted-by":"crossref","unstructured":"Nicolae B, Moody A, Gonsiorowski E, Mohror K, Cappello F. VeloC: towards high performance adaptive asynchronous checkpointing at large scale. In: Proceedings of IEEE International Parallel and Distributed Processing Symposium. 2019, 911\u2013920","DOI":"10.1109\/IPDPS.2019.00099"},{"key":"190_CR31","doi-asserted-by":"publisher","first-page":"450","DOI":"10.1016\/j.future.2018.09.041","volume":"91","author":"N Losada","year":"2019","unstructured":"Losada N, Bosilca G, Bouteiller A, Gonzalez P, Martin M J. Local rollback for resilient MPI applications with application-level checkpointing and message logging. Future Generation Computer Systems, 2019, 91: 450\u2013464","journal-title":"Future Generation Computer Systems"},{"key":"190_CR32","doi-asserted-by":"publisher","first-page":"1647","DOI":"10.1109\/TC.2008.90","volume":"57","author":"Z L Lan","year":"2008","unstructured":"Lan Z L, Li Y W. Adaptive fault management of parallel applications for high performance computing. IEEE Transactions on Computers, 2008, 57: 1647\u20131660","journal-title":"IEEE Transactions on Computers"},{"key":"190_CR33","doi-asserted-by":"crossref","unstructured":"Ibtesham D, Arnold D, Bridges P G, Ferreira K B, Brightwell R. On the viability of compression for reducing the overheads of checkpoint\/restart-based fault tolerance. In: Proceedings of the 41st International Conference on Parallel Processing. 2012, 148\u2013157","DOI":"10.1109\/ICPP.2012.45"},{"key":"190_CR34","unstructured":"Zhou P, Liu W, Fei L, Fei L, Lu S, Qin F, Zhou Y Y, Midkiff S, Torrellas G. Accmon: automatically detecting memory-related bugs via program counter-based invariants. In: Proceedings of the 37th Annual IEEE\/ACM International Symposium on Microarchitecture. 2004, 269\u2013280"},{"key":"190_CR35","doi-asserted-by":"crossref","unstructured":"Zheng Z, Li Y, Lan Z L. Anomaly localization in large-scale clusters. In: Proceedings of 2007 IEEE International Conference on Cluster Computing. 2007, 322\u2013330","DOI":"10.1109\/CLUSTR.2007.4629246"},{"issue":"1","key":"190_CR36","doi-asserted-by":"publisher","first-page":"1902","DOI":"10.1109\/TPDS.2015.2475741","volume":"27","author":"L Yu","year":"2016","unstructured":"Yu L, Lan Z L. A scalable, non-parametric method for detecting performance anomaly in large scale computing. IEEE Transactions on Parallel and Distributed Systems, 2016, 27(1): 1902\u20131914","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"190_CR37","doi-asserted-by":"crossref","unstructured":"Zheng Z, Yu L, Tang W, Lan Z L, Gupta R, Desai N, Coghlan S, Buettner D. Co-analysis of RAS log and job log on Blue Gene\/P. In: Proceddings of 2011 IEEE International Parallel & Distributed Processing Symposium. 2011, 840\u2013851","DOI":"10.1109\/IPDPS.2011.83"},{"key":"190_CR38","doi-asserted-by":"crossref","unstructured":"Berrocal E, Yu L, Wallace S, Papka M E, Lan Z L. Exploring void search for fault detection on extreme scale systems. In: Proceedings of IEEE International Conference on Cluster Computing. 2014, 1\u20139","DOI":"10.1109\/CLUSTER.2014.6968757"},{"key":"190_CR39","doi-asserted-by":"crossref","unstructured":"Gujrati P, Li Y, Lan Z L, Thakur R, White J. A meta-learning failure predictor for Blue Gene\/L systems. In: Proceedings of 2007 International Conference on Parallel Processing. 2007","DOI":"10.1109\/ICPP.2007.9"},{"key":"190_CR40","doi-asserted-by":"crossref","unstructured":"Zheng Z, Lan Z L, Park B H, Geist A. System log pre-processing to improve failure prediction. In: Proceedings of 2009 IEEE\/IFIP International Conference on Dependable Systems & Networks. 2009, 572\u2013577","DOI":"10.1109\/DSN.2009.5270289"}],"container-title":["Frontiers of Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-020-0190-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11704-020-0190-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-020-0190-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,7]],"date-time":"2024-09-07T16:04:59Z","timestamp":1725725099000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11704-020-0190-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,9,1]]},"references-count":40,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2021,12]]}},"alternative-id":["190"],"URL":"https:\/\/doi.org\/10.1007\/s11704-020-0190-y","relation":{},"ISSN":["2095-2228","2095-2236"],"issn-type":[{"type":"print","value":"2095-2228"},{"type":"electronic","value":"2095-2236"}],"subject":[],"published":{"date-parts":[[2021,9,1]]},"assertion":[{"value":"10 May 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 November 2020","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 September 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"156107"}}