{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T00:55:02Z","timestamp":1740099302809,"version":"3.37.3"},"publisher-location":"Cham","reference-count":39,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030206550"},{"type":"electronic","value":"9783030206567"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-20656-7_14","type":"book-chapter","created":{"date-parts":[[2019,6,4]],"date-time":"2019-06-04T23:02:40Z","timestamp":1559689360000},"page":"271-290","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["End-to-End Resilience for HPC Applications"],"prefix":"10.1007","author":[{"given":"Arash","family":"Rezaei","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Harsh","family":"Khetawat","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Onkar","family":"Patil","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0258-0294","authenticated-orcid":false,"given":"Frank","family":"Mueller","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Paul","family":"Hargrove","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eric","family":"Roman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,5,17]]},"reference":[{"issue":"1","key":"14_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1279711.1279713","volume":"3","author":"JH Anderson","year":"2006","unstructured":"Anderson, J.H., Calandrino, J.M.: Parallel task scheduling on multicore platforms. SIGBED Rev. 3(1), 1\u20136 (2006)","journal-title":"SIGBED Rev."},{"key":"14_CR2","doi-asserted-by":"crossref","unstructured":"Biswas, S., Supinski, B.R.D., Schulz, M., Franklin, D., Sherwood, T., Chong, F.T.: Exploiting data similarity to reduce memory footprints. In: IPDPS, pp. 152\u2013163 (2011)","DOI":"10.1109\/IPDPS.2011.24"},{"key":"14_CR3","doi-asserted-by":"crossref","unstructured":"Blumofe, R.D., Joerg, C.F., Kuszmaul, B.C., Leiserson, C.E., Randall, K.H., Zhou, Y.: Cilk: an efficient multithreaded runtime system. In: PPoPP, pp. 207\u2013216 (1995)","DOI":"10.1145\/209937.209958"},{"key":"14_CR4","doi-asserted-by":"crossref","unstructured":"B\u00f6hm, S., Engelmann, C.: File I\/O for MPI applications in redundant execution scenarios. In: Parallel, Distributed, and Network-Based Processing, February 2012","DOI":"10.1109\/PDP.2012.22"},{"issue":"6","key":"14_CR5","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1109\/MCSE.2013.98","volume":"15","author":"G Bosilca","year":"2013","unstructured":"Bosilca, G., Bouteiller, A., Danalis, A., Faverge, M., Herault, T., Dongarra, J.: PaRSEC: exploiting heterogeneity to enhance scalability. Comput. Sci. Eng. 15(6), 36\u201345 (2013)","journal-title":"Comput. Sci. Eng."},{"key":"14_CR6","doi-asserted-by":"crossref","unstructured":"Cao, C., Herault, T., Bosilca, G., Dongarra, J.: Design for a soft error resilient dynamic task-based runtime. In: IPDPS, pp. 765\u2013774, May 2015","DOI":"10.1109\/IPDPS.2015.81"},{"key":"14_CR7","doi-asserted-by":"crossref","unstructured":"Chen, S., et al.: Scheduling threads for constructive cache sharing on CMPs. In: SPAA, pp. 105\u2013115 (2007)","DOI":"10.1145\/1248377.1248396"},{"key":"14_CR8","unstructured":"Chen, Z., Wu, P.: Fail-stop failure algorithm-based fault tolerance for cholesky decomposition. IEEE TPDS 99(PrePrints), 1 (2014)"},{"key":"14_CR9","doi-asserted-by":"crossref","unstructured":"Chung, J., et al.: Containment domains: a scalable, efficient, and flexible resilience scheme for exascale systems. In: Supercomputing, pp. 58:1\u201358:11 (2012)","DOI":"10.1109\/SC.2012.36"},{"issue":"12","key":"14_CR10","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1109\/MC.2009.385","volume":"42","author":"C Dave","year":"2009","unstructured":"Dave, C., Bae, H., Min, S.J., Lee, S., Eigenmann, R., Midkiff, S.: Cetus: a source-to-source compiler infrastructure for multicores. Computer 42(12), 36\u201342 (2009)","journal-title":"Computer"},{"key":"14_CR11","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"660","DOI":"10.1007\/978-3-319-58943-5_53","volume-title":"Euro-Par 2016: Parallel Processing Workshops","author":"PC Diniz","year":"2017","unstructured":"Diniz, P.C., Liao, C., Quinlan, D.J., Lucas, R.F.: Pragma-controlled source-to-source code transformations for robust application execution. In: Desprez, F., et al. (eds.) Euro-Par 2016. LNCS, vol. 10104, pp. 660\u2013670. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-58943-5_53"},{"key":"14_CR12","doi-asserted-by":"crossref","unstructured":"Du, P., Bouteiller, A., Bosilca, G., Herault, T., Dongarra, J.: Algorithm-based fault tolerance for dense matrix factorizations. In: PPoPP, pp. 225\u2013234 (2012)","DOI":"10.1145\/2370036.2145845"},{"key":"14_CR13","doi-asserted-by":"crossref","unstructured":"Duell, J.: The design and implementation of Berkeley Labs Linux Checkpoint\/Restart. Technical report, LBNL (2003)","DOI":"10.2172\/793773"},{"issue":"2","key":"14_CR14","doi-asserted-by":"publisher","first-page":"173","DOI":"10.1142\/S0129626411000151","volume":"21","author":"A Duran","year":"2011","unstructured":"Duran, A., et al.: OmpSs: a proposal for programming heterogeneous multi-core architectures. Parall. Process. Lett. 21(2), 173\u2013193 (2011)","journal-title":"Parall. Process. Lett."},{"key":"14_CR15","doi-asserted-by":"crossref","unstructured":"Elliott, J., Hoemmen, M., Mueller, F.: Evaluating the impact of SDC on the GMRES iterative solver. In: IPDPS, pp. 1193\u20131202 (2014)","DOI":"10.1109\/IPDPS.2014.123"},{"key":"14_CR16","doi-asserted-by":"crossref","unstructured":"Elliott, J., Kharbas, K., Fiala, D., Mueller, F., Ferreira, K., Engelmann, C.: Combining partial redundancy and checkpointing for HPC. In: ICDCS, 18\u201321 June 2012","DOI":"10.1109\/ICDCS.2012.56"},{"key":"14_CR17","doi-asserted-by":"crossref","unstructured":"Fiala, D., Mueller, F., Engelmann, C., Ferreira, K., Brightwell, R.: Detection and correction of silent data corruption for large-scale high-performance computing. In: Supercomputing (2012)","DOI":"10.2172\/1081941"},{"key":"14_CR18","unstructured":"Geist, A.: How to kill a supercomputer: dirty power, cosmic rays, and bad solder. In: IEEE Spectrum, February 2016"},{"issue":"6","key":"14_CR19","doi-asserted-by":"publisher","first-page":"518","DOI":"10.1109\/TC.1984.1676475","volume":"C\u201333","author":"KH Huang","year":"1984","unstructured":"Huang, K.H., Abraham, J.: Algorithm-based fault tolerance for matrix operations. IEEE Trans. Comput. C\u201333(6), 518\u2013528 (1984)","journal-title":"IEEE Trans. Comput."},{"key":"14_CR20","doi-asserted-by":"crossref","unstructured":"Islam, T.Z., Mohror, K., Bagchi, S., Moody, A., de Supinski, B.R., Eigenmann, R.: MCREngine: a scalable checkpointing system using data-aware aggregation and compression. In: Supercomputing, pp. 17:1\u201317:11 (2012)","DOI":"10.1109\/SC.2012.77"},{"issue":"10","key":"14_CR21","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1145\/167962.165874","volume":"28","author":"Laxmikant V. Kale","year":"1993","unstructured":"Kale, L.V., Krishnan, S.: Charm++: a portable concurrent object oriented system based on c++. In: OOPSLA, pp. 91\u2013108 (1993)","journal-title":"ACM SIGPLAN Notices"},{"key":"14_CR22","doi-asserted-by":"crossref","unstructured":"Kiczales, G., et al.: Aspect-oriented programming. In: ECOOP, pp. 220\u2013242 (1997)","DOI":"10.1007\/BFb0053381"},{"key":"14_CR23","doi-asserted-by":"crossref","unstructured":"Li, S., Sridharan, V., Gurumurthi, S., Yalamanchili, S.: Software-based dynamic reliability management for GPU applications. In: Workshop in Silicon Errors in Logic System Effects (2015)","DOI":"10.1109\/IRPS.2016.7574507"},{"key":"14_CR24","doi-asserted-by":"crossref","unstructured":"Martsinkevich, T., Subasi, O., Unsal, O., Cappello, F., Labarta, J.: Fault-tolerant protocol for hybrid task-parallel message-passing applications. In: Cluster Computing, pp. 563\u2013570, September 2015","DOI":"10.1109\/CLUSTER.2015.104"},{"key":"14_CR25","unstructured":"Min, S., Iancu, C., Yelick, K.: Hierarchical work stealing on manycore clusters. In: Partitioned Global Address Space Programming Models (2011)"},{"key":"14_CR26","doi-asserted-by":"crossref","unstructured":"Moody, A., Bronevetsky, G., Mohror, K., Supinski, B.R.D.: Design, modeling, and evaluation of a scalable multi-level checkpointing system. In: Supercomputing, pp. 1\u201311 (2010)","DOI":"10.2172\/984082"},{"key":"14_CR27","unstructured":"Panzer-Steindel, B.: Data integrity. Technical report, 1.3, CERN (2007)"},{"issue":"7","key":"14_CR28","doi-asserted-by":"publisher","first-page":"789","DOI":"10.1002\/spe.4380250705","volume":"25","author":"T Parr","year":"1995","unstructured":"Parr, T., Quong, R.: ANTLR: a predicated. Softw. Pract. Exp. 25(7), 789\u2013810 (1995)","journal-title":"Softw. Pract. Exp."},{"key":"14_CR29","unstructured":"Schroeder, B., Gibson, G.A.: A large-scale study of failures in high-performance computing systems. In: DSN, pp. 249\u2013258 (2006)"},{"issue":"1","key":"14_CR30","doi-asserted-by":"crossref","first-page":"193","DOI":"10.1145\/2492101.1555372","volume":"37","author":"B Schroeder","year":"2009","unstructured":"Schroeder, B., Pinheiro, E., Weber, W.D.: Dram errors in the wild: a large-scale field study. SIGMETRICS Perform. Eval. Rev. 37(1), 193\u2013204 (2009)","journal-title":"SIGMETRICS Perform. Eval. Rev."},{"key":"14_CR31","doi-asserted-by":"crossref","unstructured":"Shantharam, M., Srinivasmurthy, S., Raghavan, P.: Fault tolerant preconditioned conjugate gradient for sparse linear system solution. In: Supercomputing, pp. 69\u201378 (2012)","DOI":"10.1145\/2304576.2304588"},{"key":"14_CR32","unstructured":"Simon, T.A., Dorband, J.: Improving application resilience through probabilistic task replication. In: Workshop on Algorithmic and Application Error Resilience, June 2013"},{"key":"14_CR33","unstructured":"Snir, M., et al.: Addressing failures in exascale computing. Int. J. High Perform. Comput. (2013)"},{"key":"14_CR34","doi-asserted-by":"crossref","unstructured":"Sridharan, V., Kaeli, D.: Eliminating microarchitectural dependency from Architectural Vulnerability. In: HPCA, pp. 117\u2013128, February 2009","DOI":"10.1109\/HPCA.2009.4798243"},{"key":"14_CR35","doi-asserted-by":"crossref","unstructured":"Sridharan, V., et al.: Memory errors in modern systems: the good, the bad, and the ugly. In: ASPLOS, pp. 297\u2013310 (2015)","DOI":"10.1145\/2775054.2694348"},{"key":"14_CR36","doi-asserted-by":"crossref","unstructured":"Yim, K.S., Pham, C., Saleheen, M., Kalbarczyk, Z., Iyer, R.: Hauberk: lightweight silent data corruption error detector for GPGPU. In: IPDPS, pp. 287\u2013300 (2011)","DOI":"10.1109\/IPDPS.2011.36"},{"key":"14_CR37","doi-asserted-by":"crossref","unstructured":"Yu, L., Li, D., Mittal, S., Vetter, J.S.: Quantitatively modeling application resilience with the data vulnerability factor. In: Supercomputing, pp. 695\u2013706 (2014)","DOI":"10.1109\/SC.2014.62"},{"key":"14_CR38","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Mueller, F., Cui, X., Potok, T.: Large-scale multi-dimensional document clustering on GPU clusters. In: IPDPS, pp. 1\u201310, April 2010","DOI":"10.1109\/IPDPS.2010.5470429"},{"key":"14_CR39","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"124","DOI":"10.1007\/978-3-319-17353-5_11","volume-title":"High Performance Computing for Computational Science \u2013 VECPAR 2014","author":"Z Zheng","year":"2015","unstructured":"Zheng, Z., Chien, A.A., Teranishi, K.: Fault tolerance in an inner-outer solver: a GVR-enabled case study. In: Dayd\u00e9, M., Marques, O., Nakajima, K. (eds.) VECPAR 2014. LNCS, vol. 8969, pp. 124\u2013132. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-17353-5_11"}],"container-title":["Lecture Notes in Computer Science","High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-20656-7_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,19]],"date-time":"2022-09-19T13:38:31Z","timestamp":1663594711000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-20656-7_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030206550","9783030206567"],"references-count":39,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-20656-7_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"17 May 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ISC High Performance","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on High Performance Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Frankfurt","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 June 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 June 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"34","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"supercomputing2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.isc-hpc.com\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"Linklings","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"70","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"17","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"24% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"4-5","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"n\/a","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}}]}}