{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T04:53:53Z","timestamp":1750308833856,"version":"3.41.0"},"publisher-location":"Cham","reference-count":31,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319589428"},{"type":"electronic","value":"9783319589435"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-58943-5_50","type":"book-chapter","created":{"date-parts":[[2017,5,27]],"date-time":"2017-05-27T08:42:07Z","timestamp":1495874527000},"page":"623-634","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Horseshoes and Hand Grenades: The Case for Approximate Coordination in Local Checkpointing Protocols"],"prefix":"10.1007","author":[{"given":"Patrick M.","family":"Widener","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kurt B.","family":"Ferreira","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Scott","family":"Levy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,5,28]]},"reference":[{"key":"50_CR1","unstructured":"Adiga, N., et al.: An overview of the BlueGene\/L supercomputer. In: ACM\/IEEE 2002 Conference Supercomputing, p. 60, November 2002"},{"key":"50_CR2","doi-asserted-by":"crossref","unstructured":"Alm\u00e1si, G., Heidelberger, P., Archer, C.J., Martorell, X., Erway, C.C., Moreira, J.E., Steinmacher-Burow, B., Zheng, Y.: Optimization of MPI collective communication on BlueGene\/L systems. In: Proceedings of the 19th Annual International Conference on Supercomputing, ICS 2005, NY, USA, pp. 253\u2013262 (2005). http:\/\/doi.acm.org\/10.1145\/1088149.1088183","DOI":"10.1145\/1088149.1088183"},{"key":"50_CR3","unstructured":"Alvisi, L., Elnozahy, E., Rao, S., Husain, S., de Mel, A.: An analysis of communication induced checkpointing. In: Twenty-Ninth Annual International Symposium on Fault-Tolerant Computing, 1999, Digest of Papers, pp. 242\u2013249 (1999)"},{"issue":"2","key":"50_CR4","doi-asserted-by":"publisher","first-page":"149","DOI":"10.1109\/32.666828","volume":"24","author":"L Alvisi","year":"1998","unstructured":"Alvisi, L., Marzullo, K.: Message logging: pessimistic, optimistic, causal, and optimal. IEEE Trans. Softw. Eng. 24(2), 149\u2013159 (1998)","journal-title":"IEEE Trans. Softw. Eng."},{"key":"50_CR5","unstructured":"Bergman, K., Borkar, S., Campbell, D., Carlson, W., Dally, W., Denneau, M., Franzon, P., Harrod, W., Hill, K., Hiller, J., Karp, S., Keckler, S., Klein, D., Kogge, P., Lucas, R., Richards, M., Scarpelli, A., Scott, S., Snavely, A., Sterling, T., Williams, R.S., Yelick, K.: Exascale computing study: technology challenges in achieving exascale systems, September 2008. http:\/\/www.science.energy.gov\/ascr\/Research\/CS\/DARPA\/exascale-hardware(2008).pdf"},{"key":"50_CR6","doi-asserted-by":"crossref","unstructured":"Bronevetsky, G.: Communication-sensitive static dataflow for parallel message passing applications. In: Proceedings of the 7th Annual IEEE\/ACM International Symposium on Code Generation and Optimization, pp. 1\u201312. IEEE Computer Society (2009)","DOI":"10.1109\/CGO.2009.32"},{"issue":"3","key":"50_CR7","first-page":"212","volume":"23","author":"F Cappello","year":"2009","unstructured":"Cappello, F.: Fault tolerance in petascale\/exascale systems: current knowledge, challenges and research opportunities. IJHPCA 23(3), 212\u2013226 (2009)","journal-title":"IJHPCA"},{"issue":"1","key":"50_CR8","doi-asserted-by":"publisher","first-page":"63","DOI":"10.1145\/214451.214456","volume":"3","author":"KM Chandy","year":"1985","unstructured":"Chandy, K.M., Lamport, L.: Distributed snapshots: determining global states of distributed systems. ACM Trans. Comp. Syst. 3(1), 63\u201375 (1985)","journal-title":"ACM Trans. Comp. Syst."},{"issue":"7","key":"50_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/173284.155333","volume":"28","author":"D Culler","year":"1993","unstructured":"Culler, D., Karp, R., Patterson, D., Sahay, A., Schauser, K.E., Santos, E., Subramonian, R., von Eicken, T.: LogP: towards a realistic model of parallel computation. SIGPLAN Not. 28(7), 1\u201312 (1993)","journal-title":"SIGPLAN Not."},{"key":"50_CR10","doi-asserted-by":"crossref","unstructured":"De, P., Kothari, R., Mann, V.: A trace-driven emulation framework to predict scalability of large clusters in presence of OS jitter. In: 2008 IEEE International Conference on Cluster Computing, pp. 232\u2013241. IEEE (2008)","DOI":"10.1109\/CLUSTR.2008.4663776"},{"key":"50_CR11","doi-asserted-by":"crossref","unstructured":"Hertel Jr., E.S., Bell, R.L., Elrick, M.G., Farnsworth, A.V., Kerley, G.I., McGlaun, J.M., Petney, S.V., Silling, S.A., Taylor, P.A., Yarrington, L.: CTH: a software family for multi-dimensional shock physics analysis. In: Proceedings of the 19th International Symposium on Shock Waves, pp. 377\u2013382 July 1993","DOI":"10.1007\/978-3-642-78829-1_61"},{"issue":"3","key":"50_CR12","doi-asserted-by":"publisher","first-page":"375","DOI":"10.1145\/568522.568525","volume":"34","author":"EN Elnozahy","year":"2002","unstructured":"Elnozahy, E.N., Alvisi, L., Wang, Y.M., Johnson, D.B.: A survey of rollback-recovery protocols in message-passing systems. ACM Comput. Surv. 34(3), 375\u2013408 (2002)","journal-title":"ACM Comput. Surv."},{"key":"50_CR13","unstructured":"Engelmann, C.: Investigating operating system noise in extreme-scale high-performance computing systems using simulation. In: Proceedings of the 11th IASTED International Conference on Parallel and Distributed Computing and Networks (PDCN 2013), pp. 11\u201313 (2013)"},{"key":"50_CR14","unstructured":"Exascale Co-Design Center for Materials in Extreme Environments (ExMatEx). http:\/\/exmatex.lanl.gov\/. Accessed 10 June 2013"},{"key":"50_CR15","doi-asserted-by":"crossref","unstructured":"Ferreira, K., Riesen, R., Stearley, J., Laros, J.H., Oldfield, R., Pedretti, K., Bridges, P., Arnold, D., Brightwell, R.: Evaluating the viability of process replication reliability for exascale systems. In: Proceedings of the ACM\/IEEE International Conference on High Performance Computing, Networking, Storage, and Analysis, (SC 2011), November 2011","DOI":"10.1145\/2063384.2063443"},{"key":"50_CR16","doi-asserted-by":"crossref","unstructured":"Ferreira, K.B., Brightwell, R., Bridges, P.G.: Characterizing application sensitivity to OS interference using kernel-level noise injection. In: Proceedings of the 2008 ACM\/IEEE Conference on Supercomputing (SC 2008), November 2008","DOI":"10.1109\/SC.2008.5219920"},{"key":"50_CR17","doi-asserted-by":"crossref","unstructured":"Ferreira, K.B., Widener, P., Levy, S., Arnold, D., Hoefler, T.: Understanding the effects of communication and coordination on checkpointing at scale. In: Proceedings of the 2014 International Conference for High Performance Computing, Networking, Storage and Analysis (Supercomputing) (2014)","DOI":"10.1109\/SC.2014.77"},{"key":"50_CR18","doi-asserted-by":"crossref","unstructured":"Guermouche, A., Ropars, T., Brunet, E., Snir, M., Cappello, F.: Uncoordinated checkpointing without domino effect for send-deterministic MPI applications. In: International Parallel Distributed Processing Symposium (IPDPS), pp. 989\u20131000, May 2011","DOI":"10.1109\/IPDPS.2011.95"},{"key":"50_CR19","unstructured":"Heroux, M.A., Doerfler, D.W., Crozier, P.S., Willenbring, J.M., Edwards, H.C., Williams, A., Rajan, M., Keiter, E.R., Thornquist, H.K., Numrich, R.W.: Improving performance via mini-applications. Technical report, SAND2009-5574, Sandia National Laboratories (2009)"},{"key":"50_CR20","doi-asserted-by":"crossref","unstructured":"Hoefler, T., Schneider, T., Lumsdaine, A.: Characterizing the influence of system noise on large-scale applications by simulation. In: Proceedings of the 2010 ACM\/IEEE International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201311. IEEE Computer Society (2010)","DOI":"10.1109\/SC.2010.12"},{"key":"50_CR21","doi-asserted-by":"crossref","unstructured":"Hoefler, T., Schneider, T., Lumsdaine, A.: LogGOPSim: simulating large-scale applications in the LogGOPS model. In: Proceedings of the 19th ACM International Symposium on High Performance Distributed Computing, pp. 597\u2013604. ACM (2010)","DOI":"10.1145\/1851476.1851564"},{"key":"50_CR22","doi-asserted-by":"crossref","unstructured":"Johnson, D.B., Zwaenepoel, W.: Recovery in distributed systems using asynchronous message logging and checkpointing. In: Proceedings of the Seventh Annual ACM Symposium on Principles of Distributed Computing, pp. 171\u2013181 (1988)","DOI":"10.1145\/62546.62575"},{"key":"50_CR23","doi-asserted-by":"crossref","unstructured":"Karlin, I., Bhatele, A., Chamberlain, B.L., Cohen, J., Devito, Z., Gokhale, M., Haque, R., Hornung, R., Keasler, J., Laney, D., Luke, E., Lloyd, S., McGraw, J., Neely, R., Richards, D., Schulz, M., Still, C.H., Wang, F., Wong, D.: LULESH programming model and performance ports overview. Technical report LLNL-TR-608824, Lawrence Livermore National Laboratory, December 2012","DOI":"10.2172\/1059462"},{"key":"50_CR24","doi-asserted-by":"crossref","unstructured":"Levy, S., Topp, B., Ferreira, K.B., Arnold, D., Hoefler, T., Widener, P.: Using simulation to evaluate the performance of resilience strategies at scale. In: 2013 SC Companion: High Performance Computing, Networking, Storage and Analysis (SCC). IEEE (2013)","DOI":"10.1007\/978-3-319-10214-6_5"},{"key":"50_CR25","doi-asserted-by":"crossref","unstructured":"Levy, S., Topp, B., Ferreira, K.B., Arnold, D., Widener, P., Hoefler, T.: Using simulation to evaluate the performance of resilience strategies and process failures. Technical report SAND2014-0688, Sandia National Laboratories (2014)","DOI":"10.2172\/1204092"},{"issue":"12","key":"50_CR26","doi-asserted-by":"publisher","first-page":"1632","DOI":"10.1002\/cpe.1413","volume":"21","author":"A Maloney","year":"2009","unstructured":"Maloney, A., Goscinski, A.: A survey and review of the current state of rollback-recovery for cluster systems. Concurr. Comput. Pract. Exp. 21(12), 1632\u20131666 (2009). doi:10.1002\/cpe.1413. ISSN 1532-0634","journal-title":"Concurr. Comput. Pract. Exp."},{"issue":"10","key":"50_CR27","doi-asserted-by":"publisher","first-page":"1482","DOI":"10.1109\/26.103043","volume":"39","author":"DL Mills","year":"1991","unstructured":"Mills, D.L.: Internet time synchronization: the network time protocol. IEEE Trans. Commun. 39(10), 1482\u20131493 (1991)","journal-title":"IEEE Trans. Commun."},{"key":"50_CR28","doi-asserted-by":"crossref","unstructured":"Monnet, S., Morin, C., Badrinath, R.: A hierarchical checkpointing protocol for parallel applications in cluster federations. In: 2004 Proceedings of 18th International Parallel and Distributed Processing Symposium, p. 211. IEEE (2004)","DOI":"10.1109\/IPDPS.2004.1303242"},{"key":"50_CR29","doi-asserted-by":"crossref","unstructured":"Murta, C.D., Torres Jr., P.R.T., Mohapatra, P.: Characterizing quality of time and topology in a time synchronization network. In: IEEE Globecom 2006, pp. 1\u20135, November 2006","DOI":"10.1109\/GLOCOM.2006.467"},{"issue":"1","key":"50_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1006\/jcph.1995.1039","volume":"117","author":"S Plimpton","year":"1995","unstructured":"Plimpton, S.: Fast parallel algorithms for short-range molecular dynamics. J. Comput. Phys. 117(1), 1\u201319 (1995)","journal-title":"J. Comput. Phys."},{"key":"50_CR31","unstructured":"Sandia National Laboratory: Mantevo project home page, January 2014. http:\/\/mantevo.org"}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2016: Parallel Processing Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-58943-5_50","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T20:39:10Z","timestamp":1750279150000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-58943-5_50"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319589428","9783319589435"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-58943-5_50","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2017]]},"assertion":[{"value":"28 May 2017","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"Euro-Par","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Grenoble","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2016","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 August 2016","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 August 2016","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"europar2016","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/europar2016.inria.fr\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}