{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T06:18:29Z","timestamp":1743142709220,"version":"3.40.3"},"publisher-location":"Cham","reference-count":21,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319232362"},{"type":"electronic","value":"9783319232379"}],"license":[{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015]]},"DOI":"10.1007\/978-3-319-23237-9_18","type":"book-chapter","created":{"date-parts":[[2015,8,24]],"date-time":"2015-08-24T09:48:07Z","timestamp":1440409687000},"page":"201-208","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Epidemic Fault Tolerance for Extreme-Scale Parallel Computing"],"prefix":"10.1007","author":[{"given":"Amogh","family":"Katti","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giuseppe","family":"Di Fatta","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,8,25]]},"reference":[{"key":"18_CR1","volume-title":"A proposal for User-Level Failure Mitigation in the MPI-3 standard","author":"W Bland","year":"2012","unstructured":"Bland, W., Bosilca, G., Bouteiller, A., Herault, T., Dongarra, J.: A proposal for User-Level Failure Mitigation in the MPI-3 standard. University of Tennessee, Department of Electrical Engineering and Computer Science (2012)"},{"key":"18_CR2","doi-asserted-by":"crossref","unstructured":"Bland, W., Bouteiller, A., Herault, T., Bosilca, G., Dongarra, J.J.: Post-failure recovery of MPI communication capability: Design and rationale. Int. J. High Perform. Comput. Appl. (2013)","DOI":"10.1177\/1094342013488238"},{"key":"18_CR3","unstructured":"Blasa, F., Cafiero, S., Fortino, G., Di Fatta, G.: Symmetric push-sum protocol for decentralised aggregation (2011)"},{"key":"18_CR4","doi-asserted-by":"crossref","unstructured":"Buntinas, D.: Scalable distributed consensus to support MPI fault tolerance. In: 26th IEEE International Conference on Parallel & Distributed Processing Symposium (IPDPS), May 2012, pp. 1240\u20131249 (2012)","DOI":"10.1109\/IPDPS.2012.113"},{"issue":"2","key":"18_CR5","doi-asserted-by":"publisher","first-page":"225","DOI":"10.1145\/226643.226647","volume":"43","author":"TD Chandra","year":"1996","unstructured":"Chandra, T.D., Toueg, S.: Unreliable failure detectors for reliable distributed systems. J. ACM (JACM) 43(2), 225\u2013267 (1996)","journal-title":"J. ACM (JACM)"},{"key":"18_CR6","unstructured":"Daly, J.T., Lead, R.: Application resilience for truculent systems. In: Workshop on Fault Tolerance for Extreme-Scale Computing, Albuquerque, NM \u2013 19\u201320 March 2009, ANL\/MCS-TM-312 (2009)"},{"key":"18_CR7","unstructured":"Daly, J., Harrod, B., Hoang, T., Nowell, L., Adolf, B., Borkar, S., Wu, J.: Inter-Agency Workshop on HPC resilience at extreme scale. In: National Security Agency Advanced Computing Systems, February 2012 (2012)"},{"issue":"3","key":"18_CR8","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1016\/j.jpdc.2012.09.009","volume":"73","author":"G Di Fatta","year":"2013","unstructured":"Di Fatta, G., Blasa, F., Cafiero, S., Fortino, G.: Fault tolerant decentralised K-Means clustering for asynchronous large-scale networks. J. Parallel Distrib. Comput. 73(3), 317\u2013329 (2013)","journal-title":"J. Parallel Distrib. Comput."},{"key":"18_CR9","unstructured":"Fault Tolerance Working Group. Run-though stabilization interfaces and semantics. In: svn. mpi-forum. org\/trac\/mpi-forum-web\/wiki\/ft\/run through stabilization (2012)"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Gupta, I., Chandra, T.D., Goldszmidt, G.S.: On scalable and efficient distributed failure detectors. In: Proceedings of the Twentieth Annual ACM Symposium on Principles of Distributed Computing, August 2001, pp. 170\u2013179. ACM (2001)","DOI":"10.1145\/383962.384010"},{"issue":"6","key":"18_CR11","doi-asserted-by":"publisher","first-page":"518","DOI":"10.1109\/TC.1984.1676475","volume":"100","author":"KH Huang","year":"1984","unstructured":"Huang, K.H., Abraham, J.A.: Algorithm-based fault tolerance for matrix operations. IEEE Trans. Comput. 100(6), 518\u2013528 (1984)","journal-title":"IEEE Trans. Comput."},{"key":"18_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"255","DOI":"10.1007\/978-3-642-24449-0_29","volume-title":"Recent Advances in the Message Passing Interface","author":"J Hursey","year":"2011","unstructured":"Hursey, J., Naughton, T., Vallee, G., Graham, R.L.: A log-scaling fault tolerant agreement algorithm for a fault tolerant MPI. In: Cotronis, Y., Danalis, A., Nikolopoulos, D.S., Dongarra, J. (eds.) EuroMPI 2011. LNCS, vol. 6960, pp. 255\u2013263. Springer, Heidelberg (2011)"},{"key":"18_CR13","unstructured":"Message Passing Interface Forum: MPI: A Message Passing Interface. In: Proceedings of Supercomputing 1993, pp. 878\u2013883. IEEE Computer Society Press (1993)"},{"key":"18_CR14","doi-asserted-by":"crossref","unstructured":"Montresor, A., Jelasity, M.: PeerSim: A scalable P2P simulator. In: IEEE Ninth International Conference on Peer-to-Peer Computing, P2P 2009, pp. 99\u2013100. IEEE (2009)","DOI":"10.1109\/P2P.2009.5284506"},{"key":"18_CR15","unstructured":"Process Recovery Proposal. https:\/\/svn.mpi-forum.org\/trac\/mpi-forum-web\/wiki\/ft\/process_recovery_2. Accessed: 14 May 2015"},{"issue":"3","key":"18_CR16","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1023\/A:1011494323443","volume":"4","author":"S Ranganathan","year":"2001","unstructured":"Ranganathan, S., George, A.D., Todd, R.W., Chidester, M.C.: Gossip-style failure detection and distributed consensus for scalable heterogeneous clusters. Cluster Comput. 4(3), 197\u2013209 (2001)","journal-title":"Cluster Comput."},{"key":"18_CR17","doi-asserted-by":"crossref","unstructured":"Schroeder, B., Gibson, G.A.: Understanding failures in petascale computers. In: Journal of Physics: Conference Series, vol. 78(1), p. 012022. IOP Publishing, July 2007","DOI":"10.1088\/1742-6596\/78\/1\/012022"},{"key":"18_CR18","doi-asserted-by":"crossref","unstructured":"Soltero, P., Bridges, P., Arnold, D., Lang, M.: A Gossip-based approach to exascale system services. In: Proceedings of the 3rd International Workshop on Runtime and Operating Systems for Supercomputers, p. 3. ACM, June 2013","DOI":"10.1145\/2491661.2481428"},{"key":"18_CR19","doi-asserted-by":"crossref","unstructured":"Song, H., Leangsuksun, C., Nassar, R., Gottumukkala, N.R., Scott, S.: Availability modeling and analysis on high performance cluster computing systems. In: The First International Conference on Availability, Reliability and Security, ARES 2006, April 2006, p.8. IEEE (2006)","DOI":"10.1109\/ARES.2006.37"},{"key":"18_CR20","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1016\/j.procs.2013.05.182","volume":"18","author":"H Strakov\u00e1","year":"2013","unstructured":"Strakov\u00e1, H., Niederbrucker, G., Gansterer, W.N.: Fault tolerance properties of gossip-based distributed orthogonal iteration methods. Procedia Comput. Sci. 18, 189\u2013198 (2013)","journal-title":"Procedia Comput. Sci."},{"key":"18_CR21","unstructured":"Taerat, N., Nakisinehaboon, N., Chandler, C., Elliot, J., Leangsuksun, C., Ostrouchov, G., Scott, S.L.: Using log information to perform statistical analysis on failures encountered by large-scale HPC deployments. In: Proceedings of the 2008 High Availability and Performance Computing Workshop, vol. 4, pp. 29\u201343 (2008)"}],"container-title":["Lecture Notes in Computer Science","Internet and Distributed Computing Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-23237-9_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,8]],"date-time":"2023-02-08T16:33:12Z","timestamp":1675873992000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-23237-9_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015]]},"ISBN":["9783319232362","9783319232379"],"references-count":21,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-23237-9_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2015]]},"assertion":[{"value":"25 August 2015","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}