{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T01:48:27Z","timestamp":1761788907574,"version":"3.37.3"},"reference-count":45,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2010,4]]},"DOI":"10.1109\/ipdpsw.2010.5470868","type":"proceedings-article","created":{"date-parts":[[2010,5,28]],"date-time":"2010-05-28T18:25:42Z","timestamp":1275071142000},"page":"1-8","source":"Crossref","is-referenced-by-count":7,"title":["Failure prediction for autonomic management of networked computer systems with availability assurance"],"prefix":"10.1109","author":[{"family":"Ziming Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Song","family":"Fu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2006.5"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2006.1639672"},{"key":"ref33","article-title":"Software failures and the road to a petafiop machine","author":"philp","year":"2005","journal-title":"Proc of the Symposium on High-Performance Computer Architecture"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2007.103"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2005.80"},{"key":"ref30","article-title":"Proactive fault tolerance for HPC with Xen virtualization","author":"nagaraj","year":"2007","journal-title":"Proceedings of ACM International Conference on Super computing (ICS)"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2004.1311948"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/956750.956799"},{"key":"ref35","article-title":"Big systems and big reliability challenges","author":"reed","year":"2003","journal-title":"Proceedings of Parallel Computing"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2005.02.003"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/568522.568525"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1088\/1742-6596\/78\/1\/012022"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1147\/rd.483.0519"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2009.21"},{"key":"ref13","doi-asserted-by":"crossref","DOI":"10.1016\/j.jpdc.2010.01.002","article-title":"Failure-aware resource management for high-availability computing clusters with distributed virtual ma-chines","author":"fu","year":"2010","journal-title":"Journal of Parallel and Distributed Computing In Press"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/1362622.1362678"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/SRDS.2007.18"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ARES.2009.13"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-2329-1"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TR.2006.884587"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2008.17"},{"key":"ref28","article-title":"Exploiting availability prediction in distributed systems","author":"mickens","year":"2006","journal-title":"Proceedings of USENIX Symposium on Networked Systems Design and Implementation (NSDI)"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1198\/016214501753382282"},{"key":"ref27","article-title":"Filtering failure logs for a BlueGene\/L prototype","author":"liang","year":"2005","journal-title":"Conference on Dependable Systems and Networks (DSN)"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.peva.2007.09.001"},{"key":"ref6","doi-asserted-by":"crossref","DOI":"10.1145\/1345206.1345253","article-title":"Compiler-enhanced incremental checkpointing for openmp applications","author":"bronevetsky","year":"2008","journal-title":"Proceedings of ACM Symposium on Principles and Practice of Parallel Programming (PPoPP)"},{"key":"ref29","doi-asserted-by":"crossref","first-page":"1135","DOI":"10.1109\/TSE.1987.232855","article-title":"on the reliability of the ibm mvs\/xa operating system","volume":"se 13","author":"mourad","year":"1987","journal-title":"IEEE Transactions on Software Engineering"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2005.70"},{"key":"ref8","article-title":"Proactive fault tolerance in MPI applications via task migration","author":"chakravorty","year":"2006","journal-title":"In Proceedings of IEEE International Conference on High Performance Computing"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2009.5270331"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2002.10017"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/71.993209"},{"journal-title":"Machine Learning Software in Java","year":"0","key":"ref1"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2007.72"},{"key":"ref45","article-title":"Performance implications of failures in large-scale cluster scheduling","author":"zhang","year":"2004","journal-title":"Proceedings of the 10th Workshop on Job Scheduling Strategies for Parallel Processing"},{"key":"ref22","article-title":"A power-aware run-time system for high-performance computing","author":"hsu","year":"2005","journal-title":"Proc ACM\/IEEE Conf Supercomputing (SC)"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/511334.511362"},{"key":"ref42","article-title":"A proactive fault-detection mechanism in large-scale cluster systems","author":"wu","year":"2006","journal-title":"In Proceedings of IEEE International Parallel and Distributed Processing Symposium (IPDPS)"},{"key":"ref24","article-title":"Exploit failure prediction for adaptive fault-tolerance in cluster computing","author":"li","year":"2006","journal-title":"Proc IEEE Int Symp Cluster Comput and the Grid (CCGrid)"},{"key":"ref41","article-title":"Proactive process-level live migration in HPC environments","author":"wang","year":"2008","journal-title":"Proc ACM\/IEEE Conf Supercomputing (SC)"},{"key":"ref23","article-title":"Transparent checkpoint-restart of multiple processes on commodity operating systems","author":"laadan","year":"2007","journal-title":"Proceedings of the USENIX Annual Technical Conference (USENIX"},{"key":"ref44","article-title":"Beyond availability: Towards a deeper understanding of machine failure characteristics in large distributed systems","author":"yalagandula","year":"2004","journal-title":"Proceedings of USENIX WORLDS"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2006.18"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/1362622.1362687"},{"key":"ref25","article-title":"Fast restart mechanism for check-point\/recovery protocols in networked environments","author":"li","year":"2008","journal-title":"In Proceedings of IEEE International Conference on Dependable Systems and Networks (DSN)"}],"event":{"name":"Distributed Processing, Workshops and Phd Forum (IPDPSW)","start":{"date-parts":[[2010,4,19]]},"location":"Atlanta, GA, USA","end":{"date-parts":[[2010,4,23]]}},"container-title":["2010 IEEE International Symposium on Parallel &amp; Distributed Processing, Workshops and Phd Forum (IPDPSW)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx5\/5465895\/5470678\/05470868.pdf?arnumber=5470868","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T10:32:30Z","timestamp":1740133950000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/5470868\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2010,4]]},"references-count":45,"URL":"https:\/\/doi.org\/10.1109\/ipdpsw.2010.5470868","relation":{},"subject":[],"published":{"date-parts":[[2010,4]]}}}