{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,3]],"date-time":"2025-05-03T19:26:58Z","timestamp":1746300418396,"version":"3.28.0"},"reference-count":30,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2010,6]]},"DOI":"10.1109\/iscc.2010.5546715","type":"proceedings-article","created":{"date-parts":[[2010,8,18]],"date-time":"2010-08-18T14:16:52Z","timestamp":1282141012000},"page":"790-795","source":"Crossref","is-referenced-by-count":8,"title":["Dependability enhancement for coalition clusters with autonomic failure management"],"prefix":"10.1109","author":[{"given":"Song","family":"Fu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2010.5470868"},{"key":"ref10","doi-asserted-by":"crossref","DOI":"10.1016\/j.jpdc.2010.06.010","article-title":"Quantifying event correlations for proactive failure management in networked computing systems","author":"fu","year":"2010","journal-title":"Journal of Parallel and Distributed Computing"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-2329-1"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TR.2006.884587"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2008.17"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/511334.511362"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2006.18"},{"key":"ref16","article-title":"Filtering failure logs for a BlueGene\/L prototype","author":"liang","year":"2005","journal-title":"Proc of Conf on Dependable Systems and Networks (DSN)"},{"key":"ref17","article-title":"Exploiting availability prediction in distributed systems","author":"mickens","year":"2006","journal-title":"Proc of USENIX Symp on Networked Systems Design and Implementation (NSDI)"},{"key":"ref18","doi-asserted-by":"crossref","first-page":"1135","DOI":"10.1109\/TSE.1987.232855","article-title":"on the reliability of the ibm mvs\/xa operating system","volume":"se 13","author":"mourad","year":"1987","journal-title":"IEEE Transactions on Software Engineering"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/1274971.1274978"},{"key":"ref28","article-title":"Networked Windows NT system field failure data analysis","author":"xu","year":"1999","journal-title":"Proc Pacific Rim Int Symp Dependable Computing (PRDC 01)"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/568522.568525"},{"key":"ref27","article-title":"Proactive processlevel live migration in HPC environments","author":"wang","year":"2008","journal-title":"Proc ACM\/IEEE Conf Supercomputing (SC)"},{"key":"ref3","article-title":"Proactive fault tolerance in MPI applications via task migration","author":"chakravorty","year":"2006","journal-title":"Proc of IEEE Intl Conf on High Performance Computing"},{"key":"ref6","doi-asserted-by":"crossref","first-page":"384","DOI":"10.1016\/j.jpdc.2010.01.002","article-title":"Failure-aware resource management for high-availability computing clusters with distributed virtual machines","volume":"70","author":"fu","year":"2010","journal-title":"Journal of Parallel and Distributed Computing"},{"key":"ref29","article-title":"Beyond availability: Towards a deeper understanding of machine failure characteristics in large distributed systems","author":"yalagandula","year":"2004","journal-title":"Proc of Usenix WORLDS"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2009.21"},{"key":"ref8","article-title":"Exploring event correlaton for failure prediction in coalitions of clusters","author":"fu","year":"2007","journal-title":"Proc of ACM\/IEEE Supercomputing Conference"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/SRDS.2007.18"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2009.5270331"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ARES.2009.13"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1198\/016214501753382282"},{"key":"ref20","article-title":"Software failures and the road to a petaflop machine","author":"philp","year":"2005","journal-title":"Proc of Symp on High Performance Computer Architecture Workshop"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2004.1311948"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/956750.956799"},{"key":"ref24","article-title":"Impact of correlated failures on dependability in a VAXcluster system","author":"tang","year":"1991","journal-title":"Proc of IFIP Working Conf on Dependable Computing for Critical Applications"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2006.5"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/378420.378434"},{"key":"ref25","article-title":"Failure analysis and modelling of a VAXcluster system","author":"tang","year":"1990","journal-title":"Proc of IEEE Intl Symp on FaultTolerant Computing (FTCS)"}],"event":{"name":"2010 IEEE Symposium on Computers and Communications (ISCC)","start":{"date-parts":[[2010,6,22]]},"location":"Riccione, Italy","end":{"date-parts":[[2010,6,25]]}},"container-title":["The IEEE symposium on Computers and Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx5\/5532341\/5546495\/05546715.pdf?arnumber=5546715","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2017,6,19]],"date-time":"2017-06-19T08:49:46Z","timestamp":1497862186000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/5546715\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2010,6]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/iscc.2010.5546715","relation":{},"subject":[],"published":{"date-parts":[[2010,6]]}}}