{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T16:21:16Z","timestamp":1784391676466,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":23,"publisher":"ACM","license":[{"start":{"date-parts":[[2017,5,7]],"date-time":"2017-05-07T00:00:00Z","timestamp":1494115200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2017,5,7]]},"DOI":"10.1145\/3102980.3103005","type":"proceedings-article","created":{"date-parts":[[2017,7,20]],"date-time":"2017-07-20T17:51:38Z","timestamp":1500573098000},"page":"150-155","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":124,"title":["Gray Failure"],"prefix":"10.1145","author":[{"given":"Peng","family":"Huang","sequence":"first","affiliation":[{"name":"Microsoft Research, Johns Hopkins University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chuanxiong","family":"Guo","sequence":"additional","affiliation":[{"name":"Microsoft Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lidong","family":"Zhou","sequence":"additional","affiliation":[{"name":"Microsoft Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jacob R.","family":"Lorch","sequence":"additional","affiliation":[{"name":"Microsoft Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingnong","family":"Dang","sequence":"additional","affiliation":[{"name":"Microsoft Azure"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Murali","family":"Chintalapati","sequence":"additional","affiliation":[{"name":"Microsoft Azure"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Randolph","family":"Yao","sequence":"additional","affiliation":[{"name":"Microsoft Azure"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2017,5,7]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1402958.1402967"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.5555\/800253.807732"},{"key":"e_1_3_2_1_3_1","volume-title":"AWS service outage on October 22nd","author":"Amazon","year":"2012","unstructured":"Amazon . AWS service outage on October 22nd , 2012 . https:\/\/aws.amazon.com\/message\/680342. Amazon. AWS service outage on October 22nd, 2012. https:\/\/aws.amazon.com\/message\/680342."},{"key":"e_1_3_2_1_4_1","volume-title":"Introducing Data Center Fabric","author":"Andreyev A.","year":"2014","unstructured":"Andreyev , A. Introducing Data Center Fabric , The Next-generation Facebook Data Center Network . https:\/\/code.facebook.com\/posts\/360346274145943\/, Nov. 2014 . Andreyev, A. Introducing Data Center Fabric, The Next-generation Facebook Data Center Network. https:\/\/code.facebook.com\/posts\/360346274145943\/, Nov. 2014."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1755913.1755926"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.5555\/296806.296824"},{"key":"e_1_3_2_1_7_1","first-page":"217","volume-title":"Proceedings of the 11th USENIX Conference on Operating Systems Design and Implementation (OSDI) (Oct.","author":"Chow M.","year":"2014","unstructured":"Chow , M. , Meisner , D. , Flinn , J. , Peek , D. , and Wenisch , T. F . The mystery machine: End-to-end performance analysis of large-scale Internet services . In Proceedings of the 11th USENIX Conference on Operating Systems Design and Implementation (OSDI) (Oct. 2014 ), pp. 217 -- 231 . Chow, M., Meisner, D., Flinn, J., Peek, D., and Wenisch, T. F. The mystery machine: End-to-end performance analysis of large-scale Internet services. In Proceedings of the 11th USENIX Conference on Operating Systems Design and Implementation (OSDI) (Oct. 2014), pp. 217--231."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.5555\/1558977.1558988"},{"key":"e_1_3_2_1_9_1","first-page":"16","volume-title":"Proceedings of the 6th Conference on Symposium on Operating Systems Design and Implementation (OSDI)","author":"Cohen I.","year":"2004","unstructured":"Cohen , I. , Goldszmidt , M. , Kelly , T. , Symons , J. , and Chase , J. S . Correlating instrumentation data to system states: A building block for automated diagnosis and control . In Proceedings of the 6th Conference on Symposium on Operating Systems Design and Implementation (OSDI) ( 2004 ), pp. 16 -- 16 . Cohen, I., Goldszmidt, M., Kelly, T., Symons, J., and Chase, J. S. Correlating instrumentation data to system states: A building block for automated diagnosis and control. In Proceedings of the 6th Conference on Symposium on Operating Systems Design and Implementation (OSDI) (2004), pp. 16--16."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/1095810.1095821"},{"key":"e_1_3_2_1_11_1","first-page":"3","volume-title":"Proc. Symposium on Reliability in Distributed Software and Database Systems","author":"Gray J.","year":"1986","unstructured":"Gray , J. Why do computers stop and what can be done about it ? In Proc. Symposium on Reliability in Distributed Software and Database Systems ( 1986 ), pp. 3 -- 12 . Gray, J. Why do computers stop and what can be done about it? In Proc. Symposium on Reliability in Distributed Software and Database Systems (1986), pp. 3--12."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1592568.1592576"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2987550.2987583"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2785956.2787496"},{"key":"e_1_3_2_1_15_1","volume-title":"Why does a cloud-scale service fail despite fault-tolerance? Unpublished internal document","author":"Huang P.","year":"2014","unstructured":"Huang , P. , Jin , X. , Bolosky , W. J. , and Zhou , Y . Why does a cloud-scale service fail despite fault-tolerance? Unpublished internal document ( 2014 ). Huang, P., Jin, X., Bolosky, W. J., and Zhou, Y. Why does a cloud-scale service fail despite fault-tolerance? Unpublished internal document (2014)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/279227.279229"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.5555\/2482626.2482667"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2043556.2043583"},{"key":"e_1_3_2_1_19_1","volume-title":"Office 365 service incident on November 13th","author":"Microsoft","year":"2013","unstructured":"Microsoft . Office 365 service incident on November 13th , 2013 . https:\/\/blogs.office.com\/2012\/11\/13\/update-on-recent-customer-issues\/. Microsoft. Office 365 service incident on November 13th, 2013. https:\/\/blogs.office.com\/2012\/11\/13\/update-on-recent-customer-issues\/."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.5555\/1251460.1251461"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/50202.50214"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2785956.2787508"},{"key":"e_1_3_2_1_23_1","first-page":"91","volume-title":"Proceedings of the 6th Conference on Symposium on Operating Systems Design (OSDI) (Dec.","author":"van Renesse R.","year":"2004","unstructured":"van Renesse , R. , and Schneider , F. B . Chain replication for supporting high throughput and availability . In Proceedings of the 6th Conference on Symposium on Operating Systems Design (OSDI) (Dec. 2004 ), pp. 91 -- 104 . van Renesse, R., and Schneider, F. B. Chain replication for supporting high throughput and availability. In Proceedings of the 6th Conference on Symposium on Operating Systems Design (OSDI) (Dec. 2004), pp. 91--104."}],"event":{"name":"HotOS '17: Workshop on Hot Topics in Operating Systems","location":"Whistler BC Canada","acronym":"HotOS '17","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 16th Workshop on Hot Topics in Operating Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3102980.3103005","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3102980.3103005","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T03:37:06Z","timestamp":1750217826000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3102980.3103005"}},"subtitle":["The Achilles' Heel of Cloud-Scale Systems"],"short-title":[],"issued":{"date-parts":[[2017,5,7]]},"references-count":23,"alternative-id":["10.1145\/3102980.3103005","10.1145\/3102980"],"URL":"https:\/\/doi.org\/10.1145\/3102980.3103005","relation":{},"subject":[],"published":{"date-parts":[[2017,5,7]]},"assertion":[{"value":"2017-05-07","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}