{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,18]],"date-time":"2026-05-18T10:00:31Z","timestamp":1779098431238,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,5,5]],"date-time":"2025-05-05T00:00:00Z","timestamp":1746403200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,5]]},"DOI":"10.1145\/3676151.3719372","type":"proceedings-article","created":{"date-parts":[[2025,5,3]],"date-time":"2025-05-03T00:57:09Z","timestamp":1746233829000},"page":"69-80","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["An Empirical Characterization of Outages and Incidents in Public Services for Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0353-7968","authenticated-orcid":false,"given":"Xiaoyu","family":"Chu","sequence":"first","affiliation":[{"name":"Vrije Universiteit, Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3461-4919","authenticated-orcid":false,"given":"Sacheendra","family":"Talluri","sequence":"additional","affiliation":[{"name":"Vrije Universiteit, Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3289-9638","authenticated-orcid":false,"given":"Qingxian","family":"Lu","sequence":"additional","affiliation":[{"name":"Vrije Universiteit, Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8030-9398","authenticated-orcid":false,"given":"Alexandru","family":"Iosup","sequence":"additional","affiliation":[{"name":"Vrije Universiteit, Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,5,5]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2025. Azure OpenAI Service. https:\/\/azure.microsoft.com\/en-us\/products\/aiservices\/openai-service."},{"key":"e_1_3_2_1_2_1","unstructured":"2025. Scaling Character.AI. https:\/\/cloud.google.com\/blog\/products\/databases\/why-characterai-chose-spanner-and-alloydb-for-postgresql."},{"key":"e_1_3_2_1_3_1","unstructured":"2025. Use Anthropic's Claude models. https:\/\/cloud.google.com\/vertex-ai\/generative-ai\/docs\/partner-models\/use-claude."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356172"},{"key":"e_1_3_2_1_5_1","unstructured":"Anthropic. 2024. Release notes of Anthropic API. https:\/\/docs.anthropic.com\/en\/release-notes\/api."},{"key":"e_1_3_2_1_6_1","unstructured":"Anthropic. 2024. Release notes of Claude. https:\/\/docs.anthropic.com\/en\/releasenotes\/claude-apps."},{"key":"e_1_3_2_1_7_1","unstructured":"Internet Archive. 2024. Archived page of OpenAI status at 2024\/04\/10. https:\/\/web. archive.org\/web\/20240410235249\/https:\/\/downdetector.com\/status\/openai\/ Accessed:2024-09--30."},{"key":"e_1_3_2_1_8_1","unstructured":"Atlassian. 2024. Atlassian Support. https:\/\/support.atlassian.com\/statuspage\/docs\/display-historical-uptime-of-components\/."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TDSC.2004.2"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2019.2949986"},{"key":"e_1_3_2_1_11_1","volume-title":"17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020","author":"Burnett Sam","year":"2020","unstructured":"Sam Burnett, Lily Chen, Douglas A. Creager, Misha Efimov, Ilya Grigorik, Ben Jones, Harsha V. Madhyastha, Pavlos Papageorge, Brian Rogan, Charles Stahl, and Julia Tuttle. 2020. Network Error Logging: Client-side measurement of endto- end web service reliability. In 17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020, Santa Clara, CA, USA, February 25--27, 2020, Ranjita Bhagwan and George Porter (Eds.). USENIX Association, 985--998. https:\/\/www.usenix.org\/conference\/nsdi20\/presentation\/burnett"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/PDSW-DISCS.2018.00011"},{"key":"e_1_3_2_1_13_1","unstructured":"Xiaoyu Chu Daniel Hofst\u00e4tter Shashikant Ilager Sacheendra Talluri Duncan Kampert Damian Podareanu Dmitry Duplyakin Ivona Brandic and Alexandru Iosup. 2024. Generic and ML Workloads in an HPC Datacenter: Node Energy Job Failures and Node-Job Analysis. arXiv:2409.08949 [cs.DC] https:\/\/arxiv.org\/abs\/2409.08949"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3578245.3584726"},{"key":"e_1_3_2_1_15_1","unstructured":"DownDetector. 2024. OpenAI user reports. https:\/\/downdetector.com\/status\/openai\/."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2014.78"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2018.00021"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/HASE.2014.24"},{"key":"e_1_3_2_1_19_1","volume-title":"Google Cluster Traces","year":"2019","unstructured":"Google. 2019. Google Cluster Traces 2019. https:\/\/github.com\/google\/clusterdata?tab=readme-ov-file."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126937"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458336.3465297"},{"key":"e_1_3_2_1_22_1","volume-title":"17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020","author":"Hu Jiyao","year":"2020","unstructured":"Jiyao Hu, Zhenyu Zhou, Xiaowei Yang, Jacob Malone, and Jonathan W. Williams. 2020. CableMon: Improving the Reliability of Cable Broadband Networks via Proactive Network Maintenance. In 17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020, Santa Clara, CA, USA, February 25--27, 2020, Ranjita Bhagwan and George Porter (Eds.). USENIX Association, 619--632. https:\/\/www.usenix.org\/conference\/nsdi20\/presentation\/hu-jiyao"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476223"},{"key":"e_1_3_2_1_24_1","volume-title":"Characterization of Large Language Model Development in the Datacenter. In 21st USENIX Symposium on Networked Systems Design and Implementation, NSDI 2024","author":"Hu Qinghao","year":"2024","unstructured":"Qinghao Hu, Zhisheng Ye, Zerui Wang, Guoteng Wang, Meng Zhang, Qiaoling Chen, Peng Sun, Dahua Lin, Xiaolin Wang, Yingwei Luo, Yonggang Wen, and Tianwei Zhang. 2024. Characterization of Large Language Model Development in the Datacenter. In 21st USENIX Symposium on Networked Systems Design and Implementation, NSDI 2024, Santa Clara, CA, April 15--17, 2024, Laurent Vanbever and Irene Zhang (Eds.). USENIX Association, 709--729. https:\/\/www.usenix.org\/conference\/nsdi24\/presentation\/hu"},{"key":"e_1_3_2_1_25_1","unstructured":"Singularity Hub. 2022. OpenAI Says DALL-E Is Generating Over 2 Million Images a Day. https:\/\/singularityhub.com\/2022\/10\/03\/openai-says-dall-e-is-generatingover-2-million-images-a-day-and-thats-just-table-stakes\/."},{"key":"e_1_3_2_1_26_1","unstructured":"IDC. 2024. A Deep Dive Into Global AI and Generative AI Spending. https:\/\/blogs.idc.com\/2024\/08\/16\/a-deep-dive-into-idcs-global-ai-andgenerative-ai-spending\/."},{"key":"e_1_3_2_1_27_1","unstructured":"Anthropic Incident. 2024. https:\/\/status.anthropic.com\/history."},{"key":"e_1_3_2_1_28_1","unstructured":"Character.AI Incident. 2024. https:\/\/status.character.ai\/history."},{"key":"e_1_3_2_1_29_1","unstructured":"OpenAI Incident. 2024. https:\/\/status.openai.com\/history."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1016\/J.JPDC.2013.04.002"},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the 2019 USENIX Annual Technical Conference, USENIX ATC 2019","author":"Jeon Myeongjae","year":"2019","unstructured":"Myeongjae Jeon, Shivaram Venkataraman, Amar Phanishayee, Junjie Qian, Wencong Xiao, and Fan Yang. 2019. Analysis of Large-Scale Multi-Tenant GPU Clusters for DNN Training Workloads. In Proceedings of the 2019 USENIX Annual Technical Conference, USENIX ATC 2019, Renton, WA, USA, July 10--12, 2019, Dahlia Malkhi and Dan Tsafrir (Eds.). USENIX Association, 947--960. https:\/\/www.usenix.org\/conference\/atc19\/presentation\/jeon"},{"key":"e_1_3_2_1_32_1","volume-title":"Generating Images with Multimodal Language Models. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023","author":"Koh Jing Yu","year":"2023","unstructured":"Jing Yu Koh, Daniel Fried, and Russ Salakhutdinov. 2023. Generating Images with Multimodal Language Models. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023, Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (Eds.). http:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/43a69d143273bd8215578bde887bb552-Abstract-Conference.html"},{"key":"e_1_3_2_1_33_1","volume-title":"LLM-Pilot: Characterize and Optimize Performance of your LLM Inference Services. arXiv preprint arXiv:2410.02425","author":"\u0141azuka Ma\u0142gorzata","year":"2024","unstructured":"Ma\u0142gorzata \u0141azuka, Andreea Anghel, and Thomas Parnell. 2024. LLM-Pilot: Characterize and Optimize Performance of your LLM Inference Services. arXiv preprint arXiv:2410.02425 (2024)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00093"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126964"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLOUD60044.2023.00064"},{"key":"e_1_3_2_1_37_1","volume-title":"Rigorous Evaluation of Large Language Models for Code Generation. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023","author":"Liu Jiawei","year":"2023","unstructured":"Jiawei Liu, Chunqiu Steven Xia, Yuyao Wang, and Lingming Zhang. 2023. Is Your Code Generated by ChatGPT Really Correct? Rigorous Evaluation of Large Language Models for Code Generation. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023, Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (Eds.). http:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/43e9d647ccd3e4b7b5baab53f0368686-Abstract-Conference.html"},{"key":"e_1_3_2_1_38_1","volume-title":"Xin Eric Wang, and William Yang Wang","author":"Lu Yujie","year":"2023","unstructured":"Yujie Lu, Xianjun Yang, Xiujun Li, Xin Eric Wang, and William Yang Wang. 2023. LLMScore: Unveiling the Power of Large Language Models in Text-to-Image Synthesis Evaluation. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023, Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (Eds.). http:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/47f30d67bce3e9824928267e9355420f-Abstract-Conference.html"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2014.62"},{"key":"e_1_3_2_1_40_1","unstructured":"OpenAI. 2024. Elevated errors in ChatGPT. https:\/\/status.openai.com\/incidents\/w20mcckg1748 Accessed: 2024-09--30."},{"key":"e_1_3_2_1_41_1","volume-title":"Start using ChatGPT instantly (Apr 1","author":"AI.","year":"2024","unstructured":"OpenAI. 2024. Start using ChatGPT instantly (Apr 1, 2024). https:\/\/help.openai.com\/en\/articles\/6825453-chatgpt-release-notes."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/2523616.2523638"},{"key":"e_1_3_2_1_43_1","volume-title":"18th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2021","author":"Primorac Mia","year":"2021","unstructured":"Mia Primorac, Katerina J. Argyraki, and Edouard Bugnion. 2021. When to Hedge in Interactive Services. In 18th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2021, April 12--14, 2021, James Mickens and Renata Teixeira (Eds.). USENIX Association, 373--387. https:\/\/www.usenix.org\/conference\/nsdi21\/presentation\/primorac"},{"key":"e_1_3_2_1_44_1","unstructured":"Similarweb Pro. 2024. https:\/\/pro.similarweb.com\/."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/2391229.2391236"},{"key":"e_1_3_2_1_46_1","unstructured":"Reuters. 2023. ChatGPT sets record for fastest-growing user base - analyst note. https:\/\/www.reuters.com\/technology\/chatgpt-sets-record-fastestgrowing-user-base-analyst-note-2023-02-01\/"},{"key":"e_1_3_2_1_47_1","unstructured":"Reuters. 2024. Google-backed Anthropic releases Claude chatbot across Europe. https:\/\/www.reuters.com\/technology\/google-backed-anthropic-releasesclaude-chatbot-across-europe-2024-05--13\/."},{"key":"e_1_3_2_1_48_1","unstructured":"Reuters. 2024. OpenAI says ChatGPT's weekly users have grown to 200 million. https:\/\/www.reuters.com\/technology\/artificial-intelligence\/openai-sayschatgpts-weekly-users-have-grown-200-million-2024-08--29\/."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2015.139"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.3390\/fi15060192"},{"key":"e_1_3_2_1_51_1","unstructured":"Selenium. 2024. WebDriver Documentation. https:\/\/www.selenium.dev\/documentation\/webdriver\/."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2015.58"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297663.3310302"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACSOS52086.2021.00039"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2015.7056044"},{"key":"e_1_3_2_1_56_1","unstructured":"Anthropic Uptime. 2024. https:\/\/status.anthropic.com\/uptime."},{"key":"e_1_3_2_1_57_1","unstructured":"Character.AI Uptime. 2024. https:\/\/status.character.ai\/uptime."},{"key":"e_1_3_2_1_58_1","unstructured":"OpenAI Uptime. 2024. https:\/\/status.openai.com\/uptime."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491101.3519665"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1016\/J.FUTURE.2022.12.022"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2401.08329"},{"key":"e_1_3_2_1_62_1","volume-title":"Amelie Chi Zhou, and Xiaowen Chu","author":"Wang Yuxin","year":"2024","unstructured":"Yuxin Wang, Yuhan Chen, Zeyu Li, Xueze Kang, Zhenheng Tang, Xin He, Rui Guo, Xin Wang, Qiang Wang, Amelie Chi Zhou, and Xiaowen Chu. 2024. BurstGPT: A Real-world Workload Dataset to Optimize LLM Serving Systems. arXiv:2401.17644"},{"key":"e_1_3_2_1_63_1","volume-title":"MLaaS in the Wild: Workload Analysis and Scheduling in Large-Scale Heterogeneous GPU Clusters. In 19th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2022","author":"Weng Qizhen","year":"2022","unstructured":"Qizhen Weng, Wencong Xiao, Yinghao Yu, Wei Wang, Cheng Wang, Jian He, Yong Li, Liping Zhang, Wei Lin, and Yu Ding. 2022. MLaaS in the Wild: Workload Analysis and Scheduling in Large-Scale Heterogeneous GPU Clusters. In 19th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2022, Renton, WA, USA, April 4--6, 2022, Amar Phanishayee and Vyas Sekar (Eds.). USENIX Association, 945--960. https:\/\/www.usenix.org\/conference\/nsdi22\/presentation\/weng"},{"key":"e_1_3_2_1_64_1","unstructured":"WIRED. 2024. The obsession with Character AI is becoming more common. https:\/\/wired.me\/technology\/character-ai-obsession\/."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3649506"},{"key":"e_1_3_2_1_67_1","unstructured":"FOX5 New York. 2024. ChatGPT recovers following outage affecting thousands of users. https:\/\/www.fox5ny.com\/news\/chatgpt-down-users-openai-internet Accessed: 2024-09--30."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3236024.3236033"}],"event":{"name":"ICPE '25: 16th ACM\/SPEC International Conference on Performance Engineering","location":"Toronto ON Canada","acronym":"ICPE '25","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering","SIGMETRICS ACM Special Interest Group on Measurement and Evaluation"]},"container-title":["Proceedings of the 16th ACM\/SPEC International Conference on Performance Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3676151.3719372","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3676151.3719372","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T16:23:00Z","timestamp":1758817380000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3676151.3719372"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,5]]},"references-count":68,"alternative-id":["10.1145\/3676151.3719372","10.1145\/3676151"],"URL":"https:\/\/doi.org\/10.1145\/3676151.3719372","relation":{},"subject":[],"published":{"date-parts":[[2025,5,5]]},"assertion":[{"value":"2025-05-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}