{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:06:51Z","timestamp":1784300811513,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":79,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFB3107100"],"award-info":[{"award-number":["2023YFB3107100"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372297"],"award-info":[{"award-number":["62372297"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,5]]},"DOI":"10.1145\/3803437.3805245","type":"proceedings-article","created":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:27:39Z","timestamp":1784298459000},"page":"731-742","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Reflex: Event-Driven Automated Fault Localization for Large-Scale LLM Training"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7845-4995","authenticated-orcid":false,"given":"Hua","family":"Ding","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9159-3107","authenticated-orcid":false,"given":"Yun","family":"Zhang","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6975-9184","authenticated-orcid":false,"given":"Bo","family":"Zhang","sequence":"additional","affiliation":[{"name":"China Electric Power Research Institute, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4482-2566","authenticated-orcid":false,"given":"Wenxiao","family":"Wang","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3236-4805","authenticated-orcid":false,"given":"Libo","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8102-5055","authenticated-orcid":false,"given":"Huan","family":"Yu","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6311-0018","authenticated-orcid":false,"given":"Zhe","family":"Nan","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7576-6162","authenticated-orcid":false,"given":"Zuquan","family":"Song","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-7390-0581","authenticated-orcid":false,"given":"Weiqiang","family":"Lou","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2551-7879","authenticated-orcid":false,"given":"Gaohong","family":"Liu","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2195-8834","authenticated-orcid":false,"given":"Xi","family":"Yang","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-5310-9523","authenticated-orcid":false,"given":"Yuhan","family":"Li","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5406-6354","authenticated-orcid":false,"given":"Qinlong","family":"Wang","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4249-4092","authenticated-orcid":false,"given":"Shuguang","family":"Wang","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3043-522X","authenticated-orcid":false,"given":"Wencong","family":"Xiao","sequence":"additional","affiliation":[{"name":"ByteDance Seed, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0767-2307","authenticated-orcid":false,"given":"Shenghong","family":"Li","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,17]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2025. DCGM Diagnostics. https:\/\/docs.nvidia.com\/datacenter\/dcgm\/latest\/user-guide\/dcgm-diagnostics.html."},{"key":"e_1_3_2_1_2_1","unstructured":"Josh Achiam Steven Adler et al. 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_3_1","volume-title":"Meet Claude. https:\/\/www.anthropic.com\/claude.","year":"2025","unstructured":"Anthropic. 2025. Meet Claude. https:\/\/www.anthropic.com\/claude."},{"key":"e_1_3_2_1_4_1","unstructured":"Karim Buzdar. 2020. An Introduction to Linux's dmesg Command. https:\/\/linuxhint.com\/dmesg_tutorial\/."},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the 22nd acm sigkdd international conference on knowledge discovery and data mining. 785\u2013794","author":"Chen Tianqi","year":"2016","unstructured":"Tianqi Chen and Carlos Guestrin. 2016. Xgboost: A scalable tree boosting system. In Proceedings of the 22nd acm sigkdd international conference on knowledge discovery and data mining. 785\u2013794."},{"key":"e_1_3_2_1_6_1","unstructured":"Gheorghe Comanici Eric Bieber et al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:2507.06261 (2025)."},{"key":"e_1_3_2_1_7_1","unstructured":"GitHub Copilot. 2023. Github copilot."},{"key":"e_1_3_2_1_8_1","unstructured":"Shengkun Cui et al. 2025. Characterizing gpu resilience and impact on ai\/hpc systems. arXiv preprint arXiv:2503.11901 (2025)."},{"key":"e_1_3_2_1_9_1","volume-title":"22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25)","author":"Deng Yangtao","year":"2025","unstructured":"Yangtao Deng, Xiang Shi, Zhuo Jiang, Xingjian Zhang, Lei Zhang, Zhang Zhang, Bo Li, Zuquan Song, Hang Zhu, Gaohong Liu, et al. 2025. Minder: Faulty machine detection for large-scale distributed model training. In 22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25). 505\u2013521."},{"key":"e_1_3_2_1_10_1","volume-title":"2023 IEEE 29th International Symposium on On-Line Testing and Robust System Design (IOLTS). 1\u20132. 10","author":"Dixit Harish","year":"2023","unstructured":"Harish Dixit. 2023. Keytone: Silent Data Corruptions at Scale. In 2023 IEEE 29th International Symposium on On-Line Testing and Robust System Design (IOLTS). 1\u20132. 10.1109\/IOLTS59296.2023.10224872"},{"key":"e_1_3_2_1_11_1","volume-title":"22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25)","author":"Dong Jianbo","year":"2025","unstructured":"Jianbo Dong, Kun Qian, Pengcheng Zhang, Zhilong Zheng, Liang Chen, Fei Feng, Yichi Xu, Yikai Zhu, Gang Lu, Xue Li, et al. 2025. Evolution of Aegis: Fault Diagnosis for {AI} Model Training Service in Production. In 22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25). 865\u2013881."},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 2017 ACM SIGSAC conference on computer and communications security. 1285\u20131298","author":"Du Min","year":"2017","unstructured":"Min Du, Feifei Li, Guineng Zheng, and Vivek Srikumar. 2017. Deeplog: Anomaly detection and diagnosis from system logs through deep learning. In Proceedings of the 2017 ACM SIGSAC conference on computer and communications security. 1285\u20131298."},{"key":"e_1_3_2_1_13_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv e-prints (2024) arXiv-2407."},{"key":"e_1_3_2_1_14_1","volume-title":"19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22)","author":"Eisenman Assaf","year":"2022","unstructured":"Assaf Eisenman, Kiran Kumar Matam, Steven Ingram, Dheevatsa Mudigere, Raghuraman Krishnamoorthi, Krishnakumar Nair, Misha Smelyanskiy, and Murali Annavaram. 2022. {Check-N-Run}: A checkpointing system for training deep learning recommendation models. In 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22). 929\u2013943."},{"key":"e_1_3_2_1_15_1","volume-title":"Vertex AI: Unified Platform for Generative AI and Machine Learning. https:\/\/cloud.google.com\/vertex-ai. Accessed: 2025-12-07.","author":"Cloud Google","year":"2025","unstructured":"Google Cloud. 2025. Vertex AI: Unified Platform for Generative AI and Machine Learning. https:\/\/cloud.google.com\/vertex-ai. Accessed: 2025-12-07."},{"key":"e_1_3_2_1_16_1","volume-title":"Logllm: Log-based anomaly detection using large language models. arXiv preprint arXiv:2411.08561","author":"Guan Wei","year":"2024","unstructured":"Wei Guan, Jian Cao, Shiyou Qian, Jianqi Gao, and Chun Ouyang. 2024. Logllm: Log-based anomaly detection using large language models. arXiv preprint arXiv:2411.08561 (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"FeadSeq: A Personalized Federated Anomaly Detection Framework for Discrete Event Sequences. ACM Transactions on Knowledge Discovery from Data 19, 6","author":"Guan Wei","year":"2025","unstructured":"Wei Guan, Jian Cao, Haiyan Zhao, Yang Gu, and Shiyou Qian. 2025. FeadSeq: A Personalized Federated Anomaly Detection Framework for Discrete Event Sequences. ACM Transactions on Knowledge Discovery from Data 19, 6 (2025), 1\u201326."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the 25th ACM international on conference on information and knowledge management. 1573\u20131582","author":"Hamooni Hossein","year":"2016","unstructured":"Hossein Hamooni, Biplob Debnath, Jianwu Xu, Hui Zhang, Guofei Jiang, and Abdullah Mueen. 2016. Logmine: Fast pattern recognition for log analytics. In Proceedings of the 25th ACM international on conference on information and knowledge management. 1573\u20131582."},{"key":"e_1_3_2_1_19_1","volume-title":"2016 IEEE 27th international symposium on software reliability engineering (ISSRE). IEEE, 207\u2013218","author":"He Shilin","year":"2016","unstructured":"Shilin He, Jieming Zhu, Pinjia He, and Michael R Lyu. 2016. Experience report: System log analysis for anomaly detection. In 2016 IEEE 27th international symposium on software reliability engineering (ISSRE). IEEE, 207\u2013218."},{"key":"e_1_3_2_1_20_1","volume-title":"Unicron: Economizing self-healing llm training at scale. arXiv preprint arXiv:2401.00134","author":"He Tao","year":"2023","unstructured":"Tao He, Xue Li, Zhibin Wang, Kun Qian, Jingbo Xu, Wenyuan Yu, and Jingren Zhou. 2023. Unicron: Economizing self-healing llm training at scale. arXiv preprint arXiv:2401.00134 (2023)."},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the Workshop on Hot Topics in Operating Systems. 9\u201316","author":"Hochschild Peter H","year":"2021","unstructured":"Peter H Hochschild, Paul Turner, Jeffrey C Mogul, Rama Govindaraju, Parthasarathy Ranganathan, David E Culler, and Amin Vahdat. 2021. Cores that don't count. In Proceedings of the Workshop on Hot Topics in Operating Systems. 9\u201316."},{"key":"e_1_3_2_1_22_1","unstructured":"Qinghao Hu et al. 2024. Characterization of large language model development in the datacenter. In NSDI 24. 709\u2013729."},{"key":"e_1_3_2_1_23_1","volume-title":"Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems 32","author":"Yanping Huang","year":"2019","unstructured":"Yanping Huang et al. 2019. Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_2_1_24_1","unstructured":"Sam Ade Jacobs et al. 2023. Deepspeed ulysses: System optimizations for enabling training of extreme long sequence transformer models. arXiv preprint arXiv:2309.14509 (2023)."},{"key":"e_1_3_2_1_25_1","volume-title":"Microsoft Research","author":"Jeon Myeongjae","year":"2018","unstructured":"Myeongjae Jeon, Shivaram Venkataraman, Junjie Qian, Amar Phanishayee, Wencong Xiao, and Fan Yang. 2018. Multi-tenant GPU clusters for deep learning workloads: Analysis and implications. Technical report, Microsoft Research (2018)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Zhihan Jiang et al. 2025. LLMPrism: Black-box Performance Diagnosis for Production LLM Training Platforms. arXiv preprint arXiv:2505.00342 (2025).","DOI":"10.1109\/DSN-S65789.2025.00034"},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 33rd ACM International Conference on the Foundations of Software Engineering. 51\u201363","author":"Jiang Zhihan","year":"2025","unstructured":"Zhihan Jiang, Junjie Huang, Guangba Yu, Zhuangbin Chen, Yichen Li, Renyi Zhong, Cong Feng, Yongqiang Yang, Zengyin Yang, and Michael Lyu. 2025. L4: Diagnosing large-scale llm training failures via automated log analysis. In Proceedings of the 33rd ACM International Conference on the Foundations of Software Engineering. 51\u201363."},{"key":"e_1_3_2_1_28_1","volume-title":"Scaling laws for neural language models. arXiv preprint arXiv:2001.08361","author":"Kaplan Jared","year":"2020","unstructured":"Jared Kaplan, Sam McCandlish, Tom Henighan, Tom B Brown, Benjamin Chess, Rewon Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei. 2020. Scaling laws for neural language models. arXiv preprint arXiv:2001.08361 (2020)."},{"key":"e_1_3_2_1_29_1","volume-title":"2021 36th IEEE\/ACM International Conference on Automated Software Engineering (ASE). IEEE, 492\u2013504","author":"Le Van-Hoang","year":"2021","unstructured":"Van-Hoang Le and Hongyu Zhang. 2021. Log-based anomaly detection without log parsing. In 2021 36th IEEE\/ACM International Conference on Automated Software Engineering (ASE). IEEE, 492\u2013504."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 44th international conference on software engineering. 1356\u20131367","author":"Le Van-Hoang","year":"2022","unstructured":"Van-Hoang Le and Hongyu Zhang. 2022. Log-based anomaly detection with deep learning: How far are we?. In Proceedings of the 44th international conference on software engineering. 1356\u20131367."},{"key":"e_1_3_2_1_31_1","volume-title":"Modeling and Maximizing Network Reliability in Large Scale Infrastructure Networks: A Heat Conduction Model Perspective","author":"Li Beibei","year":"2025","unstructured":"Beibei Li, Wei Hu, Yiwei Li, and Lemei Da. 2025. Modeling and Maximizing Network Reliability in Large Scale Infrastructure Networks: A Heat Conduction Model Perspective. IEEE Transactions on Network and Service Management (2025)."},{"key":"e_1_3_2_1_32_1","unstructured":"Liqun Li Xu Zhang Xin Zhao Hongyu Zhang Yu Kang Pu Zhao Bo Qiao Shilin He Pochian Lee Jeffrey Sun et al. 2021. Fighting the fog of war: Automated incident detection for cloud systems. In USENIX ATC 21. 131\u2013146."},{"key":"e_1_3_2_1_33_1","volume-title":"Communication efficient distributed machine learning with the parameter server. Advances in neural information processing systems 27","author":"Li Mu","year":"2014","unstructured":"Mu Li, David G Andersen, Alexander Smola, and Kai Yu. 2014. Communication efficient distributed machine learning with the parameter server. Advances in neural information processing systems 27 (2014)."},{"key":"e_1_3_2_1_34_1","volume-title":"PyTorch Distributed: Experiences on Accelerating Data Parallel Training. Proceedings of the VLDB Endowment 13","author":"Li Shen","unstructured":"Shen Li, Yanli Zhao, Rohan Varma, Omkar Salpekar, Pieter Noordhuis, Teng Li, Adam Paszke, Jeff Smith, Brian Vaughan, Pritam Damania, et al. [n. d.]. PyTorch Distributed: Experiences on Accelerating Data Parallel Training. Proceedings of the VLDB Endowment 13, 12 ([n. d.])."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","first-page":"3005","DOI":"10.14778\/3415478.3415530","article-title":"PyTorch distributed: experiences on accelerating data parallel training","volume":"13","author":"Li Shen","year":"2020","unstructured":"Shen Li, Yanli Zhao, Rohan Varma, Omkar Salpekar, Pieter Noordhuis, Teng Li, Adam Paszke, Jeff Smith, Brian Vaughan, Pritam Damania, et al. 2020. PyTorch distributed: experiences on accelerating data parallel training. Proceedings of the VLDB Endowment 13, 12 (2020), 3005\u20133018.","journal-title":"Proceedings of the VLDB Endowment"},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the 38th international conference on software engineering companion. 102\u2013111","author":"Qingwei","unstructured":"Qingwei Lin et al. 2016. Log clustering based problem identification for online service systems. In Proceedings of the 38th international conference on software engineering companion. 102\u2013111."},{"key":"e_1_3_2_1_37_1","unstructured":"Aixin Liu Aoxue Mei Bangcai Lin Bing Xue Bingxuan Wang Bingzheng Xu Bochao Wu Bowei Zhang Chaofan Lin Chen Dong et al. 2025. Deepseek-v3. 2: Pushing the frontier of open large language models. arXiv preprint arXiv:2512.02556 (2025)."},{"key":"e_1_3_2_1_38_1","volume-title":"2021 IEEE\/ACM 43rd International Conference on Software Engineering: Software Engineering in Practice (ICSE-SEIP). IEEE, 338\u2013347","author":"Liu Dewei","year":"2021","unstructured":"Dewei Liu, Chuan He, Xin Peng, Fan Lin, Chenxi Zhang, Shengfang Gong, Ziang Li, Jiayu Ou, and Zheshun Wu. 2021. Microhecl: High-efficient root cause localization in large-scale microservice systems. In 2021 IEEE\/ACM 43rd International Conference on Software Engineering: Software Engineering in Practice (ICSE-SEIP). IEEE, 338\u2013347."},{"key":"e_1_3_2_1_39_1","volume-title":"Kai Ming Ting, and Zhi-Hua Zhou","author":"Liu Fei Tony","year":"2008","unstructured":"Fei Tony Liu, Kai Ming Ting, and Zhi-Hua Zhou. 2008. Isolation forest. In 2008 eighth ieee international conference on data mining. IEEE, 413\u2013422."},{"key":"e_1_3_2_1_40_1","volume-title":"RingAttention with Blockwise Transformers for Near-Infinite Context. In The Twelfth International Conference on Learning Representations.","author":"Liu Hao","unstructured":"Hao Liu, Matei Zaharia, and Pieter Abbeel. [n. d.]. RingAttention with Blockwise Transformers for Near-Infinite Context. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_41_1","volume-title":"2010 USENIX Annual Technical Conference (USENIX ATC 10)","author":"Lou Jian-Guang","year":"2010","unstructured":"Jian-Guang Lou, Qiang Fu, Shenqi Yang, Ye Xu, and Jiang Li. 2010. Mining invariants from console logs for system problem detection. In 2010 USENIX Annual Technical Conference (USENIX ATC 10)."},{"key":"e_1_3_2_1_42_1","volume-title":"A unified approach to interpreting model predictions. Advances in neural information processing systems 30","author":"Lundberg Scott M","year":"2017","unstructured":"Scott M Lundberg and Su-In Lee. 2017. A unified approach to interpreting model predictions. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","volume":"3","author":"Ma Dongning","year":"2024","unstructured":"Dongning Ma, Fred Lin, Alban Desmaison, Joel Coburn, Daniel Moore, Sriram Sankar, and Xun Jiao. 2024. Dr. DNA: Combating silent data corruptions in deep learning using distribution of neuron activations. In Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 3. 239\u2013252."},{"key":"e_1_3_2_1_44_1","volume-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 20372\u201320394","author":"Ma Jeffrey Jian","year":"2025","unstructured":"Jeffrey Jian Ma, Hengzhi Pei, Leonard Lausen, and George Karypis. 2025. Understanding silent data corruption in LLM training. In Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 20372\u201320394."},{"key":"e_1_3_2_1_45_1","volume-title":"Adaptivelog: An adaptive log analysis framework with the collaboration of large and small language model. ACM Transactions on Software Engineering and Methodology","author":"Ma Lipeng","year":"2025","unstructured":"Lipeng Ma, Weidong Yang, Yixuan Li, Ben Fei, Mingjie Zhou, Shuhao Li, Sihang Jiang, Bo Xu, and Yanghua Xiao. 2025. Adaptivelog: An adaptive log analysis framework with the collaboration of large and small language model. ACM Transactions on Software Engineering and Methodology (2025)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","first-page":"1176","DOI":"10.14778\/3389133.3389136","article-title":"Diagnosing root causes of intermittent slow queries in cloud databases","volume":"13","author":"Ma Minghua","year":"2020","unstructured":"Minghua Ma, Zheng Yin, Shenglin Zhang, Sheng Wang, Christopher Zheng, Xinhao Jiang, Hanwen Hu, Cheng Luo, Yilin Li, Nengjun Qiu, et al. 2020. Diagnosing root causes of intermittent slow queries in cloud databases. Proceedings of the VLDB Endowment 13, 8 (2020), 1176\u20131189.","journal-title":"Proceedings of the VLDB Endowment"},{"key":"e_1_3_2_1_47_1","first-page":"4739","article-title":"Loganomaly: Unsupervised detection of sequential and quantitative anomalies in unstructured logs","volume":"19","author":"Meng Weibin","year":"2019","unstructured":"Weibin Meng, Ying Liu, Yichen Zhu, Shenglin Zhang, Dan Pei, Yuqing Liu, Yihao Chen, Ruizhi Zhang, Shimin Tao, Pei Sun, et al. 2019. Loganomaly: Unsupervised detection of sequential and quantitative anomalies in unstructured logs.. In IJCAI, Vol. 19. 4739\u20134745.","journal-title":"IJCAI"},{"key":"e_1_3_2_1_48_1","volume-title":"2020 IEEE\/ACM 28th International Symposium on Quality of Service (IWQoS). IEEE, 1\u201310","author":"Meng Yuan","year":"2020","unstructured":"Yuan Meng, Shenglin Zhang, Yongqian Sun, Ruru Zhang, Zhilong Hu, Yiyin Zhang, Chenyang Jia, Zhaogang Wang, and Dan Pei. 2020. Localizing failure root causes in a microservice through causality inference. In 2020 IEEE\/ACM 28th International Symposium on Quality of Service (IWQoS). IEEE, 1\u201310."},{"key":"e_1_3_2_1_49_1","unstructured":"Meta AI. 2025. Llama 4: Multimodal Intelligence. https:\/\/ai.meta.com\/blog\/llama-4-multimodal-intelligence\/. Accessed: 2025-11-20."},{"key":"e_1_3_2_1_50_1","unstructured":"Microsoft Corporation. 2025. Microsoft Azure Monitor: Full Observability for Applications Infrastructure and Network. https:\/\/azure.microsoft.com\/en-us\/products\/monitor\/. Accessed: 2025-12-07."},{"key":"e_1_3_2_1_51_1","volume-title":"19th USENIX Conference on File and Storage Technologies (FAST 21)","author":"Mohan Jayashree","year":"2021","unstructured":"Jayashree Mohan, Amar Phanishayee, and Vijay Chidambaram. 2021. {CheckFreq}: Frequent,{Fine-Grained} {DNN} Checkpointing. In 19th USENIX Conference on File and Storage Technologies (FAST 21). 203\u2013216."},{"key":"e_1_3_2_1_52_1","volume-title":"9th USENIX Symposium on Networked Systems Design and Implementation (NSDI 12)","author":"Nagaraj Karthik","year":"2012","unstructured":"Karthik Nagaraj, Charles Killian, and Jennifer Neville. 2012. Structured comparative analysis of systems logs to diagnose performance problems. In 9th USENIX Symposium on Networked Systems Design and Implementation (NSDI 12). 353\u2013366."},{"key":"e_1_3_2_1_53_1","volume-title":"Proceedings of the 27th ACM symposium on operating systems principles. 1\u201315","author":"Deepak","unstructured":"Deepak Narayanan et al. 2019. PipeDream: Generalized pipeline parallelism for DNN training. In Proceedings of the 27th ACM symposium on operating systems principles. 1\u201315."},{"key":"e_1_3_2_1_54_1","volume-title":"Proceedings of the international conference for high performance computing, networking, storage and analysis. 1\u201315","author":"Deepak","unstructured":"Deepak Narayanan et al. 2021. Efficient large-scale language model training on gpu clusters using megatron-lm. In Proceedings of the international conference for high performance computing, networking, storage and analysis. 1\u201315."},{"key":"e_1_3_2_1_55_1","unstructured":"NVIDIA Corporation. 2020. NVIDIA A100 Tensor Core GPU Architecture. https:\/\/images.nvidia.com\/aem-dam\/en-zz\/Solutions\/data-center\/nvidia-ampere-architecture-whitepaper.pdf Accessed: 2025-01-01."},{"key":"e_1_3_2_1_56_1","unstructured":"NVIDIA Corporation. 2025. CUDA Toolkit. https:\/\/developer.nvidia.com\/cuda-toolkit."},{"key":"e_1_3_2_1_57_1","unstructured":"NVIDIA Corporation. 2025. Xid Errors. https:\/\/docs.nvidia.com\/deploy\/xid-errors\/."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1016\/j.jpdc.2008.09.002","article-title":"Bandwidth optimal all-reduce algorithms for clusters of workstations","volume":"69","author":"Patarasuk Pitch","year":"2009","unstructured":"Pitch Patarasuk and Xin Yuan. 2009. Bandwidth optimal all-reduce algorithms for clusters of workstations. J. Parallel and Distrib. Comput. 69, 2 (2009), 117\u2013124.","journal-title":"J. Parallel and Distrib. Comput."},{"key":"e_1_3_2_1_59_1","unstructured":"PCI-SIG. 2019. PCI Express Base Specification Revision 5.0. PCI-SIG. https:\/\/pcisig.com\/specifications."},{"key":"e_1_3_2_1_60_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Penghui","unstructured":"Penghui Qi et al. 2024. Zero bubble (almost) pipeline parallelism. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_61_1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Rajbhandari Samyam","year":"2020","unstructured":"Samyam Rajbhandari, Jeff Rasley, Olatunji Ruwase, and Yuxiong He. 2020. ZeRO: Memory Optimizations Toward Training Trillion Parameter Models. In Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis (Atlanta, Georgia). IEEE Press, 1\u201316."},{"key":"e_1_3_2_1_62_1","unstructured":"Baptiste Roziere et al. 2023. Code llama: Open foundation models for code. arXiv preprint arXiv:2308.12950 (2023)."},{"key":"e_1_3_2_1_63_1","volume-title":"Proceedings of the Fifteenth European Conference on Computer Systems. 1\u201316","author":"Rzadca Krzysztof","year":"2020","unstructured":"Krzysztof Rzadca, Pawel Findeisen, Jacek Swiderski, Przemyslaw Zych, Przemyslaw Broniek, Jarek Kusmierek, Pawel Nowak, Beata Strack, Piotr Witusowski, Steven Hand, et al. 2020. Autopilot: workload autoscaling at google. In Proceedings of the Fifteenth European Conference on Computer Systems. 1\u201316."},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"crossref","first-page":"3731","DOI":"10.14778\/3685800.3685802","article-title":"Clickhouse-lightning fast analytics for everyone","volume":"17","author":"Schulze Robert","year":"2024","unstructured":"Robert Schulze, Tom Schreiber, Ilya Yatsishin, Ryadh Dahimene, and Alexey Milovidov. 2024. Clickhouse-lightning fast analytics for everyone. Proceedings of the VLDB Endowment 17, 12 (2024), 3731\u20133744.","journal-title":"Proceedings of the VLDB Endowment"},{"key":"e_1_3_2_1_65_1","volume-title":"Proceedings of the 11th ACM Conference on Emerging Networking Experiments and Technologies. 1\u201313","author":"Sharma Dhruv","year":"2015","unstructured":"Dhruv Sharma, Rishabh Poddar, Kshiteej Mahajan, Mohan Dhawan, and Vijay Mann. 2015. Hansel: Diagnosing faults in openStack. In Proceedings of the 11th ACM Conference on Emerging Networking Experiments and Technologies. 1\u201313."},{"key":"e_1_3_2_1_66_1","volume-title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. In International Conference on Learning Representations.","author":"Shazeer Noam","year":"2017","unstructured":"Noam Shazeer, Azalia Mirhoseini, Krzysztof Maziarz, Andy Davis, Quoc Le, Geoffrey Hinton, and Jeff Dean. 2017. Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_67_1","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053","author":"Mohammad Shoeybi","year":"2019","unstructured":"Mohammad Shoeybi et al. 2019. Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053 (2019)."},{"key":"e_1_3_2_1_68_1","volume-title":"Splunk Enterprise: The Platform for Operational Intelligence. https:\/\/www.splunk.com\/en_us\/products\/splunk-enterprise.html. Accessed: 2025-12-07.","author":"Splunk Inc.","year":"2025","unstructured":"Splunk Inc. 2025. Splunk Enterprise: The Platform for Operational Intelligence. https:\/\/www.splunk.com\/en_us\/products\/splunk-enterprise.html. Accessed: 2025-12-07."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"crossref","first-page":"1413","DOI":"10.1038\/s42256-024-00953-0","article-title":"Towards a personalized AI assistant to learn machine learning","volume":"6","author":"Wallisch Pascal","year":"2024","unstructured":"Pascal Wallisch and Ibrahim Sheikh. 2024. Towards a personalized AI assistant to learn machine learning. Nature Machine Intelligence 6, 12 (2024), 1413\u20131414.","journal-title":"Nature Machine Intelligence"},{"key":"e_1_3_2_1_70_1","volume-title":"Proceedings of the ACM SIGOPS 31st Symposium on Operating Systems Principles. 186\u2013203","author":"Borui","unstructured":"Borui Wan et al. 2025. Robust llm training infrastructure at bytedance. In Proceedings of the ACM SIGOPS 31st Symposium on Operating Systems Principles. 186\u2013203."},{"key":"e_1_3_2_1_71_1","volume-title":"22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25)","author":"Wan Borui","year":"2025","unstructured":"Borui Wan, Mingji Han, Yiyao Sheng, Yanghua Peng, Haibin Lin, Mofan Zhang, Zhichao Lai, Menghan Yu, Junda Zhang, Zuquan Song, et al. 2025. {ByteCheckpoint}: A Unified Checkpointing System for Large Foundation Model Development. In 22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25). 559\u2013578."},{"key":"e_1_3_2_1_72_1","volume-title":"Proceedings of the 39th IEEE\/ACM International Conference on Automated Software Engineering","author":"Wang Yidan","year":"2024","unstructured":"Yidan Wang, Zhouruixing Zhu, Qiuai Fu, Yuchi Ma, and Pinjia He. 2024. MRCA: Metric-level Root Cause Analysis for Microservices via Multi-Modal Data. In Proceedings of the 39th IEEE\/ACM International Conference on Automated Software Engineering (Sacramento, CA, USA) (ASE '24). Association for Computing Machinery, New York, NY, USA, 1057\u20131068. 10.1145\/3691620.3695485"},{"key":"e_1_3_2_1_73_1","volume-title":"Proceedings of the ACM SIGOPS 22nd symposium on Operating systems principles. 117\u2013132","author":"Xu Wei","year":"2009","unstructured":"Wei Xu, Ling Huang, Armando Fox, David Patterson, and Michael I Jordan. 2009. Detecting large-scale system problems by mining console logs. In Proceedings of the ACM SIGOPS 22nd symposium on Operating systems principles. 117\u2013132."},{"key":"e_1_3_2_1_74_1","volume-title":"8th USENIX Symposium on Networked Systems Design and Implementation (NSDI 11)","author":"Yu Minlan","year":"2011","unstructured":"Minlan Yu, Albert Greenberg, Dave Maltz, Jennifer Rexford, Lihua Yuan, Srikanth Kandula, and Changhoon Kim. 2011. Profiling network performance for multi-tier data center applications. In 8th USENIX Symposium on Networked Systems Design and Implementation (NSDI 11)."},{"key":"e_1_3_2_1_75_1","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics. 3764\u20133777","author":"Yuan Ruifeng","year":"2025","unstructured":"Ruifeng Yuan, Shichao Sun, et al. 2025. Personalized large language model assistant with evolving conditional memory. In Proceedings of the 31st International Conference on Computational Linguistics. 3764\u20133777."},{"key":"e_1_3_2_1_76_1","volume-title":"Flashrecovery: Fast and low-cost recovery from failures for large-scale training of llms. arXiv preprint arXiv:2509.03047","author":"Haijun Zhang","year":"2025","unstructured":"Haijun Zhang et al. 2025. Flashrecovery: Fast and low-cost recovery from failures for large-scale training of llms. arXiv preprint arXiv:2509.03047 (2025)."},{"key":"e_1_3_2_1_77_1","volume-title":"Opt: Open pre-trained transformer language models. arXiv preprint arXiv:2205.01068","author":"Susan Zhang","year":"2022","unstructured":"Susan Zhang et al. 2022. Opt: Open pre-trained transformer language models. arXiv preprint arXiv:2205.01068 (2022)."},{"key":"e_1_3_2_1_78_1","volume-title":"Proceedings of the 2019 27th ACM joint meeting on European software engineering conference and symposium on the foundations of software engineering. 807\u2013817","author":"Zhang Xu","year":"2019","unstructured":"Xu Zhang, Yong Xu, Qingwei Lin, Bo Qiao, Hongyu Zhang, Yingnong Dang, Chunyu Xie, Xinsheng Yang, Qian Cheng, Ze Li, et al. 2019. Robust log-based anomaly detection on unstable log data. In Proceedings of the 2019 27th ACM joint meeting on European software engineering conference and symposium on the foundations of software engineering. 807\u2013817."},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"crossref","first-page":"3848","DOI":"10.14778\/3611540.3611569","article-title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","volume":"16","author":"Yanli Zhao","year":"2023","unstructured":"Yanli Zhao et al. 2023. PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel. Proceedings of the VLDB Endowment 16, 12 (2023), 3848\u20133860.","journal-title":"Proceedings of the VLDB Endowment"}],"event":{"name":"FSE Companion '26: 34th ACM International Conference on the Foundations of Software Engineering","location":"Concordia University Montreal QC Canada","acronym":"FSE Companion '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 34th ACM International Conference on the Foundations of Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3803437.3805245","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:40:02Z","timestamp":1784299202000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3803437.3805245"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":79,"alternative-id":["10.1145\/3803437.3805245","10.1145\/3803437"],"URL":"https:\/\/doi.org\/10.1145\/3803437.3805245","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}