{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T16:19:25Z","timestamp":1783613965688,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62325205"],"award-info":[{"award-number":["62325205"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Key Program of the Natural Science Foundation of Jiangsu Province","award":["BK20243053"],"award-info":[{"award-number":["BK20243053"]}]},{"name":"Key Program of the Natural Science Foundation of Jiangsu Province","award":["BK20243059"],"award-info":[{"award-number":["BK20243059"]}]},{"name":"Gusu Innovation Project for People","award":["ZXL2024360"],"award-info":[{"award-number":["ZXL2024360"]}]},{"name":"Nanjing University-China Mobile Communications Group Co., Ltd. Joint Institute"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,9,8]]},"DOI":"10.1145\/3718958.3750521","type":"proceedings-article","created":{"date-parts":[[2025,8,27]],"date-time":"2025-08-27T16:54:11Z","timestamp":1756313651000},"page":"609-625","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["Astral: A Datacenter Infrastructure for Large Language Model Training at Scale"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5394-8450","authenticated-orcid":false,"given":"Qingkai","family":"Meng","sequence":"first","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2663-4146","authenticated-orcid":false,"given":"Hao","family":"Zheng","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9381-2019","authenticated-orcid":false,"given":"Zhenhui","family":"Zhang","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0214-664X","authenticated-orcid":false,"given":"ChonLam","family":"Lao","sequence":"additional","affiliation":[{"name":"Harvard University, Cambridge, Massachusetts, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6079-6579","authenticated-orcid":false,"given":"Chengyuan","family":"Huang","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-6714-5014","authenticated-orcid":false,"given":"Baojia","family":"Li","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9354-9117","authenticated-orcid":false,"given":"Ziyuan","family":"Zhu","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4554-8554","authenticated-orcid":false,"given":"Hao","family":"Lu","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2390-7978","authenticated-orcid":false,"given":"Weizhen","family":"Dang","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7982-8894","authenticated-orcid":false,"given":"Zitong","family":"Lin","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3595-6750","authenticated-orcid":false,"given":"Weifeng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3574-2052","authenticated-orcid":false,"given":"Lingfeng","family":"Liu","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-0992-6017","authenticated-orcid":false,"given":"Yuanyuan","family":"Gong","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6128-0384","authenticated-orcid":false,"given":"Chunzhi","family":"He","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8233-1143","authenticated-orcid":false,"given":"Xiaoyuan","family":"Hu","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2816-9777","authenticated-orcid":false,"given":"Yinben","family":"Xia","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0556-5290","authenticated-orcid":false,"given":"Xiang","family":"Li","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8020-0595","authenticated-orcid":false,"given":"Zekun","family":"He","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3669-1202","authenticated-orcid":false,"given":"Yachen","family":"Wang","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5370-8162","authenticated-orcid":false,"given":"Xianneng","family":"Zou","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1979-4420","authenticated-orcid":false,"given":"Kun","family":"Yang","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6063-4975","authenticated-orcid":false,"given":"Gianni","family":"Antichi","sequence":"additional","affiliation":[{"name":"Politecnico di Milano and Queen Mary University of London, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6934-1685","authenticated-orcid":false,"given":"Guihai","family":"Chen","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2710-7628","authenticated-orcid":false,"given":"Chen","family":"Tian","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,8,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2025. ASTRA-sim. https:\/\/astra-sim.github.io\/ Accessed on 2025-01-19."},{"key":"e_1_3_2_1_2_1","unstructured":"2025. Doubling all2all Performance with NVIDIA Collective Communication Library 2.12. https:\/\/developer.nvidia.com\/blog\/doubling-all2all-performance-with-nvidia-collective-communication-library-2-12\/ Accessed on 2025-01-19."},{"key":"e_1_3_2_1_3_1","unstructured":"2025. GitHub Copilot. https:\/\/github.com\/features\/copilot. Accessed on 2025-01-19."},{"key":"e_1_3_2_1_4_1","unstructured":"2025. GPU Burn. https:\/\/github.com\/wilicc\/gpu-burn Accessed on 2025-01-19."},{"key":"e_1_3_2_1_5_1","unstructured":"2025. Load Balancing on Aggregated Ethernet Interfaces. https:\/\/www.juniper.net\/documentation\/us\/en\/software\/junos\/high-availability\/topics\/topic-map\/load-balancing-aggregated-ethernet-interfaces.html Accessed on 2025-01-19."},{"key":"e_1_3_2_1_6_1","unstructured":"2025. NVLink and NVSwitch. https:\/\/www.nvidia.com\/en-us\/data-center\/nvlink\/ Accessed on 2025-01-19."},{"key":"e_1_3_2_1_7_1","unstructured":"2025. SimAI. https:\/\/ennanzhai.github.io\/pub\/nsdi25spring-simai.pdf Accessed on 2025-01-19."},{"key":"e_1_3_2_1_8_1","unstructured":"2025. Z-Score: Meaning and Formula. https:\/\/www.investopedia.com\/terms\/z\/zscore.asp#:~:text=Z%2Dscore%20is%20a%20statistical traders%20to%20help%20determine%20volatility Accessed on 2025-01-19."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3603269.3604878"},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"Alayrac Jean-Baptiste","year":"2022","unstructured":"Jean-Baptiste Alayrac, Jeff Donahue, Pauline Luc, Antoine Miech, Iain Barr, Yana Hasson, Karel Lenc, Arthur Mensch, Katie Millicah, Malcolm Reynolds, Roman Ring, Eliza Rutherford, Serkan Cabi, Tengda Han, Zhitao Gong, Sina Samangooei, Marianne Monteiro, Jacob Menick, Sebastian Borgeaud, Andrew Brock, Aida Nematzadeh, Sahand Sharifzadeh, Mikolaj Binkowski, Ricardo Barreira, Oriol Vinyals, Andrew Zisserman, and Karen Simonyan. 2022. Flamingo: a visual language model for few-shot learning. In Proceedings of the 36th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS '22). Curran Associates Inc., Red Hook, NY, USA, Article 1723, 21 pages."},{"key":"e_1_3_2_1_11_1","unstructured":"Alibaba. 2025. Evolution of Aegis: Fault Diagnosis for AI Model Training Cloud Service in Production. In available on request."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629574"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3495883"},{"key":"e_1_3_2_1_14_1","volume-title":"EnvPipe: Performance-preserving DNN Training Framework for Saving Energy. In 2023 USENIX Annual Technical Conference (USENIX ATC 23)","author":"Choi Sangjin","year":"2023","unstructured":"Sangjin Choi, Inhoe Koo, Jeongseob Ahn, Myeongjae Jeon, and Youngjin Kwon. 2023. EnvPipe: Performance-preserving DNN Training Framework for Saving Energy. In 2023 USENIX Annual Technical Conference (USENIX ATC 23). USENIX Association, Boston, MA, 851\u2013864. https:\/\/www.usenix.org\/conference\/atc23\/presentation\/choi"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695970"},{"key":"e_1_3_2_1_16_1","volume-title":"Minder: Faulty Machine Detection for Large-scale Distributed Model Training. arXiv:2411.01791 [cs.DC] https:\/\/arxiv.org\/abs\/2411.01791","author":"Deng Yangtao","year":"2024","unstructured":"Yangtao Deng, Xiang Shi, Zhuo Jiang, Xingjian Zhang, Lei Zhang, Zhang Zhang, Bo Li, Zuquan Song, Hang Zhu, Gaohong Liu, Fuliang Li, Shuguang Wang, Haibin Lin, Jianxi Ye, and Minlan Yu. 2024. Minder: Faulty Machine Detection for Large-scale Distributed Model Training. arXiv:2411.01791 [cs.DC] https:\/\/arxiv.org\/abs\/2411.01791"},{"key":"e_1_3_2_1_17_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arXiv:1810.04805 [cs.CL] https:\/\/arxiv.org\/abs\/1810.04805","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arXiv:1810.04805 [cs.CL] https:\/\/arxiv.org\/abs\/1810.04805"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2024.3443255"},{"key":"e_1_3_2_1_19_1","volume-title":"Echo: Simulating Distributed Training At Scale. arXiv:2412.12487 [cs.LG] https:\/\/arxiv.org\/abs\/2412.12487","author":"Feng Yicheng","year":"2024","unstructured":"Yicheng Feng, Yuetao Chen, Kaiwen Chen, Jingzong Li, Tianyuan Wu, Peng Cheng, Chuan Wu, Wei Wang, Tsung-Yi Ho, and Hong Xu. 2024. Echo: Simulating Distributed Training At Scale. arXiv:2412.12487 [cs.LG] https:\/\/arxiv.org\/abs\/2412.12487"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672233"},{"key":"e_1_3_2_1_21_1","unstructured":"Talia Gershon Seetharami Seelam Brian Belgodere Milton Bonilla Lan Hoang Danny Barnett I-Hsin Chung Apoorve Mohan Ming-Hung Chen Lixiang Luo Robert Walkup Constantinos Evangelinos Shweta Salaria Marc Dombrowa Yoonho Park Apo Kayi Liran Schour Alim Alim Ali Sydney Pavlos Maniotis Laurent Schares Bernard Metzler Bengi Karacali-Akyamac Sophia Wen Tatsuhiro Chiba Sunyanan Choochotkaew Takeshi Yoshimura Claudia Misale Tonia Elengikal Kevin O Connor Zhuoran Liu Richard Molina Lars Schneidenbach James Caden Christopher Laibinis Carlos Fonseca Vasily Tarasov Swaminathan Sundararaman Frank Schmuck Scott Guthridge Jeremy Cohn Marc Eshel Paul Muench Runyu Liu William Pointer Drew Wyskida Bob Krull Ray Rose Brent Wolfe William Cornejo John Walter Colm Malone Clifford Perucci Frank Franco Nigel Hinds Bob Calio Pavel Druyan Robert Kilduff John Kienle Connor McStay Andrew Figueroa Matthew Connolly Edie Fost Gina Roma Jake Fonseca Ido Levy Michele Payne Ryan Schenkel Amir Malki Lion Schneider Aniruddha Narkhede Shekeba Moshref Alexandra Kisin Olga Dodin Bill Rippon Henry Wrieth John Ganci Johnny Colino Donna Habeger-Rose Rakesh Pandey Aditya Gidh Aditya Gaur Dennis Patterson Samsuddin Salmani Rambilas Varma Rumana Rumana Shubham Sharma Aditya Gaur Mayank Mishra Rameswar Panda Aditya Prasad Matt Stallone Gaoyuan Zhang Yikang Shen David Cox Ruchir Puri Dakshi Agrawal Drew Thorstensen Joel Belog Brent Tang Saurabh Kumar Gupta Amitabha Biswas Anup Maheshwari Eran Gampel Jason Van Patten Matthew Runion Sai Kaki Yigal Bogin Brian Reitz Steve Pritko Shahan Najam Surya Nambala Radhika Chirra Rick Welp Frank DiMitri Felipe Telles Amilcar Arvelo King Chu Ed Seminaro Andrew Schram Felix Eickhoff William Hanson Eric Mckeever Michael Light Dinakaran Joseph Piyush Chaudhary Piyush Shivam Puneet Chaudhary Wesley Jones Robert Guthrie Chris Bostic Rezaul Islam Steve Duersch Wayne Sawdon John Lewars Matthew Klos Michael Spriggs Bill McMillan George Gao Ashish Kamra Gaurav Singh Marc Curry Tushar Katarki Joe Talerico Zenghui Shi Sai Sindhur Malleni and Erwan Gallen. 2025. The infrastructure powering IBM's Gen AI model development. arXiv:2407.05467 [cs.DC] https:\/\/arxiv.org\/abs\/2407.05467"},{"key":"e_1_3_2_1_22_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri et al. 2024. The Llama 3 Herd of Models. arXiv:2407.21783 [cs.AI] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/2829988.2787496"},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of Machine Learning and Systems, D. Marculescu, Y. Chi, and C. Wu (Eds.)","volume":"4","author":"Hu Hanpeng","year":"2022","unstructured":"Hanpeng Hu, Chenyu Jiang, Yuchen Zhong, Yanghua Peng, Chuan Wu, Yibo Zhu, Haibin Lin, and Chuanxiong Guo. 2022. dPRO: A Generic Performance Diagnosis and Optimization Toolkit for Expediting Distributed DNN Training. In Proceedings of Machine Learning and Systems, D. Marculescu, Y. Chi, and C. Wu (Eds.), Vol. 4. 623\u2013637. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2022\/file\/b422680f3db0986ddd7f8f126baaf0fa-Paper.pdf"},{"key":"e_1_3_2_1_25_1","volume-title":"DISTMM: Accelerating Distributed Multimodal Model Training. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Huang Jun","year":"2024","unstructured":"Jun Huang, Zhen Zhang, Shuai Zheng, Feng Qin, and Yida Wang. 2024. DISTMM: Accelerating Distributed Multimodal Model Training. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24). USENIX Association, Santa Clara, CA, 1157\u20131171. https:\/\/www.usenix.org\/conference\/nsdi24\/presentation\/huang"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of Machine Learning and Systems, A. Talwalkar, V. Smith, and M. Zaharia (Eds.)","volume":"1","author":"Jia Zhihao","year":"2019","unstructured":"Zhihao Jia, Matei Zaharia, and Alex Aiken. 2019. Beyond Data and Model Parallelism for Deep Neural Networks.. In Proceedings of Machine Learning and Systems, A. Talwalkar, V. Smith, and M. Zaharia (Eds.), Vol. 1. 1\u201313. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2019\/file\/b422680f3db0986ddd7f8f126baaf0fa-Paper.pdf"},{"key":"e_1_3_2_1_27_1","volume-title":"21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Jiang Ziheng","year":"2024","unstructured":"Ziheng Jiang, Haibin Lin, Yinmin Zhong, Qi Huang, Yangrui Chen, Zhi Zhang, Yanghua Peng, Xiang Li, Cong Xie, Shibiao Nong, Yulu Jia, Sun He, Hongmin Chen, Zhihao Bai, Qi Hou, Shipeng Yan, Ding Zhou, Yiyao Sheng, Zhuo Jiang, Haohan Xu, Haoran Wei, Zhang Zhang, Pengfei Nie, Leqi Zou, Sida Zhao, Liang Xiang, Zherui Liu, Zhe Li, Xiaoying Jia, Jianxi Ye, Xin Jin, and Xin Liu. 2024. MegaScale: Scaling Large Language Model Training to More Than 10,000 GPUs. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24). USENIX Association, Santa Clara, CA, 745\u2013760. https:\/\/www.usenix.org\/conference\/nsdi24\/presentation\/jiang-ziheng"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In Proceedings of the 40th International Conference on Machine Learning (Honolulu, Hawaii, USA) (ICML'23). JMLR.org, Article 814, 13 pages."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3603269.3604836"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3603269.3604869"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672264"},{"key":"e_1_3_2_1_32_1","volume-title":"Hostping: Diagnosing Intra-host Network Bottlenecks in RDMA Servers. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Liu Kefei","year":"2023","unstructured":"Kefei Liu, Zhuo Jiang, Jiao Zhang, Haoran Wei, Xiaolong Zhong, Lizhuang Tan, Tian Pan, and Tao Huang. 2023. Hostping: Diagnosing Intra-host Network Bottlenecks in RDMA Servers. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). USENIX Association, Boston, MA, 15\u201329. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/liu-kefei"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3587135.3592200"},{"key":"e_1_3_2_1_34_1","volume-title":"21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Matam Kiran Kumar","year":"2024","unstructured":"Kiran Kumar Matam, Hani Ramezani, Fan Wang, Zeliang Chen, Yue Dong, Maomao Ding, Zhiwei Zhao, Zhengyu Zhang, Ellie Wen, and Assaf Eisenman. 2024. QuickUpdate: a Real-Time Personalization System for Large-Scale Recommendation Models. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24). USENIX Association, Santa Clara, CA, 731\u2013744. https:\/\/www.usenix.org\/conference\/nsdi24\/presentation\/matam"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_2_1_36_1","unstructured":"Maxim Naumov Dheevatsa Mudigere Hao-Jun Michael Shi Jianyu Huang Narayanan Sundaraman Jongsoo Park Xiaodong Wang Udit Gupta Carole-Jean Wu Alisson G. Azzolini Dmytro Dzhulgakov Andrey Mallevich Ilia Cherniavskii Yinghai Lu Raghuraman Krishnamoorthi Ansha Yu Volodymyr Kondratenko Stephanie Pereira Xianjie Chen Wenlin Chen Vijay Rao Bill Jia Liang Xiong and Misha Smelyanskiy. 2019. Deep Learning Recommendation Model for Personalization and Recommendation Systems. arXiv:1906.00091 [cs.IR] https:\/\/arxiv.org\/abs\/1906.00091"},{"key":"e_1_3_2_1_37_1","unstructured":"OpenAI Josh Achiam Steven Adler et al. 2024. GPT-4 Technical Report. arXiv:2303.08774 [cs.CL] https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2018.2801475"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672265"},{"key":"e_1_3_2_1_40_1","unstructured":"Alec Radford Jeff Wu Rewon Child David Luan Dario Amodei and Ilya Sutskever. 2019. Language Models are Unsupervised Multitask Learners. (2019)."},{"key":"e_1_3_2_1_41_1","volume-title":"CASSINI: Network-Aware Job Scheduling in Machine Learning Clusters. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Rajasekaran Sudarsanan","year":"2024","unstructured":"Sudarsanan Rajasekaran, Manya Ghobadi, and Aditya Akella. 2024. CASSINI: Network-Aware Job Scheduling in Machine Learning Clusters. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24). USENIX Association, Santa Clara, CA. https:\/\/www.usenix.org\/conference\/nsdi24\/presentation\/rajasekaran"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"e_1_3_2_1_43_1","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. 2019. Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053 (2019)."},{"key":"e_1_3_2_1_44_1","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar Aurelien Rodriguez Armand Joulin Edouard Grave and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. arXiv:2302.13971 [cs.CL] https:\/\/arxiv.org\/abs\/2302.13971"},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of Machine Learning and Systems, I. Dhillon, D. Papailiopoulos, and V. Sze (Eds.)","volume":"2","author":"Wang Guanhua","year":"2020","unstructured":"Guanhua Wang, Shivaram Venkataraman, Amar Phanishayee, Nikhil Devanur, Jorgen Thelin, and Ion Stoica. 2020. Blink: Fast and Generic Collectives for Distributed ML. In Proceedings of Machine Learning and Systems, I. Dhillon, D. Papailiopoulos, and V. Sze (Eds.), Vol. 2. 172\u2013186. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2020\/file\/cd3a9a55f7f3723133fa4a13628cdf03-Paper.pdf"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/HOTI63208.2024.00013"},{"key":"e_1_3_2_1_47_1","volume-title":"TopoOpt: Co-optimizing Network Topology and Parallelization Strategy for Distributed Training Jobs. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Wang Weiyang","year":"2023","unstructured":"Weiyang Wang, Moein Khazraee, Zhizhen Zhong, Manya Ghobadi, Zhihao Jia, Dheevatsa Mudigere, Ying Zhang, and Anthony Kewitsch. 2023. TopoOpt: Co-optimizing Network Topology and Parallelization Strategy for Distributed Training Jobs. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). USENIX Association, Boston, MA, 739\u2013767. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/wang-weiyang"},{"key":"e_1_3_2_1_48_1","volume-title":"TRANSOM: An Efficient Fault-Tolerant System for Training LLMs. arXiv:2310.10046 [cs.DC] https:\/\/arxiv.org\/abs\/2310.10046","author":"Wu Baodong","year":"2023","unstructured":"Baodong Wu, Lei Xia, Qingping Li, Kangyu Li, Xu Chen, Yongqiang Guo, Tieyao Xiang, Yuheng Chen, and Shigang Li. 2023. TRANSOM: An Efficient Fault-Tolerant System for Training LLMs. arXiv:2310.10046 [cs.DC] https:\/\/arxiv.org\/abs\/2310.10046"},{"key":"e_1_3_2_1_49_1","volume-title":"Zeus: Understanding and Optimizing GPU Energy Consumption of DNN Training. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"You Jie","year":"2023","unstructured":"Jie You, Jae-Won Chung, and Mosharaf Chowdhury. 2023. Zeus: Understanding and Optimizing GPU Energy Consumption of DNN Training. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). USENIX Association, Boston, MA, 119\u2013139. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/you"},{"key":"e_1_3_2_1_50_1","volume-title":"Hashing Linearity Enables Relative Path Control in Data Centers. In 2021 USENIX Annual Technical Conference (USENIX ATC 21)","author":"Zhang Zhehui","year":"2021","unstructured":"Zhehui Zhang, Haiyang Zheng, Jiayao Hu, Xiangning Yu, Chenchen Qi, Xuemei Shi, and Guohui Wang. 2021. Hashing Linearity Enables Relative Path Control in Data Centers. In 2021 USENIX Annual Technical Conference (USENIX ATC 21). USENIX Association, 855\u2013862. https:\/\/www.usenix.org\/conference\/atc21\/presentation\/zhang-zhehui"},{"key":"e_1_3_2_1_51_1","volume-title":"Hashing Linearity Enables Relative Path Control in Data Centers. In 2021 USENIX Annual Technical Conference (USENIX ATC 21)","author":"Zhang Zhehui","year":"2021","unstructured":"Zhehui Zhang, Haiyang Zheng, Jiayao Hu, Xiangning Yu, Chenchen Qi, Xuemei Shi, and Guohui Wang. 2021. Hashing Linearity Enables Relative Path Control in Data Centers. In 2021 USENIX Annual Technical Conference (USENIX ATC 21). USENIX Association, 855\u2013862. https:\/\/www.usenix.org\/conference\/atc21\/presentation\/zhang-zhehui"},{"key":"e_1_3_2_1_52_1","volume-title":"Daydream: Accurately Estimating the Efficacy of Optimizations for DNN Training. In 2020 USENIX Annual Technical Conference (USENIX ATC 20)","author":"Zhu Hongyu","year":"2020","unstructured":"Hongyu Zhu, Amar Phanishayee, and Gennady Pekhimenko. 2020. Daydream: Accurately Estimating the Efficacy of Optimizations for DNN Training. In 2020 USENIX Annual Technical Conference (USENIX ATC 20). USENIX Association, 337\u2013352. https:\/\/www.usenix.org\/conference\/atc20\/presentation\/zhu-hongyu"}],"event":{"name":"SIGCOMM '25: ACM SIGCOMM 2025 Conference","location":"S\u00e3o Francisco Convent Coimbra Portugal","acronym":"SIGCOMM '25","sponsor":["SIGCOMM ACM Special Interest Group on Data Communication"]},"container-title":["Proceedings of the ACM SIGCOMM 2025 Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3718958.3750521","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T15:07:31Z","timestamp":1759244851000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3718958.3750521"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,27]]},"references-count":52,"alternative-id":["10.1145\/3718958.3750521","10.1145\/3718958"],"URL":"https:\/\/doi.org\/10.1145\/3718958.3750521","relation":{},"subject":[],"published":{"date-parts":[[2025,8,27]]},"assertion":[{"value":"2025-08-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}