{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,13]],"date-time":"2026-08-13T06:28:03Z","timestamp":1786602483962,"version":"build-2736575974"},"reference-count":72,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1109\/icde65706.2026.00271","type":"proceedings-article","created":{"date-parts":[[2026,8,12]],"date-time":"2026-08-12T19:18:20Z","timestamp":1786562300000},"page":"3639-3653","source":"Crossref","is-referenced-by-count":0,"title":["DLRover-LM: LLM Pre-Training Framework With Thousands of Accelerators in AntGroup"],"prefix":"10.1109","author":[{"given":"Ziling","family":"Huang","sequence":"first","affiliation":[{"name":"Sichuan University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhengmao","family":"Ye","sequence":"additional","affiliation":[{"name":"Sichuan University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingsong","family":"Cai","sequence":"additional","affiliation":[{"name":"Sichuan University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zelong","family":"Huang","sequence":"additional","affiliation":[{"name":"Sichuan University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Sang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haitao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jian","family":"Sha","sequence":"additional","affiliation":[{"name":"Tsinghua University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tingfeng","family":"Lan","sequence":"additional","affiliation":[{"name":"University of Virginia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hui","family":"Lu","sequence":"additional","affiliation":[{"name":"The University of Texas at Arlington"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuanchun","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingjie","family":"Tang","sequence":"additional","affiliation":[{"name":"Sichuan University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.14778\/3641204.3641221"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.14778\/3401960.3401970"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.14778\/3583140.3583165"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.14778\/3681954.3682003"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE60146.2024.00009"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383496"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/s11431-020-1632-x"},{"key":"ref8","volume-title":"Introducing chatgpt"},{"key":"ref9","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023","journal-title":"arXiv preprint"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640366"},{"key":"ref11","article-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020","journal-title":"arXiv preprint"},{"key":"ref12","article-title":"Scaling laws for autoregressive generative modeling","author":"Henighan","year":"2020","journal-title":"arXiv preprint"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE65448.2025.00029"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-012-9338-y"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651379"},{"key":"ref16","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3458817.3476209","article-title":"Efficient large-scale language model training on gpu clusters using megatron-lm","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Narayanan","year":"2021"},{"key":"ref17","article-title":"Characterization of large language model development in the datacenter","volume-title":"Proceedingsof the 21st USENIX Symposium on Networked Systems Design and Implementation, ser. NSDI\u201924","author":"Hu"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.14778\/3415478.3415530"},{"key":"ref19","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism","author":"Shoeybi","year":"2020"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.48550\/arxiv.1811.06965"},{"key":"ref21","doi-asserted-by":"crossref","DOI":"10.1109\/HPCA61900.2025.00096","volume-title":"Revisiting reliability in large-scale machine learning research clusters","author":"Kokolis","year":"2025"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.14778\/3685800.3685832"},{"key":"ref23","article-title":"Opt: Open pre-trained transformer language models","author":"Zhang","year":"2022","journal-title":"arXiv preprint"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.14778\/3611540.3611569"},{"key":"ref25","article-title":"Pipedream: Fast and efficient pipeline parallel dnn training","author":"Harlap","year":"2018","journal-title":"arXiv preprint"},{"key":"ref26","volume-title":"Simplefsdp: Simpler fully sharded data parallel with torch.compile","author":"Zhang","year":"2024"},{"key":"ref27","article-title":"Gshard: Scaling giant models with conditional computation and automatic sharding","author":"Lepikhin","year":"2020","journal-title":"arXiv preprint"},{"key":"ref28","first-page":"559","article-title":"Alpa: Automating inter-and \\{Intra-Operator\\} parallelism for distributed deep learning","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng","year":"2022"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-1479-4"},{"key":"ref30","volume-title":"Google\u2019s operations research tools"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1189"},{"key":"ref32","article-title":"FlashAttention-2: Faster attention with better parallelism and work partitioning","volume-title":"International Conference on Learning Representations (ICLR)","author":"Dao","year":"2024"},{"key":"ref33","volume-title":"Xla"},{"key":"ref34","article-title":"NVIDIA Corporation","volume-title":"NVIDIA Nsight Systems","year":"2025"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/3315508.3329973"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.14778\/3675034.3675043"},{"key":"ref37","article-title":"Xputimer: Anomaly diagnostics for divergent 1lm training in gpu clusters of thousand-plus scale","author":"Cui","year":"2025","journal-title":"arXiv preprint"},{"key":"ref38","volume-title":"Qwen2 technical report","author":"Yang","year":"2024"},{"key":"ref39","volume-title":"Qwen3 technical report","author":"Yang","year":"2025"},{"key":"ref40","article-title":"Every flop counts: Scaling a 300b mixture-of-experts ling 1lm without premium gpus","author":"Team","year":"2025","journal-title":"arXiv preprint"},{"key":"ref41","volume-title":"Pytorch fsdp: Experiences on scaling fully sharded data parallel","author":"Zhao","year":"2023"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.14778\/3415478.3415530"},{"key":"ref43","volume-title":"Gspmd: General and scalable parallelization for ml computation graphs","author":"Xu","year":"2021"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441593"},{"key":"ref46","first-page":"7937","article-title":"Memory-efficient pipeline-parallel dnn training","volume-title":"International Conference on Machine Learning. PMLR","author":"Narayanan","year":"2021"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"ref49","first-page":"18332","article-title":"Deepspeed-moe: Advancing mixture-of-experts inference and training to power next-generation ai scale","volume-title":"International conference on machine learning. PMLR","author":"Rajbhandari","year":"2022"},{"key":"ref50","article-title":"Deepseek-r1: Incentivizing reasoning capability in 1lms via reinforcement learning","author":"Guo","year":"2025","journal-title":"arXiv preprint"},{"key":"ref51","volume-title":"Deepseek1lm: Scaling open-source language models with longtermism","author":"Bi","year":"2024"},{"key":"ref52","author":"Guo","year":"2025","journal-title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning"},{"key":"ref53","volume-title":"Deepseek-v3 technical report","author":"Liu","year":"2025"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.14778\/3570690.3570697"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1145\/3472883.3486987"},{"key":"ref56","first-page":"347","article-title":"nnScaler: Constraint-Guided parallelization plan generation for deep learning training","volume-title":"8th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Lin"},{"key":"ref57","article-title":"Moe parallel folding: Heterogeneous parallelism mappings for efficient large-scale moe model training with megatron core","author":"Liu","year":"2025","journal-title":"arXiv preprint"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1145\/3767295.3769325"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507778"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1145\/3567955.3567959"},{"key":"ref62","volume-title":"Flux: Fast software-based communication overlap on gpus through kernel fusion","author":"Chang","year":"2024"},{"key":"ref63","first-page":"265","article-title":"Tensorflow: a system for large-scale machine learning","volume-title":"Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation, ser. OSDI\u201916. USA: USENIX Association","author":"Abadi","year":"2016"},{"key":"ref64","article-title":"Tensorflow: Large-scale machine learning on heterogeneous distributed systems","author":"Abadi","year":"2016","journal-title":"arXiv preprint"},{"key":"ref65","article-title":"Fault tolerance in distributed paradigms","volume-title":"In2011 International Conference on Computer Communication and Management, Proc. of CSIT","volume":"5","author":"Haider","year":"2011"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1145\/3190508.3190517"},{"issue":"28","key":"ref67","first-page":"1","article-title":"Distributed systems principles and paradigms","volume":"2","author":"Van Steen","year":"2002","journal-title":"Network"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3519584"},{"key":"ref69","first-page":"497","article-title":"Bamboo: Making preemptible instances resilient for affordable training of large DNNs","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Thorpe"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640375"},{"key":"ref71","article-title":"Unicron: Economizing self-healing 1lm training at scale","author":"He","year":"2023","journal-title":"arXiv preprint"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE60146.2024.00394"}],"event":{"name":"2026 IEEE 42nd International Conference on Data Engineering (ICDE)","location":"Montreal, QC, Canada","start":{"date-parts":[[2026,5,4]]},"end":{"date-parts":[[2026,5,8]]}},"container-title":["2026 IEEE 42nd International Conference on Data Engineering (ICDE)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11629178\/11629165\/11629553.pdf?arnumber=11629553","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,13]],"date-time":"2026-08-13T05:50:45Z","timestamp":1786600245000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11629553\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":72,"URL":"https:\/\/doi.org\/10.1109\/icde65706.2026.00271","relation":{},"subject":[],"published":{"date-parts":[[2026,5]]}}}