{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T16:50:07Z","timestamp":1782924607120,"version":"3.54.5"},"reference-count":45,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,25]],"date-time":"2026-05-25T00:00:00Z","timestamp":1779667200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,25]],"date-time":"2026-05-25T00:00:00Z","timestamp":1779667200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,25]]},"DOI":"10.1109\/ipdps65963.2026.00062","type":"proceedings-article","created":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T20:55:16Z","timestamp":1782852916000},"page":"669-682","source":"Crossref","is-referenced-by-count":0,"title":["From Skew to Symmetry: Node-Interconnect Multi-Path Balancing with Execution-time Planning for Modern GPU Clusters"],"prefix":"10.1109","author":[{"given":"Jinghan","family":"Yao","sequence":"first","affiliation":[{"name":"The Ohio State University,Department of Computer Science and Engineering,Columbus,OH,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kaushik","family":"Kandadi","sequence":"additional","affiliation":[{"name":"The Ohio State University,Department of Computer Science and Engineering,Columbus,OH,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bharath","family":"Ramesh","sequence":"additional","affiliation":[{"name":"The Ohio State University,Department of Computer Science and Engineering,Columbus,OH,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hari","family":"Subramoni","sequence":"additional","affiliation":[{"name":"The Ohio State University,Department of Computer Science and Engineering,Columbus,OH,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dhabaleswar K.","family":"Panda","sequence":"additional","affiliation":[{"name":"The Ohio State University,Department of Computer Science and Engineering,Columbus,OH,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Nvidia nvlink & nvlink switch","year":"2024"},{"key":"ref2","article-title":"NVIDIA NVLink Switch System (NVLink4\/NVSwitch3) \u2014 Introduction","volume-title":"per-link raw bandwidth and NVSwitch porting details","year":"2023"},{"key":"ref3","volume-title":"AMD Instinct MI300X Platform \u2014 Data Sheet","year":"2024"},{"key":"ref4","volume-title":"NDR Overview \u2014 NVIDIA DGX SuperPOD Design Guide","year":"2024"},{"key":"ref5","volume-title":"RDMA over Converged Ethernet (RoCE) version 2 \u2014 Configuration Guide"},{"issue":"120","key":"ref6","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus","year":"2022","journal-title":"Journal of Machine Learning Research"},{"key":"ref7","first-page":"18 332","article-title":"Deepspeed-MoE: Advancing mixture-of-experts inference and training to power next-generation AI scale","volume-title":"Proceedings of the 39th International Conference on Machine Learning (ICML)","volume":"162","author":"Rajbhandari"},{"key":"ref8","first-page":"269","article-title":"Tutel: Adaptive mixture-of-experts at scale","volume-title":"Proceedings of Machine Learning and Systems","volume":"5","author":"Hwang"},{"key":"ref9","article-title":"Deep learning recommendation model (dlrm) for personalization and recommendation systems","author":"Naumov","year":"2019"},{"key":"ref10","first-page":"193","article-title":"{DistServe}: Disaggregating prefill and decoding for goodput-optimized large language model serving","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/1807167.1807184"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/1654059.1654078"},{"key":"ref13","volume-title":"Doubling all2all performance with nvidia collective communication library 2.12","year":"2022"},{"key":"ref14","volume-title":"UCX Frequently Asked Questions","year":"2025"},{"key":"ref15","volume-title":"Unified Communication X (UCX) \u2014 HPC-X Documentation","year":"2025"},{"key":"ref16","article-title":"Building multirail infiniband clusters: MPI-level designs and performance evaluation","volume-title":"Proceedings of the 2004 International Conference on High Performance Computing","author":"Liu"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/SFCS.1998.743463"},{"key":"ref18","doi-asserted-by":"crossref","DOI":"10.1145\/1806689.1806708","article-title":"Faster approximation schemes for fractional multicommodity flow problems via dynamic graph algorithms","author":"Madry","year":"2010"},{"key":"ref19","volume-title":"GPUDirect RDMA \u2014 Overview (CUDA 13.0)","year":"2025"},{"key":"ref20","volume-title":"Understanding rccl bandwidth and xgmi performance on mi300x. Topology and link description for MI300X xGMI","year":"2025"},{"key":"ref21","volume-title":"QM97X0 Series \u2014 Specifications","year":"2025"},{"key":"ref22","volume-title":"RDMA over Converged Ethernet (RoCE) \u2014 MLNX OFED Documentation","year":"2024"},{"key":"ref23","volume-title":"GPUDirect RDMA and GPUDirect Storage \u2014 GPU Operator","year":"2025"},{"key":"ref24","article-title":"How does nccl know about internode topology? (pxn discussion)","year":"2023"},{"key":"ref25","volume-title":"Does pxn apply to all-reduce for rail-optimized topology?","year":"2023"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1137\/S0097539704446232"},{"key":"ref27","article-title":"NVIDIA Collective Communications Library (NCCL)"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/2780584"},{"key":"ref29","article-title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model","year":"2024"},{"key":"ref30","article-title":"Deepseek-v3 technical report","author":"Liu","year":"2024"},{"key":"ref31","article-title":"Qwen2 technical report","author":"Yang","year":"2024"},{"key":"ref32","doi-asserted-by":"crossref","DOI":"10.52202\/079017-2670","article-title":"Toward efficient inference for mixture of experts","volume-title":"Advances in Neural Information Processing Systems (NeurIPS 2024)","author":"Huang","year":"2024"},{"key":"ref33","article-title":"Lazarus: Resilient and elastic training of mixture-of-experts models with adaptive expert placement","author":"Wu","year":"2024"},{"key":"ref34","article-title":"Pro-prophet: A systematic load balancing method for efficient parallel training of large-scale moe models","author":"Wang","year":"2024"},{"key":"ref35","article-title":"A survey on inference optimization techniques for mixture of experts models","author":"Liu","year":"2024"},{"key":"ref36","first-page":"595","article-title":"Gandiva: Introspective cluster scheduling for deep learning","volume-title":"Proceedings of the 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI \u201918)","author":"Xiao"},{"key":"ref37","article-title":"Tiresias: A gpu cluster manager for distributed deep learning","volume-title":"Proceedings of the 16th USENIX Symposium on Networked Systems Design and Implementation (NSDI \u201919)","author":"Gu"},{"key":"ref38","article-title":"Pollux: Co-adaptive cluster scheduling for goodput-optimized deep learning","volume-title":"Proceedings of the 15th USENIX Symposium on Operating Systems Design and Implementation (OSDI \u201921)","author":"Qiao"},{"key":"ref39","first-page":"533","article-title":"Antman: Dynamic scaling on GPU clusters for deep learning","volume-title":"Proceedings of the 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI \u201920)","author":"Xiao"},{"key":"ref40","first-page":"400","article-title":"Salus: Fine-grained GPU sharing primitives for deep learning applications","volume-title":"Proceedings of Machine Learning and Systems (MLSys 2020)","author":"Yu"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629578"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/2785956.2787484"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3341302.3342085"},{"key":"ref44","first-page":"593","article-title":"TACCL: Guiding collective algorithm synthesis using communication sketches","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Shah"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2019.00087"}],"event":{"name":"2026 IEEE International Parallel and Distributed Processing Symposium (IPDPS)","location":"New Orleans, LA, USA","start":{"date-parts":[[2026,5,25]]},"end":{"date-parts":[[2026,5,29]]}},"container-title":["2026 IEEE International Parallel and Distributed Processing Symposium (IPDPS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11575315\/11575316\/11575365.pdf?arnumber=11575365","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T16:01:29Z","timestamp":1782921689000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11575365\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,25]]},"references-count":45,"URL":"https:\/\/doi.org\/10.1109\/ipdps65963.2026.00062","relation":{},"subject":[],"published":{"date-parts":[[2026,5,25]]}}}