{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T05:23:15Z","timestamp":1783056195103,"version":"3.54.6"},"reference-count":35,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T00:00:00Z","timestamp":1780444800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T00:00:00Z","timestamp":1780444800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,6,3]]},"DOI":"10.1109\/secon68281.2026.11579096","type":"proceedings-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T19:41:25Z","timestamp":1783021285000},"page":"61-69","source":"Crossref","is-referenced-by-count":0,"title":["TorBinPack: Topology-Aware Online Scheduler for Minimizing Congestion in AI Superclusters"],"prefix":"10.1109","author":[{"given":"R\u00e9mi Lei","family":"Chen","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University,Shanghai Key Laboratory of Scalable Computing and Systems, School of Computer Science,Shanghai,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuang","family":"Zhao","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University,Shanghai Key Laboratory of Scalable Computing and Systems, School of Computer Science,Shanghai,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tin Ping","family":"Chan","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University,Shanghai Key Laboratory of Scalable Computing and Systems, School of Computer Science,Shanghai,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yinuo","family":"Li","sequence":"additional","affiliation":[{"name":"Huawei,Shenzhen,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wen","family":"Peng","sequence":"additional","affiliation":[{"name":"Huawei,Shenzhen,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaofeng","family":"Gao","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University,Shanghai Key Laboratory of Scalable Computing and Systems, School of Computer Science,Shanghai,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guihai","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University,Shanghai Key Laboratory of Scalable Computing and Systems, School of Computer Science,Shanghai,China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in neural information processing systems (NeurIPS)"},{"key":"ref2","article-title":"Pangu-\u03a3: Towards trillion parameter language model with sparse heterogeneous computing","author":"Ren","year":"2023","journal-title":"arXiv preprint arXiv:2303.10845"},{"key":"ref3","article-title":"Nvidia ethernet networking accelerates world\u2019s largest ai supercomputer, built by xai","year":"2024"},{"key":"ref4","article-title":"Building meta\u2019s genai infrastructure","author":"Lee","year":"2024"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672265"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.48550\/arxiv.1811.06965"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.1206"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1002\/j.1538-7305.1953.tb01433.x"},{"key":"ref9","article-title":"An rdma protocol specification (version 1.0)","author":"Recio","year":"2002"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672239"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3617232.3624863"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM48880.2022.9796688"},{"key":"ref14","article-title":"Nvidia h100 price guide 2025: Detailed costs, comparisons & expert insights - jarvislabs.ai docs","year":"2025"},{"key":"ref15","article-title":"Quobyte brings gpu-converged storage to ai clusters","author":"Mellor","year":"2025"},{"key":"ref16","first-page":"595","article-title":"Gandiva: Introspective cluster scheduling for deep learning","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Xiao"},{"key":"ref17","first-page":"515","article-title":"{HiveD}: Sharing a {GPU} cluster for deep learning with guarantees","volume-title":"14th USENIX symposium on operating systems design and implementation (OSDI 20)","author":"Zhao"},{"key":"ref18","first-page":"593","article-title":"{TACCL}: Guiding collective algorithm synthesis using communication sketches","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Shah"},{"key":"ref19","first-page":"1403","article-title":"Cassini: Network-aware job scheduling in machine learning clusters","author":"Rajasekaran","year":"2024","journal-title":"USENIX Networked Systems Design and Implementation (NSDI)"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2025.105138"},{"key":"ref21","volume-title":"Computers and Intractability: A Guide to the Theory of NP-Completeness","author":"Garey","year":"1979"},{"key":"ref22","article-title":"Volcano: A cloud native batch system","year":"2025"},{"key":"ref23","first-page":"485","article-title":"Tiresias: A gpu cluster manager for distributed deep learning","author":"Gu","year":"2019","journal-title":"USENIX Networked Systems Design and Implementation (NSDI)"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3190508.3190517"},{"key":"ref25","first-page":"289","article-title":"Themis: Fair and efficient gpu cluster scheduling","volume-title":"USENIX Symposium on Networked Systems Design and Implementation (NSDI)","author":"Mahajan"},{"key":"ref26","first-page":"945","article-title":"Mlaas in the wild: Workload analysis and scheduling in large-scale heterogeneous gpu clusters","volume-title":"USENIX Symposium on Networked Systems Design and Implementation (NSDI)","author":"Weng"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3567499"},{"key":"ref28","first-page":"481","article-title":"Heterogeneity-aware cluster scheduling policies for deep learning workloads","volume-title":"USENIX Symposium on Operating Systems Design and Implementation (OSDI)","author":"Narayanan"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613175"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/SECON64284.2024.10934901"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/SECON64284.2024.10934850"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/SECON64284.2024.10934959"},{"key":"ref33","first-page":"1","article-title":"Topology-Aware gpu scheduling for learning workloads in cloud environments","volume-title":"International Conference for High Performance Computing, Networking, Storage and Analysis (SC)","author":"Amaral"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TCC.2020.3040312"},{"key":"ref35","article-title":"Communication contention aware scheduling of multiple deep learning training jobs","author":"Wang","year":"2020","journal-title":"arXiv preprint arXiv:2002.10105"}],"event":{"name":"2026 22nd Annual IEEE International Conference on Sensing, Communication, and Networking (SECON)","location":"Pisa, Italy","start":{"date-parts":[[2026,6,3]]},"end":{"date-parts":[[2026,6,5]]}},"container-title":["2026 22nd Annual IEEE International Conference on Sensing, Communication, and Networking (SECON)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11578950\/11578436\/11579096.pdf?arnumber=11579096","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T05:15:01Z","timestamp":1783055701000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11579096\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,3]]},"references-count":35,"URL":"https:\/\/doi.org\/10.1109\/secon68281.2026.11579096","relation":{},"subject":[],"published":{"date-parts":[[2026,6,3]]}}}