{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,30]],"date-time":"2025-08-30T05:40:08Z","timestamp":1756532408196,"version":"3.44.0"},"reference-count":34,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,8,4]],"date-time":"2025-08-04T00:00:00Z","timestamp":1754265600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,8,4]],"date-time":"2025-08-04T00:00:00Z","timestamp":1754265600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012492","name":"Youth Innovation Promotion Association","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012492","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,8,4]]},"DOI":"10.1109\/icccn65249.2025.11133724","type":"proceedings-article","created":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T17:39:20Z","timestamp":1756489160000},"page":"1-9","source":"Crossref","is-referenced-by-count":0,"title":["Network Resource-Aware Multi-Job Deployment in Deep Learning Clusters"],"prefix":"10.1109","author":[{"given":"Ai","family":"Zhong","sequence":"first","affiliation":[{"name":"University of Science and Technology of China,School of Computer Science and Technology,Hefei,China,230026"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gongming","family":"Zhao","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China,School of Computer Science and Technology,Hefei,China,230026"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongli","family":"Xu","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China,School of Computer Science and Technology,Hefei,China,230026"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jin","family":"Fang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China,School of Computer Science and Technology,Hefei,China,230026"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiawei","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China,School of Computer Science and Technology,Hefei,China,230026"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peng","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China,School of Computer Science and Technology,Hefei,China,230026"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.5555\/2999134.2999257"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/n19\u20131423"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"volume-title":"Cloud gpus on gcp","key":"ref5"},{"year":"2022","key":"ref6","article-title":"Gpu-accelerated microsoft azure"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3452296.3472904"},{"key":"ref8","first-page":"1421","article-title":"Towards {Domain-Specific} network transport for distributed {DNN} training","volume-title":"21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Wang"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3411029.3411037"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672265"},{"key":"ref11","article-title":"Maximize system throughput with nvidia nvlink"},{"journal-title":"Synergy: Resource sensitive dnn scheduling in multi-tenant clusters","year":"2021","author":"Mohan","key":"ref12"},{"key":"ref13","first-page":"485","article-title":"Tiresias: A {GPU} cluster manager for distributed deep learning","volume-title":"16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 19)","author":"Gu"},{"article-title":"Pollux: Co-adaptive cluster scheduling for goodput-optimized deep learning","volume-title":"15th USENIX Symposium on Operating Systems Design and Implementation ({OSDI}21)","author":"Qiao","key":"ref14"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575721"},{"key":"ref16","first-page":"481","article-title":"{Heterogeneity-Aware} cluster scheduling policies for deep learning workloads","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Narayanan"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629583"},{"key":"ref18","first-page":"515","article-title":"{HiveD} : Sharing a {GPU} cluster for deep learning with guarantees","volume-title":"14th USENIX symposium on operating systems design and implementation (OSDI 20)","author":"Zhao"},{"key":"ref19","first-page":"721","article-title":"Elastic resource sharing for distributed deep learning","volume-title":"18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21)","author":"Hwang"},{"key":"ref20","first-page":"595","article-title":"Gandiva: Introspective cluster scheduling for deep learning","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Xiao"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3405671.3405810"},{"article-title":"Plink: Discovering and exploiting datacenter network locality for efficient cloud-based distributed training","volume-title":"Proc. of MLSys","author":"Luo","key":"ref22"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid51090.2021.00049"},{"journal-title":"Accurate, large minibatch sgd: Training imagenet in 1 hour","year":"2017","author":"Goyal","key":"ref24"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2021.3091475"},{"journal-title":"Bandwidth reduction using importance weighted pruning on ring allreduce","year":"2019","author":"Cheng","key":"ref26"},{"key":"ref27","article-title":"Nvidia collective communications library (nccl)"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1016\/S0166-218X(01)00343-2"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/321958.321975"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/springerreference_6235"},{"volume-title":"ElasticFlow Traces","key":"ref31"},{"volume-title":"Acme traces from the Shanghai AI Lab","key":"ref32"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/HOTI.2013.23"},{"journal-title":"Megascale: Scaling large language model training to more than 10, 000 gpus","year":"2024","author":"Jiang","key":"ref34"}],"event":{"name":"2025 34th International Conference on Computer Communications and Networks (ICCCN)","start":{"date-parts":[[2025,8,4]]},"location":"Tokyo, Japan","end":{"date-parts":[[2025,8,7]]}},"container-title":["2025 34th International Conference on Computer Communications and Networks (ICCCN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11133715\/11133717\/11133724.pdf?arnumber=11133724","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,30]],"date-time":"2025-08-30T05:17:05Z","timestamp":1756531025000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11133724\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,4]]},"references-count":34,"URL":"https:\/\/doi.org\/10.1109\/icccn65249.2025.11133724","relation":{},"subject":[],"published":{"date-parts":[[2025,8,4]]}}}