{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T01:19:29Z","timestamp":1782955169276,"version":"3.54.5"},"reference-count":82,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,5,19]]},"DOI":"10.1109\/ccgrid64434.2025.00038","type":"proceedings-article","created":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T17:36:08Z","timestamp":1751304968000},"page":"1-12","source":"Crossref","is-referenced-by-count":1,"title":["Deep Learning Training Job Scheduling for Proactive Straggler Reduction"],"prefix":"10.1109","author":[{"given":"Haiying","family":"Shen","sequence":"first","affiliation":[{"name":"University of Virginia,Department of Computer Science,Charlottesville,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zeyu","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Virginia,Department of Computer Science,Charlottesville,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476209"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/s11227-020-03241-x"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS51616.2021.00057"},{"key":"ref5","first-page":"1145","article-title":"Asynchronous Byzantine machine learning (the case of SGD)","volume-title":"In Proc. of the 35th International Conference on Machine Learning, ser. Proc. of Machine Learning Research","volume":"80","author":"Damaskinos","year":"2018"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3064176.3064182"},{"key":"ref7","first-page":"631","article-title":"Litz: Elastic framework for High-Performance distributed machine learning","volume-title":"Proc. of 2018 USENIX Annual Technical Conference (USENIX ATC 18)","author":"Qiao"},{"key":"ref8","first-page":"400","article-title":"Resource elasticity in distributed deep learning","volume-title":"In Proc. of Machine Learning and Systems, I. Dhillon, D. Papailiopoulos, and V. Sze, Eds.","volume":"2","author":"Or","year":"2020"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM42981.2021.9488815"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/2987550.2987554"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737587"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421307"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421299"},{"key":"ref14","article-title":"Heterogeneity-aware cluster scheduling policies for deep learning workloads","volume-title":"Proc. of OSDI","author":"Narayanan","year":"2020"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2019.00150"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304009"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2019.00028"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICAC.2016.42"},{"key":"ref19","first-page":"533","article-title":"AntMan: Dynamic scaling on GPU clusters for deep learning","volume-title":"Proc. of 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20). USENIX Association","author":"Xiao","year":"2020"},{"key":"ref20","first-page":"947","article-title":"Analysis of large-scale multi-tenant gpu clusters for dnn training workloads","volume-title":"Proc. of the 2019 USENIX Conference on Usenix Annual Technical Conference","author":"Jeon"},{"key":"ref21","article-title":"Gandiva: Introspective cluster scheduling for deep learning","volume-title":"Proc. of OSDI","author":"Xiao","year":"2018"},{"key":"ref22","first-page":"418","volume-title":"Tictac: Accelerating distributed deep learning with communication scheduling","volume":"1","author":"Hashemi","year":"2019"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/2741948.2741964"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2018.8486422"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3190508.3190517"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3127479.3127490"},{"key":"ref27","article-title":"Tiresias: A gpu cluster manager for distributed deep learning","volume-title":"Proc. of NSDI","author":"Gu","year":"2019"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737460"},{"key":"ref29","article-title":"Themis: Fair and efficient GPU cluster scheduling","volume-title":"Proc. of NSDI","author":"Mahajan","year":"2020"},{"key":"ref30","article-title":"GRAPHENE: Packing and dependency-aware scheduling for dataparallel clusters","volume-title":"Proc. of OSDI","author":"Grandl","year":"2016"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3342195.3387555"},{"key":"ref32","article-title":"Revisiting distributed synchronous SGD","author":"Chen","year":"2016","journal-title":"arXiv preprint"},{"key":"ref33","article-title":"Large scale distributed deep networks","volume-title":"Proc. of Advances in Neural Information Processing Systems, F. Pereira, C. Burges, L. Bottou, and K. Weinberger, Eds.","volume":"25","author":"Dean","year":"2012"},{"key":"ref34","volume-title":"Source Code","year":"2024"},{"key":"ref35","volume-title":"Kubernetes.","year":"2024"},{"key":"ref36","volume-title":"Mlbench: Distributed machine learning benchmark.","year":"2024"},{"key":"ref37","volume-title":"Porpular machine learning model in pytorch.","year":"2024"},{"key":"ref38","volume-title":"BookCorpus.","year":"2024"},{"key":"ref39","volume-title":"ImageNet-1K.","year":"2019"},{"key":"ref40","first-page":"1191","article-title":"THC: Accelerating distributed deep learning using tensor homomorphic compression","volume-title":"21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Li"},{"key":"ref41","volume-title":"Microsoft ElasticFlow Trace.","year":"2023"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575721"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3005745.3005750"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/3341302.3342080"},{"key":"ref45","first-page":"1","article-title":"Pollux: Co-adaptive cluster scheduling for goodput-optimized deep learning","volume-title":"Proc. of 15th USENIX Symposium on Operating Systems Design and Implementation (OSDI 21). USENIX Association","author":"Qiao"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS47924.2020.00033"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2019.1800386"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1145\/3544216.3544224"},{"key":"ref49","first-page":"559","article-title":"Alpa: Automating inter- and Intra-Operator parallelism for distributed deep learning","volume-title":"Proc. of 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng"},{"key":"ref50","article-title":"GPipe: Efficient Training of Giant Neural Networks Using Pipeline Parallelism.","volume-title":"Red Hook","author":"Huang","year":"2019"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/BigData47090.2019.9006011"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1145\/3603269.3604830"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611974782.51"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1145\/3098822.3098843"},{"key":"ref56","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. of ICML","author":"Mnih","year":"2016"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"key":"ref58","volume-title":"The llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359642"},{"key":"ref60","article-title":"Pytorch: An imperative style, high-performance deep learning library","author":"Paszke","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1145\/3544216.3544224"},{"key":"ref62","first-page":"69","article-title":"Transparent GPU sharing in container clouds for deep learning workloads","volume-title":"Proc. of 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Wu"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/tc.2023.3303988"},{"key":"ref64","article-title":"A Unified Architecture for Accelerating Distributed DNN Training in Heterogeneous GPU\/CPU Clusters","volume-title":"Proc. of the 14th USENIX Conference on Operating Systems Design and Implementation, ser. OSDI\u201920. USA: USENIX Association","author":"Jiang","year":"2020"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613175"},{"key":"ref66","article-title":"CASSINI: Network-Aware job scheduling in machine learning clusters","volume-title":"Proc. of the 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Rajasekaran"},{"key":"ref67","article-title":"Mlaas in the wild: Workload analysis and scheduling in large-scale heterogeneous gpu clusters","volume-title":"Proc. of 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22), ser. NSDI\u201922","author":"Weng"},{"key":"ref68","article-title":"Accelerating collective communication in data parallel training across deep learning frameworks","volume-title":"Proc. of 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22), ser. NSDI\u201922","author":"Romero"},{"key":"ref69","first-page":"809","article-title":"Better together: Jointly optimizing ML collective scheduling and execution planning using SYNDICATE","volume-title":"Proc. of 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Mahajan"},{"key":"ref70","first-page":"719","article-title":"Effectively scheduling computational graphs of deep neural networks toward their Domain-Specific accelerators","volume-title":"Proc. of the 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Zhao"},{"key":"ref71","first-page":"779","article-title":"MGG: Accelerating graph neural networks with Fine-Grained Intra-Kernel Communication-Computation pipelining on Multi-GPU platforms","volume-title":"Proc. of the 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Wang"},{"key":"ref72","first-page":"681","article-title":"Cocktailer: Analyzing and optimizing dynamic control flow in deep learning","volume-title":"Proc. of the 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Zhang"},{"key":"ref73","first-page":"663","article-title":"AlpaServe: Statistical multiplexing with model parallelism for deep learning serving","volume-title":"Proc. of 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Li"},{"key":"ref74","first-page":"267","article-title":"Unity: Accelerating DNN training through joint optimization of algebraic transformations and parallelization","volume-title":"Proc. of 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Unger"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS57955.2024.00035"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS57955.2024.00036"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1145\/3328740"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/BigData55660.2022.10021101"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2022.3205723"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1145\/3578933"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid54584.2022.00084"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1145\/2670979.2671005"}],"event":{"name":"2025 IEEE 25th International Symposium on Cluster, Cloud and Internet Computing (CCGrid)","location":"Troms\u00f8, Norway","start":{"date-parts":[[2025,5,19]]},"end":{"date-parts":[[2025,5,22]]}},"container-title":["2025 IEEE 25th International Symposium on Cluster, Cloud and Internet Computing (CCGrid)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11044421\/11044790\/11044824.pdf?arnumber=11044824","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T05:33:49Z","timestamp":1751348029000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11044824\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,19]]},"references-count":82,"URL":"https:\/\/doi.org\/10.1109\/ccgrid64434.2025.00038","relation":{},"subject":[],"published":{"date-parts":[[2025,5,19]]}}}