{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T17:08:56Z","timestamp":1783184936634,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":64,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,4]],"date-time":"2024-08-04T00:00:00Z","timestamp":1722729600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,4]]},"DOI":"10.1145\/3651890.3672239","type":"proceedings-article","created":{"date-parts":[[2024,7,31]],"date-time":"2024-07-31T13:11:43Z","timestamp":1722431503000},"page":"1-15","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":42,"title":["Crux: GPU-Efficient Communication Scheduling for Deep Learning Training"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-5468-7366","authenticated-orcid":false,"given":"Jiamin","family":"Cao","sequence":"first","affiliation":[{"name":"Alibaba Cloud, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0726-3933","authenticated-orcid":false,"given":"Yu","family":"Guan","sequence":"additional","affiliation":[{"name":"Alibaba Cloud, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9882-9279","authenticated-orcid":false,"given":"Kun","family":"Qian","sequence":"additional","affiliation":[{"name":"Alibaba Cloud, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3548-2030","authenticated-orcid":false,"given":"Jiaqi","family":"Gao","sequence":"additional","affiliation":[{"name":"Alibaba Cloud, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3043-522X","authenticated-orcid":false,"given":"Wencong","family":"Xiao","sequence":"additional","affiliation":[{"name":"Alibaba Cloud, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0939-8943","authenticated-orcid":false,"given":"Jianbo","family":"Dong","sequence":"additional","affiliation":[{"name":"Alibaba Cloud, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1213-0554","authenticated-orcid":false,"given":"Binzhang","family":"Fu","sequence":"additional","affiliation":[{"name":"Alibaba Cloud, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7272-8143","authenticated-orcid":false,"given":"Dennis","family":"Cai","sequence":"additional","affiliation":[{"name":"Alibaba Cloud, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4352-7497","authenticated-orcid":false,"given":"Ennan","family":"Zhai","sequence":"additional","affiliation":[{"name":"Alibaba Cloud, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,8,4]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2022. AMD uProf. https:\/\/www.amd.com\/en\/developer\/uprof.html."},{"key":"e_1_3_2_1_2_1","unstructured":"2022. Equal-cost multi-path routing (ECMP). https:\/\/en.wikipedia.org\/wiki\/Equal-cost_multi-path_routing."},{"key":"e_1_3_2_1_3_1","unstructured":"2022. Intel Performance Counter Monitor. https:\/\/github.com\/intel\/pcm."},{"key":"e_1_3_2_1_4_1","unstructured":"2022. NVIDIA Collective Communications Library (NCCL). https:\/\/developer.nvidia.com\/nccl."},{"key":"e_1_3_2_1_5_1","unstructured":"2022. PyTorch. https:\/\/pytorch.org\/."},{"key":"e_1_3_2_1_6_1","unstructured":"2022. RDMA over Converged Ethernet. https:\/\/en.wikipedia.org\/wiki\/RDMA_over_Converged_Ethernet."},{"key":"e_1_3_2_1_7_1","unstructured":"2022. X-DeepLearning. https:\/\/github.com\/alibaba\/x-deeplearning."},{"key":"e_1_3_2_1_8_1","unstructured":"2023. Adobe Firefly. https:\/\/www.adobe.com\/sensei\/generative-ai\/firefly.html."},{"key":"e_1_3_2_1_9_1","volume-title":"Alibaba GPU Cluster Trace","year":"2023","unstructured":"2023. Alibaba GPU Cluster Trace 2023. https:\/\/github.com\/alibaba\/alibaba-lingjun-dataset-2023."},{"key":"e_1_3_2_1_10_1","unstructured":"2023. Breadth-first search. https:\/\/en.wikipedia.org\/wiki\/Breadth-first_search."},{"key":"e_1_3_2_1_11_1","unstructured":"2023. Github Copilot. https:\/\/github.com\/features\/copilot."},{"key":"e_1_3_2_1_12_1","unstructured":"2023. Microsoft365. https:\/\/www.microsoft.com\/en-us\/microsoft-365."},{"key":"e_1_3_2_1_13_1","unstructured":"2024. Megatron GPT3 MODEL. https:\/\/github.com\/NVIDIA\/Megatron-LM\/tree\/main\/examples\/gpt3."},{"key":"e_1_3_2_1_14_1","unstructured":"2024. Multi-commodity flow problem. https:\/\/en.wikipedia.org\/wiki\/Multi-commodity_flow_problem."},{"key":"e_1_3_2_1_15_1","unstructured":"Martin Abadi Paul Barham Jianmin Chen Zhifeng Chen Andy Davis Jeffrey Dean Matthieu Devin Sanjay Ghemawat Geoffrey Irving Michael Isard Manjunath Kudlur Josh Levenberg Rajat Monga Sherry Moore Derek G. Murray Benoit Steiner Paul Tucker Vijay Vasudevan Pete Warden Martin Wicke Yuan Yu and Xiaoqiang Zheng. 2016. TensorFlow: A system for large-scale machine learning. In OSDI."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3230543.3230569"},{"key":"e_1_3_2_1_17_1","volume-title":"Information-Agnostic Flow Scheduling for Commodity Data Centers. In 12th USENIX Symposium on Networked Systems Design and Implementation, NSDI 15","author":"Bai Wei","year":"2015","unstructured":"Wei Bai, Kai Chen, Hao Wang, Li Chen, Dongsu Han, and Chen Tian. 2015. Information-Agnostic Flow Scheduling for Commodity Data Centers. In 12th USENIX Symposium on Networked Systems Design and Implementation, NSDI 15, Oakland, CA, USA, May 4--6, 2015. USENIX Association, 455--468. https:\/\/www.usenix.org\/conference\/nsdi15\/technical-sessions\/presentation\/bai"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","unstructured":"Ran Ben Basat Sivaramakrishnan Ramanathan Yuliang Li Gianni Antichi Minlan Yu and Michael Mitzenmacher. 2020. PINT: Probabilistic In-band Network Telemetry. In SIGCOMM '20: Proceedings of the 2020 Annual conference of the ACM Special Interest Group on Data Communication on the applications technologies architectures and protocols for computer communication Virtual Event USA August 10--14 2020. ACM 662--680. 10.1145\/3387514.3405894","DOI":"10.1145\/3387514.3405894"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/2934872.2934888"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2872362.2872368"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421307"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2390231.2390237"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/2785956.2787480"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2018436.2018448"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2619239.2626315"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2619239.2626322"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00056"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2021.3091475"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/1080776.1080792"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","unstructured":"Robert Grandl Ganesh Ananthanarayanan Srikanth Kandula Sriram Rao and Aditya Akella. 2014. Multi-resource packing for cluster schedulers. (2014) 455--466. 10.1145\/2619239.2626334","DOI":"10.1145\/2619239.2626334"},{"key":"e_1_3_2_1_31_1","volume-title":"GRAPHENE: Packing and Dependency-Aware Scheduling for Data-Parallel Clusters. In 12th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2016","author":"Grandl Robert","year":"2016","unstructured":"Robert Grandl, Srikanth Kandula, Sriram Rao, Aditya Akella, and Janardhan Kulkarni. 2016. GRAPHENE: Packing and Dependency-Aware Scheduling for Data-Parallel Clusters. In 12th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2016, Savannah, GA, USA, November 2--4, 2016. USENIX Association, 81--97. https:\/\/www.usenix.org\/conference\/osdi16\/technical-sessions\/presentation\/grandl_graphene"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/1592568.1592576"},{"key":"e_1_3_2_1_33_1","volume-title":"Tiresias: A GPU Cluster Manager for Distributed Deep Learning. In 16th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2019","author":"Gu Juncheng","year":"2019","unstructured":"Juncheng Gu, Mosharaf Chowdhury, Kang G. Shin, Yibo Zhu, Myeongjae Jeon, Junjie Qian, Hongqiang Harry Liu, and Chuanxiong Guo. 2019. Tiresias: A GPU Cluster Manager for Distributed Deep Learning. In 16th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2019, Boston, MA, February 26--28, 2019. USENIX Association, 485--500. https:\/\/www.usenix.org\/conference\/nsdi19\/presentation\/gu"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-8191(06)80021-9"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS47924.2020.00113"},{"key":"e_1_3_2_1_36_1","volume-title":"Analysis of Large-Scale Multi-Tenant GPU Clusters for DNN Training Workloads. In 2019 USENIX Annual Technical Conference, USENIX ATC 2019","author":"Jeon Myeongjae","year":"2019","unstructured":"Myeongjae Jeon, Shivaram Venkataraman, Amar Phanishayee, Junjie Qian, Wencong Xiao, and Fan Yang. 2019. Analysis of Large-Scale Multi-Tenant GPU Clusters for DNN Training Workloads. In 2019 USENIX Annual Technical Conference, USENIX ATC 2019, Renton, WA, USA, July 10--12, 2019. USENIX Association, 947--960. https:\/\/www.usenix.org\/conference\/atc19\/presentation\/jeon"},{"key":"e_1_3_2_1_37_1","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2020","author":"Jiang Yimin","year":"2020","unstructured":"Yimin Jiang, Yibo Zhu, Chang Lan, Bairen Yi, Yong Cui, and Chuanxiong Guo. 2020. A Unified Architecture for Accelerating Distributed DNN Training in Heterogeneous GPU\/CPU Clusters. In 14th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2020, Virtual Event, November 4--6, 2020. USENIX Association, 463--479. https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/jiang"},{"key":"e_1_3_2_1_38_1","volume-title":"Towards An Application Objective-Aware Network Interface. In 12th USENIX Workshop on Hot Topics in Cloud Computing, HotCloud 2020","author":"Jyothi Sangeetha Abdu","year":"2020","unstructured":"Sangeetha Abdu Jyothi, Sayed Hadi Hashemi, Roy H. Campbell, and Brighten Godfrey. 2020. Towards An Application Objective-Aware Network Interface. In 12th USENIX Workshop on Hot Topics in Cloud Computing, HotCloud 2020, July 13--14, 2020. USENIX Association. https:\/\/www.usenix.org\/conference\/hotcloud20\/presentation\/jyothi"},{"key":"e_1_3_2_1_39_1","unstructured":"Changhoon Kim Anirudh Sivaraman Naga Katta Antonin Bas Advait Dixit and Lawrence J Wobker. 2015. In-band network telemetry via programmable dataplanes. In SIGCOMM."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.21437\/INTERSPEECH.2017-556"},{"key":"e_1_3_2_1_41_1","volume-title":"ATP: In-network Aggregation for Multitenant Learning. In 18th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2021","author":"Lao ChonLam","year":"2021","unstructured":"ChonLam Lao, Yanfang Le, Kshiteej Mahajan, Yixi Chen, Wenfei Wu, Aditya Akella, and Michael M. Swift. 2021. ATP: In-network Aggregation for Multitenant Learning. In 18th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2021, April 12--14, 2021. USENIX Association, 741--761. https:\/\/www.usenix.org\/conference\/nsdi21\/presentation\/lao"},{"key":"e_1_3_2_1_42_1","volume-title":"Accelerating Distributed MoE Training and Inference with Lina. In 2023 USENIX Annual Technical Conference, USENIX ATC 2023","author":"Li Jiamin","year":"2023","unstructured":"Jiamin Li, Yimin Jiang, Yibo Zhu, Cong Wang, and Hong Xu. 2023. Accelerating Distributed MoE Training and Inference with Lina. In 2023 USENIX Annual Technical Conference, USENIX ATC 2023, Boston, MA, USA, July 10--12, 2023. USENIX Association, 945--959. https:\/\/www.usenix.org\/conference\/atc23\/presentation\/li-jiamin"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341302.3342085"},{"key":"e_1_3_2_1_44_1","volume-title":"Hostping: Diagnosing Intra-host Network Bottlenecks in RDMA Servers. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023","author":"Liu Kefei","year":"2023","unstructured":"Kefei Liu, Zhuo Jiang, Jiao Zhang, Haoran Wei, Xiaolong Zhong, Lizhuang Tan, Tian Pan, and Tao Huang. 2023. Hostping: Diagnosing Intra-host Network Bottlenecks in RDMA Servers. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023, Boston, MA, April 17--19, 2023. USENIX Association, 15--29. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/liu-kefei"},{"key":"e_1_3_2_1_45_1","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023","author":"Liu Tianfeng","year":"2023","unstructured":"Tianfeng Liu, Yangrui Chen, Dan Li, Chuan Wu, Yibo Zhu, Jun He, Yanghua Peng, Hongzheng Chen, Hongzhi Chen, and Chuanxiong Guo. 2023. BGL: GPU-Efficient GNN Training by Optimizing Graph Data I\/O and Preprocessing. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023, Boston, MA, April 17--19, 2023. USENIX Association, 103--118. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/liu-tianfeng"},{"key":"e_1_3_2_1_46_1","volume-title":"Themis: Fair and Efficient GPU Cluster Scheduling. In 17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020","author":"Mahajan Kshiteej","year":"2020","unstructured":"Kshiteej Mahajan, Arjun Balasubramanian, Arjun Singhvi, Shivaram Venkataraman, Aditya Akella, Amar Phanishayee, and Shuchi Chawla. 2020. Themis: Fair and Efficient GPU Cluster Scheduling. In 17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020, Santa Clara, CA, USA, February 25--27, 2020. USENIX Association, 289--304. https:\/\/www.usenix.org\/conference\/nsdi20\/presentation\/mahajan"},{"key":"e_1_3_2_1_47_1","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023","author":"Mahajan Kshiteej","year":"2023","unstructured":"Kshiteej Mahajan, Ching-Hsiang Chu, Srinivas Sridharan, and Aditya Akella. 2023. Better Together: Jointly Optimizing ML Collective Scheduling and Execution Planning using SYNDICATE. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023, Boston, MA, April 17--19, 2023. USENIX Association, 809--824. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/mahajan"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3563766.3564096"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1016\/J.JPDC.2008.09.002"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359642"},{"key":"e_1_3_2_1_51_1","volume-title":"CASSINI: Network-Aware Job Scheduling in Machine Learning Clusters.","author":"Rajasekaran Sudarsanan","year":"2024","unstructured":"Sudarsanan Rajasekaran, Manya Ghobadi, and Aditya Akella. 2024. CASSINI: Network-Aware Job Scheduling in Machine Learning Clusters. (2024). https:\/\/www.usenix.org\/conference\/nsdi24\/presentation\/rajasekaran"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3563766.3564115"},{"key":"e_1_3_2_1_53_1","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023","author":"Shah Aashaka","year":"2023","unstructured":"Aashaka Shah, Vijay Chidambaram, Meghan Cowan, Saeed Maleki, Madan Musuvathi, Todd Mytkowicz, Jacob Nelson, and Olli Saarikivi. 2023. TACCL: Guiding Collective Algorithm Synthesis using Communication Sketches. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023, Boston, MA, April 17--19, 2023. USENIX Association, 593--612. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/shah"},{"key":"e_1_3_2_1_54_1","volume-title":"Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4--9, 2017, Long Beach, CA, USA. 5998--6008. https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html"},{"key":"e_1_3_2_1_55_1","volume-title":"Zhuang Wang, Ang Chen, and T. S. Eugene Ng.","author":"Wang Weitao","year":"2021","unstructured":"Weitao Wang, Sushovan Das, Xinyu Crystal Wu, Zhuang Wang, Ang Chen, and T. S. Eugene Ng. 2021. MXDAG: A Hybrid Abstraction for Cluster Applications. CoRR abs\/2107.07442 (2021). arXiv:2107.07442 https:\/\/arxiv.org\/abs\/2107.07442"},{"key":"e_1_3_2_1_56_1","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023","author":"Wang Weitao","year":"2023","unstructured":"Weitao Wang, Masoud Moshref, Yuliang Li, Gautam Kumar, T. S. Eugene Ng, Neal Cardwell, and Nandita Dukkipati. 2023. Poseidon: Efficient, Robust, and Practical Datacenter CC via Deployable INT. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023, Boston, MA, April 17--19, 2023. USENIX Association, 255--274. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/wang-weitao"},{"key":"e_1_3_2_1_57_1","volume-title":"Gandiva: Introspective Cluster Scheduling for Deep Learning. In 13th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2018","author":"Xiao Wencong","year":"2018","unstructured":"Wencong Xiao, Romil Bhardwaj, Ramachandran Ramjee, Muthian Sivathanu, Nipun Kwatra, Zhenhua Han, Pratyush Patel, Xuan Peng, Hanyu Zhao, Quanlu Zhang, Fan Yang, and Lidong Zhou. 2018. Gandiva: Introspective Cluster Scheduling for Deep Learning. In 13th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2018, Carlsbad, CA, USA, October 8--10, 2018. USENIX Association, 595--610. https:\/\/www.usenix.org\/conference\/osdi18\/presentation\/xiao"},{"key":"e_1_3_2_1_58_1","volume-title":"AntMan: Dynamic Scaling on GPU Clusters for Deep Learning. In 14th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2020","author":"Xiao Wencong","year":"2020","unstructured":"Wencong Xiao, Shiru Ren, Yong Li, Yang Zhang, Pengyang Hou, Zhi Li, Yihui Feng, Wei Lin, and Yangqing Jia. 2020. AntMan: Dynamic Scaling on GPU Clusters for Deep Learning. In 14th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2020, Virtual Event, November 4--6, 2020. USENIX Association, 533--548. https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/xiao"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/800141.804691"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/2741948.2741957"},{"key":"e_1_3_2_1_61_1","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2020","author":"Zhao Hanyu","year":"2020","unstructured":"Hanyu Zhao, Zhenhua Han, Zhi Yang, Quanlu Zhang, Fan Yang, Lidong Zhou, Mao Yang, Francis C. M. Lau, Yuqi Wang, Yifan Xiong, and Bin Wang. 2020. HiveD: Sharing a GPU Cluster for Deep Learning with Guarantees. In 14th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2020, Virtual Event, November 4--6, 2020. USENIX Association, 515--532. https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/zhao-hanyu"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3533044"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544216.3544224"},{"key":"e_1_3_2_1_64_1","volume-title":"Shockwave: Fair and Efficient Cluster Scheduling for Dynamic Adaptation in Machine Learning. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023","author":"Zheng Pengfei","year":"2023","unstructured":"Pengfei Zheng, Rui Pan, Tarannum Khan, Shivaram Venkataraman, and Aditya Akella. 2023. Shockwave: Fair and Efficient Cluster Scheduling for Dynamic Adaptation in Machine Learning. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023, Boston, MA, April 17--19, 2023. USENIX Association, 703--723. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/zheng"}],"event":{"name":"ACM SIGCOMM '24: ACM SIGCOMM 2024 Conference","location":"Sydney NSW Australia","acronym":"ACM SIGCOMM '24","sponsor":["SIGCOMM ACM Special Interest Group on Data Communication"]},"container-title":["Proceedings of the ACM SIGCOMM 2024 Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3651890.3672239","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3651890.3672239","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T17:49:12Z","timestamp":1750268952000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3651890.3672239"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,4]]},"references-count":64,"alternative-id":["10.1145\/3651890.3672239","10.1145\/3651890"],"URL":"https:\/\/doi.org\/10.1145\/3651890.3672239","relation":{},"subject":[],"published":{"date-parts":[[2024,8,4]]},"assertion":[{"value":"2024-08-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}