{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T12:16:05Z","timestamp":1780661765158,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,26]],"date-time":"2026-04-26T00:00:00Z","timestamp":1777161600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Key R&D Program of Shandong Province","award":["2024CXGC010113"],"award-info":[{"award-number":["2024CXGC010113"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,27]]},"DOI":"10.1145\/3767295.3803603","type":"proceedings-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T20:20:04Z","timestamp":1777062004000},"page":"1160-1180","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["HARP: Orchestrating Automated Parallel Training on Heterogeneous GPU Clusters"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-2841-7334","authenticated-orcid":false,"given":"Antian","family":"Liang","sequence":"first","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4144-8587","authenticated-orcid":false,"given":"Zhigang","family":"Zhao","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7518-5466","authenticated-orcid":false,"given":"Kai","family":"Zhang","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4451-8550","authenticated-orcid":false,"given":"Xuri","family":"Shi","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2725-7277","authenticated-orcid":false,"given":"Chuantao","family":"Li","sequence":"additional","affiliation":[{"name":"Shandong Computer Science Center (National Supercomputer Center in Jinan), Jinan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-6617-0924","authenticated-orcid":false,"given":"Chunxiao","family":"Wang","sequence":"additional","affiliation":[{"name":"Shandong Computer Science Center (National Supercomputer Center in Jinan), Jinan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2926-4814","authenticated-orcid":false,"given":"Zhenying","family":"He","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1169-8032","authenticated-orcid":false,"given":"Yinan","family":"Jing","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9059-3713","authenticated-orcid":false,"given":"X. Sean","family":"Wang","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,26]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al. 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877\u20131901."},{"key":"e_1_3_2_1_3_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Dao Tri","year":"2024","unstructured":"Tri Dao. 2024. FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Tri Dao Daniel Y. Fu Stefano Ermon Atri Rudra and Christopher R\u00e9. 2022. FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness. In Advances in Neural Information Processing Systems (NeurIPS).","DOI":"10.52202\/068431-1189"},{"key":"e_1_3_2_1_5_1","volume-title":"Matthew James Johnson, and Chris Leary","author":"Frostig Roy","year":"2018","unstructured":"Roy Frostig, Matthew James Johnson, and Chris Leary. 2018. Compiling machine learning programs via high-level tracing. Systems for Machine Learning 4, 9 (2018)."},{"key":"e_1_3_2_1_6_1","unstructured":"Google. 2026. Google Cloud Provider. https:\/\/cloud.google.com\/ Last accessed on 2026-02-28."},{"key":"e_1_3_2_1_7_1","volume-title":"Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems 32","author":"Huang Yanping","year":"2019","unstructured":"Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Dehao Chen, Mia Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc V Le, Yonghui Wu, et al. 2019. Gpipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of Machine Learning and Systems 5","author":"Korthikanti Vijay Anand","year":"2023","unstructured":"Vijay Anand Korthikanti, Jared Casper, Sangkug Lym, Lawrence McAfee, Michael Andersch, Mohammad Shoeybi, and Bryan Catanzaro. 2023. Reducing activation recomputation in large transformer models. Proceedings of Machine Learning and Systems 5 (2023)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0480"},{"key":"e_1_3_2_1_10_1","volume-title":"Hetu v2: A General and Scalable Deep Learning System with Hierarchical and Heterogeneous Single Program Multiple Data Annotations. arXiv preprint arXiv:2504.20490","author":"Li Haoyang","year":"2025","unstructured":"Haoyang Li, Fangcheng Fu, Hao Ge, Sheng Lin, Xuanyu Wang, Jiawen Niu, Xupeng Miao, and Bin Cui. 2025. Hetu v2: A General and Scalable Deep Learning System with Hierarchical and Heterogeneous Single Program Multiple Data Annotations. arXiv preprint arXiv:2504.20490 (2025)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476145"},{"key":"e_1_3_2_1_12_1","unstructured":"Lianmin Zheng. 2022. Github repository: alpa-projects\/alpa. https:\/\/github.com\/alpa-projects\/alpa Last accessed on 2025-01-05."},{"key":"e_1_3_2_1_13_1","volume-title":"13th USENIX symposium on operating systems design and implementation (OSDI 18)","author":"Moritz Philipp","year":"2018","unstructured":"Philipp Moritz, Robert Nishihara, Stephanie Wang, Alexey Tumanov, Richard Liaw, Eric Liang, Melih Elibol, Zongheng Yang, William Paul, Michael I Jordan, et al. 2018. Ray: A distributed framework for emerging {AI} applications. In 13th USENIX symposium on operating systems design and implementation (OSDI 18). 561\u2013577."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_2_1_15_1","volume-title":"International Conference on Machine Learning. PMLR, 7937\u20137947","author":"Narayanan Deepak","year":"2021","unstructured":"Deepak Narayanan, Amar Phanishayee, Kaiyu Shi, Xie Chen, and Matei Zaharia. 2021. Memory-efficient pipeline-parallel dnn training. In International Conference on Machine Learning. PMLR, 7937\u20137947."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_2_1_17_1","unstructured":"Nvidia. 2025. Nvida SuperNIC. https:\/\/www.nvidia.cn\/networking\/products\/ethernet\/supernic\/ Last accessed on 2025-09-21."},{"key":"e_1_3_2_1_18_1","unstructured":"Nvidia. 2025. The nvidia collective communication library. https:\/\/github.com\/openxla\/xla Last accessed on 2025-09-21."},{"key":"e_1_3_2_1_19_1","volume-title":"Zero bubble pipeline parallelism. arXiv preprint arXiv:2401.10241","author":"Qi Penghui","year":"2023","unstructured":"Penghui Qi, Xinyi Wan, Guangxing Huang, and Min Lin. 2023. Zero bubble pipeline parallelism. arXiv preprint arXiv:2401.10241 (2023)."},{"key":"e_1_3_2_1_20_1","unstructured":"Alec Radford Karthik Narasimhan Tim Salimans Ilya Sutskever et al. 2018. Improving language understanding by generative pre-training. (2018)."},{"key":"e_1_3_2_1_21_1","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et al. 2019. Language models are unsupervised multitask learners. OpenAI blog 1 8 (2019) 9."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"e_1_3_2_1_23_1","volume-title":"International Conference on Machine Learning. PMLR, 29416\u201329440","author":"Ryabinin Max","year":"2023","unstructured":"Max Ryabinin, Tim Dettmers, Michael Diskin, and Alexander Borzunov. 2023. Swarm parallelism: Training large models can be surprisingly communication-efficient. In International Conference on Machine Learning. PMLR, 29416\u201329440."},{"key":"e_1_3_2_1_24_1","volume-title":"Megatron-lm: Training multibillion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. 2019. Megatron-lm: Training multibillion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053 (2019)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731569.3764839"},{"key":"e_1_3_2_1_26_1","volume-title":"2024 USENIX Annual Technical Conference (USENIX ATC 24)","author":"Um Taegeon","year":"2024","unstructured":"Taegeon Um, Byungsoo Oh, Minyoung Kang, Woo-Yeon Lee, Goeun Kim, Dongseob Kim, Youngtaek Kim, Mohd Muzzammil, and Myeongjae Jeon. 2024. Metis: Fast Automatic Distributed Training on Heterogeneous {GPUs}. In 2024 USENIX Annual Technical Conference (USENIX ATC 24). 563\u2013578."},{"key":"e_1_3_2_1_27_1","volume-title":"ATOM: Asynchronous Training of Massive Models for Deep Learning in a Decentralized Environment. arXiv preprint arXiv:2403.10504","author":"Wu Xiaofeng","year":"2024","unstructured":"Xiaofeng Wu, Jia Rao, and Wei Chen. 2024. ATOM: Asynchronous Training of Massive Models for Deep Learning in a Decentralized Environment. arXiv preprint arXiv:2403.10504 (2024)."},{"key":"e_1_3_2_1_28_1","unstructured":"XLA and TensorFlow teams. 2017. XLA \u2014 TensorFlow compiled. https:\/\/tensorflow.google.cn\/xla?hl=zh-cn#inspect_compiled_programs Last accessed on 2024-05-07."},{"key":"e_1_3_2_1_29_1","unstructured":"Si Xu Zixiao Huang Yan Zeng Shengen Yan Xuefei Ning Haolin Ye Sipei Gu Chunsheng Shui Zhezheng Lin Hao Zhang et al. 2024. HetHub: A Heterogeneous distributed hybrid training system for large-scale models. arXiv e-prints (2024) arXiv-2405."},{"key":"e_1_3_2_1_30_1","unstructured":"Yuanzhong Xu HyoukJoong Lee Dehao Chen Blake Hechtman Yanping Huang Rahul Joshi Maxim Krikun Dmitry Lepikhin Andy Ly Marcello Maggioni et al. 2021. GSPMD: general and scalable parallelization for ML computation graphs. arXiv preprint arXiv:2105.04663 (2021)."},{"key":"e_1_3_2_1_31_1","volume-title":"HexiScale: Accommodating Large Language Model Training over Heterogeneous Environment. arXiv preprint arXiv:2409.01143","author":"Yan Ran","year":"2024","unstructured":"Ran Yan, Youhe Jiang, Xiaonan Nie, Fangcheng Fu, Bin Cui, and Binhang Yuan. 2024. HexiScale: Accommodating Large Language Model Training over Heterogeneous Environment. arXiv preprint arXiv:2409.01143 (2024)."},{"key":"e_1_3_2_1_32_1","first-page":"25464","article-title":"Decentralized training of foundation models in heterogeneous environments","volume":"35","author":"Yuan Binhang","year":"2022","unstructured":"Binhang Yuan, Yongjun He, Jared Davis, Tianyi Zhang, Tri Dao, Beidi Chen, Percy S Liang, Christopher Re, and Ce Zhang. 2022. Decentralized training of foundation models in heterogeneous environments. Advances in Neural Information Processing Systems 35 (2022), 25464\u201325477.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.126661"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629580"},{"key":"e_1_3_2_1_35_1","volume-title":"Poplar: Efficient Scaling of Distributed DNN Training on Heterogeneous GPU Clusters. arXiv preprint arXiv:2408.12596","author":"Zhang WenZheng","year":"2024","unstructured":"WenZheng Zhang, Yang Hu, Jing Shi, and Xiaoying Bai. 2024. Poplar: Efficient Scaling of Distributed DNN Training on Heterogeneous GPU Clusters. arXiv preprint arXiv:2408.12596 (2024)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","unstructured":"WenZheng Zhang Yang Hu Jing Shi and Xiaoying Bai. 2025. Poplar: efficient scaling of distributed DNN training on heterogeneous GPU clusters. In Proceedings of the Thirty-Ninth AAAI Conference on Artificial Intelligence and Thirty-Seventh Conference on Innovative Applications of Artificial Intelligence and Fifteenth Symposium on Educational Advances in Artificial Intelligence (AAAI'25\/IAAI'25\/EAAI'25). AAAI Press Article 2519 9 pages. 10.1609\/aaai.v39i21.34417","DOI":"10.1609\/aaai.v39i21.34417"},{"key":"e_1_3_2_1_37_1","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng Lianmin","year":"2022","unstructured":"Lianmin Zheng, Zhuohan Li, Hao Zhang, Yonghao Zhuang, Zhifeng Chen, Yanping Huang, Yida Wang, Yuanzhong Xu, Danyang Zhuo, Eric P Xing, et al. 2022. Alpa: Automating inter-and {Intra-Operator} parallelism for distributed deep learning. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22). 559\u2013578."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of Machine Learning and Systems 5","author":"Zhuang Yonghao","year":"2023","unstructured":"Yonghao Zhuang, Lianmin Zheng, Zhuohan Li, Eric Xing, Qirong Ho, Joseph Gonzalez, Ion Stoica, Hao Zhang, and Hexu Zhao. 2023. On optimizing the communication of model parallelism. Proceedings of Machine Learning and Systems 5 (2023)."}],"event":{"name":"EUROSYS '26: 21st European Conference on Computer Systems","location":"McEwan Hall\/The University of Edinburgh Edinburgh Scotland UK","acronym":"EUROSYS '26","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 21st European Conference on Computer Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3767295.3803603","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T11:59:16Z","timestamp":1780660756000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3767295.3803603"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,26]]},"references-count":38,"alternative-id":["10.1145\/3767295.3803603","10.1145\/3767295"],"URL":"https:\/\/doi.org\/10.1145\/3767295.3803603","relation":{},"subject":[],"published":{"date-parts":[[2026,4,26]]},"assertion":[{"value":"2026-04-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}