{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T09:02:22Z","timestamp":1784624542404,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T00:00:00Z","timestamp":1785888000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"the National Key Research and Development Program","award":["No. 2024YFB4505604"],"award-info":[{"award-number":["No. 2024YFB4505604"]}]},{"name":"the National Natural Science Foundation of China","award":["No. 62402025"],"award-info":[{"award-number":["No. 62402025"]}]},{"name":"Huawei-BUAA Joint Lab","award":["TC20220105488-2025-03"],"award-info":[{"award-number":["TC20220105488-2025-03"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,6]]},"DOI":"10.1145\/3820441.3820473","type":"proceedings-article","created":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T08:11:31Z","timestamp":1784621491000},"page":"217-224","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["HyLink: Harnessing PCIe and Dedicated Interconnects for Efficient Collective Communication"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-5443-2066","authenticated-orcid":false,"given":"Yuezheng","family":"Liu","sequence":"first","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5274-5512","authenticated-orcid":false,"given":"Menghao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2773-3742","authenticated-orcid":false,"given":"Xuebin","family":"Song","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-9055-2798","authenticated-orcid":false,"given":"Juner","family":"Shen","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3473-9703","authenticated-orcid":false,"given":"Chunming","family":"Hu","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8865-3055","authenticated-orcid":false,"given":"Xudong","family":"Liu","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,8,5]]},"reference":[{"key":"e_1_3_3_1_2_2","volume-title":"ROCm Collective Communication Library (RCCL)","year":"2018","unstructured":"AMD. 2018. ROCm Collective Communication Library (RCCL). https:\/\/github.com\/ROCm\/rccl"},{"key":"e_1_3_3_1_3_2","volume-title":"4th Gen AMD EPYC Processor Architecture","year":"2023","unstructured":"AMD. 2023. 4th Gen AMD EPYC Processor Architecture. White Paper. AMD. https:\/\/www.amd.com\/content\/dam\/amd\/en\/documents\/products\/epyc\/4th-gen-epyc-processor-architecture-white-paper.pdf Accessed: 2026-03-08."},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441620"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3718958.3750499"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1145\/1122971.1122975"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651379"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3669940.3707223"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1147\/JRD.2019.2947013"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575724"},{"key":"e_1_3_3_1_11_2","volume-title":"Huawei Collective Communication Library (HCCL)","year":"2024","unstructured":"Huawei. 2024. Huawei Collective Communication Library (HCCL). https:\/\/gitee.com\/ascend\/cann-hccl"},{"key":"e_1_3_3_1_12_2","volume-title":"Atlas 800 Training Server Maintenance and Service Guide (Model 9000, Air Cooling) 19 - Technical Specifications","author":"Ltd. Huawei Technologies Co.,","year":"2024","unstructured":"Huawei Technologies Co., Ltd.2024. Atlas 800 Training Server Maintenance and Service Guide (Model 9000, Air Cooling) 19 - Technical Specifications. Huawei Enterprise Support. https:\/\/support.huawei.com\/enterprise\/en\/doc\/EDOC1100141951\/43611d26\/technical-specifications Accessed: 2026-03-02."},{"key":"e_1_3_3_1_13_2","volume-title":"Source Code for HyLink","author":"authors HyLink","year":"2026","unstructured":"HyLink authors. 2026. Source Code for HyLink. https:\/\/github.com\/Networked-System-and-Security-Group\/HyLink"},{"key":"e_1_3_3_1_14_2","volume-title":"Intel\u00ae Xeon\u00ae Processor Scalable Family Technical Overview","author":"Corporation Intel","year":"2022","unstructured":"Intel Corporation. 2022. Intel\u00ae Xeon\u00ae Processor Scalable Family Technical Overview. Technical Overview 673025. Intel Corporation. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/articles\/technical\/xeon-processor-scalable-family-technical-overview.html Accessed: 2024-05-20."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507778"},{"key":"e_1_3_3_1_16_2","volume-title":"Massively Scale Your Deep Learning Training with NCCL 2.4","author":"Jeaugey Sylvain","unstructured":"Sylvain Jeaugey. [n. d.]. Massively Scale Your Deep Learning Training with NCCL 2.4. NVIDIA Corporation. https:\/\/developer.nvidia.com\/blog\/massively-scale-deep-learning-training-nccl-2-4\/"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651362"},{"key":"e_1_3_3_1_18_2","unstructured":"Jayacharan Kolla Pedram Alizadeh and Gilbert Lee. 2025. Understanding RCCL Bandwidth and xGMI Performance on AMD Instinct MI300X. AMD ROCm Blogs. https:\/\/rocm.blogs.amd.com\/software-tools-optimization\/mi300x-rccl-xgmi\/README.html Accessed: 2026-02-25."},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00071"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00114"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3718958.3750514"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672249"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_3_1_24_2","volume-title":"NVIDIA Collective Communication Library (NCCL)","year":"2015","unstructured":"NVIDIA. 2015. NVIDIA Collective Communication Library (NCCL). https:\/\/github.com\/NVIDIA\/nccl Accessed: 2024-05-21."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731025"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3527382"},{"key":"e_1_3_3_1_27_2","series-title":"(OSDI \u201925)","volume-title":"Proceedings of the 19th USENIX Conference on Operating Systems Design and Implementation","author":"Ren Zhenghang","year":"2025","unstructured":"Zhenghang Ren, Yuxuan Li, Zilong Wang, Xinyang Huang, Wenxue Li, Kaiqiang Xu, Xudong Liao, Yijun Sun, Bowen Liu, Han Tian, Junxue Zhang, Mingfei Wang, Zhizhen Zhong, Guyue Liu, Ying Zhang, and Kai Chen. 2025. Enabling efficient GPU communication over multiple NICs with FuseLink. In Proceedings of the 19th USENIX Conference on Operating Systems Design and Implementation (Boston, MA, USA) (OSDI \u201925). USENIX Association, USA, Article 6, 18\u00a0pages. https:\/\/dl.acm.org\/doi\/10.5555\/3767901.3767907"},{"key":"e_1_3_3_1_28_2","first-page":"1445","volume-title":"21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Sensi Daniele\u00a0De","year":"2024","unstructured":"Daniele\u00a0De Sensi, Tommaso Bonato, David Saam, and Torsten Hoefler. 2024. Swing: Short-cutting Rings for Higher Bandwidth Allreduce. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24). USENIX Association, Santa Clara, CA, 1445\u20131462. https:\/\/www.usenix.org\/conference\/nsdi24\/presentation\/de-sensi"},{"key":"e_1_3_3_1_29_2","first-page":"593","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Shah Aashaka","year":"2023","unstructured":"Aashaka Shah, Vijay Chidambaram, Meghan Cowan, Saeed Maleki, Madan Musuvathi, Todd Mytkowicz, Jacob Nelson, Olli Saarikivi, and Rachee Singh. 2023. TACCL: Guiding Collective Algorithm Synthesis using Communication Sketches. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). USENIX Association, Boston, MA, 593\u2013612. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/shah"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","unstructured":"Debendra\u00a0Das Sharma. 2024. PCI-Express: Evolution of a Ubiquitous Load-Store Interconnect Over Two Decades and the Path Forward for the Next Two Decades. IEEE Circuits and Systems Magazine 24 2 (2024) 47\u201361. 10.1109\/MCAS.2024.3373556","DOI":"10.1109\/MCAS.2024.3373556"},{"key":"e_1_3_3_1_31_2","first-page":"172","volume-title":"Proceedings of Machine Learning and Systems","volume":"2","author":"Wang Guanhua","year":"2020","unstructured":"Guanhua Wang, Shivaram Venkataraman, Amar Phanishayee, Nikhil Devanur, Jorgen Thelin, and Ion Stoica. 2020. Blink: Fast and Generic Collectives for Distributed ML. In Proceedings of Machine Learning and Systems , I.\u00a0Dhillon, D.\u00a0Papailiopoulos, and V.\u00a0Sze (Eds.), Vol.\u00a02. 172\u2013186. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2020\/file\/cd3a9a55f7f3723133fa4a13628cdf03-Paper.pdf"},{"key":"e_1_3_3_1_32_2","unstructured":"Guanhua Wang Chengming Zhang Zheyu Shen Ang Li and Olatunji Ruwase. 2024. Domino: Eliminating Communication in LLM Training via Generic Tensor Slicing and Overlapping. arxiv:https:\/\/arXiv.org\/abs\/2409.15241\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2409.15241"},{"key":"e_1_3_3_1_33_2","first-page":"739","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Wang Weiyang","year":"2023","unstructured":"Weiyang Wang, Moein Khazraee, Zhizhen Zhong, Manya Ghobadi, Zhihao Jia, Dheevatsa Mudigere, Ying Zhang, and Anthony Kewitsch. 2023. TopoOpt: Co-optimizing Network Topology and Parallelization Strategy for Distributed Training Jobs. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). USENIX Association, Boston, MA, 739\u2013767. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/wang-weiyang"},{"key":"e_1_3_3_1_34_2","first-page":"945","volume-title":"19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22)","author":"Weng Qizhen","year":"2022","unstructured":"Qizhen Weng, Wencong Xiao, Yinghao Yu, Wei Wang, Cheng Wang, Jian He, Yong Li, Liping Zhang, Wei Lin, and Yu Ding. 2022. MLaaS in the Wild: Workload Analysis and Scheduling in Large-Scale Heterogeneous GPU Clusters. In 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22). USENIX Association, Renton, WA, 945\u2013960. https:\/\/www.usenix.org\/conference\/nsdi22\/presentation\/weng"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00068"},{"key":"e_1_3_3_1_36_2","first-page":"667","volume-title":"22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25)","author":"Xu Guanbin","year":"2025","unstructured":"Guanbin Xu, Zhihao Le, Yinhe Chen, Zhiqi Lin, Zewen Jin, Youshan Miao, and Cheng Li. 2025. AutoCCL: Automated Collective Communication Tuning for Accelerating Distributed and Parallel DNN Training. In 22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25). USENIX Association, Philadelphia, PA, 667\u2013683. https:\/\/www.usenix.org\/conference\/nsdi25\/presentation\/xu-guanbin"},{"key":"e_1_3_3_1_37_2","unstructured":"Wayne\u00a0Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong Yifan Du Chen Yang Yushuo Chen Zhipeng Chen Jinhao Jiang Ruiyang Ren Yifan Li Xinyu Tang Zikang Liu Peiyu Liu Jian-Yun Nie and Ji-Rong Wen. 2025. A Survey of Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2303.18223\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2303.18223"},{"key":"e_1_3_3_1_38_2","unstructured":"Yang Zhou Zhongjie Chen Ziming Mao ChonLam Lao Shuo Yang Pravein\u00a0Govindan Kannan Jiaqi Gao Yilong Zhao Yongji Wu Kaichao You Fengyuan Ren Zhiying Xu Costin Raiciu and Ion Stoica. 2025. An Extensible Software Transport Layer for GPU Networking. arxiv:https:\/\/arXiv.org\/abs\/2504.17307\u00a0[cs.NI] https:\/\/arxiv.org\/abs\/2504.17307"}],"event":{"name":"APNet 2026: The 10th Asia-Pacific Workshop on Networking","location":"Singapore Singapore","acronym":"APNet '26"},"container-title":["Proceedings of the 10th Asia-Pacific Workshop on Networking"],"original-title":[],"deposited":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T08:15:57Z","timestamp":1784621757000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3820441.3820473"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,5]]},"references-count":37,"alternative-id":["10.1145\/3820441.3820473","10.1145\/3820441"],"URL":"https:\/\/doi.org\/10.1145\/3820441.3820473","relation":{},"subject":[],"published":{"date-parts":[[2026,8,5]]},"assertion":[{"value":"2026-08-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}