{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,11]],"date-time":"2026-08-11T19:28:16Z","timestamp":1786476496727,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":15,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,8,11]],"date-time":"2026-08-11T00:00:00Z","timestamp":1786406400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Natural Science Foundation of China","award":["62402025"],"award-info":[{"award-number":["62402025"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,17]]},"DOI":"10.1145\/3789240.3830303","type":"proceedings-article","created":{"date-parts":[[2026,8,11]],"date-time":"2026-08-11T18:33:22Z","timestamp":1786473202000},"page":"2265-2267","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Weaver: Diagnosing Extra Kernel and Synchronization Interference in GPU Workloads"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-2027-2903","authenticated-orcid":false,"given":"Jingyuan","family":"Zhu","sequence":"first","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5274-5512","authenticated-orcid":false,"given":"Menghao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8635-5438","authenticated-orcid":false,"given":"Yuxuan","family":"Chen","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,8,11]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Flare: Anomaly Diagnostics for Divergent LLM Training in GPU Clusters of Thousand-Plus Scale.","author":"Cui Wei","year":"2025","unstructured":"Wei Cui, Ji Zhang, Han Zhao, Chao Liu, Jian Sha, Bingsheng He, Minyi Guo, and Quan Chen. 2025. Flare: Anomaly Diagnostics for Divergent LLM Training in GPU Clusters of Thousand-Plus Scale."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"S\u00e9bastien Darche and Michel Dagenais. 2024. Low-Overhead Trace Collection and Profiling on GPU Compute Kernels. ACM Trans. Parallel Comput.","DOI":"10.1145\/3649510"},{"key":"e_1_3_2_1_3_1","volume-title":"Minlan Yu, and Hong Xu.","author":"Deng Yangtao","year":"2025","unstructured":"Yangtao Deng, Lei Zhang, Qinlong Wang, Xiaoyun Zhi, Xinlei Zhang, Zhuo Jiang, Haohan Xu, Lei Wang, Zuquan Song, Gaohong Liu, Yangyang Bai, Shuguang Wang, Wencong Xiao, Jia jun Ye, Minlan Yu, and Hong Xu. 2025. Mycroft: Tracing Dependencies in Collective Communication Towards Reliable LLM Training. In ACM SOSP 2025."},{"key":"e_1_3_2_1_4_1","volume-title":"Xin Jin, and Xin Liu.","author":"Jiang Ziheng","year":"2024","unstructured":"Ziheng Jiang, Haibin Lin, Yinmin Zhong, Qi Huang, Yangrui Chen, Zhi Zhang, Yanghua Peng, Xiang Li, Cong Xie, Shibiao Nong, Yulu Jia, Sun He, Hongmin Chen, Zhihao Bai, Qi Hou, Shipeng Yan, Ding Zhou, Yiyao Sheng, Zhuo Jiang, Haohan Xu, Haoran Wei, Zhang Zhang, Pengfei Nie, Leqi Zou, Sida Zhao, Liang Xiang, Zherui Liu, Zhe Li, Xiaoying Jia, Jia jun Ye, Xin Jin, and Xin Liu. 2024. MegaScale: Scaling Large Language Model Training to More Than 10,000 GPUs. In USENIX NSDI 2024."},{"key":"e_1_3_2_1_5_1","volume-title":"Characterizing Compute-Communication Overlap in GPU-Accelerated Distributed Deep Learning: Performance and Power Implications. In IEEE ISPASS","author":"Lee Seonho","year":"2025","unstructured":"Seonho Lee, Jihwan Oh, Junkyum Kim, Seokjin Go, Jongse Park, and Divya Mahajan. 2025. Characterizing Compute-Communication Overlap in GPU-Accelerated Distributed Deep Learning: Performance and Power Implications. In IEEE ISPASS 2025."},{"key":"e_1_3_2_1_6_1","unstructured":"NVIDIA. 2025. Nsight Systems. https:\/\/developer.nvidia.com\/nsightsystems."},{"key":"e_1_3_2_1_7_1","volume-title":"Optimizing Distributed ML Communication with Fused Computation-Collective Operations. In SC","author":"Punniyamurthy Kishore","year":"2024","unstructured":"Kishore Punniyamurthy, Khaled Hamidouche, and Bradford M. Beckmann. 2023. Optimizing Distributed ML Communication with Fused Computation-Collective Operations. In SC 2024."},{"key":"e_1_3_2_1_8_1","unstructured":"PyTorch. 2025. PyTorch Profiler. https:\/\/pytorch.org\/docs\/stable\/profiler.html."},{"key":"e_1_3_2_1_9_1","volume-title":"Enabling Compute-Communication Overlap in Distributed Deep Learning Training Platforms. In ACM\/IEEE ISCA","author":"Rashidi Saeed","year":"2020","unstructured":"Saeed Rashidi, Matthew Denton, Srinivas Sridharan, Sudarshan M. Srinivasan, Amoghavarsha Suresh, Jade Nie, and Tushar Krishna. 2020. Enabling Compute-Communication Overlap in Distributed Deep Learning Training Platforms. In ACM\/IEEE ISCA 2021."},{"key":"e_1_3_2_1_10_1","unstructured":"WEAVER. 2026. GPU Kernel Diagnosis. https:\/\/github.com\/Networked-System-and-Security-Group\/Weaver.git."},{"key":"e_1_3_2_1_11_1","volume-title":"Patterson","author":"Williams Samuel","year":"2009","unstructured":"Samuel Williams, Andrew Waterman, and David A. Patterson. 2009. Roofline: an insightful visual performance model for multicore architectures. Commun. ACM."},{"key":"e_1_3_2_1_12_1","volume-title":"Transparent GPU Sharing in Container Clouds for Deep Learning Workloads. In USENIX NSDI","author":"Wu Bingyang","year":"2023","unstructured":"Bingyang Wu, Zili Zhang, Zhihao Bai, Xuanzhe Liu, and Xin Jin. 2023. Transparent GPU Sharing in Container Clouds for Deep Learning Workloads. In USENIX NSDI 2023."},{"key":"e_1_3_2_1_13_1","volume-title":"USENIX ATC","author":"Wu Tianyuan","year":"2025","unstructured":"Tianyuan Wu, Wei Wang, Yinghao Yu, Siran Yang, Wenchao Wu, Qinkai Duan, Guodong Yang, Jiamang Wang, Lin Qu, and Liping Zhang. 2025. GREYHOUND: Hunting Fail-Slows in Hybrid-Parallel Training at Scale. In USENIX ATC 2025."},{"key":"e_1_3_2_1_14_1","volume-title":"Holmes: Localizing Irregularities in LLM Training with Mega-scale GPU Clusters. In USENIX NSDI","author":"Yao Zhiyi","year":"2025","unstructured":"Zhiyi Yao, Pengbo Hu, Congcong Miao, Xuya Jia, Zuning Liang, Yuedong Xu, Chunzhi He, Hao Lu, Mingzhuo Chen, Xiang Li, Zekun He, Yachen Wang, Xianneng Zou, and Junchen Jiang. 2025. Holmes: Localizing Irregularities in LLM Training with Mega-scale GPU Clusters. In USENIX NSDI 2025."},{"key":"e_1_3_2_1_15_1","volume-title":"Orca: A Distributed Serving System for Transformer-Based Generative Models. In USENIX Symposium on Operating Systems Design and Implementation.","author":"Yu Gyeong-In","year":"2022","unstructured":"Gyeong-In Yu and Joo Seong Jeong. 2022. Orca: A Distributed Serving System for Transformer-Based Generative Models. In USENIX Symposium on Operating Systems Design and Implementation."}],"event":{"name":"SIGCOMM '26: ACM SIGCOMM 2026 Conference","location":"Colorado Convention Center Denver CO USA","acronym":"SIGCOMM '26","sponsor":["SIGCOMM ACM Special Interest Group on Data Communication"]},"container-title":["Proceedings of the ACM SIGCOMM 2026 Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3789240.3830303","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,11]],"date-time":"2026-08-11T18:46:57Z","timestamp":1786474017000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3789240.3830303"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,11]]},"references-count":15,"alternative-id":["10.1145\/3789240.3830303","10.1145\/3789240"],"URL":"https:\/\/doi.org\/10.1145\/3789240.3830303","relation":{},"subject":[],"published":{"date-parts":[[2026,8,11]]},"assertion":[{"value":"2026-08-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}