{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T09:02:09Z","timestamp":1784624529369,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":15,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T00:00:00Z","timestamp":1785888000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100019491","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2024YFB2907000"],"award-info":[{"award-number":["2024YFB2907000"]}],"id":[{"id":"10.13039\/501100019491","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,6]]},"DOI":"10.1145\/3820441.3820458","type":"proceedings-article","created":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T08:11:31Z","timestamp":1784621491000},"page":"115-121","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Semantic-Aware Loss Recovery for Cross-Datacenter Model Training"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-4742-2629","authenticated-orcid":false,"given":"Xingjian","family":"Zhang","sequence":"first","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6029-3589","authenticated-orcid":false,"given":"Yutong","family":"Zhao","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2940-622X","authenticated-orcid":false,"given":"Jiaxue","family":"Liu","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6826-4596","authenticated-orcid":false,"given":"Lizhuang","family":"Tan","sequence":"additional","affiliation":[{"name":"Shandong Computer Science Center (National Supercomputer Center in Ji'nan), Jinan, Shandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7245-1298","authenticated-orcid":false,"given":"Shangguang","family":"Wang","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3555-0155","authenticated-orcid":false,"given":"Yiran","family":"Zhang","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,8,5]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Tom\u00a0B. Brown Benjamin Mann Nick Ryder et\u00a0al. 2020. Language Models are Few-Shot Learners. arxiv:https:\/\/arXiv.org\/abs\/2005.14165\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2005.14165"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3696348.3696880"},{"key":"e_1_3_3_1_4_2","unstructured":"Jacob Devlin Ming-Wei Chang Kenton Lee and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arxiv:https:\/\/arXiv.org\/abs\/1810.04805\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/1810.04805"},{"key":"e_1_3_3_1_5_2","unstructured":"Arthur Douillard Qixuan Feng Andrei\u00a0A. Rusu Rachita Chhaparia Yani Donchev Adhiguna Kuncoro Marc\u2019Aurelio Ranzato Arthur Szlam and Jiajun Shen. 2024. DiLoCo: Distributed Low-Communication Training of Language Models. arxiv:https:\/\/arXiv.org\/abs\/2311.08105\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2311.08105"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long Short-Term Memory. Neural Comput. 9 8 (Nov. 1997) 1735\u20131780. 10.1162\/neco.1997.9.8.1735","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/3712285.3759902"},{"key":"e_1_3_3_1_8_2","unstructured":"Yujun Lin Song Han Huizi Mao Yu Wang and William\u00a0J. Dally. 2020. Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training. arxiv:https:\/\/arXiv.org\/abs\/1712.01887\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1712.01887"},{"key":"e_1_3_3_1_9_2","unstructured":"Stephen Merity Caiming Xiong James Bradbury and Richard Socher. 2016. Pointer Sentinel Mixture Models. arxiv:https:\/\/arXiv.org\/abs\/1609.07843\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/1609.07843"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3230543.3230557"},{"key":"e_1_3_3_1_11_2","volume-title":"Turbocharge LLM Training Across Long-Haul Data Center Networks with NVIDIA NeMo Framework","year":"2024","unstructured":"NVIDIA. 2024. Turbocharge LLM Training Across Long-Haul Data Center Networks with NVIDIA NeMo Framework. NVIDIA Developer Blog. https:\/\/developer.nvidia.com\/blog\/turbocharge-llm-training-across-long-haul-data-center-networks-with-nvidia-nemo-framework\/ Accessed: 2026-03-11."},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3642970.3655843"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.5555\/3691825.3691904"},{"key":"e_1_3_3_1_14_2","first-page":"541","volume-title":"22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25)","author":"Wang Xizheng","year":"2025","unstructured":"Xizheng Wang, Qingxu Li, Yichi Xu, Gang Lu, Dan Li, Li Chen, Heyang Zhou, Linkang Zheng, Sen Zhang, Yikai Zhu, Yang Liu, Pengcheng Zhang, Kun Qian, Kunling He, Jiaqi Gao, Ennan Zhai, Dennis Cai, and Binzhang Fu. 2025. SimAI: Unifying Architecture Design and Performance Tuning for Large-Scale Large Language Model Training with Scalability and Precision. In 22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25). USENIX Association, Philadelphia, PA, 541\u2013558. https:\/\/www.usenix.org\/conference\/nsdi25\/presentation\/wang-xizheng-simai"},{"key":"e_1_3_3_1_15_2","unstructured":"An Yang Anfeng Li Baosong Yang et\u00a0al. 2025. Qwen3 Technical Report. arxiv:https:\/\/arXiv.org\/abs\/2505.09388\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2505.09388"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/IWQOS61813.2024.10682853"}],"event":{"name":"APNet 2026: The 10th Asia-Pacific Workshop on Networking","location":"Singapore Singapore","acronym":"APNet '26"},"container-title":["Proceedings of the 10th Asia-Pacific Workshop on Networking"],"original-title":[],"deposited":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T08:12:21Z","timestamp":1784621541000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3820441.3820458"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,5]]},"references-count":15,"alternative-id":["10.1145\/3820441.3820458","10.1145\/3820441"],"URL":"https:\/\/doi.org\/10.1145\/3820441.3820458","relation":{},"subject":[],"published":{"date-parts":[[2026,8,5]]},"assertion":[{"value":"2026-08-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}