{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T05:45:06Z","timestamp":1782798306434,"version":"3.54.5"},"reference-count":41,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,18]],"date-time":"2026-05-18T00:00:00Z","timestamp":1779062400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,18]],"date-time":"2026-05-18T00:00:00Z","timestamp":1779062400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,18]]},"DOI":"10.1109\/infocom59046.2026.11571341","type":"proceedings-article","created":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:38:15Z","timestamp":1782761895000},"page":"1-10","source":"Crossref","is-referenced-by-count":0,"title":["Compass: Dissecting Communication and Computation Operators for Efficient LLM Training"],"prefix":"10.1109","author":[{"given":"Guangyu","family":"Xiang","sequence":"first","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou),Data Science and Analytics Thrust"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lin","family":"Zhang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology,Department of Computer Science and Engineering"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haoxuan","family":"Yu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology,School of Computer Science and Technology,Shenzhen"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinglin","family":"Pan","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou),Data Science and Analytics Thrust"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shaohuai","family":"Shi","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology,School of Computer Science and Technology,Shenzhen"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaowen","family":"Chu","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou),Data Science and Analytics Thrust"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"issue":"240","key":"ref2","first-page":"1","article-title":"Palm: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2023","journal-title":"Journal of Machine Learning Research"},{"key":"ref3","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref4","first-page":"1","article-title":"Efficient large-scale language model training on gpu clusters using megatron-lm","volume-title":"Proceedings of the international conference for high performance computing, networking, storage and analysis","author":"Narayanan"},{"key":"ref5","article-title":"An analysis of deep neural network models for practical applications","author":"Canziani","year":"2016"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.14778\/3415478.3415530"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.48550\/arxiv.1811.06965"},{"key":"ref8","article-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism","author":"Shoeybi","year":"2019"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/iiswc59245.2023.00026"},{"key":"ref10","article-title":"Flux: Fast software-based communication overlap on gpus through kernel fusion","author":"Chang","year":"2024"},{"key":"ref11","article-title":"Comet: Fine-grained computationcommunication overlapping for mixture-of-experts","author":"Zhang","year":"2025"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507778"},{"key":"ref13","first-page":"745","article-title":"Megascale: Scaling large language model training to more than 10,000 gpus","volume-title":"21st USENIX Symposium on Networked Systems Design and Implementation","author":"Jiang"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3567955.3567959"},{"key":"ref15","first-page":"341","article-title":"Reducing activation recomputation in large transformer models","volume-title":"Proceedings of Machine Learning and Systems","volume":"5","author":"Korthikanti"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.011.2000530"},{"key":"ref17","first-page":"401","article-title":"Towards scalable distributed training of deep learning on public cloud clusters","volume-title":"Proceedings of Machine Learning and Systems","volume":"3","author":"Shi"},{"key":"ref18","first-page":"48","article-title":"Breadth-first pipeline parallelism","volume-title":"Proceedings of Machine Learning and Systems","volume":"5","author":"Lamy-Poirier"},{"key":"ref19","first-page":"593","article-title":"Taccl: Guiding collective algorithm synthesis using communication sketches","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation","author":"Shah"},{"key":"ref20","article-title":"Pipedream: Fast and efficient pipeline parallel dnn training","author":"Harlap","year":"2018"},{"key":"ref21","article-title":"Nvidia collective communications library (nccl)","year":"2025"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3582016.3582065"},{"key":"ref23","article-title":"NVIDIA RTX A6000","year":"2025"},{"key":"ref24","article-title":"Atlas 900 A3 SuperPoD","year":"2025"},{"key":"ref25","first-page":"31","article-title":"GPUDirect RDMA: A case for efficient and scalable non-contiguous data transfers","volume-title":"2019 IEEE International Parallel and Distributed Processing Symposium","author":"Jena"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441620"},{"key":"ref27","article-title":"CUTLASS: CUDA Templates for Linear Algebra Subroutines","author":"Corporation","year":"2025"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM53939.2023.10228874"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508418"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.3354\/cr030079"},{"key":"ref31","article-title":"Megatron-LM","author":"Corporation","year":"2025"},{"key":"ref32","article-title":"Training GPT with Predefined Configurations","author":"Corporation","year":"2025"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.09.002"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ExaMPI49596.2019.00008"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM42981.2021.9488803"},{"key":"ref37","article-title":"Horovod: fast and easy distributed deep learning in TensorFlow","author":"Sergeev","year":"2018"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737367"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3052862"},{"key":"ref40","article-title":"Deepseek-v3 technical report","author":"Liu","year":"2025"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651379"}],"event":{"name":"IEEE INFOCOM 2026 - IEEE Conference on Computer Communications","location":"Tokyo, Japan","start":{"date-parts":[[2026,5,18]]},"end":{"date-parts":[[2026,5,21]]}},"container-title":["IEEE INFOCOM 2026 - IEEE Conference on Computer Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11571071\/11571169\/11571341.pdf?arnumber=11571341","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T05:19:35Z","timestamp":1782796775000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11571341\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,18]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/infocom59046.2026.11571341","relation":{},"subject":[],"published":{"date-parts":[[2026,5,18]]}}}