{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T07:10:00Z","timestamp":1784099400278,"version":"3.55.0"},"reference-count":20,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,24]],"date-time":"2026-05-24T00:00:00Z","timestamp":1779580800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,24]],"date-time":"2026-05-24T00:00:00Z","timestamp":1779580800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,24]]},"DOI":"10.1109\/icc59461.2026.11588168","type":"proceedings-article","created":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T19:38:09Z","timestamp":1784057889000},"page":"1-6","source":"Crossref","is-referenced-by-count":0,"title":["Communication Bottleneck Analysis for Distributed MoE Training"],"prefix":"10.1109","author":[{"given":"Li","family":"He","sequence":"first","affiliation":[{"name":"University of Victoria,Department of Computer Science,Victoria,Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenjun","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Victoria,Department of Computer Science,Victoria,Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianping","family":"Pan","sequence":"additional","affiliation":[{"name":"University of Victoria,Department of Computer Science,Victoria,Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"issue":"120","key":"ref1","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus","year":"2022","journal-title":"Journal of Machine Learning Research"},{"key":"ref2","first-page":"5547","article-title":"GLaM: Efficient scaling of language models with mixture-of-experts","volume-title":"International Conference on Machine Learning","author":"Du"},{"key":"ref3","author":"Jiang","year":"2024","journal-title":"Mixtral of experts"},{"key":"ref4","author":"Lepikhin","year":"2020","journal-title":"GShard: Scaling giant models with conditional computation and automatic sharding"},{"key":"ref5","first-page":"18332","article-title":"DeepSpeed-MoE: Advancing mixture-of-experts inference and training to power next-generation AI scale","volume-title":"International Conference on Machine Learning","author":"Rajbhandari"},{"key":"ref6","first-page":"288","article-title":"Megablocks: Efficient sparse training with mixture-of-experts","volume-title":"Proceedings of Machine Learning and Systems","volume":"5","author":"Gale"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1991.3.1.79"},{"key":"ref8","author":"Shazeer","year":"2017","journal-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer"},{"key":"ref9","author":"Shoeybi","year":"2019","journal-title":"Megatron-LM: Training multi-billion parameter language models using model parallelism"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"ref11","first-page":"1","article-title":"Efficient large-scale language model training on GPU clusters using megatron-LM","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Narayanan"},{"key":"ref12","year":"2024","journal-title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1177\/1094342005051521"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.09.002"},{"key":"ref15","author":"Ma","year":"2025","journal-title":"MoE-GPS: Guidlines for prediction strategy for dynamic expert duplication in MoE load balancing"},{"key":"ref16","author":"Doucet","year":"2025","journal-title":"HarMoEny: Efficient multi-GPU inference of MoE models"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"ref18","first-page":"269","article-title":"Tutel: Adaptive mixture-of-experts at scale","volume-title":"Proceedings of Machine Learning and Systems","volume":"5","author":"Hwang"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM53939.2023.10228874"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM55648.2025.11044508"}],"event":{"name":"ICC 2026 - IEEE International Conference on Communications","location":"Glasgow, United Kingdom","start":{"date-parts":[[2026,5,24]]},"end":{"date-parts":[[2026,5,28]]}},"container-title":["ICC 2026 - IEEE International Conference on Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11586754\/11586037\/11588168.pdf?arnumber=11588168","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T06:42:00Z","timestamp":1784097720000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11588168\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,24]]},"references-count":20,"URL":"https:\/\/doi.org\/10.1109\/icc59461.2026.11588168","relation":{},"subject":[],"published":{"date-parts":[[2026,5,24]]}}}