{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T16:32:57Z","timestamp":1781886777312,"version":"3.54.5"},"reference-count":32,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,22]],"date-time":"2025-06-22T00:00:00Z","timestamp":1750550400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,22]],"date-time":"2025-06-22T00:00:00Z","timestamp":1750550400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100017413","name":"Innovation Fund","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100017413","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,22]]},"DOI":"10.1109\/dac63849.2025.11132853","type":"proceedings-article","created":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T17:35:41Z","timestamp":1757957741000},"page":"1-7","source":"Crossref","is-referenced-by-count":1,"title":["BirdMoE: Reducing Communication Costs for Mixture-of-Experts Training Using Load-Aware Bi-random Quantization"],"prefix":"10.1109","author":[{"given":"Donglei","family":"Wu","sequence":"first","affiliation":[{"name":"Guangzhou University,Cyberspace Institute of Advanced Technology,Guangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weihao","family":"Yang","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology,School of Computer Science and Technology,Shenzhen,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangyu","family":"Zou","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology,School of Computer Science and Technology,Shenzhen,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinda","family":"Jia","sequence":"additional","affiliation":[{"name":"Indiana University,Luddy School of Informatics, Computing, and Engineering,Bloomington,IN,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dingwen","family":"Tao","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences,Beijing,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wen","family":"Xia","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology,School of Computer Science and Technology,Shenzhen,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhihong","family":"Tian","sequence":"additional","affiliation":[{"name":"Guangzhou University,Cyberspace Institute of Advanced Technology,Guangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proc. ICLR\u201921","author":"Dosovitskiy"},{"key":"ref2","first-page":"4171","article-title":"BERT: pre-training of deep bidirectional transformers for language understanding","volume-title":"Proc. NAACL-HLT\u201919","author":"Devlin"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2023.3307459"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/DAC56929.2023.10248003"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i4.20345"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3617688"},{"key":"ref7","article-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer","volume-title":"Proc. ICLR","volume":"2017","author":"Shazeer"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508418"},{"key":"ref9","article-title":"Accelerating distributed moe training and inference with lina","volume-title":"Proc. USENIX ATC 2023","author":"Li"},{"key":"ref10","article-title":"Smartmoe: Efficiently training sparsely-activated models through combining offline and online parallelization","volume-title":"Proc. USENIX ATC 2023","author":"Zhai"},{"key":"ref11","article-title":"Gshard: Scaling giant models with conditional computation and automatic sharding","volume-title":"Proc. ICLR 2021","author":"Lepikhin"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20858"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2024.3447221"},{"key":"ref14","article-title":"QSGD: communication-efficient SGD via gradient quantization and encoding","volume-title":"Proc. NIPS\u201917","author":"Alistarh"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD58817.2023.00096"},{"key":"ref16","article-title":"Deep gradient compression: Reducing the communication bandwidth for distributed training","volume-title":"Proc. ICLR\u201918","author":"Lin"},{"key":"ref17","first-page":"14236","article-title":"Powersgd: Practical low-rank gradient compression for distributed optimization","volume-title":"Proc. NIPS\u201919","author":"Vogels"},{"key":"ref18","article-title":"Unbiased compression saves communication in distributed optimization: When and how much?","volume-title":"Proc. NIPS\u201923","author":"He"},{"key":"ref19","article-title":"EF-BV: A unified theory of error feedback and variance reduction mechanisms for biased and unbiased compression in distributed optimization","volume-title":"Proc. NIPS\u201922","author":"Condat"},{"key":"ref20","first-page":"21 150","article-title":"Deepreduce: A sparsetensor communication framework for federated deep learning","volume-title":"Proc. NIPS\u201921","author":"Xu"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref22","article-title":"Language models are few-shot learners","volume-title":"Proc. NIPS\u201920","author":"Brown"},{"key":"ref23","article-title":"Scaling vision with sparse mixture of experts","volume-title":"Proc. NIPS\u201921","author":"Riquelme"},{"key":"ref24","article-title":"Towards scalable distributed training of deep learning on public cloud clusters","volume-title":"Proc. MLSys\u201921","author":"Shi"},{"key":"ref25","first-page":"10921","article-title":"Topkapi: Parallel and fast sketches for finding top-k frequent elements","volume-title":"Proc. NIPS\u201918","author":"Mandal"},{"key":"ref26","article-title":"Pointer sentinel mixture models","volume-title":"Proc. ICLR\u201917","author":"Merity"},{"key":"ref27","article-title":"Learning multiple layers of features from tiny images","volume-title":"Tech. Rep","author":"Krizhevsky","year":"2009"},{"key":"ref28","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. ICLR\u201915","author":"Kingma"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58536-5_5"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358295"},{"key":"ref32","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume-title":"Proc. NIPS\u201919","author":"Paszke"}],"event":{"name":"2025 62nd ACM\/IEEE Design Automation Conference (DAC)","location":"San Francisco, CA, USA","start":{"date-parts":[[2025,6,22]]},"end":{"date-parts":[[2025,6,25]]}},"container-title":["2025 62nd ACM\/IEEE Design Automation Conference (DAC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11132383\/11132091\/11132853.pdf?arnumber=11132853","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,16]],"date-time":"2025-09-16T05:24:36Z","timestamp":1758000276000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11132853\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,22]]},"references-count":32,"URL":"https:\/\/doi.org\/10.1109\/dac63849.2025.11132853","relation":{},"subject":[],"published":{"date-parts":[[2025,6,22]]}}}