{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T05:44:55Z","timestamp":1782798295292,"version":"3.54.5"},"reference-count":47,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,18]],"date-time":"2026-05-18T00:00:00Z","timestamp":1779062400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,18]],"date-time":"2026-05-18T00:00:00Z","timestamp":1779062400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100003012","name":"Impact Fund","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100003012","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,18]]},"DOI":"10.1109\/infocom59046.2026.11571522","type":"proceedings-article","created":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:38:15Z","timestamp":1782761895000},"page":"1-10","source":"Crossref","is-referenced-by-count":0,"title":["Director: Accelerating Distributed MoE Serving via Online Proactive Expert Placement"],"prefix":"10.1109","author":[{"given":"Qianli","family":"Liu","sequence":"first","affiliation":[{"name":"The Hong Kong University of Science and Technology,Department of Computer Science and Engineering,Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kaibin","family":"Guo","sequence":"additional","affiliation":[{"name":"Sun Yat-Sen University,School of Software Engineering,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zicong","family":"Hong","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology,Department of Computer Science and Engineering,Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Li","sequence":"additional","affiliation":[{"name":"Xi&#x2019;an Jiaotong University,School of Cyber Science and Engineering,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fahao","family":"Chen","sequence":"additional","affiliation":[{"name":"Shandong University,School of Artificial Intelligence,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haodong","family":"Wang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology,Department of Computer Science and Engineering,Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jian","family":"Lin","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology,Department of Computer Science and Engineering,Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Song","family":"Guo","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology,Department of Computer Science and Engineering,Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"The llama 4 herd: Native multimodality with a mixture-of-experts architecture","year":"2025"},{"key":"ref2","article-title":"Qwen technical report","author":"Bai","year":"2023"},{"key":"ref3","article-title":"Qwen2 technical report","year":"2024"},{"key":"ref4","article-title":"Qwen3 technical report","author":"Yang","year":"2025"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.70"},{"key":"ref6","article-title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model","author":"Liu","year":"2024"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-025-09422-z"},{"key":"ref8","article-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer","author":"Shazeer","year":"2017"},{"key":"ref9","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","author":"Fedus","year":"2022","journal-title":"Journal of Machine Learning Research"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/OJCS.2024.3380828"},{"key":"ref11","article-title":"Gshard: Scaling giant models with conditional computation and automatic sharding","author":"Lepikhin","year":"2020"},{"key":"ref12","article-title":"Expert Parallelism Load Balancer","year":"2025"},{"key":"ref13","article-title":"Moetuner: Optimized mixture of expert serving with balanced expert placement and token routing","author":"Go","year":"2025"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508418"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3588964"},{"key":"ref16","article-title":"Tutel: Adaptive mixture-of-experts at scale","author":"Hwang","year":"2023","journal-title":"MLSys"},{"key":"ref17","article-title":"Smartmoe: Efficiently training sparsely-activated models through combining offline and online parallelization","volume-title":"ATC","author":"Zhai"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3603269.3604869"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1287\/opre.45.6.831"},{"key":"ref20","article-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer","volume-title":"ICLR","author":"Shazeer"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.334"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.334"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1611"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3650083"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3669940.3707272"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM53939.2023.10228874"},{"key":"ref27","article-title":"Klotski: Efficient mixture-of-expert inference via expertaware multi-batch pipeline","volume-title":"ASPLOS","author":"Fang"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM52122.2024.10621327"},{"key":"ref29","article-title":"Expertflow: Optimized expert activation and token allocation for efficient mixture-of-experts inference","author":"He","year":"2024"},{"key":"ref30","article-title":"Netmoe: Accelerating moe training through dynamic sample placement","volume-title":"ICLR","author":"Liu"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TON.2025.3585359"},{"key":"ref32","article-title":"{PopFetcher}: Towards accelerated {Mixture-of-Experts} training via popularity based {Expert-Wise} prefetch","volume-title":"ATC","author":"Zhang"},{"key":"ref33","article-title":"Pointer sentinel mixture models","author":"Merity","year":"2016"},{"key":"ref34","article-title":"Let\u2019s verify step by step","volume-title":"ICLR","author":"Lightman"},{"key":"ref35","article-title":"Livecodebench: Holistic and contamination free evaluation of large language models for code","author":"Jain","year":"2024"},{"key":"ref36","article-title":"Sida: Sparsity-inspired data-aware serving for efficient and scalable large mixture-of-experts models","author":"Du","year":"2024"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/3680207.3723493"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1214\/aop\/1176996452"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-33461-5_31"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/FOCS.2010.7"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1137\/130929400"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511977152"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3605573.3605613"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1038\/s41592-019-0686-2"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1145\/3714983.3714987"},{"key":"ref46","article-title":"Mixtral of experts","author":"Jiang","year":"2024"},{"key":"ref47","article-title":"Deepspeed-moe: Advancing mixture-of-experts inference and training to power next-generation ai scale","volume-title":"ICML","author":"Rajbhandari"}],"event":{"name":"IEEE INFOCOM 2026 - IEEE Conference on Computer Communications","location":"Tokyo, Japan","start":{"date-parts":[[2026,5,18]]},"end":{"date-parts":[[2026,5,21]]}},"container-title":["IEEE INFOCOM 2026 - IEEE Conference on Computer Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11571071\/11571169\/11571522.pdf?arnumber=11571522","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T05:15:52Z","timestamp":1782796552000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11571522\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,18]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/infocom59046.2026.11571522","relation":{},"subject":[],"published":{"date-parts":[[2026,5,18]]}}}