{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T16:55:48Z","timestamp":1782924948868,"version":"3.54.5"},"reference-count":45,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,25]],"date-time":"2026-05-25T00:00:00Z","timestamp":1779667200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,25]],"date-time":"2026-05-25T00:00:00Z","timestamp":1779667200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,25]]},"DOI":"10.1109\/ipdps65963.2026.00069","type":"proceedings-article","created":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T20:55:16Z","timestamp":1782852916000},"page":"760-774","source":"Crossref","is-referenced-by-count":0,"title":["HarMoEny: Efficient Inference of MoE Models"],"prefix":"10.1109","author":[{"given":"Zachary","family":"Doucet","sequence":"first","affiliation":[{"name":"McGill University,Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rishi","family":"Sharma","sequence":"additional","affiliation":[{"name":"EPFL,Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Martijn","family":"de Vos","sequence":"additional","affiliation":[{"name":"EPFL,Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rafael","family":"Pires","sequence":"additional","affiliation":[{"name":"EPFL,Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Anne-Marie","family":"Kermarrec","sequence":"additional","affiliation":[{"name":"EPFL,Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Oana","family":"Balmau","sequence":"additional","affiliation":[{"name":"McGill University,Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3649506"},{"key":"ref3","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref4","article-title":"GPT-4 technical report","author":"Achiam","year":"2023"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/mm.2024.3426478"},{"key":"ref6","article-title":"AWS to offer nvidia\u2019s t4 GPUs for AI inferencing","author":"Leopold","year":"2019"},{"key":"ref7","article-title":"Amazon EC2 update \u2013 inf1 instances with AWS inferentia chips for high performance cost-effective inferencing","author":"Barr","year":"2019"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1991.3.1.79"},{"key":"ref9","article-title":"Gshard: Scaling giant models with conditional computation and automatic sharding","volume-title":"ICLR","author":"Lepikhin","year":"2021"},{"issue":"120","key":"ref10","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume-title":"Journal of Machine Learning Research","volume":"23","author":"Fedus","year":"2022"},{"key":"ref11","article-title":"Mixtral of experts","author":"Jiang","year":"2024"},{"key":"ref12","article-title":"Qwen2. 5 technical report","author":"Yang","year":"2024"},{"key":"ref13","article-title":"Deepseek-v3 technical report","author":"Liu","year":"2024"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ipdps57955.2024.00086"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508418"},{"key":"ref16","article-title":"Fastmoe: A fast mixture-of-expert training system","author":"He","year":"2021"},{"key":"ref17","article-title":"Accelerating distributed MoE training and inference with lina","volume-title":"USENIX ATC","author":"Li"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/2541940.2541941"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/3342195.3387517"},{"key":"ref20","article-title":"DeepSpeed-MoE: Advancing mixture-of-experts inference and training to power next-generation ai scale","volume-title":"ICML","author":"Rajbhandari"},{"key":"ref21","article-title":"Attention is all you need","volume-title":"NeurIPS","author":"Vaswani"},{"key":"ref22","doi-asserted-by":"crossref","DOI":"10.1007\/3-540-33486-6_6","article-title":"A neural probabilistic language model","volume-title":"NeurIPS","author":"Bengio"},{"key":"ref23","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/P18-1007","article-title":"Subword regularization: Improving neural network translation models with multiple subword candidates","volume-title":"ACL","author":"Kudo"},{"key":"ref24","article-title":"Character-based NMT with transformer","author":"Gupta","year":"2019"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3706418"},{"key":"ref26","article-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer","volume-title":"ICLR","author":"Shazeer"},{"key":"ref27","article-title":"Large scale distributed deep networks","volume-title":"NeurIPS","volume":"25","author":"Dean"},{"key":"ref28","article-title":"Learning multiple layers of features from tiny images","volume-title":"Technical Report","author":"Krizhevsky","year":"2009"},{"key":"ref29","article-title":"Megatron-LM: Training multi-billion parameter language models using model parallelism","author":"Shoeybi","year":"2020"},{"key":"ref30","first-page":"499","article-title":"Hawk: Hybrid datacenter scheduling","volume-title":"2015 USENIX Annual Technical Conference (USENIX ATC 15)","author":"Delgado"},{"key":"ref31","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume-title":"NeurIPS","author":"Paszke"},{"issue":"140","key":"ref32","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume-title":"Journal of machine learning research","volume":"21","author":"Raffel","year":"2020"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"ref34","doi-asserted-by":"crossref","DOI":"10.1109\/ICCV.2015.11","article-title":"Aligning books and movies: Towards story-like visual explanations by watching movies and reading books","author":"Zhu","year":"2015"},{"key":"ref35","article-title":"Pointer sentinel mixture models","author":"Merity","year":"2016"},{"key":"ref36","volume-title":"Acl 2019 fourth conference on machine translation (wmt19), shared task: Machine translation of news"},{"key":"ref37","article-title":"Tutel: Adaptive mixture-of-experts at scale","volume-title":"MLSys","author":"Hwang","year":"2023"},{"key":"ref38","article-title":"Floe: On-the-fly moe inference on memory-constrained GPU","volume-title":"Forty-second International Conference on Machine Learning","author":"Zhou"},{"key":"ref39","article-title":"Deepspeed-mii: Mii makes low-latency and high-throughput inference possible, powered by deepspeed","year":"2022"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref41","article-title":"Megablocks: Efficient sparse training with mixture-of-experts","volume-title":"MLSys","author":"Gale","year":"2023"},{"key":"ref42","article-title":"SmartMoE: Efficiently training Sparsely-Activated models through combining offline and online parallelization","volume-title":"USENIX ATC","author":"Zhai"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/cluster52292.2023.00015"},{"key":"ref44","article-title":"Netmoe: Accelerating moe training through dynamic sample placement","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Liu"},{"key":"ref45","first-page":"1053","article-title":"{PopFetcher}: Towards accelerated {Mixture-of-Experts} training via popularity based {Expert-Wise} prefetch","volume-title":"2025 USENIX Annual Technical Conference (USENIX ATC 25)","author":"Zhang"}],"event":{"name":"2026 IEEE International Parallel and Distributed Processing Symposium (IPDPS)","location":"New Orleans, LA, USA","start":{"date-parts":[[2026,5,25]]},"end":{"date-parts":[[2026,5,29]]}},"container-title":["2026 IEEE International Parallel and Distributed Processing Symposium (IPDPS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11575315\/11575316\/11575403.pdf?arnumber=11575403","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T16:29:31Z","timestamp":1782923371000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11575403\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,25]]},"references-count":45,"URL":"https:\/\/doi.org\/10.1109\/ipdps65963.2026.00069","relation":{},"subject":[],"published":{"date-parts":[[2026,5,25]]}}}