{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:50:57Z","timestamp":1782996657955,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,6]]},"DOI":"10.1145\/3797905.3807834","type":"proceedings-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T11:50:37Z","timestamp":1782993037000},"page":"881-892","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["LayerScope: Predictive Cross-Layer Scheduling for Efficient Multi-Batch MoE Inference on Legacy Servers"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2661-0889","authenticated-orcid":false,"given":"Enda","family":"Yu","sequence":"first","affiliation":[{"name":"National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6243-8479","authenticated-orcid":false,"given":"Dezun","family":"Dong","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9320-2798","authenticated-orcid":false,"given":"Zhaoning","family":"Zhang","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-6006-3005","authenticated-orcid":false,"given":"Zhe","family":"Bai","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7167-4086","authenticated-orcid":false,"given":"Weiling","family":"Yang","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4605-148X","authenticated-orcid":false,"given":"Haojie","family":"Wang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9743-2034","authenticated-orcid":false,"given":"Dongsheng","family":"Li","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6651-7032","authenticated-orcid":false,"given":"Yongwei","family":"Wu","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6125-3330","authenticated-orcid":false,"given":"Xiangke","family":"Liao","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_3_2_2_2","first-page":"715","volume-title":"ACM ASPLOS","author":"Cao Shiyi","year":"2025","unstructured":"Shiyi Cao, Shu Liu, Tyler Griggs, Peter Schafhalter, Xiaoxuan Liu, Ying Sheng, Joseph\u00a0E Gonzalez, et\u00a0al. 2025. Moe-lightning: High-throughput moe inference on memory-constrained gpus. In ACM ASPLOS. 715\u2013730."},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3731569.3764843"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Le Chen Dahu Feng Erhu Feng Rong Zhao Yingrui Wang Yubin Xia Haibo Chen and Pinjie Xu. 2025. HeteroLLM: Accelerating Large Language Model Inference on Mobile SoCs platform with Heterogeneous AI Accelerators. arxiv:https:\/\/arXiv.org\/abs\/2501.14794","DOI":"10.20944\/preprints202501.0901.v1"},{"key":"e_1_3_3_2_5_2","unstructured":"Peizhuang Cong Aomufei Yuan Shimao Chen Yuxuan Tian Bowen Ye and Tong Yang. 2024. Prediction is all moe needs: Expert load distribution goes from fluctuating to stabilizing. arxiv:https:\/\/arXiv.org\/abs\/2404.16914"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.1145\/3721146.3721961"},{"key":"e_1_3_3_2_7_2","unstructured":"Zhixu Du Shiyu Li Yuhao Wu Xiangyu Jiang Jingwei Sun Qilin Zheng Yongkai Wu Ang Li Hai Li and Yiran Chen. 2024. Sida: Sparsity-inspired data-aware serving for efficient and scalable large mixture-of-experts models. MLSys 6 (2024) 224\u2013238."},{"key":"e_1_3_3_2_8_2","unstructured":"Haojie Duanmu Xiuhong Li Zhihang Yuan Size Zheng Jiangfei Duan Xingcheng Zhang and Dahua Lin. 2025. MxMoE: Mixed-precision Quantization for MoE with Accuracy and Performance Co-Design. arxiv:https:\/\/arXiv.org\/abs\/2505.05799"},{"key":"e_1_3_3_2_9_2","unstructured":"Hugging Face. 2024. ShareGPT-V3-unfiltered-cleaned-split. https:\/\/huggingface.co\/datasets\/learnanything\/sharegpt_v3_unfiltered_cleaned_split."},{"key":"e_1_3_3_2_10_2","unstructured":"Zhiyuan Fang Zicong Hong Yuegui Huang et\u00a0al. 2025. Fate: Fast Edge Inference of Mixture-of-Experts Models via Cross-Layer Gate. arxiv:https:\/\/arXiv.org\/abs\/2502.12224"},{"key":"e_1_3_3_2_11_2","first-page":"574","volume-title":"ACM ASPLOS","author":"Fang Zhiyuan","year":"2025","unstructured":"Zhiyuan Fang, Yuegui Huang, Zicong Hong, Yufeng Lyu, Wuhui Chen, Yue Yu, Fan Yu, and Zibin Zheng. 2025. Klotski: Efficient Mixture-of-Expert Inference via Expert-Aware Multi-Batch Pipeline. In ACM ASPLOS. 574\u2013588."},{"key":"e_1_3_3_2_12_2","unstructured":"William Fedus Barret Zoph and Noam Shazeer. 2022. Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity. JMLR120 (2022) 1\u201339."},{"key":"e_1_3_3_2_13_2","unstructured":"Elias Frantar and Dan Alistarh. 2023. Qmoe: Practical sub-1-bit compression of trillion-parameter models. arxiv:https:\/\/arXiv.org\/abs\/2310.16795"},{"key":"e_1_3_3_2_14_2","unstructured":"Yongxin Guo Zhenglin Cheng Xiaoying Tang Zhaopeng Tu and Tao Lin. 2024. Dynamic mixture of experts: An auto-tuning approach for efficient transformer models. arxiv:https:\/\/arXiv.org\/abs\/2405.14297"},{"key":"e_1_3_3_2_15_2","unstructured":"Vima Gupta Kartik Sinha Ada Gavrilovska and Anand\u00a0Padmanabha Iyer. 2024. Lynx: Enabling Efficient MoE Inference through Dynamic Batch-Aware Expert Selection. arxiv:https:\/\/arXiv.org\/abs\/2411.08982"},{"key":"e_1_3_3_2_16_2","unstructured":"Xin He Shunkang Zhang Yuxin Wang Haiyan Yin Zihao Zeng Shaohuai Shi Zhenheng Tang Xiaowen Chu Ivor Tsang and Ong\u00a0Yew Soon. 2024. Expertflow: Optimized expert activation and token allocation for efficient mixture-of-experts inference. arxiv:https:\/\/arXiv.org\/abs\/2410.17954"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"crossref","unstructured":"Huanqi Hu Bowen Xiao Shixuan Sun Jianian Yin Zhexi Zhang Xiang Luo Chengquan Jiang Weiqi Xu Xiaoying Jia Xin Liu et\u00a0al. 2025. LiquidGEMM: Hardware-Efficient W4A8 GEMM Kernel for High-Performance LLM Serving. arxiv:https:\/\/arXiv.org\/abs\/2509.01229","DOI":"10.1145\/3712285.3759852"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Haiyang Huang Newsha Ardalani Anna Sun Liu Ke Shruti Bhosale Hsien-Hsin Lee Carole-Jean Wu and Benjamin Lee. 2024. Toward efficient inference for mixture of experts. NIPS 37 (2024) 84033\u201384059.","DOI":"10.52202\/079017-2670"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00078"},{"key":"e_1_3_3_2_20_2","unstructured":"Albert\u00a0Q Jiang Alexandre Sablayrolles Antoine Roux Arthur Mensch Blanche Savary Chris Bamford Devendra\u00a0Singh Chaplot Diego de\u00a0las Casas Emma\u00a0Bou Hanna Florian Bressand et\u00a0al. 2024. Mixtral of experts. arxiv:https:\/\/arXiv.org\/abs\/2401.04088"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i17.33945"},{"key":"e_1_3_3_2_22_2","first-page":"56099","volume-title":"ICLR","author":"Kamahori Keisuke","year":"2025","unstructured":"Keisuke Kamahori, Tian Tang, Yile Gu, Kan Zhu, and Baris Kasikci. 2025. Fiddler: CPU-GPU Orchestration for Fast Inference of Mixture-of-Experts Models. In ICLR. 56099\u201356115."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_24_2","unstructured":"Xinlu Lai. 2024. The DPO Dataset for Chinese and English with emoji. https:\/\/huggingface.co\/datasets\/shareAI\/DPO-zh-en-emoji."},{"key":"e_1_3_3_2_25_2","first-page":"155","volume-title":"USENIX OSDI","author":"Lee Wonbeom","year":"2024","unstructured":"Wonbeom Lee, Jungi Lee, Junghwan Seo, and Jaewoong Sim. 2024. InfiniGen: Efficient generative inference of large language models with dynamic KV cache management. In USENIX OSDI. 155\u2013172."},{"key":"e_1_3_3_2_26_2","first-page":"945","volume-title":"USENIX ATC 23","author":"Li Jiamin","year":"2023","unstructured":"Jiamin Li, Yimin Jiang, Yibo Zhu, Cong Wang, and Hong Xu. 2023. Accelerating distributed MoE training and inference with lina. In USENIX ATC 23. 945\u2013959."},{"key":"e_1_3_3_2_27_2","unstructured":"Aixin Liu Bei Feng Bin Wang Bingxuan Wang Bo Liu Chenggang Zhao Chengqi Dengr Chong Ruan Damai Dai Daya Guo et\u00a0al. 2024. Deepseek-v2: A strong economical and efficient mixture-of-experts language model. arxiv:https:\/\/arXiv.org\/abs\/2405.04434"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"crossref","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong\u00a0Jae Lee. 2023. Visual instruction tuning. NIPS 36 (2023) 34892\u201334916.","DOI":"10.52202\/075280-1516"},{"key":"e_1_3_3_2_29_2","unstructured":"Jingyuan Liu Jianlin Su Xingcheng Yao Zhejun Jiang Guokun Lai Yulun Du Yidao Qin et\u00a0al. 2025. Muon is Scalable for LLM Training. arxiv:https:\/\/arXiv.org\/abs\/2502.16982"},{"key":"e_1_3_3_2_30_2","unstructured":"Xudong Lu Qi Liu Yuhui Xu Aojun Zhou Siyuan Huang Bo Zhang Junchi Yan and Hongsheng Li. 2024. Not all experts are equal: Efficient expert pruning and skipping for mixture-of-experts large language models. arxiv:https:\/\/arXiv.org\/abs\/2402.14800"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731057"},{"key":"e_1_3_3_2_32_2","unstructured":"OpenAI Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia\u00a0Leoni Aleman et\u00a0al. 2024. GPT-4 Technical Report. arxiv:https:\/\/arXiv.org\/abs\/2303.08774"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00113"},{"key":"e_1_3_3_2_34_2","first-page":"18332","volume-title":"ICML","author":"Rajbhandari Samyam","year":"2022","unstructured":"Samyam Rajbhandari, Conglong Li, Zhewei Yao, Minjia Zhang, Reza\u00a0Yazdani Aminabadi, Ammar\u00a0Ahmad Awan, Jeff Rasley, and Yuxiong He. 2022. Deepspeed-moe: Advancing mixture-of-experts inference and training to power next-generation ai scale. In ICML. 18332\u201318346."},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"crossref","unstructured":"Jay Shah Ganesh Bikshandi Ying Zhang Vijay Thakkar Pradeep Ramani and Tri Dao. 2024. Flashattention-3: Fast and accurate attention with asynchrony and low-precision. NIPS 37 (2024) 68658\u201368685.","DOI":"10.52202\/079017-2193"},{"key":"e_1_3_3_2_37_2","unstructured":"Xiaoniu Song Zihang Zhong Rong Chen and Haibo Chen. 2024. Promoe: Fast moe-based llm serving using proactive caching. arxiv:https:\/\/arXiv.org\/abs\/2410.22134"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695964"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"crossref","unstructured":"Ruslan Svirschevski Avner May Zhuoming Chen Beidi Chen Zhihao Jia and Max Ryabinin. 2024. Specexec: Massively parallel speculative decoding for interactive llm inference on consumer devices. NIPS 37 (2024) 16342\u201316368.","DOI":"10.52202\/079017-0522"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"crossref","unstructured":"Wei Tao Haocheng Lu Xiaoyang Qu Bin Zhang Kai Lu Jiguang Wan and Jianzong Wang. 2025. MoQAE: Mixed-Precision Quantization for Long-Context LLM Inference via Mixture of Quantization-Aware Experts. arxiv:https:\/\/arXiv.org\/abs\/2506.07533","DOI":"10.18653\/v1\/2025.acl-long.531"},{"key":"e_1_3_3_2_41_2","unstructured":"Qwen Team. 2025. Qwen3 Technical Report. arxiv:https:\/\/arXiv.org\/abs\/2505.09388"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00096"},{"key":"e_1_3_3_2_43_2","unstructured":"Tairan Xu Leyang Xue Zhan Lu Adrian Jackson and Luo Mai. 2025. MoE-Gen: High-Throughput MoE Inference on a Single GPU with Module-Based Batching. arxiv:https:\/\/arXiv.org\/abs\/2503.09716"},{"key":"e_1_3_3_2_44_2","first-page":"55625","volume-title":"ICML","author":"Xue Fuzhao","year":"2024","unstructured":"Fuzhao Xue, Zian Zheng, Yao Fu, Jinjie Ni, Zangwei Zheng, Wangchunshu Zhou, and Yang You. 2024. OpenMoE: an early effort on open mixture-of-experts language models. In ICML. 55625\u201355655."},{"key":"e_1_3_3_2_45_2","unstructured":"Leyang Xue Yao Fu Zhan Lu Luo Mai and Mahesh Marina. 2025. MoE-Infinity: Efficient MoE Inference on Personal Machines with Sparsity-Aware Expert Cache. arxiv:https:\/\/arXiv.org\/abs\/2401.14361"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS57955.2024.00086"},{"key":"e_1_3_3_2_47_2","unstructured":"Hanfei Yu Xingqi Cui Hong Zhang and Hao Wang. 2025. fMoE: Fine-Grained Expert Offloading for Large Mixture-of-Experts Serving. arxiv:https:\/\/arXiv.org\/abs\/2502.05370"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.879"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.23919\/DATE64628.2025.10992741"},{"key":"e_1_3_3_2_50_2","unstructured":"Xuanlei Zhao Bin Jia Haotian Zhou Ziming Liu Shenggan Cheng and Yang You. 2024. Hetegen: Efficient heterogeneous parallel inference for large language models on resource-constrained devices. MLSys 6 (2024) 162\u2013172."},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3676536.3676741"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","DOI":"10.1109\/DAC63849.2025.11133274"}],"event":{"name":"ICS '26: 2026 International Conference on Supercomputing","location":"Belfast United Kingdom","acronym":"ICS '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 40th ACM International Conference on Supercomputing"],"original-title":[],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:38:14Z","timestamp":1782995894000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797905.3807834"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":51,"alternative-id":["10.1145\/3797905.3807834","10.1145\/3797905"],"URL":"https:\/\/doi.org\/10.1145\/3797905.3807834","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}