{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:55:17Z","timestamp":1776930917641,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":57,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFB4503400"],"award-info":[{"award-number":["2023YFB4503400"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62402456,62402457"],"award-info":[{"award-number":["62402456,62402457"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhejiang Provincial Natural Science Foundation of China","award":["LQ24F020027"],"award-info":[{"award-number":["LQ24F020027"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,16]]},"DOI":"10.1145\/3712285.3759903","type":"proceedings-article","created":{"date-parts":[[2025,11,12]],"date-time":"2025-11-12T16:04:47Z","timestamp":1762963487000},"page":"1951-1965","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Diff-MoE: Efficient Batched MoE Inference with Priority-Driven Differential Expert Caching"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-8431-0864","authenticated-orcid":false,"given":"Kexin","family":"Li","sequence":"first","affiliation":[{"name":"National Engineering Research Center for Big Data Technology and System, Services Computing Technology and System Lab, Cluster and Grid Computing Lab, School of Computer Science and Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0581-522X","authenticated-orcid":false,"given":"Wenkan","family":"Huang","sequence":"additional","affiliation":[{"name":"National Engineering Research Center for Big Data Technology and System, Services Computing Technology and System Lab, Cluster and Grid Computing Lab, School of Computer Science and Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9951-3345","authenticated-orcid":false,"given":"Qinggang","family":"Wang","sequence":"additional","affiliation":[{"name":"National Engineering Research Center for Big Data Technology and System, Services Computing Technology and System Lab, Cluster and Grid Computing Lab, School of Computer Science and Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7903-2061","authenticated-orcid":false,"given":"Long","family":"Zheng","sequence":"additional","affiliation":[{"name":"National Engineering Research Center for Big Data Technology and System, Services Computing Technology and System Lab, Cluster and Grid Computing Lab, School of Computer Science and Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6302-813X","authenticated-orcid":false,"given":"Xiaofei","family":"Liao","sequence":"additional","affiliation":[{"name":"National Engineering Research Center for Big Data Technology and System, Services Computing Technology and System Lab, Cluster and Grid Computing Lab, School of Computer Science and Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3934-7605","authenticated-orcid":false,"given":"Hai","family":"Jin","sequence":"additional","affiliation":[{"name":"National Engineering Research Center for Big Data Technology and System, Services Computing Technology and System Lab, Cluster and Grid Computing Lab, School of Computer Science and Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0380-3506","authenticated-orcid":false,"given":"Jingling","family":"Xue","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, University of New South Wales, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,15]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00073"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","unstructured":"Hyung\u00a0Won Chung Le Hou Shayne Longpre Barret Zoph Yi Tay William Fedus Eric Li Xuezhi Wang Mostafa Dehghani Siddhartha Brahma Albert Webson Shixiang\u00a0Shane Gu Zhuyun Dai Mirac Suzgun Xinyun Chen Aakanksha Chowdhery Sharan Narang Gaurav Mishra Adams Yu Vincent\u00a0Y. Zhao Yanping Huang Andrew\u00a0M. Dai Hongkun Yu Slav Petrov Ed\u00a0H. Chi Jeff Dean Jacob Devlin Adam Roberts Denny Zhou Quoc\u00a0V. Le and Jason Wei. 2022. Scaling Instruction-Finetuned Language Models. CoRR abs\/2210.11416 (2022). 10.48550\/arXiv.2210.11416","DOI":"10.48550\/arXiv.2210.11416"},{"key":"e_1_3_3_2_5_2","unstructured":"Junyoung Chung \u00c7aglar G\u00fcl\u00e7ehre KyungHyun Cho and Yoshua Bengio. 2014. Empirical Evaluation of Gated Recurrent Neural Networks on Sequence Modeling. CoRR abs\/1412.3555 (2014). http:\/\/arxiv.org\/abs\/1412.3555"},{"key":"e_1_3_3_2_6_2","first-page":"613","volume-title":"Proceedings of the 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI\u201917), Boston, MA, USA, March 27-29, 2017","author":"Crankshaw Daniel","year":"2017","unstructured":"Daniel Crankshaw, Xin Wang, Giulio Zhou, Michael\u00a0J. Franklin, Joseph\u00a0E. Gonzalez, and Ion Stoica. 2017. Clipper: A Low-Latency Online Prediction Serving System. In Proceedings of the 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI\u201917), Boston, MA, USA, March 27-29, 2017. USENIX Association, 613\u2013627. https:\/\/www.usenix.org\/conference\/nsdi17\/technical-sessions\/presentation\/crankshaw"},{"key":"e_1_3_3_2_7_2","first-page":"797","volume-title":"Proceedings of the 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201923), Boston, MA, USA, July 10-12, 2023","author":"Cui Weihao","year":"2023","unstructured":"Weihao Cui, Zhenhua Han, Lingji Ouyang, Yichuan Wang, Ningxin Zheng, Lingxiao Ma, Yuqing Yang, Fan Yang, Jilong Xue, Lili Qiu, Lidong Zhou, Quan Chen, Haisheng Tan, and Minyi Guo. 2023. Optimizing Dynamic Neural Networks with Brainstorm. In Proceedings of the 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201923), Boston, MA, USA, July 10-12, 2023. USENIX Association, 797\u2013815. https:\/\/www.usenix.org\/conference\/osdi23\/presentation\/cui"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.70"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/n19-1423"},{"key":"e_1_3_3_2_10_2","first-page":"224","volume-title":"Proceedings of the Seventh Annual Conference on Machine Learning and Systems (MLSys\u201924), Santa Clara, CA, USA, May 13-16, 2024","author":"Du Zhixu","year":"2024","unstructured":"Zhixu Du, Shiyu Li, Yuhao Wu, Xiangyu Jiang, Jingwei Sun, Qilin Zheng, Yongkai Wu, Ang Li, Hai Li, and Yiran Chen. 2024. SiDA: Sparsity-Inspired Data-Aware Serving for Efficient and Scalable Large Mixture-of-Experts Models. In Proceedings of the Seventh Annual Conference on Machine Learning and Systems (MLSys\u201924), Santa Clara, CA, USA, May 13-16, 2024. mlsys.org, 224\u2013238. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2024\/hash\/698cfaf72a208aef2e78bcac55b74328-Abstract-Conference.html"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","unstructured":"Artyom Eliseev and Denis Mazur. 2023. Fast Inference of Mixture-of-Experts Language Models with Offloading. CoRR abs\/2312.17238 (2023). 10.48550\/arXiv.2312.17238","DOI":"10.48550\/arXiv.2312.17238"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","unstructured":"Zhiyuan Fang Yuegui Huang Zicong Hong Yufeng Lyu Wuhui Chen Yue Yu Fan Yu and Zibin Zheng. 2025. Klotski: Efficient Mixture-of-Expert Inference via Expert-Aware Multi-Batch Pipeline. CoRR abs\/2502.06888 (2025). 10.48550\/arXiv.2502.06888","DOI":"10.48550\/arXiv.2502.06888"},{"key":"e_1_3_3_2_13_2","unstructured":"William Fedus Barret Zoph and Noam Shazeer. 2022. Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity. Journal of Machine Learning Research 23 (2022) 1\u201339. https:\/\/jmlr.org\/papers\/v23\/21-0998.html"},{"key":"e_1_3_3_2_14_2","first-page":"439","volume-title":"Proceedings of the Seventh Annual Conference on Machine Learning and Systems (MLSys\u201924), Santa Clara, CA, USA, May 13-16, 2024","author":"Frantar Elias","year":"2024","unstructured":"Elias Frantar and Dan Alistarh. 2024. QMoE: Sub-1-Bit Compression of Trillion Parameter Models. In Proceedings of the Seventh Annual Conference on Machine Learning and Systems (MLSys\u201924), Santa Clara, CA, USA, May 13-16, 2024. mlsys.org, 439\u2013451. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2024\/hash\/c74b624843218d9b6713fcf299d6d5e4-Abstract-Conference.html"},{"key":"e_1_3_3_2_15_2","unstructured":"Georgi Gerganov.2024. llama.cpp. https:\/\/github.com\/ggml-org\/llama.cpp"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","unstructured":"Hao Gu Wei Li Lujun Li Qiyuan Zhu Mark\u00a0G. Lee Shengjie Sun Wei Xue and Yike Guo. 2025. Delta Decompression for MoE-based LLMs Compression. CoRR abs\/2502.17298 (2025). 10.48550\/arXiv.2502.17298","DOI":"10.48550\/arXiv.2502.17298"},{"key":"e_1_3_3_2_17_2","unstructured":"HuggingFace. 2022. HuggingFace accelerate. https:\/\/huggingface.co\/docs\/accelerate\/index"},{"key":"e_1_3_3_2_18_2","first-page":"269","volume-title":"Proceedings of the Sixth Conference on Machine Learning and Systems (MLSys\u201923), Miami, FL, USA, June 4-8, 2023","author":"Hwang Changho","year":"2023","unstructured":"Changho Hwang, Wei Cui, Yifan Xiong, Ziyue Yang, Ze Liu, Han Hu, Zilong Wang, Rafael Salas, Jithin Jose, Prabhat Ram, HoYuen Chau, Peng Cheng, Fan Yang, Mao Yang, and Yongqiang Xiong. 2023. Tutel: Adaptive Mixture-of-Experts at Scale. In Proceedings of the Sixth Conference on Machine Learning and Systems (MLSys\u201923), Miami, FL, USA, June 4-8, 2023. mlsys.org, 269\u2013287. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2023\/hash\/5616d34cf8ff73942cfd5aa922842556-Abstract-mlsys2023.html"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00078"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","unstructured":"Albert\u00a0Q. Jiang Alexandre Sablayrolles Antoine Roux Arthur Mensch Blanche Savary Chris Bamford Devendra\u00a0Singh Chaplot Diego de Las\u00a0Casas Emma\u00a0Bou Hanna Florian Bressand Gianna Lengyel Guillaume Bour Guillaume Lample L\u00e9lio\u00a0Renard Lavaud Lucile Saulnier Marie-Anne Lachaux Pierre Stock Sandeep Subramanian Sophia Yang Szymon Antoniak Teven\u00a0Le Scao Th\u00e9ophile Gervet Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William\u00a0El Sayed. 2024. Mixtral of Experts. CoRR abs\/2401.04088 (2024). 10.48550\/arXiv.2401.04088","DOI":"10.48550\/arXiv.2401.04088"},{"key":"e_1_3_3_2_21_2","first-page":"1","volume-title":"Proceedings of the 13th International Conference on Learning Representations (ICLR\u201925), Singapore, April 24-28, 2025","author":"Kamahori Keisuke","year":"2025","unstructured":"Keisuke Kamahori, Tian Tang, Yile Gu, Kan Zhu, and Baris Kasikci. 2025. Fiddler: CPU-GPU Orchestration for Fast Inference of Mixture-of-Experts Models. In Proceedings of the 13th International Conference on Learning Representations (ICLR\u201925), Singapore, April 24-28, 2025. OpenReview.net, 1\u201317. https:\/\/openreview.net\/forum?id=N5fVv6PZGz"},{"key":"e_1_3_3_2_22_2","unstructured":"Nitish\u00a0Shirish Keskar Bryan McCann Lav\u00a0R. Varshney Caiming Xiong and Richard Socher. 2019. CTRL: A Conditional Transformer Language Model for Controllable Generation. CoRR abs\/1909.05858 (2019). http:\/\/arxiv.org\/abs\/1909.05858"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","unstructured":"Young\u00a0Jin Kim Raffy Fahim and Hany\u00a0Hassan Awadalla. 2023. Mixture of Quantized Experts (MoQE): Complementary Effect of Low-bit Quantization and Robustness. CoRR abs\/2310.02410 (2023). 10.48550\/arXiv.2310.02410","DOI":"10.48550\/arXiv.2310.02410"},{"key":"e_1_3_3_2_24_2","first-page":"1","volume-title":"Proceedings of the 9th International Conference on Learning Representations (ICLR\u201921), Virtual Event, Austria, May 3-7, 2021","author":"Lepikhin Dmitry","year":"2021","unstructured":"Dmitry Lepikhin, HyoukJoong Lee, Yuanzhong Xu, Dehao Chen, Orhan Firat, Yanping Huang, Maxim Krikun, Noam Shazeer, and Zhifeng Chen. 2021. GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding. In Proceedings of the 9th International Conference on Learning Representations (ICLR\u201921), Virtual Event, Austria, May 3-7, 2021. OpenReview.net, 1\u201323. https:\/\/openreview.net\/forum?id=qrwe7XHTmYb"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"e_1_3_3_2_26_2","first-page":"87","volume-title":"Proceedings of the Seventh Annual Conference on Machine Learning and Systems (MLSys\u201924), Santa Clara, CA, USA, May 13-16, 2024","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Wei-Ming Chen, Wei-Chen Wang, Guangxuan Xiao, Xingyu Dang, Chuang Gan, and Song Han. 2024. AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration. In Proceedings of the Seventh Annual Conference on Machine Learning and Systems (MLSys\u201924), Santa Clara, CA, USA, May 13-16, 2024. mlsys.org, 87\u2013100. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2024\/hash\/42a452cbafa9dd64e9ba4aa95cc1ef21-Abstract-Conference.html"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3649329.3656507"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.334"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-1206"},{"key":"e_1_3_3_2_30_2","unstructured":"NVIDIA. 2019. FasterTransformer. https:\/\/github.com\/NVIDIA\/FasterTransformer"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","unstructured":"OpenAI. 2023. GPT-4 Technical Report. CoRR abs\/2303.08774 (2023). 10.48550\/arXiv.2303.08774","DOI":"10.48550\/arXiv.2303.08774"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","unstructured":"Narendra Patwardhan Stefano Marrone and Carlo Sansone. 2023. Transformers in the Real World: A Survey on NLP Applications. Information 14 4 (2023) 242. 10.3390\/INFO14040242","DOI":"10.3390\/INFO14040242"},{"key":"e_1_3_3_2_33_2","unstructured":"Alec Radford Karthik Narasimhan Tim Salimans Ilya Sutskever et\u00a0al. 2018. Improving language understanding by generative pre-training. OpenAI blog (2018) 1\u201312. https:\/\/cdn.openai.com\/research-covers\/language-unsupervised\/language_understanding_paper.pdf"},{"key":"e_1_3_3_2_34_2","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et\u00a0al. 2019. Language models are unsupervised multitask learners. OpenAI blog 1 8 (2019) 1\u201324. https:\/\/cdn.openai.com\/better-language-models\/language_models_are_unsupervised_multitask_learners.pdf"},{"key":"e_1_3_3_2_35_2","unstructured":"Colin Raffel Noam Shazeer Adam Roberts Katherine Lee Sharan Narang Michael Matena Yanqi Zhou Wei Li and Peter\u00a0J. Liu. 2020. Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer. Journal of Machine Learning Research 21 (2020) 1\u201367. https:\/\/jmlr.org\/papers\/v21\/20-074.html"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/D16-1264"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"Siva Reddy Danqi Chen and Christopher\u00a0D. Manning. 2019. CoQA: A Conversational Question Answering Challenge. Transactions of the Association for Computational Linguistics 7 (2019) 249\u2013266. https:\/\/aclanthology.org\/Q19-1016","DOI":"10.1162\/tacl_a_00266"},{"key":"e_1_3_3_2_38_2","first-page":"8583","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems (NIPS\u201921), December 6-14, 2021, virtual","author":"Riquelme Carlos","year":"2021","unstructured":"Carlos Riquelme, Joan Puigcerver, Basil Mustafa, Maxim Neumann, Rodolphe Jenatton, Andr\u00e9\u00a0Susano Pinto, Daniel Keysers, and Neil Houlsby. 2021. Scaling Vision with Sparse Mixture of Experts. In Proceedings of the Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems (NIPS\u201921), December 6-14, 2021, virtual. 8583\u20138595. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/48237d9f2dea8c74c2a72126cf63d933-Abstract.html"},{"key":"e_1_3_3_2_39_2","unstructured":"Victor Sanh Lysandre Debut Julien Chaumond and Thomas Wolf. 2019. DistilBERT A Distilled Version of BERT: Smaller Faster Cheaper and Lighter. CoRR abs\/1910.01108 (2019). http:\/\/arxiv.org\/abs\/1910.01108"},{"key":"e_1_3_3_2_40_2","first-page":"1","volume-title":"Proceedings of the 5th International Conference on Learning Representations (ICLR\u201917), Toulon, France, April 24-26, 2017","author":"Shazeer Noam","year":"2017","unstructured":"Noam Shazeer, Azalia Mirhoseini, Krzysztof Maziarz, Andy Davis, Quoc\u00a0V. Le, Geoffrey\u00a0E. Hinton, and Jeff Dean. 2017. Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. In Proceedings of the 5th International Conference on Learning Representations (ICLR\u201917), Toulon, France, April 24-26, 2017. OpenReview.net, 1\u201319. https:\/\/openreview.net\/forum?id=B1ckMDqlg"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"publisher","unstructured":"Xiaoniu Song Zihang Zhong and Rong Chen. 2024. ProMoE: Fast MoE-based LLM Serving using Proactive Caching. CoRR abs\/2410.22134 (2024). 10.48550\/arXiv.2410.22134","DOI":"10.48550\/arXiv.2410.22134"},{"key":"e_1_3_3_2_42_2","first-page":"5998","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems (NIPS\u201917), December 4-9, 2017, Long Beach, CA, USA","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Proceedings of the Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems (NIPS\u201917), December 4-9, 2017, Long Beach, CA, USA. 5998\u20136008. https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3649329.3655945"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00040"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00088"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/3431920.3439288"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00022"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","unstructured":"Qinggang Wang Long Zheng Jieshan Zhao Xiaofei Liao Hai Jin and Jingling Xue. 2020. A Conflict-free Scheduler for High-performance Graph Processing on Multi-pipeline FPGAs. ACM Transactions on Architecture and Code Optimization 17 2 (2020) 1\u201326. 10.1145\/3390523","DOI":"10.1145\/3390523"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00009"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","unstructured":"Peng Xu Xiatian Zhu and David\u00a0A. Clifton. 2023. Multimodal Learning With Transformers: A Survey. IEEE Transactions on Pattern Analysis and Machine Intelligence 45 10 (2023) 12113\u201312132. 10.1109\/TPAMI.2023.3275156","DOI":"10.1109\/TPAMI.2023.3275156"},{"key":"e_1_3_3_2_51_2","first-page":"28522","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems (NIPS\u201921), December 6-14, 2021, virtual","author":"Xu Yufei","year":"2021","unstructured":"Yufei Xu, Qiming Zhang, Jing Zhang, and Dacheng Tao. 2021. ViTAE: Vision Transformer Advanced by Exploring Intrinsic Inductive Bias. In Proceedings of the Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems (NIPS\u201921), December 6-14, 2021, virtual. 28522\u201328535. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/efb76cff97aaf057654ef2f38cd77d73-Abstract.html"},{"key":"e_1_3_3_2_52_2","unstructured":"Leyang Xue Yao Fu Zhan Lu Luo Mai and Mahesh Marina. 2025. MoE-Infinity: Efficient MoE Inference on Personal Machines with Sparsity-Aware Expert Cache. CoRR abs\/2401.14361 (2025). https:\/\/arxiv.org\/abs\/2401.14361"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.612"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"publisher","unstructured":"Dianhai Yu Liang Shen Hongxiang Hao Weibao Gong HuaChao Wu Jiang Bian Lirong Dai and Haoyi Xiong. 2024. MoESys: A Distributed and Efficient Mixture-of-Experts Training and Inference System for Internet Services. IEEE Transactions on Services Computing 17 5 (2024) 2626\u20132639. 10.1109\/TSC.2024.3399654","DOI":"10.1109\/TSC.2024.3399654"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00105"},{"key":"e_1_3_3_2_56_2","first-page":"961","volume-title":"Proceedings of the 2023 USENIX Annual Technical Conference (ATC\u201923), Boston, MA, USA, July 10-12, 2023","author":"Zhai Mingshu","year":"2023","unstructured":"Mingshu Zhai, Jiaao He, Zixuan Ma, Zan Zong, Runqing Zhang, and Jidong Zhai. 2023. SmartMoE: Efficiently Training Sparsely-Activated Models through Combining Offline and Online Parallelization. In Proceedings of the 2023 USENIX Annual Technical Conference (ATC\u201923), Boston, MA, USA, July 10-12, 2023. USENIX Association, 961\u2013975. https:\/\/www.usenix.org\/conference\/atc23\/presentation\/zhai"},{"key":"e_1_3_3_2_57_2","first-page":"1049","volume-title":"Proceedings of the 2019 USENIX Annual Technical Conference (ATC\u201919), Renton, WA, USA, July 10-12, 2019","author":"Zhang Chengliang","year":"2019","unstructured":"Chengliang Zhang, Minchen Yu, Wei Wang, and Feng Yan. 2019. MArk: Exploiting Cloud Services for Cost-Effective, SLO-Aware Machine Learning Inference Serving. In Proceedings of the 2019 USENIX Annual Technical Conference (ATC\u201919), Renton, WA, USA, July 10-12, 2019. USENIX Association, 1049\u20131062. https:\/\/www.usenix.org\/conference\/atc19\/presentation\/zhang-chengliang"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","unstructured":"Bin Zhu Peng Jin Munan Ning Bin Lin Jinfa Huang Qi Song Jiaxi Cui Junwu Zhang Zhenyu Tang Mingjun Pan Xing Zhou and Li Yuan. 2024. LLMBind: A Unified Modality-Task Integration Framework. CoRR abs\/2402.14891 (2024). 10.48550\/arXiv.2402.14891","DOI":"10.48550\/arXiv.2402.14891"}],"event":{"name":"SC '25: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St. Louis MO USA","acronym":"SC '25","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3712285.3759903","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T18:29:21Z","timestamp":1773253761000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3712285.3759903"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,15]]},"references-count":57,"alternative-id":["10.1145\/3712285.3759903","10.1145\/3712285"],"URL":"https:\/\/doi.org\/10.1145\/3712285.3759903","relation":{},"subject":[],"published":{"date-parts":[[2025,11,15]]},"assertion":[{"value":"2025-11-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}