{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T05:42:27Z","timestamp":1782538947114,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T00:00:00Z","timestamp":1743292800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100001858","name":"Vinnova","doi-asserted-by":"publisher","award":["2023-03003"],"award-info":[{"award-number":["2023-03003"]}],"id":[{"id":"10.13039\/501100001858","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004359","name":"Vetenskapsr\u00e5det","doi-asserted-by":"publisher","award":["2021-04212"],"award-info":[{"award-number":["2021-04212"]}],"id":[{"id":"10.13039\/501100004359","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,3,30]]},"DOI":"10.1145\/3721146.3721956","type":"proceedings-article","created":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T17:42:05Z","timestamp":1743529325000},"page":"132-138","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Priority-Aware Preemptive Scheduling for Mixed-Priority Workloads in MoE Inference"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2600-9025","authenticated-orcid":false,"given":"Mohammad","family":"Siavashi","sequence":"first","affiliation":[{"name":"KTH Royal Institute of Technology, Stockholm, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2161-0437","authenticated-orcid":false,"given":"Faezeh","family":"Keshmiri Dindarloo","sequence":"additional","affiliation":[{"name":"Unaffiliated, Stockholm, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1256-1070","authenticated-orcid":false,"given":"Dejan","family":"Kostic","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology, Stockholm, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9675-9729","authenticated-orcid":false,"given":"Marco","family":"Chiesa","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology, Stockholm, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,4]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Towards a Human-like Open-Domain Chatbot. arXiv preprint arXiv:2001.09977","author":"Adiwardana Daniel","year":"2020","unstructured":"Daniel Adiwardana, Minh-Thang Luong, David R So, Jamie Hall, Noah Fiedel, Romal Thoppilan, Zi Yang, Apoorv Kulshreshtha, Gaurav Nemade, Yifeng Lu, and Quoc V Le. 2020. Towards a Human-like Open-Domain Chatbot. arXiv preprint arXiv:2001.09977 (2020). https:\/\/arxiv.org\/abs\/2001.09977"},{"key":"e_1_3_2_1_2_1","volume-title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI","author":"Agrawal Amey","year":"2024","unstructured":"Amey Agrawal, Nitin Kedia, Ashish Panwar, Jayashree Mohan, Nipun Kwatra, Bhargav S. Gulavani, Alexey Tumanov, and Ramachandran Ramjee. 2024. Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 2024). Santa Clara, CA, 117--134. https:\/\/www.usenix.org\/conference\/osdi24\/presentation\/agrawal"},{"key":"e_1_3_2_1_3_1","volume-title":"Memory Layers at Scale. arXiv preprint arXiv:2412.09764","author":"Berges Vincent-Pierre","year":"2024","unstructured":"Vincent-Pierre Berges, Barlas O\u011fuz, Daniel Haziza, Wen-tau Yih, Luke Zettlemoyer, and Gargi Ghosh. 2024. Memory Layers at Scale. arXiv preprint arXiv:2412.09764 (2024). https:\/\/arxiv.org\/abs\/2412.09764 Accessed: 2025-02-11."},{"key":"e_1_3_2_1_4_1","unstructured":"Mark Chen Jerry Tworek Heewoo Jun Qiming Yuan Henrique Ponde de Oliveira Pinto Jared Kaplan Harri Edwards Yuri Burda Nicholas Joseph Greg Brockman Alex Ray Raul Puri Gretchen Krueger Michael Petrov Heidy Khlaaf Girish Sastry Pamela Mishkin Brooke Chan Scott Gray Nick Ryder Mikhail Pavlov Alethea Power Lukasz Kaiser Mohammad Bavarian Clemens Winter Philippe Tillet Felipe Petroski Such Dave Cummings Matthias Plappert Fotios Chantzis Elizabeth Barnes Ariel Herbert-Voss William Hebgen Guss Alex Nichol Alex Paino Nikolas Tezak Jie Tang Igor Babuschkin Suchir Balaji Shantanu Jain William Saunders Christopher Hesse Andrew N Carr Jan Leike Joshua Achiam Vedant Misra Evan Morikawa Alec Radford Matthew Knight Miles Brundage Mira Murati Katie Mayer Peter Welinder Bob McGrew Dario Amodei Sam McCandlish Ilya Sutskever and Wojciech Zaremba. 2021. Evaluating Large Language Models Trained on Code. arXiv preprint arXiv:2107.03374 (2021). https:\/\/arxiv.org\/abs\/2107.03374"},{"key":"e_1_3_2_1_5_1","unstructured":"Zhiyuan Dai et al. 2024. DeepSeek-MoE: Towards Ultimate Expert Specialization in Mixture-of-Experts Language Models. arXiv preprint arXiv:2406.00023 (2024)."},{"key":"e_1_3_2_1_6_1","unstructured":"DeepSeek-AI et al. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. (2025). The DeepSeek-R1 model contains 671 billion parameters.."},{"key":"e_1_3_2_1_7_1","unstructured":"Hugging Face. 2023. Text Generation Inference. https:\/\/github.com\/huggingface\/text-generation-inference. Accessed: 2025-02-06."},{"key":"e_1_3_2_1_8_1","volume-title":"LLM Inference at Scale with TGI. Hugging Face","author":"Goyanes Martin Iglesias","year":"2024","unstructured":"Martin Iglesias Goyanes. 2024. LLM Inference at Scale with TGI. Hugging Face (2024). https:\/\/huggingface.co\/blog\/martinigoyanes\/llm-inference-at-scale-with-tgi"},{"key":"e_1_3_2_1_9_1","volume-title":"Microsecond-scale Preemption for Concurrent GPU-accelerated DNN Inferences. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Han Mingcong","year":"2022","unstructured":"Mingcong Han, Hanze Zhang, Rong Chen, and Haibo Chen. 2022. Microsecond-scale Preemption for Concurrent GPU-accelerated DNN Inferences. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22). USENIX Association, Carlsbad, CA, 539--558. https:\/\/www.usenix.org\/conference\/osdi22\/presentation\/han"},{"key":"e_1_3_2_1_10_1","volume-title":"ORCA: A Fine-Grained Execution Model for Large Language Model Inference. arXiv preprint arXiv:2301.10292","author":"Minsoo Jeon","year":"2023","unstructured":"Minsoo Jeon et al. 2023. ORCA: A Fine-Grained Execution Model for Large Language Model Inference. arXiv preprint arXiv:2301.10292 (2023). https:\/\/arxiv.org\/abs\/2301.10292"},{"key":"e_1_3_2_1_11_1","unstructured":"Albert Q. Jiang Alexandre Sablayrolles Antoine Roux Arthur Mensch Blanche Savary Chris Bamford Devendra Singh Chaplot Diego de las Casas Emma Bou Hanna Florian Bressand Gianna Lengyel Guillaume Bour Guillaume Lample L\u00e9lio Renard Lavaud Lucile Saulnier Marie-Anne Lachaux Pierre Stock Sandeep Subramanian Sophia Yang Szymon Antoniak Teven Le Scao Th\u00e9ophile Gervet Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William El Sayed. 2024. Mixtral of Experts. arXiv preprint arXiv:2401.04088 (2024). https:\/\/arxiv.org\/abs\/2401.04088"},{"key":"e_1_3_2_1_12_1","volume-title":"16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 19)","author":"Kaffes Kostis","year":"2019","unstructured":"Kostis Kaffes, Timothy Chong, Jack Tigar Humphries, Adam Belay, David Mazi\u00e8res, and Christos Kozyrakis. 2019. Shinjuku: Preemptive Scheduling for microsecond-scale Tail Latency. In 16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 19). USENIX Association, Boston, MA, 345--360. https:\/\/www.usenix.org\/conference\/nsdi19\/presentation\/kaffes"},{"key":"e_1_3_2_1_13_1","volume-title":"Scaling Laws for Fine-Grained Mixture of Experts. arXiv preprint arXiv:2402.07871","author":"Krajewski Jakub","year":"2024","unstructured":"Jakub Krajewski, Jan Ludziejewski, Kamil Adamczewski, Maciej Pi\u00f3ro, Micha\u0142 Krutul, Szymon Antoniak, Kamil Ciebiera, Krystian Kr\u00f3l, Tomasz Odrzyg\u00f3\u017ad\u017a, Piotr Sankowski, Marek Cygan, and Sebastian Jaszczur. 2024. Scaling Laws for Fine-Grained Mixture of Experts. arXiv preprint arXiv:2402.07871 (2024). https:\/\/arxiv.org\/abs\/2402.07871"},{"key":"e_1_3_2_1_14_1","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Lee Wonbeom","year":"2024","unstructured":"Wonbeom Lee, Jungi Lee, Junghwan Seo, and Jaewoong Sim. 2024. {InfiniGen}: Efficient generative inference of large language models with dynamic {KV} cache management. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). 155--172."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672274"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K16-1028"},{"key":"e_1_3_2_1_17_1","unstructured":"NVIDIA. 2023. FasterTransformer: A Fast Inference Toolkit for Transformer Based Models. https:\/\/github.com\/NVIDIA\/FasterTransformer."},{"key":"e_1_3_2_1_18_1","unstructured":"NVIDIA Corporation. 2023. NVIDIA Grace Hopper Superchip Architecture In-Depth. https:\/\/developer.nvidia.com\/blog\/nvidia-grace-hopper-superchip-architecture-in-depth\/ Accessed: 2025-01-14."},{"key":"e_1_3_2_1_19_1","unstructured":"NVIDIA Corporation. 2023. NVLink & NVSwitch: Fastest HPC Data Center Platform. https:\/\/www.nvidia.com\/en-us\/data-center\/nvlink\/ Accessed: 2025-01-14."},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the 6th International Conference on Learning Representations. https:\/\/arxiv.org\/abs\/1705","author":"Paulus Romain","year":"2018","unstructured":"Romain Paulus, Caiming Xiong, and Richard Socher. 2018. A Deep Reinforced Model for Abstractive Summarization. In Proceedings of the 6th International Conference on Learning Representations. https:\/\/arxiv.org\/abs\/1705.04304"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3132747.3132780"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.24"},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 1073--1083","author":"Liu Peter J","year":"2017","unstructured":"Abigail See, Peter J Liu, and Christopher D Manning. 2017. Get To The Point: Summarization with Pointer-Generator Networks. In Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 1073--1083. https:\/\/aclanthology.org\/P17-1099\/"},{"key":"e_1_3_2_1_24_1","unstructured":"ShareGPT Team. 2023. ShareGPT. https:\/\/sharegpt.com\/. Accessed: 2025-01-19."},{"key":"e_1_3_2_1_25_1","unstructured":"Noam Shazeer et al. 2017. Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. arXiv preprint arXiv:1701.06538 (2017)."},{"key":"e_1_3_2_1_26_1","volume-title":"FastSwitch: Optimizing Context Switching Efficiency in Fairness-aware Large Language Model Serving. arXiv preprint arXiv:2411.18424","author":"Shen Ao","year":"2024","unstructured":"Ao Shen, Zhiyao Li, and Mingyu Gao. 2024. FastSwitch: Optimizing Context Switching Efficiency in Fairness-aware Large Language Model Serving. arXiv preprint arXiv:2411.18424 (2024). https:\/\/arxiv.org\/abs\/2411.18424"},{"key":"e_1_3_2_1_27_1","volume-title":"Hongyu Ren Zhang, et al","author":"Shi Diyi","year":"2023","unstructured":"Diyi Shi, Hao Zheng, Hongyu Ren Zhang, et al. 2023. vLLM: A High-Throughput and Memory-Efficient Inference Engine for Large Language Models. arXiv preprint arXiv:2305.11342 (2023). https:\/\/arxiv.org\/abs\/2305.11342"},{"key":"e_1_3_2_1_28_1","volume-title":"Llumnix: Dynamic Scheduling for Large Language Model Serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24)","author":"Sun Biao","year":"2024","unstructured":"Biao Sun, Ziming Huang, Hanyu Zhao, Wencong Xiao, Xinyi Zhang, Yong Li, and Wei Lin. 2024. Llumnix: Dynamic Scheduling for Large Language Model Serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24). https:\/\/www.usenix.org\/conference\/osdi24\/presentation\/sun-biao"},{"key":"e_1_3_2_1_29_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention Is All You Need. In Advances in Neural Information Processing Systems. 5998--6008."},{"key":"e_1_3_2_1_30_1","unstructured":"vLLM Project. 2024. vLLM's V1 Engine Architecture. https:\/\/github.com\/vllm-project\/vllm\/issues\/8779. Accessed: 2025-02-09."},{"key":"e_1_3_2_1_31_1","volume-title":"Fast Distributed Inference Serving for Large Language Models. arXiv preprint arXiv:2305.05920","author":"Wu Bingyang","year":"2023","unstructured":"Bingyang Wu, Yinmin Zhong, Zili Zhang, Gang Huang, Xuanzhe Liu, and Xin Jin. 2023. Fast Distributed Inference Serving for Large Language Models. arXiv preprint arXiv:2305.05920 (2023)."},{"key":"e_1_3_2_1_32_1","volume-title":"FastServe: Fast Distributed Inference Serving for Large Language Models. arXiv preprint arXiv:2305.05920","author":"Wu Bingyang","year":"2023","unstructured":"Bingyang Wu, Yinmin Zhong, Zili Zhang, Shengyu Liu, Fangyue Liu, Yuanhang Sun, Gang Huang, Xuanzhe Liu, and Xin Jin. 2023. FastServe: Fast Distributed Inference Serving for Large Language Models. arXiv preprint arXiv:2305.05920 (2023). https:\/\/arxiv.org\/abs\/2305.05920"},{"key":"e_1_3_2_1_33_1","volume-title":"DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong Yinmin","year":"2024","unstructured":"Yinmin Zhong, Shengyu Liu, Junda Chen, Jianbo Hu, Yibo Zhu, Xuanzhe Liu, Xin Jin, and Hao Zhang. 2024. DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). 193--210. https:\/\/www.usenix.org\/conference\/osdi24\/presentation\/zhong-yinmin"}],"event":{"name":"EuroMLSys '25: 5th Workshop on Machine Learning and Systems","location":"World Trade Center Rotterdam Netherlands","acronym":"EuroMLSys '25","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 5th Workshop on Machine Learning and Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3721146.3721956","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3721146.3721956","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:57:39Z","timestamp":1750298259000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3721146.3721956"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,30]]},"references-count":33,"alternative-id":["10.1145\/3721146.3721956","10.1145\/3721146"],"URL":"https:\/\/doi.org\/10.1145\/3721146.3721956","relation":{},"subject":[],"published":{"date-parts":[[2025,3,30]]},"assertion":[{"value":"2025-04-01","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}