{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T20:59:04Z","timestamp":1774731544961,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3774901.3778066","type":"proceedings-article","created":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T18:13:33Z","timestamp":1765995213000},"page":"25-30","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["ML Inference Scheduling with Predictable Latency"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-9864-5520","authenticated-orcid":false,"given":"Haidong","family":"Zhao","sequence":"first","affiliation":[{"name":"Inria, Paris, France and Sorbonne University, Paris, France"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5704-4889","authenticated-orcid":false,"given":"Nikolaos","family":"Georgantas","sequence":"additional","affiliation":[{"name":"Inria, Paris, France"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,17]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2025. MULTI-PROCESS SERVICE: vR555. https:\/\/docs.nvidia.com\/deploy\/pdf\/CUDA_Multi_Process_Service_Overview.pdf"},{"key":"e_1_3_2_1_2_1","unstructured":"2025. NVIDIA Nsight Compute. https:\/\/developer.nvidia.com\/nsight-compute"},{"key":"e_1_3_2_1_3_1","unstructured":"2025. NVIDIA TensorRT Documentation. https:\/\/docs.nvidia.com\/deeplearning\/tensorrt\/"},{"key":"e_1_3_2_1_4_1","unstructured":"2025. NVIDIA Triton Inference Server. https:\/\/developer.nvidia.com\/triton-inference-server"},{"key":"e_1_3_2_1_5_1","unstructured":"2025. TorchServe. https:\/\/pytorch.org\/serve\/"},{"key":"e_1_3_2_1_6_1","unstructured":"Arthur Chiao. 2023. Understanding NVIDIA GPU Performance: Utilization vs. Saturation. https:\/\/arthurchiao.art\/blog\/understanding-gpu-performance\/"},{"key":"e_1_3_2_1_7_1","volume-title":"Serving Heterogeneous Machine Learning Models on Multi-GPU Servers with Spatio-Temporal Sharing. In 2022 USENIX Annual Technical Conference (USENIX ATC 22)","author":"Choi Seungbeom","year":"2022","unstructured":"Seungbeom Choi, Sunho Lee, Yeonjae Kim, Jongse Park, Youngjin Kwon, and Jaehyuk Huh. 2022. Serving Heterogeneous Machine Learning Models on Multi-GPU Servers with Spatio-Temporal Sharing. In 2022 USENIX Annual Technical Conference (USENIX ATC 22). USENIX Association, Carlsbad, CA, 199\u2013216. https:\/\/www.usenix.org\/conference\/atc22\/presentation\/choi-seungbeom"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/2959100.2959190"},{"key":"e_1_3_2_1_9_1","volume-title":"Clipper: A Low-Latency Online Prediction Serving System. In 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17)","author":"Crankshaw Daniel","year":"2017","unstructured":"Daniel Crankshaw, Xin Wang, Guilio Zhou, Michael J. Franklin, Joseph E. Gonzalez, and Ion Stoica. 2017. Clipper: A Low-Latency Online Prediction Serving System. In 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17). USENIX Association, Boston, MA, 613\u2013627. https:\/\/www.usenix.org\/conference\/nsdi17\/technical-sessions\/presentation\/crankshaw"},{"key":"e_1_3_2_1_10_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly Jakob Uszkoreit and Neil Houlsby. 2021. An Image is Worth 16\u00d716 Words: Transformers for Image Recognition at Scale. https:\/\/arxiv.org\/abs\/2010.11929"},{"key":"e_1_3_2_1_11_1","unstructured":"GigaSpaces. 2023. Amazon Found Every 100ms of Latency Cost them 1% in Sales. https:\/\/www.gigaspaces.com\/blog\/amazon-found-every-100ms-of-latency-cost-them-1-in-sales"},{"key":"e_1_3_2_1_12_1","unstructured":"GPUnet. 2024. GPU vs CPU Cost Analysis in Shared Hosting Environments. https:\/\/medium.com\/@GPUnet\/gpu-vs-cpu-cost-analysis-in-shared-hosting-environments-bda82a65a1df"},{"key":"e_1_3_2_1_13_1","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Hao Mingzhe","unstructured":"Mingzhe Hao, Levent Toksoz, Nanqinqin Li, Edward Edberg Halim, Henry Hoffmann, and Haryadi S. Gunawi. 2020. LinnOS: Predictability on Unpredictable Flash Storage with a Light Neural Network. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20). USENIX Association, 173\u2013190. https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/hao"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_15_1","unstructured":"Chip Huyen. 2020. Machine Learning is Going Real-Time. https:\/\/huyenchip.com\/2020\/12\/27\/real-time-machine-learning.html"},{"key":"e_1_3_2_1_16_1","unstructured":"Glenn Jocher Ayush Chaurasia and Jing Qiu. 2023. Ultralytics YOLOv8. https:\/\/github.com\/ultralytics\/ultralytics"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10586-021-03299-z"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD63220.2024.00038"},{"key":"e_1_3_2_1_19_1","unstructured":"Inc. Kinara. 2023. Optimizing Latency for Edge AI Deployments: Designing the Optimal System for Low Latency Video Analytics Applications. Technical Report. https:\/\/kinara.ai\/wp-content\/themes\/kinara\/files\/Latency-on-the-Edge-WP-v2.pdf"},{"key":"e_1_3_2_1_20_1","unstructured":"Yinhan Liu Myle Ott Naman Goyal Jingfei Du Mandar Joshi Danqi Chen Omer Levy Mike Lewis Luke Zettlemoyer and Veselin Stoyanov. 2019. RoBERTa: A Robustly Optimized BERT Pretraining Approach. https:\/\/arxiv.org\/abs\/1907.11692"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437984.3458837"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.rser.2020.109725"},{"key":"e_1_3_2_1_24_1","unstructured":"Pandu Nayak. 2019. Understanding searches better than ever before. https:\/\/blog.google\/products\/search\/search-language-understanding-bert\/"},{"key":"e_1_3_2_1_25_1","volume-title":"High-Performance ML Serving. In Workshop on ML Systems at NIPS","author":"Olston Christopher","year":"2017","unstructured":"Christopher Olston, Fangwei Li, Jeremiah Harmsen, Jordan Soyke, Kiril Gorovoy, Li Lao, Noah Fiedel, Sukriti Ramesh, and Vinu Rajashekhar. 2017. TensorFlow-Serving: Flexible, High-Performance ML Serving. In Workshop on ML Systems at NIPS 2017."},{"key":"e_1_3_2_1_26_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. https:\/\/arxiv.org\/abs\/1409.1556"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629578"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3274808.3274820"},{"key":"e_1_3_2_1_29_1","volume-title":"Characterization and Prediction of Performance Interference on Mediated Passthrough GPUs for Interference-aware Scheduler. In 11th USENIX Workshop on Hot Topics in Cloud Computing (HotCloud 19)","author":"Xu Xin","year":"2019","unstructured":"Xin Xu, Na Zhang, Michael Cui, Michael He, and Ridhi Surana. 2019. Characterization and Prediction of Performance Interference on Mediated Passthrough GPUs for Interference-aware Scheduler. In 11th USENIX Workshop on Hot Topics in Cloud Computing (HotCloud 19). USENIX Association, Renton, WA. https:\/\/www.usenix.org\/conference\/hotcloud19\/presentation\/xu-xin"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3079202"}],"event":{"name":"MIDDLEWARE '25: 26th International Middleware Conference","location":"Nashville TN USA","acronym":"MAIoT '25","sponsor":["IFIP International Federation for Information Processing","USENIX The Advanced Computing System Association"]},"container-title":["Proceedings of the Middleware for Autonomous AIoT Systems in the Computing Continuum"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774901.3778066","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T18:13:40Z","timestamp":1765995220000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774901.3778066"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,15]]},"references-count":30,"alternative-id":["10.1145\/3774901.3778066","10.1145\/3774901"],"URL":"https:\/\/doi.org\/10.1145\/3774901.3778066","relation":{},"subject":[],"published":{"date-parts":[[2025,12,15]]},"assertion":[{"value":"2025-12-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}