{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T07:17:50Z","timestamp":1760426270275,"version":"3.30.2"},"reference-count":29,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,10,21]],"date-time":"2024-10-21T00:00:00Z","timestamp":1729468800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,10,21]],"date-time":"2024-10-21T00:00:00Z","timestamp":1729468800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,10,21]]},"DOI":"10.1109\/mascots64422.2024.10786517","type":"proceedings-article","created":{"date-parts":[[2024,12,13]],"date-time":"2024-12-13T18:50:09Z","timestamp":1734115809000},"page":"1-8","source":"Crossref","is-referenced-by-count":1,"title":["Optimizing GPU Multiplexing for Efficient and Cost-Effective Access to Diverse Large Language Models in GPU Clusters"],"prefix":"10.1109","author":[{"given":"Yue","family":"Zhu","sequence":"first","affiliation":[{"name":"IBM Research Center, Yorktown Heights,NY,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chen","family":"Wang","sequence":"additional","affiliation":[{"name":"IBM Research Center, Yorktown Heights,NY,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Max","family":"Calman","sequence":"additional","affiliation":[{"name":"IBM Research Center, Yorktown Heights,NY,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rina","family":"Nakazawa","sequence":"additional","affiliation":[{"name":"IBM Research,Tokoyo,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eun Kyung","family":"Lee","sequence":"additional","affiliation":[{"name":"IBM Research Center, Yorktown Heights,NY,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Llama: Open and Efficient Foundation Language models","author":"Touvron","year":"2023","journal-title":"arXiv preprint arXiv:2302.13971"},{"key":"ref2","article-title":"Tinystories: How Small can Language Models be and Still Speak Coherent English?","author":"Eldan","year":"2023","journal-title":"arXiv preprint arXiv:2305.07759"},{"key":"ref3","article-title":"A Comprehensive Performance Study of Large Language Models on Novel AI Accelerators","author":"Emani","year":"2023","journal-title":"arXiv preprint arXiv:2310.04607"},{"key":"ref4","article-title":"MIG introduction"},{"key":"ref5","article-title":"vLLM: Easy, Fast and Cheap LLM Serving for Everyone"},{"key":"ref6","first-page":"521","article-title":"Orca: A Distributed Serving System for {Transformer-Based} Generative Models","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Yu"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref8","article-title":"MPS introduction"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW55747.2022.00124"},{"article-title":"MIGPerf: A Comprehensive Benchmark for Deep Learning Training and Inference Workloads on Multi-Instance GPUs","year":"2023","author":"Zhang","key":"ref10"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3642970.3655827"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3642970.3655830"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3542929.3563510"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3453417.3453439"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER52292.2023.00023"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3547276.3548630"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3530390.3532734"},{"key":"ref18","article-title":"k8s-dra-driver"},{"key":"ref19","article-title":"openai-community\/gpt2-xl"},{"key":"ref20","article-title":"facebook\/opt-2.7b"},{"key":"ref21","article-title":"mistralai\/Mistral-7B-v0.1"},{"key":"ref22","article-title":"VMware\/open-llama-13b-open-instruct"},{"key":"ref23","article-title":"Dataset Card for ShareGPT90K"},{"article-title":"Stanford Alpaca","year":"2023","author":"Taori","key":"ref24"},{"key":"ref25","article-title":"Testing LLM Backends for Performance with Service Mocking"},{"key":"ref26","article-title":"LLM Has a Performance Problem Inherent to its Architecture: Latency"},{"key":"ref27","article-title":"FlexLLM: A System for Co-Serving Large Language Model Inference and Parameter-Efficient Finetuning","author":"Miao","year":"2024","journal-title":"arXiv preprint arXiv:2402.18789"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/3642970.3655832"},{"key":"ref29","article-title":"vLLM: Easy, Fast, and Cheap LLM Serving with PagedAttention"}],"event":{"name":"2024 32nd International Conference on Modeling, Analysis and Simulation of Computer and Telecommunication Systems (MASCOTS)","start":{"date-parts":[[2024,10,21]]},"location":"Krakow, Poland","end":{"date-parts":[[2024,10,23]]}},"container-title":["2024 32nd International Conference on Modeling, Analysis and Simulation of Computer and Telecommunication Systems (MASCOTS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10786488\/10786336\/10786517.pdf?arnumber=10786517","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,14]],"date-time":"2024-12-14T06:53:16Z","timestamp":1734159196000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10786517\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,21]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/mascots64422.2024.10786517","relation":{},"subject":[],"published":{"date-parts":[[2024,10,21]]}}}