{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T16:50:37Z","timestamp":1774716637929,"version":"3.50.1"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100002347","name":"Federal Ministry of Education and Research","doi-asserted-by":"publisher","award":["01IS22092"],"award-info":[{"award-number":["01IS22092"]}],"id":[{"id":"10.13039\/501100002347","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,5,19]]},"DOI":"10.1109\/ccgrid64434.2025.00040","type":"proceedings-article","created":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T17:36:08Z","timestamp":1751304968000},"page":"11-20","source":"Crossref","is-referenced-by-count":1,"title":["Draco: Dynamic Resource Allocation for Concurrent ML Applications"],"prefix":"10.1109","author":[{"given":"Theo","family":"Radig","sequence":"first","affiliation":[{"name":"Hasso-Plattner-Institute, Internet Technologies and Softwarization,Potsdam,Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Holger","family":"Karl","sequence":"additional","affiliation":[{"name":"Hasso-Plattner-Institute, Internet Technologies and Softwarization,Potsdam,Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","year":"2024","journal-title":"NVIDIA CUDA deep neural network (cuDNN)"},{"key":"ref2","article-title":"PyTorch: An imperative style, highperformance deep learning library","author":"Paszke","year":"2019","journal-title":"arXiv [cs.LG]"},{"key":"ref3","first-page":"519","article-title":"PerfIso: Performance isolation for commercial Latency-Sensitive services","volume-title":"2018 USENIX Annual Technical Conference (USENIX ATC 18)","author":"Iorgulescu","year":"2018"},{"key":"ref4","article-title":"DynamoLLM: Designing LLM inference clusters for performance and energy efficiency","author":"Stojkovic","year":"2024","journal-title":"arXiv [cs.AI]"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1512.03385"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3588195.3595940"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3642970.3655827"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER52292.2023.00023"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2019.2944602"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629578"},{"key":"ref11","first-page":"539","article-title":"Microsecond-scale preemption for concurrent GPU-accelerated DNN inferences","author":"Han","year":"2022","journal-title":"Oper Syst Des Implement"},{"key":"ref12","year":"2024","journal-title":"NVIDIA A30 tensor core GPU"},{"key":"ref13","journal-title":"NVIDIA multi-instance GPU user guide"},{"key":"ref14","year":"2024","journal-title":"1. introduction - multi-process service r555 documentation"},{"key":"ref15","author":"Harris","year":"2015","journal-title":"GPU pro tip: CUDA 7 streams simplify concurrency"},{"key":"ref16","year":"2024","journal-title":"NVIDIA CUDA green contexts"},{"key":"ref17","year":"2024","journal-title":"CUDA c++ programming guide"},{"key":"ref18","year":"2023","journal-title":"CUDA contexts"},{"key":"ref19","author":"Rennich","year":"2024","journal-title":"Cuda c\/c++ streams and concurrency"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/3453483.3454083"},{"key":"ref21","article-title":"TVM: An automated end-to-end optimizing compiler for deep learning","author":"Chen","year":"2018","journal-title":"arXiv [cs.LG]"},{"key":"ref22","year":"2024","journal-title":"ld.so(8) - linux manual page"},{"key":"ref23","year":"2024","journal-title":"NVIDIA CUDA execution control"},{"key":"ref24","year":"2024","journal-title":"NVIDIA CUDA execution control"},{"key":"ref25","year":"2024","journal-title":"2. usage - cupti 12.6 documentation"},{"key":"ref27","article-title":"Missile: Fine-grained, hardware-level GPU resource isolation for multitenant DNN inference","author":"Zhang","year":"2024","journal-title":"arXiv [cs.DC]"}],"event":{"name":"2025 IEEE 25th International Symposium on Cluster, Cloud and Internet Computing (CCGrid)","location":"Troms\u00f8, Norway","start":{"date-parts":[[2025,5,19]]},"end":{"date-parts":[[2025,5,22]]}},"container-title":["2025 IEEE 25th International Symposium on Cluster, Cloud and Internet Computing (CCGrid)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11044421\/11044790\/11044818.pdf?arnumber=11044818","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T06:02:32Z","timestamp":1751349752000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11044818\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,19]]},"references-count":26,"URL":"https:\/\/doi.org\/10.1109\/ccgrid64434.2025.00040","relation":{},"subject":[],"published":{"date-parts":[[2025,5,19]]}}}