{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T14:10:05Z","timestamp":1784211005639,"version":"3.55.0"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T00:00:00Z","timestamp":1771286400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T00:00:00Z","timestamp":1771286400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100004837","name":"Ministerio de Ciencia e Innovaci\u00f3n","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004837","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-026-08312-z","type":"journal-article","created":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T06:36:19Z","timestamp":1771310179000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["A comprehensive evaluation of spatial co-execution on GPUs using MPS and MIG technologies"],"prefix":"10.1007","volume":"82","author":[{"given":"Jorge","family":"Villarrubia","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Luis","family":"Costero","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Francisco D.","family":"Igual","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Katzalin","family":"Olcoz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,2,17]]},"reference":[{"issue":"3","key":"8312_CR1","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1038\/s42256-022-00463-x","volume":"4","author":"M Pandey","year":"2022","unstructured":"Pandey M, Fernandez M, Gentile F, Isayev O, Tropsha A, Stern AC et al (2022) The transformational role of GPU computing and deep learning in drug discovery. Nat Mach Intell 4(3):211\u2013221. https:\/\/doi.org\/10.1038\/s42256-022-00463-x","journal-title":"Nat Mach Intell"},{"key":"8312_CR2","doi-asserted-by":"publisher","DOI":"10.1145\/3729215","author":"C Silvano","year":"2025","unstructured":"Silvano C, Ielmini D, Ferrandi F, Fiorin L, Curzel S, Benini L et al (2025) A survey on deep learning hardware accelerators for heterogeneous HPC platforms. ACM Comput Surv. https:\/\/doi.org\/10.1145\/3729215","journal-title":"ACM Comput Surv"},{"issue":"16","key":"8312_CR3","doi-asserted-by":"publisher","first-page":"540","DOI":"10.34218\/IJARET_16_01_038","volume":"02","author":"A Atluri","year":"2025","unstructured":"Atluri A (2025) The evolution of NVIDIA GPUs for deep learning: from gaming to AI powerhouse. Int J Adv Res Eng Technol 02(16):540\u2013551. https:\/\/doi.org\/10.34218\/IJARET_16_01_038","journal-title":"Int J Adv Res Eng Technol"},{"key":"8312_CR4","unstructured":"Radford A, Wu J, Child R, Luan D, Amodei D, Sutskever I (2019) Language models are unsupervised multitask learners. OpenAI technical report"},{"key":"8312_CR5","unstructured":"NVIDIA. Technical documentation of the Blackwell architecture. URL https:\/\/resources.nvidia.com\/en-us-blackwell-architecture"},{"key":"8312_CR6","doi-asserted-by":"publisher","unstructured":"Adufu T, Ha J, Kim Y (2024) Exploring the diversity of multiple job deployments over GPUs for efficient resource sharing. In: International Conference on Information Networking. IEEE Computer Society, pp 777\u2013782. https:\/\/doi.org\/10.1109\/ICOIN59985.2024.10572198","DOI":"10.1109\/ICOIN59985.2024.10572198"},{"key":"8312_CR7","unstructured":"Durvasula S, Zhao A, Kiguru R, Guan Y, Chen Z, Vijaykumar N. ACS: concurrent kernel execution on irregular, input-dependent computational graphs. Available from: arXiv:https:\/\/arxiv.org\/abs\/2401.12377"},{"key":"8312_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2025.105128","volume":"204","author":"J Villarrubia","year":"2025","unstructured":"Villarrubia J, Costero L, Igual FD, Olcoz K (2025) Leveraging multi-instance GPUs through moldable task scheduling. J Parallel Distrib Comput 204:105128. https:\/\/doi.org\/10.1016\/j.jpdc.2025.105128","journal-title":"J Parallel Distrib Comput"},{"key":"8312_CR9","unstructured":"Elvinger P, Strati F, Jerger NE, Klimovic A. Measuring GPU utilization one level deeper. Available from: arXiv:https:\/\/arxiv.org\/abs\/2501.16909"},{"key":"8312_CR10","doi-asserted-by":"publisher","unstructured":"Gao Y, He Y, Li X, Zhao B, Lin H, Liang Y et al (2024) An empirical study on low GPU utilization of deep learning jobs. In: Proceedings of the IEEE\/ACM 46th International Conference on Software Engineering. ICSE \u201924. New York, NY, USA: Association for Computing Machinery. https:\/\/doi.org\/10.1145\/3597503.3639232","DOI":"10.1145\/3597503.3639232"},{"key":"8312_CR11","unstructured":"You J, Chung JW, Chowdhury M (2023) Zeus: understanding and optimizing GPU energy consumption of DNN training. In: 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). Boston, MA: USENIX Association, pp 119\u2013139. Available from: https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/you"},{"key":"8312_CR12","unstructured":"Morand C, Ligozat AL, N\u00e9v\u00e9ol A. How Green Can AI Be? A study of trends in machine learning environmental impacts. Available from: arXiv:https:\/\/arxiv.org\/abs\/2412.17376"},{"key":"8312_CR13","unstructured":"NVIDIA. Multi-Process Service. URL https:\/\/docs.nvidia.com\/deploy\/mps\/"},{"key":"8312_CR14","unstructured":"NVIDIA. Multi-Instance GPU User Guide. URL https:\/\/docs.nvidia.com\/datacenter\/tesla\/mig-user-guide\/"},{"key":"8312_CR15","doi-asserted-by":"publisher","unstructured":"Li B, Patel T, Samsi S, Gadepally V, Tiwari D (2022) MISO: exploiting multi-instance GPU capability on multi-tenant GPU clusters. In: 13th Symposium on Cloud Computing. https:\/\/doi.org\/10.1145\/3542929.3563510","DOI":"10.1145\/3542929.3563510"},{"key":"8312_CR16","doi-asserted-by":"publisher","unstructured":"Zhang B, Li S, Li Z (2024) MIGER: integrating multi-instance GPU and multi-process service for deep learning clusters. In: 53rd International Conference on Parallel Processing, pp 504\u2013513. https:\/\/doi.org\/10.1145\/3673038.3673089","DOI":"10.1145\/3673038.3673089"},{"issue":"6","key":"8312_CR17","doi-asserted-by":"publisher","first-page":"1451","DOI":"10.1109\/TPDS.2021.3115630","volume":"33","author":"C Zhao","year":"2022","unstructured":"Zhao C, Gao W, Nie F, Zhou H (2022) A Survey of GPU Multitasking Methods Supported by Hardware Architecture. IEEE Trans Parallel Distrib Syst 33(6):1451\u20131463. https:\/\/doi.org\/10.1109\/TPDS.2021.3115630","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"8312_CR18","unstructured":"NVIDIA Corporation. CUDA C++ Programming Guide. Available from: https:\/\/docs.nvidia.com\/cuda\/pdf\/CUDA_C_Programming_Guide.pdf"},{"key":"8312_CR19","unstructured":"AMD. MxGPU deployment guide. URL https:\/\/drivers.amd.com\/relnotes\/amd_mxgpu_deploymentguide_vmware.pdf"},{"key":"8312_CR20","unstructured":"Intel. iGPU SR-IOV documentation. URL https:\/\/www.intel.sg\/content\/dam\/www\/central-libraries\/us\/en\/documents\/2022-09\/intel-whitepaper2022-dfi-v11.pdf"},{"key":"8312_CR21","unstructured":"NVIDIA Corporation. NVIDIA Data Center GPU Manager (DCGM) Documentation. https:\/\/developer.nvidia.com\/dcgm"},{"key":"8312_CR22","unstructured":"NVIDIA Corporation. API Reference Guide of NVIDIA Management Library (NVML). Version vR580. Available from: https:\/\/docs.nvidia.com\/deploy\/nvml-api\/index.html"},{"key":"8312_CR23","doi-asserted-by":"publisher","unstructured":"Robroek T, Yousefzadeh-Asl-Miandoab E, T\u00f6z\u00fcn P (2024) An analysis of collocation on GPUs for deep learning training. In: Proceedings of the 4th Workshop on Machine Learning and Systems. EuroMLSys \u201924. New York, NY, USA: Association for Computing Machinery, pp 81\u201390. https:\/\/doi.org\/10.1145\/3642970.3655827","DOI":"10.1145\/3642970.3655827"},{"key":"8312_CR24","doi-asserted-by":"publisher","unstructured":"Weaver A, Kavi K, Milojicic D, Enriquez RPH, Hogade N, Mishra A et al (2024) Granularity- and Interference-Aware GPU Sharing with MPS. In: SC24-W: Workshops of the International Conference for High Performance Computing, Networking, Storage and Analysis, pp 1630\u20131637. https:\/\/doi.org\/10.1109\/SCW63240.2024.00203","DOI":"10.1109\/SCW63240.2024.00203"},{"key":"8312_CR25","unstructured":"Zhao Y, Liu X, Liu S, Li X, Zhu Y, Huang G et al. MuxFlow: efficient and safe GPU sharing in large-scale production deep learning clusters. Available from: arXiv:2303.13803"},{"issue":"1","key":"8312_CR26","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1007\/s11634-023-00574-2","volume":"19","author":"P Giudici","year":"2025","unstructured":"Giudici P, Raffinetti E (2025) RGA: a unified measure of predictive accuracy. Adv Data Anal Classif 19(1):67\u201393. https:\/\/doi.org\/10.1007\/s11634-023-00574-2","journal-title":"Adv Data Anal Classif"},{"issue":"2","key":"8312_CR27","doi-asserted-by":"publisher","first-page":"330","DOI":"10.1080\/02331888.2024.2434904","volume":"59","author":"P Giudici","year":"2025","unstructured":"Giudici P, Raffinetti E, Toscani G (2025) Measuring multidimensional inequality: a new proposal based on the Fourier transform. Statistics 59(2):330\u2013353. https:\/\/doi.org\/10.1080\/02331888.2024.2434904","journal-title":"Statistics"},{"key":"8312_CR28","doi-asserted-by":"publisher","unstructured":"Hu B, Rossbach CJ (2020) Altis: modernizing GPGPU benchmarks. In: 2020 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS), pp 1\u201311. https:\/\/doi.org\/10.1109\/ISPASS48437.2020.00011","DOI":"10.1109\/ISPASS48437.2020.00011"},{"key":"8312_CR29","doi-asserted-by":"publisher","unstructured":"Danalis A, Marin G, McCurdy C, Meredith JS, Roth PC, Spafford K et al (2010) The Scalable Heterogeneous Computing (SHOC) benchmark suite. In: Proceedings of the 3rd Workshop on General-Purpose Computation on Graphics Processing Units. GPGPU-3. New York, NY, USA: Association for Computing Machinery, pp. 63\u201374. https:\/\/doi.org\/10.1145\/1735688.1735702","DOI":"10.1145\/1735688.1735702"},{"key":"8312_CR30","doi-asserted-by":"publisher","unstructured":"Che S, Boyer M, Meng J, Tarjan D, Sheaffer JW, Lee SH et al (2009) Rodinia: a benchmark suite for heterogeneous computing. In: 2009 IEEE International Symposium on Workload Characterization (IISWC), pp 44\u201354. https:\/\/doi.org\/10.1109\/IISWC.2009.5306797","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"8312_CR31","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"8312_CR32","doi-asserted-by":"publisher","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics. Minneapolis, Minnesota: Association for Computational Linguistics, pp 4171\u20134186. https:\/\/doi.org\/10.18653\/v1\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"issue":"10","key":"8312_CR33","doi-asserted-by":"publisher","first-page":"1196","DOI":"10.1038\/s41592-021-01252-x","volume":"18","author":"\u017d Avsec","year":"2021","unstructured":"Avsec \u017d, Agarwal V, Visentin D, Ledsam JR, Grabska-Barwinska A, Taylor KR et al (2021) Effective gene expression prediction from sequence by integrating long-range interactions. Nat Methods 18(10):1196\u20131203. https:\/\/doi.org\/10.1038\/s41592-021-01252-x","journal-title":"Nat Methods"},{"key":"8312_CR34","unstructured":"Research I. IBM Materials: AI for Materials Science. GitHub. Model: pos-egnn.v1-6M. https:\/\/github.com\/IBM\/materials"},{"key":"8312_CR35","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2025.107813","volume":"171","author":"E de la Calle","year":"2025","unstructured":"de la Calle E (2025) Evaluation of Juliana tool: a translator for Julia\u2019s CUDA.jl code into KernelAbstraction.jl. Future Gener Comput Syst 171:107813. https:\/\/doi.org\/10.1016\/j.future.2025.107813","journal-title":"Future Gener Comput Syst"},{"key":"8312_CR36","doi-asserted-by":"publisher","unstructured":"Saiz A, Prieto P, Abad P, Gregorio JA, Puente V (2022) Top-down performance profiling on NVIDIA\u2019s GPUs. In: 2022 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp 179\u2013189. https:\/\/doi.org\/10.1109\/IPDPS53621.2022.00026","DOI":"10.1109\/IPDPS53621.2022.00026"},{"key":"8312_CR37","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2025.108145","volume":"176","author":"J Villarrubia","year":"2026","unstructured":"Villarrubia J, Costero L, Igual FD, Olcoz K (2026) Solving the task scheduling and GPU reconfiguration problem on MIG devices via deep reinforcement learning. Future Gener Comput Syst 176:108145. https:\/\/doi.org\/10.1016\/j.future.2025.108145","journal-title":"Future Gener Comput Syst"},{"key":"8312_CR38","unstructured":"Xiao W, Bhardwaj R, Ramjee R, Sivathanu M, Kwatra N, Han Z, et al (2018) Gandiva: introspective cluster scheduling for deep learning. In: 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18). Carlsbad, CA: USENIX Association, pp 595\u2013610. https:\/\/www.usenix.org\/conference\/osdi18\/presentation\/xiao"},{"key":"8312_CR39","unstructured":"Sanh V, Debut L, Chaumond J, Wolf T. DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter. arXiv:1910.01108"},{"key":"8312_CR40","unstructured":"Wende F, Steinke T, Cordes F (2014) Multi-threaded Kernel offloading to GPGPU using hyper-Q on kepler architecture. Takustr. 7, 14195 Berlin: ZIB. 14\u201319"},{"key":"8312_CR41","unstructured":"Zhang H, Li Y, Xiao W, Huang Y, Di X, Yin J, et al. MIGPerf: A Comprehensive Benchmark for Deep Learning Training and Inference Workloads on Multi-instance GPUs. arXiv:2301.00407"},{"key":"8312_CR42","doi-asserted-by":"publisher","unstructured":"Zhao W, Jayarajan A, Pekhimenko G (2025) Tally: non-intrusive performance isolation for concurrent deep learning workloads. In: Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 1. ASPLOS \u201925, pp 1052\u20131068. https:\/\/doi.org\/10.1145\/3669940.3707282","DOI":"10.1145\/3669940.3707282"},{"key":"8312_CR43","unstructured":"Wu B, Zhang Z, Bai Z, Liu X, Jin X (2023) Transparent GPU sharing in container clouds for deep learning workloads. In: 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23), pp 69\u201385. Boston, MA: USENIX Association. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/wu"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08312-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-026-08312-z","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08312-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T06:36:24Z","timestamp":1771310184000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-026-08312-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,17]]},"references-count":43,"journal-issue":{"issue":"4","published-online":{"date-parts":[[2026,3]]}},"alternative-id":["8312"],"URL":"https:\/\/doi.org\/10.1007\/s11227-026-08312-z","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,17]]},"assertion":[{"value":"13 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 January 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}],"article-number":"175"}}