{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T08:12:38Z","timestamp":1759133558232,"version":"3.41.0"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2016,10,13]],"date-time":"2016-10-13T00:00:00Z","timestamp":1476316800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100000780","name":"European Commission","doi-asserted-by":"publisher","award":["644312 - RAPID - H2020-ICT-2014\/H2020-ICT-2014-1"],"award-info":[{"award-number":["644312 - RAPID - H2020-ICT-2014\/H2020-ICT-2014-1"]}],"id":[{"id":"10.13039\/501100000780","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2017,10]]},"DOI":"10.1007\/s10766-016-0462-1","type":"journal-article","created":{"date-parts":[[2016,10,13]],"date-time":"2016-10-13T07:17:13Z","timestamp":1476343033000},"page":"1142-1163","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":23,"title":["On the Virtualization of CUDA Based GPU Remoting on ARM and X86 Machines in the GVirtuS Framework"],"prefix":"10.1007","volume":"45","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4767-2045","authenticated-orcid":false,"given":"Raffaele","family":"Montella","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giulio","family":"Giunta","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giuliano","family":"Laccetti","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Marco","family":"Lapegna","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Carlo","family":"Palmieri","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Carmine","family":"Ferraro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Valentina","family":"Pelliccia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cheol-Ho","family":"Hong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ivor","family":"Spence","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dimitrios S.","family":"Nikolopoulos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,10,13]]},"reference":[{"key":"462_CR1","doi-asserted-by":"crossref","unstructured":"Armand, F., Gien, M., Maign, G., Mardinian, G.: Shared device driver model for virtualized mobile handsets. In: Proceedings of the First Workshop on Virtualization in Mobile Computing, pp. 12\u201316. ACM (2008)","DOI":"10.1145\/1622103.1622104"},{"key":"462_CR2","doi-asserted-by":"crossref","unstructured":"Bairoch, A.M., Apweiler, R., Wu, C.H., Barker, W.C., Boeckmann, B., Ferro Rojas, S., Gasteiger, E., et al.: The universal protein resource (UniProt). Nucleic Acids Res. 33(Database issue), D154\u2013D159 (2005)","DOI":"10.1093\/nar\/gki070"},{"key":"462_CR3","unstructured":"Bell, N., Garland, M.: Efficient sparse matrix-vector multiplication on CUDA. NVIDIA Technical Report NVR-2008-004, Nvidia Corporation (2008)"},{"key":"462_CR4","doi-asserted-by":"crossref","unstructured":"Caruso P.G. Laccetti, Lapegna, M.: A performance contract system in a grid enabling, component based programming environment. In: Advances in Grid Computing-EGC 2005, LNCS, vol. 3470, pp. 982\u2013992. Springer (2005)","DOI":"10.1007\/11508380_100"},{"key":"462_CR5","unstructured":"Castello, A., Duato, J., Mayo, R., Pena, A.J., Quintana-Ort, E.S., Roca, V., Silla, F.: On the use of remote GPUs and low-power processors for the acceleration of scientific applications. In: The Fourth International Conference on Smart Grids, Green Communications and IT Energy-aware Technologies (ENERGY), pp. 57\u201362 (2014)"},{"issue":"1","key":"462_CR6","doi-asserted-by":"crossref","first-page":"46","DOI":"10.1109\/99.660313","volume":"5","author":"L Dagum","year":"1998","unstructured":"Dagum, L., Enon, R.: OpenMP: an industry standard API for shared-memory programming. IEEE Comput. Sci. Eng. 5(1), 46\u201355 (1998)","journal-title":"IEEE Comput. Sci. Eng."},{"key":"462_CR7","doi-asserted-by":"crossref","unstructured":"Di Lauro, R., Giannone, F., Ambrosio, L., Montella, R.: Virtualizing general purpose GPUs for high performance cloud computing: an application to a fluid simulator. In: IEEE 10th International Symposium on Proceedings of Parallel and Distributed Processing with Applications (ISPA), pp. 863\u2013864 (2012)","DOI":"10.1109\/ISPA.2012.136"},{"key":"462_CR8","doi-asserted-by":"crossref","unstructured":"Di Lauro, R., Lucarelli, F., Montella, R.: SIaaS-sensing instrument as a service using cloud computing to turn physical instrument into ubiquitous service. In: 2012 IEEE 10th International Symposium on Parallel and Distributed Processing with Applications. IEEE, pp. 861\u2013862 (2012)","DOI":"10.1109\/ISPA.2012.135"},{"key":"462_CR9","doi-asserted-by":"crossref","unstructured":"Foster, I., Zhao, Y., Raicu, I., Lu, S.: Cloud computing and grid computing 360-degree compared. In: IEEE Grid Computing Environments Workshop GCE 08, pp. 1\u201310 (2008)","DOI":"10.1109\/GCE.2008.4738445"},{"issue":"1","key":"462_CR10","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1016\/j.envsoft.2006.05.024","volume":"22","author":"G Giunta","year":"2007","unstructured":"Giunta, G., Mariani, P., Montella, R., Riccio, A.: pPOM: a nested, scalable, parallel and Fortran 90 implementation of the Princeton Ocean Model. Environ. Model. Softw. 22(1), 117\u2013122 (2007)","journal-title":"Environ. Model. Softw."},{"issue":"4","key":"462_CR11","doi-asserted-by":"crossref","first-page":"13","DOI":"10.1109\/MM.2008.57","volume":"28","author":"M Garland","year":"2008","unstructured":"Garland, M., Le Grand, S., Nickolls, J., Anderson, J., Hardwick, J., Morton, S., Phillips, E., Zhang, Y., Volkov, V.: Parallel computing experiences with CUDA. IEEE Micro 28(4), 13\u201327 (2008)","journal-title":"IEEE Micro"},{"key":"462_CR12","doi-asserted-by":"crossref","unstructured":"Giunta, G., Montella, R., Agrillo, G., Coviello, G.: A GPGPU transparent virtualization component for high performance computing clouds. In: EuroPar 2010 Parallel Processing, LNCS, vol. 6271, no. 2, pp. 379\u2013391. Springer (2010)","DOI":"10.1007\/978-3-642-15277-1_37"},{"key":"462_CR13","doi-asserted-by":"crossref","unstructured":"Giunta, G., Montella, R., Laccetti, G., Isaila, F., Blas, F.: A GPU accelerated high performance cloud computing infrastructure for grid computing based virtual environmental laboratory. Adv. Grid Comput. 35\u201343 (2011)","DOI":"10.5772\/14594"},{"key":"462_CR14","doi-asserted-by":"crossref","unstructured":"Gropp, W.: MPICH2: a new start for MPI implementations. In: Recent Advances in Parallel Virtual Machine and Message Passing Interface 2002, LNCS, vol. 2474, p. 7. Springer (2002)","DOI":"10.1007\/3-540-45825-5_5"},{"key":"462_CR15","doi-asserted-by":"crossref","unstructured":"Gupta, V., Gavrilovska, A., Schwan, K., Kharche, H., Tolia, N., Talwar, V., Ranganathan, P.: GViM: GPU-accelerated virtual machines. In: Proceedings of the 3rd ACM Workshop on System-Level Virtualization for High Performance Computing, pp. 17\u201324. ACM (2009)","DOI":"10.1145\/1519138.1519141"},{"key":"462_CR16","volume-title":"NVIDIA GRID: Graphics Accelerated VDI with the Visual Performance of a Workstation","author":"A Herrera","year":"2014","unstructured":"Herrera, A.: NVIDIA GRID: Graphics Accelerated VDI with the Visual Performance of a Workstation. Nvidia Corp, Santa Clara (2014)"},{"key":"462_CR17","unstructured":"Kawai, A., Yasuoka, K., Yoshikawa, K., Narumi, T.: Distributed-shared CUDA: virtualization of large-scale GPU systems for programmability and reliability (2012)"},{"key":"462_CR18","doi-asserted-by":"crossref","unstructured":"Karunadasa, N.P., Ranasinghe, D.N.: Accelerating high performance applications with CUDA and MPI. In: 2009 International Conference on Industrial and Information Systems (ICIIS), pp. 331\u2013336. IEEE (2009)","DOI":"10.1109\/ICIINFS.2009.5429842"},{"key":"462_CR19","doi-asserted-by":"crossref","unstructured":"Kehne, J., Metter, J., Bellosa, F.: GPUswap: enabling oversubscription of GPU memory through transparent swapping. In: Proceedings of the 11th ACM SIGPLAN\/SIGOPS International Conference on Virtual Execution Environments, pp. 65\u201377. ACM (2015)","DOI":"10.1145\/2731186.2731192"},{"key":"462_CR20","doi-asserted-by":"crossref","unstructured":"Laccetti, G., Montella, R., Palmieri, C., Pelliccia, V.: The high performance internet of things: using GVirtuS to share high-end GPUs with ARM based cluster computing nodes. In: Parallel Processing and Applied Mathematics 2013, LNCS, vol. 8384, pp. 734\u2013744. Springer, Berlin, Heidelberg (2013)","DOI":"10.1007\/978-3-642-55224-3_69"},{"key":"462_CR21","doi-asserted-by":"crossref","unstructured":"Ligowski, L., Rudnicki, W.: An efficient implementation of Smith\u2013Waterman algorithm on GPU using CUDA, for massively parallel scanning of sequence databases. In: IEEE International Symposium on Parallel and Distributed Processing 2009, IPDPS 2009, pp. 1\u20138. IEEE (2009)","DOI":"10.1109\/IPDPS.2009.5160931"},{"issue":"1","key":"462_CR22","doi-asserted-by":"crossref","first-page":"93","DOI":"10.1186\/1756-0500-3-93","volume":"3","author":"Y Liu","year":"2010","unstructured":"Liu, Y., Schmidt, B., Maskell, D.L.: CUDASW++ 2.0: enhanced Smith\u2013Waterman protein database search on CUDA-enabled GPUs based on SIMT and virtualized SIMD abstractions. BMC Res. Notes 3(1), 93 (2010)","journal-title":"Notes"},{"issue":"2","key":"462_CR23","first-page":"1","volume":"9","author":"SA Manavski","year":"2008","unstructured":"Manavski, S.A., Valle, G.: CUDA compatible GPU cards as efficient hardware accelerators for Smith\u2013Waterman sequence alignment. BMC Bioinf. 9(2), 1 (2008)","journal-title":"BMC Bioinf."},{"key":"462_CR24","unstructured":"Martinez-Noriega, E.J., Josafat, E., Kawai, A., Yoshikawa, K., Yasuoka, K., Narumi, T.: CUDA Enabled for Android Tablets through DS-CUDA (2013)"},{"key":"462_CR25","doi-asserted-by":"crossref","unstructured":"Montella, R., Foster, I.: Using hybrid grid\/cloud computing technologies for environmental data elastic storage, processing, and provisioning. In: Handbook of Cloud Computing, pp. 595\u2013618. Springer, USA (2010)","DOI":"10.1007\/978-1-4419-6524-0_26"},{"key":"462_CR26","doi-asserted-by":"crossref","unstructured":"Montella, R., Coviello, G., Giunta, G., Laccetti, G., Isaila, F., Blas, J.G.: A general-purpose virtualization service for HPC on cloud computing: an application to GPUs. In: International Conference on Parallel Processing and Applied Mathematics, pp. 740\u2013749. Springer, Berlin, Heidelberg (2011)","DOI":"10.1007\/978-3-642-31464-3_75"},{"issue":"1","key":"462_CR27","doi-asserted-by":"crossref","first-page":"139","DOI":"10.1007\/s10586-013-0341-0","volume":"17","author":"R Montella","year":"2014","unstructured":"Montella, R., Giunta, G., Laccetti, G.: Virtualizing high-end GPGPUs on ARM clusters for the next generation of high performance cloud computing. Cluster Comput. 17(1), 139\u2013152 (2014)","journal-title":"Cluster Comput."},{"issue":"16","key":"462_CR28","doi-asserted-by":"crossref","first-page":"4423","DOI":"10.1002\/cpe.3540","volume":"27","author":"R Montella","year":"2015","unstructured":"Montella, R., Kelly, D., Xiong, W., Brizius, A., Elliott, J., Madduri, R., Maheshwari, K., et al.: FACE IT: A science gateway for food security research. Concurr. Comput. Pract. Exp. 27(16), 4423\u20134436 (2015)","journal-title":"Concurr. Comput. Pract. Exp."},{"key":"462_CR29","doi-asserted-by":"crossref","unstructured":"Montella, R., Giunta, G., Laccetti, G., Lapegna, M., Palmieri, C., Ferraro, C., Pelliccia, V.: Virtualizing CUDA enabled GPGPUs on ARM clusters. In: Parallel Processing in and Applied Mathematics 2015, LNCS, vol. 9574, Springer, Berlin, Heidelberg (2016)","DOI":"10.1007\/978-3-319-32152-3_1"},{"key":"462_CR30","doi-asserted-by":"crossref","unstructured":"Pham, Q., Malik, T., Foster, I., Di Lauro, R., Montella, R., SOLE: linking research papers with science objects. In: Provenance and Annotation of Data and Processes 2012, LNCS, vol. 7525, pp. 203\u2013208. Springer, Berlin, Heidelberg (2012)","DOI":"10.1007\/978-3-642-34222-6_16"},{"key":"462_CR31","doi-asserted-by":"crossref","unstructured":"Prades, J., Reao, C., Silla, F.: CUDA acceleration for Xen virtual machines in infiniband clusters with rCUDA. In: Proceedings of the 21st ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, p. 35. ACM (2016)","DOI":"10.1145\/2851141.2851181"},{"key":"462_CR32","doi-asserted-by":"crossref","first-page":"322","DOI":"10.1016\/j.future.2013.07.013","volume":"36","author":"N Rajovic","year":"2014","unstructured":"Rajovic, N., Rico, A., Puzovic, N., Adeniyi-Jones, C., Ramirez, A.: Tibidabo: making the case for an ARM-based HPC system. Fut. Gener. Comput. Syst. 36, 322\u2013334 (2014)","journal-title":"Fut. Gener. Comput. Syst."},{"key":"462_CR33","doi-asserted-by":"crossref","unstructured":"Reao, C., Mayo, R., Quintana-Orti, E.S., Silla, F., Duato, J., Pea, A.J.: Influence of InfiniBand FDR on the performance of remote GPU virtualization. In: Proceedings of the 2013 IEEE International Conference on Cluster Computing, Indianapolis, USA (2013)","DOI":"10.1109\/CLUSTER.2013.6702662"},{"key":"462_CR34","doi-asserted-by":"crossref","unstructured":"Reao, C., Silla, F., Pena, A.J., Shainer, G., Schultz, S., Castello, A., Quintana-Orti, E.S., Duato, J.: POSTER: Boosting the performance of remote GPU virtualization using InfiniBand connect-IB and PCIe 3.0. In: 2014 IEEE International Conference on Cluster Computing (CLUSTER), pp. 266\u2013267. IEEE (2014)","DOI":"10.1109\/CLUSTER.2014.6968737"},{"issue":"6","key":"462_CR35","doi-asserted-by":"crossref","first-page":"804","DOI":"10.1109\/TC.2011.112","volume":"61","author":"L Shi","year":"2012","unstructured":"Shi, L., Chen, H., Sun, J., Li, K.: vCUDA: GPU-accelerated high-performance computing in virtual machines. IEEE Trans. Comput. 61(6), 804\u2013816 (2012)","journal-title":"IEEE Trans. Comput."},{"issue":"10","key":"462_CR36","doi-asserted-by":"crossref","first-page":"1370","DOI":"10.1016\/j.jpdc.2008.05.014","volume":"68","author":"C Shuai","year":"2008","unstructured":"Shuai, C., Boyer, M., Meng, J., Tarjan, D., Sheaffer, J.W., Skadron, K.: A performance study of general-purpose applications on graphics processors using CUDA. J. Parallel Distrib. Comput. 68(10), 1370\u20131380 (2008)","journal-title":"J. Parallel Distrib. Comput."},{"key":"462_CR37","doi-asserted-by":"crossref","unstructured":"Shuai, C., Boyer, M., Meng, J., Tarjan, D., Sheaffer, J.W., Lee, S.-H., Skadron, K.: Rodinia: a benchmark suite for heterogeneous computing. In: Proceedings of the IEEE International Symposium on Workload Characterization\u2014IISWC 2009, pp. 44\u201354 (2009)","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"462_CR38","doi-asserted-by":"crossref","unstructured":"Sourouri, M., Gillberg, T., Baden, S.B., Cai, X.: Effective multi-GPU communication using multiple CUDA streams and threads. In: 2014 20th IEEE International Conference on Parallel and Distributed Systems (ICPADS), pp. 981\u2013986. IEEE (2014)","DOI":"10.1109\/PADSW.2014.7097919"},{"key":"462_CR39","unstructured":"Szafaryn, L.G., Skadron, K., Saucerman, J.J.: Experiences accelerating MATLAB systems biology applications. In: Proceedings of the workshop on biomedicine in computing: systems, architectures, and circuits (BiC) 2009. In: Conjunction with the 36th IEEE\/ACM International Symposium on Computer Architecture (ISCA) (2009)"},{"key":"462_CR40","doi-asserted-by":"crossref","unstructured":"Volkov, V., Demmel, J.W.: Benchmarking GPUs to tune dense linear algebra. In: International Conference for High Performance Computing, Networking, Storage and Analysis 2008, SC 2008, pp. 1\u201311. IEEE (2008)","DOI":"10.1109\/SC.2008.5214359"},{"issue":"1","key":"462_CR41","doi-asserted-by":"crossref","first-page":"266","DOI":"10.1016\/j.cpc.2010.06.035","volume":"182","author":"C Yang","year":"2011","unstructured":"Yang, C., Huang, C., Lin, C.: Hybrid CUDA, OpenMP, and MPI parallel programming on multicore GPU clusters. Comput. Phys. Commun. 182(1), 266\u2013269 (2011)","journal-title":"Comput. Phys. Commun."}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10766-016-0462-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-016-0462-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-016-0462-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,11]],"date-time":"2025-06-11T10:00:03Z","timestamp":1749636003000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10766-016-0462-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,10,13]]},"references-count":41,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2017,10]]}},"alternative-id":["462"],"URL":"https:\/\/doi.org\/10.1007\/s10766-016-0462-1","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"type":"print","value":"0885-7458"},{"type":"electronic","value":"1573-7640"}],"subject":[],"published":{"date-parts":[[2016,10,13]]}}}