{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T20:06:36Z","timestamp":1783713996254,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,1,26]]},"DOI":"10.1145\/3784828.3786264","type":"proceedings-article","created":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T13:19:17Z","timestamp":1769087957000},"page":"457-468","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Performance and Programmability of MPI+X Integration with CUDA, HIP, SYCL, OpenACC, and OpenMP Offloading for Supercomputing: A Case Study on Dense Matrix\u2013Vector Multiplication"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1971-4973","authenticated-orcid":false,"given":"Ezhilmathi","family":"Krishnasamy","sequence":"first","affiliation":[{"name":"University of Luxembourg, Esch-sur-Alzette, Luxembourg and University of Ljubljana, ljubljana, Slovenia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4498-020X","authenticated-orcid":false,"given":"James","family":"Throtter","sequence":"additional","affiliation":[{"name":"Simula Research Laboratory, Oslo, Norway"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3706-4414","authenticated-orcid":false,"given":"Xing","family":"Cai","sequence":"additional","affiliation":[{"name":"University of Oslo, Oslo, Norway and Simula Research Laboratory, Oslo, Norway"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7296-7817","authenticated-orcid":false,"given":"Dirk","family":"Pleiter","sequence":"additional","affiliation":[{"name":"University of Groningen, Groningen, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1790-7093","authenticated-orcid":false,"given":"Leon","family":"Kos","sequence":"additional","affiliation":[{"name":"University of Ljubljana, Ljubljana, Slovenia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7323-309X","authenticated-orcid":false,"given":"Laura","family":"Saavedra","sequence":"additional","affiliation":[{"name":"University of Santiago de Compostela, Santiago de Compostela, Spain"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9338-2834","authenticated-orcid":false,"given":"Pascal","family":"Bouvry","sequence":"additional","affiliation":[{"name":"University of Luxembourg, Esch-sur-Alzette, Luxembourg"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,1,25]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"2024. Cray Compiler Environment. https:\/\/cpe.ext.hpe.com\/docs\/cce\/man7\/intro_openacc.7.html"},{"key":"e_1_3_3_2_3_2","unstructured":"2024. OpenACC compilers profilers and debuggers are designed and available to download from multiple vendors and academic organizations.https:\/\/www.openacc.org\/tools"},{"key":"e_1_3_3_2_4_2","unstructured":"2024. The OpenMP API specification for parallel programming. https:\/\/www.openmp.org\/updates\/openmp-accelerator-support-gpus\/"},{"key":"e_1_3_3_2_5_2","unstructured":"2024. Software Technology. Web page Exascale Computing Project (ECP) Research. https:\/\/www.exascaleproject.org\/research\/#software \u201cSoftware Technology\u201d section of the ECP Research page."},{"key":"e_1_3_3_2_6_2","unstructured":"2025. European Exascale Projects. Web page EuroHPC: European Exascale section. https:\/\/exascale-projects.eu\/european-exascale\/ Overview of the network of European exascale research and innovation projects under EuroHPC JU."},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"crossref","unstructured":"Ahmad Abdelfattah David Keyes and Hatem Ltaief. 2016. Kblas: An optimized library for dense matrix-vector multiplication on gpu accelerators. ACM Transactions on Mathematical Software (TOMS) 42 3 (2016) 1\u201331.","DOI":"10.1145\/2818311"},{"key":"e_1_3_3_2_8_2","volume-title":"hipBLAS: ROCm GPU-Accelerated Basic Linear Algebra Subprograms","author":"Inc. Advanced Micro Devices,","year":"2024","unstructured":"Advanced Micro Devices, Inc.2024. hipBLAS: ROCm GPU-Accelerated Basic Linear Algebra Subprograms. AMD. https:\/\/rocm.docs.amd.com\/projects\/hipBLAS\/en\/latest\/ Part of the AMD ROCm software stack."},{"key":"e_1_3_3_2_9_2","volume-title":"hipSPARSE: ROCm GPU-Accelerated Sparse Linear Algebra Library","author":"Inc. Advanced Micro Devices,","year":"2024","unstructured":"Advanced Micro Devices, Inc.2024. hipSPARSE: ROCm GPU-Accelerated Sparse Linear Algebra Library. AMD. https:\/\/rocm.docs.amd.com\/projects\/hipSPARSE\/en\/latest\/ Part of the AMD ROCm software stack."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.5555\/323215"},{"key":"e_1_3_3_2_11_2","unstructured":"Edoardo Apra Eric\u00a0J Bylaska Wibe\u00a0A De\u00a0Jong Niranjan Govind Karol Kowalski Tjerk\u00a0P Straatsma Marat Valiev Hubertus\u00a0JJ van Dam Yuri Alexeev James Anchell et\u00a0al. 2020. NWChem: Past present and future. The Journal of chemical physics 152 18 (2020)."},{"key":"e_1_3_3_2_12_2","unstructured":"OpenMP Architecture\u00a0Review Board. 2024. OpenMP API Specification. https:\/\/www.openmp.org\/specifications\/. Accessed: 2025-10-27."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/sc41404.2022.00007"},{"key":"e_1_3_3_2_14_2","unstructured":"Sharan Chetlur Cliff Woolley Philippe Vandermersch Jonathan Cohen John Tran Bryan Catanzaro and Evan Shelhamer. 2014. cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1410.0759 (2014)."},{"key":"e_1_3_3_2_15_2","volume-title":"Intel\u00ae oneAPI Math Kernel Library - Data Parallel C++ Developer Reference","author":"Corporation Intel","year":"2025","unstructured":"Intel Corporation. 2025. Intel\u00ae oneAPI Math Kernel Library - Data Parallel C++ Developer Reference. Intel Corporation. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/onemkl\/developer-reference-dpcpp\/2025-2\/overview.html Document ID 772045."},{"key":"e_1_3_3_2_16_2","unstructured":"NVIDIA Corporation. 2025. cuBLAS Documentation. https:\/\/docs.nvidia.com\/cuda\/cublas\/. Accessed: 2025-09-15."},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-43229-4_21"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","unstructured":"Jack\u00a0J. Dongarra Jeremy Du\u00a0Croz Sven Hammarling and Iain\u00a0S. Duff. 1990. A Set of Level 3 Basic Linear Algebra Subprograms. ACM Transactions on Mathematical Software (TOMS) 16 1 (1990) 1\u201317. 10.1145\/77626.79170","DOI":"10.1145\/77626.79170"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"crossref","unstructured":"Jack\u00a0J Dongarra Piotr Luszczek and Antoine Petitet. 2003. The LINPACK Benchmark: past present and future. Concurrency and Computation: practice and experience 15 9 (2003) 803\u2013820.","DOI":"10.1002\/cpe.728"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Peter et\u00a0al. Eastman. 2017. OpenMM 7: Rapid development of high performance algorithms for molecular dynamics. PLoS computational biology 13 7 (2017) e1005659.","DOI":"10.1371\/journal.pcbi.1005659"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898717822"},{"key":"e_1_3_3_2_22_2","first-page":"131","volume-title":"2023 IEEE\/ACM 23rd International Symposium on Cluster, Cloud and Internet Computing (CCGrid)","author":"al. Chen et","year":"2023","unstructured":"Chen et al.2023. Implementing and Optimizing a GPU-Aware MPI Library for Intel GPUs: Early Experiences. In 2023 IEEE\/ACM 23rd International Symposium on Cluster, Cloud and Internet Computing (CCGrid). IEEE, 131\u2013140."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"crossref","unstructured":"Noriyuki Fujimoto. 2008. Dense matrix-vector multiplication on the CUDA architecture. Parallel Processing Letters 18 04 (2008) 511\u2013530.","DOI":"10.1142\/S0129626408003545"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/ExaMPI52011.2020.00006"},{"key":"e_1_3_3_2_25_2","volume-title":"Scientific Computing: An Introductory Survey (2nd ed.)","author":"Heath Michael\u00a0T.","year":"2001","unstructured":"Michael\u00a0T. Heath. 2001. Scientific Computing: An Introductory Survey (2nd ed.). McGraw-Hill Science\/Engineering\/Math, New York, NY. 563 pages. Includes extensive discussion of parallel numerical algorithms."},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611976724"},{"key":"e_1_3_3_2_27_2","volume-title":"Proceedings of the 10th International Conference on ICT for Intelligent Systems (ICTIS 2025)","author":"Krishnasamy Ezhilmathi","year":"2025","unstructured":"Ezhilmathi Krishnasamy and Pascal Bouvry. 2025. Comparative Performance Analysis of CUDA and OpenMP Offloading for BLAS Operations on GPU. In Proceedings of the 10th International Conference on ICT for Intelligent Systems (ICTIS 2025). New York, USA. In press. To appear December 2025."},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.1145\/3774949.3774956"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/EECT64505.2025.10966957"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACCPD56842.2022.00011"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","unstructured":"Charles\u00a0L. Lawson Richard\u00a0J. Hanson David\u00a0R. Kincaid and Fred\u00a0T. Krogh. 1979. Basic Linear Algebra Subprograms for Fortran Usage. ACM Transactions on Mathematical Software (TOMS) 5 3 (1979) 308\u2013323. 10.1145\/355841.355847","DOI":"10.1145\/355841.355847"},{"key":"e_1_3_3_2_32_2","unstructured":"Deep Learning. 2016. Ian goodfellow yoshua bengio aaron courville. The reference book for deep learning models 1 (2016)."},{"key":"e_1_3_3_2_33_2","volume-title":"NVIDIA cuBLAS Library: GPU-Accelerated Basic Linear Algebra Subprograms","author":"Corporation NVIDIA","year":"2024","unstructured":"NVIDIA Corporation. 2024. NVIDIA cuBLAS Library: GPU-Accelerated Basic Linear Algebra Subprograms. NVIDIA. https:\/\/docs.nvidia.com\/cuda\/cublas\/index.html Version 12.5, part of the NVIDIA CUDA Toolkit."},{"key":"e_1_3_3_2_34_2","volume-title":"NVIDIA cuSPARSE Library: GPU-Accelerated Sparse Matrix Operations","author":"Corporation NVIDIA","year":"2024","unstructured":"NVIDIA Corporation. 2024. NVIDIA cuSPARSE Library: GPU-Accelerated Sparse Matrix Operations. NVIDIA. https:\/\/docs.nvidia.com\/cuda\/cusparse\/index.html Version 12.5, part of the NVIDIA CUDA Toolkit."},{"key":"e_1_3_3_2_35_2","unstructured":"M Quin. 2000. parallel programming in C with MPI and OpenMP. Tata McGraw Hills edition (2000)."},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.5555\/829576"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"Ayhan\u00a0H. Saleem and Matthew\u00a0R. Norman. 2024. Accelerated Numerical Modeling of Shallow Water Flows with MPI OpenACC and GPUs. Environmental Modelling & Software 180 (2024) 106141.","DOI":"10.1016\/j.envsoft.2024.106141"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"crossref","unstructured":"Francesco Salvadore Giacomo Rossi Srikanth Sathyanarayana and Matteo Bernardini. 2024. Openmp offload toward the exascale using intel\u00ae gpu max 1550: evaluation of streams compressible solver. The Journal of Supercomputing 80 14 (2024) 21094\u201321127.","DOI":"10.1007\/s11227-024-06254-y"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2011.332"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-32152-3_14"},{"key":"e_1_3_3_2_41_2","unstructured":"AMD\u00a0ROCm Team. 2025. hipBLAS Documentation. https:\/\/rocm.docs.amd.com\/projects\/hipBLAS\/en\/latest\/index.html. Accessed: 2025-09-15."},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","unstructured":"R.\u00a0Clint Whaley and Jack\u00a0J. Dongarra. 1998. Automatically Tuned Linear Algebra Software. Proceedings of the 1998 ACM\/IEEE Conference on Supercomputing (1998) 1\u201327. 10.1145\/280759.280864","DOI":"10.1145\/280759.280864"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"crossref","unstructured":"Ao Xu and Bo-Tao Li. 2023. Multi-GPU thermal lattice Boltzmann simulations using OpenACC and MPI. International Journal of Heat and Mass Transfer 201 (2023) 123649.","DOI":"10.1016\/j.ijheatmasstransfer.2022.123649"}],"event":{"name":"SCA\/HPCAsiaWS 2026: SCA\/HPCAsia 2026 Workshops: Supercomputing Asia and International Conference on High Performance Computing in Asia Pacific Region Workshops","location":"Osaka , Japan","acronym":"SCA\/HPCAsiaWS 2026"},"container-title":["Proceedings of the Supercomputing Asia and International Conference on High Performance Computing in Asia Pacific Region Workshops"],"original-title":[],"deposited":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T13:40:00Z","timestamp":1769089200000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3784828.3786264"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,25]]},"references-count":42,"alternative-id":["10.1145\/3784828.3786264","10.1145\/3784828"],"URL":"https:\/\/doi.org\/10.1145\/3784828.3786264","relation":{},"subject":[],"published":{"date-parts":[[2026,1,25]]},"assertion":[{"value":"2026-01-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}