{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,21]],"date-time":"2026-03-21T19:23:02Z","timestamp":1774120982091,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,8,7]],"date-time":"2023-08-07T00:00:00Z","timestamp":1691366400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,8,7]]},"DOI":"10.1145\/3605573.3605640","type":"proceedings-article","created":{"date-parts":[[2023,9,13]],"date-time":"2023-09-13T16:21:16Z","timestamp":1694622076000},"page":"173-182","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Implementing OpenMP\u2019s SIMD Directive in LLVM\u2019s GPU Runtime"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1982-9092","authenticated-orcid":false,"given":"Eric","family":"Wright","sequence":"first","affiliation":[{"name":"University of Delaware, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7870-8963","authenticated-orcid":false,"given":"Johannes","family":"Doerfert","sequence":"additional","affiliation":[{"name":"Lawrence Livermore National Laboratory, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6468-6839","authenticated-orcid":false,"given":"Shilei","family":"Tian","sequence":"additional","affiliation":[{"name":"Stony Brook University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8449-8579","authenticated-orcid":false,"given":"Barbara","family":"Chapman","sequence":"additional","affiliation":[{"name":"Stony Brook University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3560-9428","authenticated-orcid":false,"given":"Sunita","family":"Chandrasekaran","sequence":"additional","affiliation":[{"name":"University of Delaware, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,9,13]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"[n. d.]. OpenACC Specification. https:\/\/www.openacc.org\/specification. Accessed: 2021-05-27."},{"key":"e_1_3_2_1_2_1","unstructured":"2021. OpenACC Programming and Best Practices Guide. openacc-standard.org. 47\u201352 pages. https:\/\/www.openacc.org\/sites\/default\/files\/inline-files\/OpenACC_Programming_Guide_0_0.pdf"},{"key":"e_1_3_2_1_3_1","volume-title":"Offloading Support for OpenMP in Clang and LLVM. In Workshop on the LLVM Compiler Infrastructure in HPC (LLVM-HPC). 1\u201311","author":"Ant\u00e3o F.","year":"2016","unstructured":"Samuel\u00a0F. Ant\u00e3o, Alexey Bataev, Arpith\u00a0C. Jacob, Gheorghe-Teodor Bercea, Alexandre\u00a0E. Eichenberger, Georgios Rokos, Matt Martineau, Tian Jin, Guray Ozen, Zehra Sura, Tong Chen, Hyojin Sung, Carlo Bertolli, and Kevin O\u2019Brien. 2016. Offloading Support for OpenMP in Clang and LLVM. In Workshop on the LLVM Compiler Infrastructure in HPC (LLVM-HPC). 1\u201311."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/2833157.2833161"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Carlo Bertolli Samuel Ant\u00e3o Alexandre\u00a0E. Eichenberger Kevin O\u2019Brien Zehra Sura Arpith\u00a0C. Jacob Tong Chen and Olivier Sallenave. 2014. Coordinating GPU Threads for OpenMP 4.0 in LLVM. In LLVM Compiler Infrastructure in HPC (LLVM-HPC). 12\u201321.","DOI":"10.1109\/LLVM-HPC.2014.10"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-69303-1_8"},{"key":"e_1_3_2_1_7_1","volume-title":"First Experiences in Performance Benchmarking with the New SPEChpc 2021 Suites. In International Symposium on Cluster, Cloud and Internet Computing (CCGrid). 675\u2013684","author":"Brunst Holger","year":"2022","unstructured":"Holger Brunst, Sunita Chandrasekaran, Florina\u00a0M Ciorba, Nick Hagerty, Robert Henschel, Guido Juckeland, Junjie Li, Ver\u00f3nica G\u00a0Melesse Vergara, Sandra Wienke, and Miguel Zavala. 2022. First Experiences in Performance Benchmarking with the New SPEChpc 2021 Suites. In International Symposium on Cluster, Cloud and Internet Computing (CCGrid). 675\u2013684."},{"key":"e_1_3_2_1_8_1","unstructured":"Diego\u00a0Luis Caballero\u00a0de Gea. 2015. SIMD@ OpenMP: a programming model approach to leverage SIMD features. (2015)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-85262-7_6"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58144-2_3"},{"key":"e_1_3_2_1_11_1","volume-title":"The TRegion Interface and Compiler Optimizations for OpenMP Target Regions. In International Workshop on OpenMP (IWOMP), Vol.\u00a011718","author":"Doerfert Johannes","year":"2019","unstructured":"Johannes Doerfert, Jose Manuel\u00a0Monsalve Diaz, and Hal Finkel. 2019. The TRegion Interface and Compiler Optimizations for OpenMP Target Regions. In International Workshop on OpenMP (IWOMP), Vol.\u00a011718. 153\u2013167."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS53621.2022.00055"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","unstructured":"Douglas Doerfler Christopher Daley and USDOE. 2020. SU3_bench: Lattice QCD SU(3) Matrix-Matrix Multiply Microbenchmark (SU3_bench) v1.0. https:\/\/doi.org\/10.11578\/dc.20200610.4","DOI":"10.11578\/dc.20200610.4"},{"key":"e_1_3_2_1_14_1","volume-title":"Accelerating Numerical Dense Linear Algebra Calculations with GPUs. Numerical Computations with GPUs","author":"Dongarra Jack","year":"2014","unstructured":"Jack Dongarra, Mark Gates, Azzam Haidar, Jakub Kurzak, Piotr Luszczek, Stanimire Tomov, and Ichitaro Yamazaki. 2014. Accelerating Numerical Dense Linear Algebra Calculations with GPUs. Numerical Computations with GPUs (2014), 1\u201326."},{"key":"e_1_3_2_1_15_1","volume-title":"Kokkos: Enabling Performance Portability Across Manycore Architectures. In Extreme Scaling Workshop (XSW). 18\u201324","author":"Edwards H\u00a0Carter","year":"2013","unstructured":"H\u00a0Carter Edwards and Christian\u00a0R Trott. 2013. Kokkos: Enabling Performance Portability Across Manycore Architectures. In Extreme Scaling Workshop (XSW). 18\u201324."},{"key":"e_1_3_2_1_16_1","volume-title":"Efficient Execution of OpenMP on GPUs. In International Symposium on Code Generation and Optimization (CGO). 41\u201352","author":"Huber Joseph","year":"2022","unstructured":"Joseph Huber, Melanie Cornelius, Giorgis Georgakoudis, Shilei Tian, Jose Manuel\u00a0Monsalve Diaz, Kuter Dinel, Barbara\u00a0M. Chapman, and Johannes Doerfert. 2022. Efficient Execution of OpenMP on GPUs. In International Symposium on Code Generation and Optimization (CGO). 41\u201352."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/HiPC.2017.00048"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-45550-1_20"},{"key":"e_1_3_2_1_19_1","volume-title":"59\u201372","author":"Klemm Michael","year":"2012","unstructured":"Michael Klemm, Alejandro Duran, Xinmin Tian, Hideki Saito, Diego Caballero, and Xavier Martorell. 2012. Extending OpenMP* with Vector Constructs for Modern Multicore SIMD Architectures.IWOMP 7312 (2012), 59\u201372."},{"key":"e_1_3_2_1_20_1","volume-title":"GPU Support. In International Workshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems (PMBS). 54\u201364","author":"Martineau Matt","year":"2016","unstructured":"Matt Martineau, Simon McIntosh-Smith, Carlo Bertolli, Arpith\u00a0C. Jacob, Samuel\u00a0F. Antao, Alexandre Eichenberger, Gheorghe-Teodor Bercea, Tong Chen, Tian Jin, Kevin O\u2019Brien, Georgios Rokos, Hyojin Sung, and Zehra Sura. 2016. Performance Analysis and Optimization of Clang\u2019s OpenMP 4.5 GPU Support. In International Workshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems (PMBS). 54\u201364."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1137\/140980260"},{"key":"e_1_3_2_1_22_1","volume-title":"OpenMP GPU Offload in Flang and LLVM. In Workshop on the LLVM Compiler Infrastructure in HPC (LLVM-HPC). 1\u20139.","author":"\u00d6zen G\u00fcray","year":"2018","unstructured":"G\u00fcray \u00d6zen, Simone Atzeni, Michael Wolfe, Annemarie Southwell, and Gary Klimowicz. 2018. OpenMP GPU Offload in Flang and LLVM. In Workshop on the LLVM Compiler Infrastructure in HPC (LLVM-HPC). 1\u20139."},{"key":"e_1_3_2_1_23_1","volume-title":"Portability and Productivity. In International Workshop on Performance, Portability and Productivity in HPC (P3HPC). 37\u201346","author":"Pennycook J","year":"2018","unstructured":"Simon\u00a0J Pennycook, Jason\u00a0D Sewall, and Jeff\u00a0R Hammond. 2018. Evaluating the Impact of Proposed OpenMP 5.0 Features on Performance, Portability and Productivity. In International Workshop on Performance, Portability and Productivity in HPC (P3HPC). 37\u201346."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-85262-7_11"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Shilei Tian Johannes Doerfert and Barbara\u00a0M. Chapman. 2020. Concurrent Execution of Deferred OpenMP Target Tasks with Hidden Helper Threads. In Languages and Compilers for Parallel Computing (LCPC). 41\u201356.","DOI":"10.1007\/978-3-030-95953-1_4"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1147\/JRD.2019.2962428"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1002\/qua.25851"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-43659-3_20"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3468267.3470576"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2016.50"}],"event":{"name":"ICPP 2023: 52nd International Conference on Parallel Processing","location":"Salt Lake City UT USA","acronym":"ICPP 2023"},"container-title":["Proceedings of the 52nd International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3605573.3605640","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3605573.3605640","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:49:04Z","timestamp":1750182544000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3605573.3605640"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,7]]},"references-count":30,"alternative-id":["10.1145\/3605573.3605640","10.1145\/3605573"],"URL":"https:\/\/doi.org\/10.1145\/3605573.3605640","relation":{},"subject":[],"published":{"date-parts":[[2023,8,7]]},"assertion":[{"value":"2023-09-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}