{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T01:35:15Z","timestamp":1773192915961,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,2,25]],"date-time":"2023-02-25T00:00:00Z","timestamp":1677283200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,2,25]]},"DOI":"10.1145\/3582514.3582517","type":"proceedings-article","created":{"date-parts":[[2023,2,24]],"date-time":"2023-02-24T17:27:41Z","timestamp":1677259661000},"page":"39-49","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Harmonic CUDA: Asynchronous Programming on GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7397-3012","authenticated-orcid":false,"given":"Jonathan D.","family":"Wapman","sequence":"first","affiliation":[{"name":"University of California, Davis, Davis, CA, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2189-4026","authenticated-orcid":false,"given":"Sean","family":"Treichler","sequence":"additional","affiliation":[{"name":"NVIDIA, Santa Clara, CA, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1523-9199","authenticated-orcid":false,"given":"Serban D.","family":"Porumbescu","sequence":"additional","affiliation":[{"name":"University of California, Davis, Davis, CA, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6582-8237","authenticated-orcid":false,"given":"John D.","family":"Owens","sequence":"additional","affiliation":[{"name":"University of California, Davis, Davis, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,2,25]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-287-134-3_7"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063400"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/2555243.2555258"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2018.022071134"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485008"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2021.3113475"},{"key":"e_1_3_2_1_7_1","volume-title":"Lee Howes, Kirk Shoop, Michael Garland, Eric Niebler, and Bryce Adelstein Lelbach.","author":"Dominiak Micha\u0142","year":"2022","unstructured":"Micha\u0142 Dominiak , Georgy Evtushenko , Lewis Baker , Lucian Radu Teodorescu , Lee Howes, Kirk Shoop, Michael Garland, Eric Niebler, and Bryce Adelstein Lelbach. 2022 . std::execution. C++ Standards Committee Papers . https:\/\/www.open-std.org\/jtc1\/sc22\/wg21\/docs\/papers\/2022\/p2300r5.html Micha\u0142 Dominiak, Georgy Evtushenko, Lewis Baker, Lucian Radu Teodorescu, Lee Howes, Kirk Shoop, Michael Garland, Eric Niebler, and Bryce Adelstein Lelbach. 2022. std::execution. C++ Standards Committee Papers. https:\/\/www.open-std.org\/jtc1\/sc22\/wg21\/docs\/papers\/2022\/p2300r5.html"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1201\/9781003033707-22"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/micro.2016.7783759"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/3294771.3294869"},{"key":"e_1_3_2_1_11_1","volume-title":"Cooperative Groups: Flexible CUDA Thread Programming. https:\/\/developer.nvidia.com\/blog\/cooperative-groups\/","author":"Harris Mark","year":"2017","unstructured":"Mark Harris and Kyrylo Perelygin . 2017 . Cooperative Groups: Flexible CUDA Thread Programming. https:\/\/developer.nvidia.com\/blog\/cooperative-groups\/ Mark Harris and Kyrylo Perelygin. 2017. Cooperative Groups: Flexible CUDA Thread Programming. https:\/\/developer.nvidia.com\/blog\/cooperative-groups\/"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358275"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2010.04.115"},{"key":"e_1_3_2_1_14_1","volume-title":"CUTLASS: Fast Linear Algebra in CUDA C++. https:\/\/devblogs.nvidia.com\/cutlass-linear-algebra-cuda\/","author":"Kerr Andrew","year":"2017","unstructured":"Andrew Kerr , Duane Merrill , Julien Demouth , and John Tran . 2017 . CUTLASS: Fast Linear Algebra in CUDA C++. https:\/\/devblogs.nvidia.com\/cutlass-linear-algebra-cuda\/ Andrew Kerr, Duane Merrill, Julien Demouth, and John Tran. 2017. CUTLASS: Fast Linear Algebra in CUDA C++. https:\/\/devblogs.nvidia.com\/cutlass-linear-algebra-cuda\/"},{"key":"e_1_3_2_1_15_1","unstructured":"Ronny Krashinsky Olivier Giroux Stephen Jones NickStam and Sridhar Ramaswamy. 2020. NVIDIA Ampere Architecture In-Depth. https:\/\/developer.nvidia.com\/blog\/nvidia-ampere-architecture-in-depth\/.  Ronny Krashinsky Olivier Giroux Stephen Jones NickStam and Sridhar Ramaswamy. 2020. NVIDIA Ampere Architecture In-Depth. https:\/\/developer.nvidia.com\/blog\/nvidia-ampere-architecture-in-depth\/."},{"key":"e_1_3_2_1_16_1","unstructured":"MathWorks Corporation. 2022. Simulink. https:\/\/www.mathworks.com\/help\/simulink\/index.html  MathWorks Corporation. 2022. Simulink. https:\/\/www.mathworks.com\/help\/simulink\/index.html"},{"key":"e_1_3_2_1_17_1","volume-title":"2013--2022","author":"Merrill Duane","year":"2013","unstructured":"Duane Merrill . 2013--2022 . CUB : Flexible Library of Cooperative Threadblock Primitives and Other Utilities for CUDA Kernel Programming . ( 2013 --2022). https:\/\/github.com\/NVIDIA\/cub. Duane Merrill. 2013--2022. CUB: Flexible Library of Cooperative Threadblock Primitives and Other Utilities for CUDA Kernel Programming. (2013--2022). https:\/\/github.com\/NVIDIA\/cub."},{"key":"e_1_3_2_1_18_1","unstructured":"National Instruments Corporation. 2022. LabVIEW Documentation. https:\/\/www.ni.com\/docs\/en-US\/bundle\/labview\/page\/lvhelp\/labview_help.html  National Instruments Corporation. 2022. LabVIEW Documentation. https:\/\/www.ni.com\/docs\/en-US\/bundle\/labview\/page\/lvhelp\/labview_help.html"},{"key":"e_1_3_2_1_19_1","unstructured":"NVIDIA Corporation. 2020. NVIDIA H100 Tensor Core GPU Architecture. https:\/\/resources.nvidia.com\/en-us-tensor-core.  NVIDIA Corporation. 2020. NVIDIA H100 Tensor Core GPU Architecture. https:\/\/resources.nvidia.com\/en-us-tensor-core."},{"key":"e_1_3_2_1_20_1","unstructured":"NVIDIA Corporation. 2022. CUDA cuBLAS Library (v11.6). http:\/\/developer.nvidia.com\/cublas.  NVIDIA Corporation. 2022. CUDA cuBLAS Library (v11.6). http:\/\/developer.nvidia.com\/cublas."},{"key":"e_1_3_2_1_21_1","unstructured":"NVIDIA Corporation. 2022. CUDA Samples. https:\/\/github.com\/NVIDIA\/cuda-samples.  NVIDIA Corporation. 2022. CUDA Samples. https:\/\/github.com\/NVIDIA\/cuda-samples."},{"key":"e_1_3_2_1_22_1","unstructured":"NVIDIA Corporation. 2022. libcu++: The C++ Standard Library for Your Entire System. https:\/\/nvidia.github.io\/libcudacxx\/ Version 1.8.1.  NVIDIA Corporation. 2022. libcu++: The C++ Standard Library for Your Entire System. https:\/\/nvidia.github.io\/libcudacxx\/ Version 1.8.1."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304025"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2491956.2462176"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45937-5_14"}],"event":{"name":"PMAM'23: 14th International Workshop on Programming Models and Applications for Multicores and Manycores","location":"Montreal QC Canada","acronym":"PMAM'23","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing","SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 14th International Workshop on Programming Models and Applications for Multicores and Manycores"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3582514.3582517","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:47:14Z","timestamp":1750178834000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3582514.3582517"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,25]]},"references-count":25,"alternative-id":["10.1145\/3582514.3582517","10.1145\/3582514"],"URL":"https:\/\/doi.org\/10.1145\/3582514.3582517","relation":{},"subject":[],"published":{"date-parts":[[2023,2,25]]},"assertion":[{"value":"2023-02-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}