{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T18:39:34Z","timestamp":1785955174123,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":59,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,2,22]],"date-time":"2022-02-22T00:00:00Z","timestamp":1645488000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000015","name":"DOE U.S. Department of Energy","doi-asserted-by":"publisher","award":["17-SC-20-SC"],"award-info":[{"award-number":["17-SC-20-SC"]}],"id":[{"id":"10.13039\/100000015","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006228","name":"Oak Ridge National Laboratory","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006228","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["CNS-2125813"],"award-info":[{"award-number":["CNS-2125813"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,2,28]]},"DOI":"10.1145\/3503222.3507708","type":"proceedings-article","created":{"date-parts":[[2022,2,22]],"date-time":"2022-02-22T20:49:01Z","timestamp":1645562941000},"page":"171-185","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":20,"title":["ValueExpert: exploring value patterns in GPU-accelerated applications"],"prefix":"10.1145","author":[{"given":"Keren","family":"Zhou","sequence":"first","affiliation":[{"name":"Rice University, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yueming","family":"Hao","sequence":"additional","affiliation":[{"name":"North Carolina State University, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"John","family":"Mellor-Crummey","sequence":"additional","affiliation":[{"name":"Rice University, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaozhu","family":"Meng","sequence":"additional","affiliation":[{"name":"Rice University, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xu","family":"Liu","sequence":"additional","affiliation":[{"name":"North Carolina State University, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,2,22]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"https:\/\/www.top500.org\/lists\/top500\/2021\/06\/ [Accessed","year":"2021","unstructured":"2020. TOP500. https:\/\/www.top500.org\/lists\/top500\/2021\/06\/ [Accessed August 4, 2021]."},{"key":"e_1_3_2_1_2_1","volume-title":"https:\/\/github.com\/ROCm-Developer-Tools\/rocprofiler [Accessed","year":"2021","unstructured":"2021. ROC-profiler. https:\/\/github.com\/ROCm-Developer-Tools\/rocprofiler [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_3_1","volume-title":"https:\/\/en.wikipedia.org\/wiki\/SHA-2 [Accessed","year":"2021","unstructured":"2021. SHA-2. https:\/\/en.wikipedia.org\/wiki\/SHA-2 [Accessed Apr 9, 2021]."},{"key":"e_1_3_2_1_4_1","volume-title":"Tensorflow: A system for large-scale machine learning. In 12th $USENIX$ symposium on operating systems design and implementation ($OSDI$ 16). 265\u2013283.","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, and Michael Isard. 2016. Tensorflow: A system for large-scale machine learning. In 12th $USENIX$ symposium on operating systems design and implementation ($OSDI$ 16). 265\u2013283."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.02513"},{"key":"e_1_3_2_1_6_1","unstructured":"Amazon Corp.. 2019. Amazon EC2 G4 Instances with NVIDIA T4 Tensor Core GPUs now available in 6 additional regions. https:\/\/aws.amazon.com\/about-aws\/whats-new\/2019\/10\/amazon-ec2-g4-instances-with-nvidia-t4-tensor-core-gpus-now-available-in-6-additional-regions [Accessed Aug 9 2021]."},{"key":"e_1_3_2_1_7_1","volume-title":"Wave propagation modules for PyTorch. https:\/\/github.com\/ar4\/deepwave [Accessed","author":"Geophysical Ausar","year":"2021","unstructured":"Ausar Geophysical. 2021. Wave propagation modules for PyTorch. https:\/\/github.com\/ar4\/deepwave [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ispass.2009.4919648"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/P3HPC49587.2019.00012"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.25080\/majora-92bf1922-003"},{"key":"e_1_3_2_1_11_1","unstructured":"Alexey Bochkovskiy Chien-Yao Wang and Hong-Yuan Mark Liao. 2020. Yolov4: Optimal speed and accuracy of object detection. arXiv preprint arXiv:2004.10934."},{"key":"e_1_3_2_1_12_1","volume-title":"The Softer Side of Exascale. https:\/\/www.nextplatform.com\/2020\/02\/14\/the-softer-side-of-exascale [Accessed","author":"Burt Jeffrey","year":"2021","unstructured":"Jeffrey Burt. 2020. The Softer Side of Exascale. https:\/\/www.nextplatform.com\/2020\/02\/14\/the-softer-side-of-exascale [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_14_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arxiv:1810.04805.","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arxiv:1810.04805."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/xsw.2013.7"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453953.3453972"},{"key":"e_1_3_2_1_17_1","volume-title":"TensorBoard: TensorFlow\u2019s visualization toolkit. https:\/\/www.tensorflow.org\/tensorboard [Accessed","author":"Google Corp.","year":"2021","unstructured":"Google Corp.. 2021. TensorBoard: TensorFlow\u2019s visualization toolkit. https:\/\/www.tensorflow.org\/tensorboard [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/tc.2019.2896628"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2016.90"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/iiswc.2015.14"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/isca45697.2020.00047"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1088\/1361-648x"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2508148.2485934"},{"key":"e_1_3_2_1_25_1","volume-title":"Scalasca, TAU, and Vampir. In Competence in High Performance Computing","author":"Kn\u00fcpfer Andreas","year":"2011","unstructured":"Andreas Kn\u00fcpfer, Christian R\u00f6ssel, Dieter Mey, Scott Biersdorff, Kai Diethelm, Dominic Eschweiler, Markus Geimer, Michael Gerndt, Daniel Lorenz, Allen Malony, Wolfgang Nagel, Yury Oleynik, Peter Philippen, Pavel Saviankou, Dirk Schmidl, Sameer Shende, Ronny Tsch\u00fcter, Michael Wagner, Bert Wesarg, and Felix Wolf. 2012. Score-P: A Joint Performance Measurement Run-Time Infrastructure for Periscope, Scalasca, TAU, and Vampir. In Competence in High Performance Computing 2011. Springer Berlin Heidelberg, 79\u201391."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2739480.2754652"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/cgo.2004.1281665"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2010.57"},{"key":"e_1_3_2_1_29_1","volume-title":"cuBLAS. https:\/\/developer.nvidia.com\/cublas [Accessed","author":"NVIDIA Corp.","year":"2021","unstructured":"NVIDIA Corp.. 2021. cuBLAS. https:\/\/developer.nvidia.com\/cublas [Accessed March 9, 2021]."},{"key":"e_1_3_2_1_30_1","volume-title":"CUDA Graph: CUDA Toolkit Documentation. https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html#cuda-graphs [Accessed","author":"NVIDIA Corp.","year":"2021","unstructured":"NVIDIA Corp.. 2021. CUDA Graph: CUDA Toolkit Documentation. https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html#cuda-graphs [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_31_1","volume-title":"NVIDIA Compute Sanitizer API. https:\/\/docs.nvidia.com\/cuda\/sanitizer-docs\/SanitizerApi\/index.html [Accessed","author":"NVIDIA Corp.","year":"2021","unstructured":"NVIDIA Corp.. 2021. NVIDIA Compute Sanitizer API. https:\/\/docs.nvidia.com\/cuda\/sanitizer-docs\/SanitizerApi\/index.html [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_32_1","volume-title":"NVIDIA CUDA Toolkit. https:\/\/developer.nvidia.com\/cuda-toolkit [Accessed","author":"NVIDIA Corp.","year":"2021","unstructured":"NVIDIA Corp.. 2021. NVIDIA CUDA Toolkit. https:\/\/developer.nvidia.com\/cuda-toolkit [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_33_1","volume-title":"NVIDIA Jetson: The AI platform for autonomous everything. https:\/\/www.nvidia.com\/en-us\/autonomous-machines\/embedded-systems [Accessed","author":"NVIDIA Corp.","year":"2021","unstructured":"NVIDIA Corp.. 2021. NVIDIA Jetson: The AI platform for autonomous everything. https:\/\/www.nvidia.com\/en-us\/autonomous-machines\/embedded-systems [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_34_1","unstructured":"NVIDIA Corp.. 2021. nvprof: CUDA Toolkit Documentation. http:\/\/docs.nvidia.com\/cuda\/profiler-users-guide\/index.html [Accessed Aug 9 2021]."},{"key":"e_1_3_2_1_35_1","volume-title":"NVIDIA Nsight Compute. https:\/\/developer.nvidia.com\/nsight-compute [Accessed","author":"NVIDIA Corporation","year":"2021","unstructured":"NVIDIA Corporation. 2021. NVIDIA Nsight Compute. https:\/\/developer.nvidia.com\/nsight-compute [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_36_1","volume-title":"NVIDIA Nsight Systems. https:\/\/developer.nvidia.com\/nsight-systems [Accessed","author":"NVIDIA Corporation","year":"2021","unstructured":"NVIDIA Corporation. 2021. NVIDIA Nsight Systems. https:\/\/developer.nvidia.com\/nsight-systems [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_37_1","volume-title":"https:\/\/docs.nvidia.com\/cuda\/cuda-compiler-driver-nvcc\/index.html [Accessed","author":"NVIDIA Corporation","year":"2021","unstructured":"NVIDIA Corporation. 2021. NVIDIA NVCC. https:\/\/docs.nvidia.com\/cuda\/cuda-compiler-driver-nvcc\/index.html [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_38_1","volume-title":"OpenMP Application Programming Interface. https:\/\/www.openmp.org\/spec-html\/5.1\/openmp.html [Accessed","author":"Architecture Review Board MP","year":"2021","unstructured":"OpenMP Architecture Review Board. 2021. OpenMP Application Programming Interface. https:\/\/www.openmp.org\/spec-html\/5.1\/openmp.html [Accessed Aug 9, 2021]."},{"key":"e_1_3_2_1_39_1","volume-title":"Pytorch: An imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703.","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, and Luca Antiga. 2019. Pytorch: An imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3296979.3192368"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1063\/5.0014475"},{"key":"e_1_3_2_1_42_1","volume-title":"Fast parallel algorithms for short-range molecular dynamics. Sandia National Labs","author":"Plimpton Steve","unstructured":"Steve Plimpton. 1993. Fast parallel algorithms for short-range molecular dynamics. Sandia National Labs., Albuquerque, NM (United States)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/lca.2014.2299539"},{"key":"e_1_3_2_1_44_1","unstructured":"Joseph Redmon. 2013\u20132016. Darknet: Open Source Neural Networks in C. http:\/\/pjreddie.com\/darknet\/ [Accessed Aug 9 2021]."},{"key":"e_1_3_2_1_45_1","volume-title":"VTune Performance Analyzer Essentials","author":"Reinders James","unstructured":"James Reinders. 2005. VTune Performance Analyzer Essentials. Intel Press."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750375"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/icse.2019.00103"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358307"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3140659.3080205"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356213"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3093336.3037729"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3173162.3177159"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/pact.2015.29"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.3354\/cr030079"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/2464996.2465022"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3378678.3391881"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.01370"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00093"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3392717.3392752"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO51591.2021.9370339"}],"event":{"name":"ASPLOS '22: 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","location":"Lausanne Switzerland","acronym":"ASPLOS '22","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGOPS ACM Special Interest Group on Operating Systems","SIGARCH ACM Special Interest Group on Computer Architecture","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503222.3507708","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503222.3507708","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503222.3507708","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:11:39Z","timestamp":1750191099000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503222.3507708"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,2,22]]},"references-count":59,"alternative-id":["10.1145\/3503222.3507708","10.1145\/3503222"],"URL":"https:\/\/doi.org\/10.1145\/3503222.3507708","relation":{},"subject":[],"published":{"date-parts":[[2022,2,22]]},"assertion":[{"value":"2022-02-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}