{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,10]],"date-time":"2026-01-10T07:55:22Z","timestamp":1768031722404,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":66,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,11,12]],"date-time":"2023-11-12T00:00:00Z","timestamp":1699747200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"Exascale Computing Project, a collaborative effort of the U.S. Department of Energy Office of Science and the National Nuclear Security Administration","award":["17-SC-20-SC"],"award-info":[{"award-number":["17-SC-20-SC"]}]},{"name":"National Energy Research Scientific Computing Center, which is supported by the Office of Science of the U.S. Department of Energy","award":["DE-AC02-05CH11231"],"award-info":[{"award-number":["DE-AC02-05CH11231"]}]},{"name":"Oak Ridge Leadership Computing Facility at the Oak Ridge National Laboratory","award":["DE-AC05-00OR22725"],"award-info":[{"award-number":["DE-AC05-00OR22725"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,11,12]]},"DOI":"10.1145\/3624062.3624177","type":"proceedings-article","created":{"date-parts":[[2023,11,10]],"date-time":"2023-11-10T13:53:39Z","timestamp":1699624419000},"page":"1007-1018","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":12,"title":["Performance Portability Evaluation of Blocked Stencil Computations on GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4596-0289","authenticated-orcid":false,"given":"Oscar","family":"Antepara","sequence":"first","affiliation":[{"name":"Lawrence Berkeley National Laboratory, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8327-5717","authenticated-orcid":false,"given":"Samuel","family":"Williams","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Laboratory, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9644-9982","authenticated-orcid":false,"given":"Hans","family":"Johansen","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Laboratory, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6347-7135","authenticated-orcid":false,"given":"Tuowen","family":"Zhao","sequence":"additional","affiliation":[{"name":"University of Utah, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5237-637X","authenticated-orcid":false,"given":"Samantha","family":"Hirsch","sequence":"additional","affiliation":[{"name":"University of Utah, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3125-9123","authenticated-orcid":false,"given":"Priya","family":"Goyal","sequence":"additional","affiliation":[{"name":"University of Utah, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3058-7573","authenticated-orcid":false,"given":"Mary","family":"Hall","sequence":"additional","affiliation":[{"name":"University of Utah, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,11,12]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"ALCF. 2023. Aurora. https:\/\/www.alcf.anl.gov\/aurora."},{"key":"e_1_3_2_2_2_1","unstructured":"AMD. 2022. AMD CDNA 2 ARCHITECTURE. https:\/\/www.amd.com\/system\/files\/documents\/amd-cdna2-white-paper.pdf."},{"key":"e_1_3_2_2_3_1","unstructured":"AMD. 2023. AMD rocProf Documentation. https:\/\/docs.amd.com\/projects\/rocprofiler\/en\/docs-5.1.0\/rocprof.html."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1155\/2009\/382638"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2015.103"},{"key":"e_1_3_2_2_6_1","unstructured":"Brick Library 2021. BrickLib Documentation. https:\/\/bricks.run\/."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2011.70"},{"key":"e_1_3_2_2_8_1","volume-title":"Auto-tuning Stencil Codes for Cache-Based Multicore Platforms. Ph.\u00a0D. Dissertation. EECS Department","author":"Datta Kaushik","unstructured":"Kaushik Datta. 2009. Auto-tuning Stencil Codes for Cache-Based Multicore Platforms. Ph.\u00a0D. Dissertation. EECS Department, University of California, Berkeley."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1137\/070693199"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"Kaushik Datta Mark Murphy Vasily Volkov Samuel Williams Jonathan Carter Leonid Oliker David Patterson John Shalf and Katherine Yelick. 2008. Stencil Computation Optimization and Auto-Tuning on State-of-the-art Multicore Architectures. In Supercomputing (SC).","DOI":"10.1109\/SC.2008.5222004"},{"key":"e_1_3_2_2_11_1","volume-title":"Introducing the Semi-stencil Algorithm. In International Conference on Parallel Processing and Applied Mathematics: Part I (PPAM). 11\u00a0pages.","author":"De\u00a0La\u00a0Cruz Ra\u00fal","year":"2010","unstructured":"Ra\u00fal De\u00a0La\u00a0Cruz, Mauricio Araya-Polo, and Jos\u00e9\u00a0Mar\u00eda Cela. 2010. Introducing the Semi-stencil Algorithm. In International Conference on Parallel Processing and Applied Mathematics: Part I (PPAM). 11\u00a0pages."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/P3HPC49587.2019.00006"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/377792.377807"},{"key":"e_1_3_2_2_14_1","first-page":"21","article-title":"Cache Optimization for Structured and Unstructured Grid Multigrid","volume":"10","author":"Douglas C.","year":"2000","unstructured":"Craig\u00a0C. Douglas, Jonathan Hu, Markus Kowarschik, Ulrich R\u00fcde, and Christian Weiss. 2000. Cache Optimization for Structured and Unstructured Grid Multigrid. Elect. Trans. Numer. Anal 10 (2000), 21\u201340.","journal-title":"Elect. Trans. Numer. Anal"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/P3HPC54578.2021.00009"},{"key":"e_1_3_2_2_16_1","volume-title":"Proc. ACM International Conference on Supercomputing (ICS).","author":"Frigo M.","unstructured":"M. Frigo and V. Strumpen. 2005. Evaluation of cache-based superscalar and cacheless vector architectures for scientific computations. In Proc. ACM International Conference on Supercomputing (ICS)."},{"key":"e_1_3_2_2_17_1","volume-title":"Compiler Construction","author":"Henretty Tom","unstructured":"Tom Henretty, Kevin Stock, Louis-No\u00ebl Pouchet, Franz Franchetti, J Ramanujam, and P Sadayappan. 2011. Data layout transformation for stencil computations on short-vector simd architectures. In Compiler Construction. Springer, 225\u2013245."},{"key":"e_1_3_2_2_18_1","volume-title":"International Conference on Supercomputing (ICS).","author":"Holewinski Justin","unstructured":"Justin Holewinski, Louis-No\u00ebl Pouchet, and P. Sadayappan. 2012. High-performance code generation for stencil computations on GPU architectures. In International Conference on Supercomputing (ICS)."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","unstructured":"Khaled\u00a0Z. Ibrahim Chao Yang and Pieter Maris. 2022. Performance Portability of Sparse Block Diagonal Matrix Multiple Vector Multiplications on GPUs. In 2022 IEEE\/ACM International Workshop on Performance Portability and Productivity in HPC (P3HPC). 58\u201367. https:\/\/doi.org\/10.1109\/P3HPC56579.2022.00011","DOI":"10.1109\/P3HPC56579.2022.00011"},{"key":"e_1_3_2_2_20_1","unstructured":"INTEL. 2023. Intel Advisor tool on Florentia-JLSE. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/tools\/oneapi\/advisor.html."},{"key":"e_1_3_2_2_21_1","unstructured":"INTEL. 2023. INTEL IRIS XE GPU ARCHITECTURE. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/oneapi\/optimization-guide-gpu\/2023-0\/intel-iris-xe-gpu-architecture.html."},{"key":"e_1_3_2_2_22_1","unstructured":"INTEL. 2023. Intel oneAPI on Florentia-JLSE. https:\/\/software.intel.com\/ONEAPI."},{"key":"e_1_3_2_2_23_1","volume-title":"A strategy for high performance in computational fluid dynamics. Ph.\u00a0D. Dissertation","author":"Jayaraj Jagan","unstructured":"Jagan Jayaraj. 2013. A strategy for high performance in computational fluid dynamics. Ph.\u00a0D. Dissertation. University of Minnesota."},{"key":"e_1_3_2_2_24_1","volume-title":"JLSE: Florentia GPU Nodes. https:\/\/www.jlse.anl.gov\/hardware-under-development","author":"JLSE.","year":"2023","unstructured":"JLSE. 2023. JLSE: Florentia GPU Nodes. https:\/\/www.jlse.anl.gov\/hardware-under-development"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2017.04.002"},{"key":"e_1_3_2_2_26_1","volume-title":"DiMEPACK - A Cache-Optimized Multigrid Library. In International Conference on Parallel and Distributed Processing Techniques and Applications (PDPTA)","author":"Kowarschik Markus","year":"2001","unstructured":"Markus Kowarschik and Christian Wei\u00df. 2001. DiMEPACK - A Cache-Optimized Multigrid Library. In International Conference on Parallel and Distributed Processing Techniques and Applications (PDPTA), volume I."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/1250734.1250761"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/P3HPC54578.2021.00008"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2019.02.005"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.7314631"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3374916"},{"key":"e_1_3_2_2_32_1","volume-title":"Technical Report DCS-TR-379. Department of Computer Science","author":"McCalpin J.","year":"1999","unstructured":"J. McCalpin and D. Wonnacott. 1999. Time skewing: A value-based approach to optimizing for memory locality. Technical Report DCS-TR-379. Department of Computer Science, Rutgers University."},{"key":"e_1_3_2_2_33_1","volume-title":"Evaluating Performance Portability of\u00a0OpenMP for SNAP on NVIDIA, Intel, and AMD GPUs Using the Roofline Methodology","author":"Mehta A.","unstructured":"Neil\u00a0A. Mehta, Rahulkumar Gayatri, Yasaman Ghadar, Christopher Knight, and Jack Deslippe. 2021. Evaluating Performance Portability of\u00a0OpenMP for SNAP on NVIDIA, Intel, and AMD GPUs Using the Roofline Methodology. In Accelerator Programming Using Directives, Sridutt Bhalachandra, Sandra Wienke, Sunita Chandrasekaran, and Guido Juckeland (Eds.). Springer International Publishing, Cham, 3\u201324."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/1513895.1513905"},{"key":"e_1_3_2_2_35_1","volume-title":"NERSC: Perlmutter GPU Nodes. https:\/\/docs.nersc.gov\/systems\/perlmutter\/","author":"NERSC.","year":"2022","unstructured":"NERSC. 2022. NERSC: Perlmutter GPU Nodes. https:\/\/docs.nersc.gov\/systems\/perlmutter\/"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2010.2"},{"key":"e_1_3_2_2_37_1","unstructured":"NVIDIA. 2020. NVIDIA A100 GPU ARCHITECTURE. https:\/\/images.nvidia.com\/aem-dam\/en-zz\/Solutions\/data-center\/nvidia-ampere-architecture-whitepaper.pdf."},{"key":"e_1_3_2_2_38_1","unstructured":"NVIDIA. 2023. NVIDIA Nsight Compute CLI Documentation. https:\/\/docs.nvidia.com\/nsight-compute\/NsightComputeCli\/index.html."},{"key":"e_1_3_2_2_39_1","unstructured":"NVIDIA. 2023. NVIDIA Nsight Documentation. https:\/\/docs.nvidia.com\/nsight-systems\/UserGuide\/index.html."},{"key":"e_1_3_2_2_40_1","volume-title":"OLCF: Crusher GPU Nodes. https:\/\/docs.olcf.ornl.gov\/systems\/crusher_quick_start_guide.html","author":"OLCF.","year":"2023","unstructured":"OLCF. 2023. OLCF: Crusher GPU Nodes. https:\/\/docs.olcf.ornl.gov\/systems\/crusher_quick_start_guide.html"},{"key":"e_1_3_2_2_41_1","unstructured":"oneAPI DPC++ Compiler 2022. DPC++ on Crusher-OLCF. https:\/\/github.com\/intel\/llvm\/releases\/tag\/2022-09."},{"key":"e_1_3_2_2_42_1","volume-title":"37th IEEE International Parallel and Distributed Processing Symposium (IPDPS).","author":"Ozturk Emin","year":"2023","unstructured":"M.\u00a0Emin Ozturk, Omid Asudeh, Gerald Sabin, P. Sadayappan, and Aravind Sukumaran-Rajam. 2023. A Performance Portability Study Using Tensor Contraction Benchmarks. In AsHES 2023: The Thirteenth International Workshop on Accelerators and Hybrid Exascale Systems (to appear). 37th IEEE International Parallel and Distributed Processing Symposium (IPDPS)."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2017.08.007"},{"key":"e_1_3_2_2_44_1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis","author":"Rawat Prashant\u00a0Singh","unstructured":"Prashant\u00a0Singh Rawat, Aravind Sukumaran-Rajam, Atanas Rountev, Fabrice Rastello, Louis-No\u00ebl Pouchet, and P. Sadayappan. 2018. Associative Instruction Reordering to Alleviate Register Pressure. In Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis (Dallas, Texas) (SC \u201918). IEEE Press, Piscataway, NJ, USA, Article 46, 13\u00a0pages. http:\/\/dl.acm.org\/citation.cfm?id=3291656.3291718"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"G. Rivera and C. Tseng. 2000. Tiling Optimizations for 3D Scientific Computations. In Supercomputing (SC).","DOI":"10.1109\/SC.2000.10015"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342004041295"},{"key":"e_1_3_2_2_47_1","volume-title":"Proc. ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI).","author":"Song Y.","unstructured":"Y. Song and Z. Li. 1999. New tiling techniques to improve cache temporal locality. In Proc. ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI)."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"crossref","unstructured":"Kevin Stock Martin Kong Tobias Grosser Louis-No\u00ebl Pouchet Fabrice Rastello Jagannathan Ramanujam and Ponnuswamy Sadayappan. 2014. A framework for enhancing data reuse via associative reordering. In ACM SIGPLAN Notices Vol.\u00a049. ACM 65\u201376.","DOI":"10.1145\/2666356.2594342"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/1989493.1989508"},{"key":"e_1_3_2_2_50_1","unstructured":"TOP 500. 2023. TOP 500 website. https:\/\/www.top500.org\/."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2017.2703149"},{"key":"e_1_3_2_2_52_1","volume-title":"TiDA: High-Level Programming Abstractions for Data Locality Management","author":"Unat Didem","unstructured":"Didem Unat, Tan Nguyen, Weiqun Zhang, Muhammed\u00a0Nufail Farooqi, Burak Bastem, George Michelogiannakis, Ann Almgren, and John Shalf. 2016. TiDA: High-Level Programming Abstractions for Data Locality Management. Springer International Publishing, Cham, 116\u2013135."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/COMPSAC.2009.82"},{"key":"e_1_3_2_2_54_1","volume-title":"Lattice Boltzmann Simulation Optimization on Leading Multicore Platforms. In Interational Conference on Parallel and Distributed Computing Systems (IPDPS).","author":"Williams S.","unstructured":"S. Williams, J. Carter, L. Oliker, J. Shalf, and K. Yelick. 2008. Lattice Boltzmann Simulation Optimization on Leading Multicore Platforms. In Interational Conference on Parallel and Distributed Computing Systems (IPDPS)."},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"crossref","unstructured":"Samuel Williams Leonid Oliker Jonathan Carter and John Shalf. 2011. Extracting ultra-scale Lattice Boltzmann performance via hierarchical and distributed auto-tuning. In Supercomputing (SC).","DOI":"10.1145\/2063384.2063458"},{"key":"e_1_3_2_2_56_1","volume-title":"Proc. Conference on Computing Frontiers.","author":"Williams S.","unstructured":"S. Williams, J. Shalf, L. Oliker, S. Kamil, P. Husbands, and K. Yelick. 2006. The potential of the Cell processor for scientific computing. In Proc. Conference on Computing Frontiers."},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498765.1498785"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2000.845979"},{"key":"e_1_3_2_2_59_1","volume-title":"Vector Folding: Improving Stencil Performance via Multi-dimensional SIMD-vector Representation. In 2015 IEEE 17th International Conference on High Performance Computing and Communications","author":"Yount C.","year":"2015","unstructured":"C. Yount. 2015. Vector Folding: Improving Stencil Performance via Multi-dimensional SIMD-vector Representation. In 2015 IEEE 17th International Conference on High Performance Computing and Communications, 2015 IEEE 7th International Symposium on Cyberspace Safety and Security, and 2015 IEEE 12th International Conference on Embedded Software and Systems. 865\u2013870."},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/WOLFHPC.2016.08"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"crossref","unstructured":"T. Zeiser G. Wellein A. Nitsure K. Iglberger U. Rude and G. Hager. 2008. Introducing a parallel cache oblivious blocking approach for the lattice Boltzmann method. Progress in Computational Fluid Dynamics 8 (2008).","DOI":"10.1504\/PCFD.2008.018088"},{"key":"e_1_3_2_2_62_1","volume-title":"Snowflake: A Lightweight Portable Stencil DSL. In 2017 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW). 795\u2013804","author":"Zhang N.","unstructured":"N. Zhang, M. Driscoll, C. Markley, S. Williams, P. Basu, and A. Fox. 2017. Snowflake: A Lightweight Portable Stencil DSL. In 2017 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW). 795\u2013804."},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/2259016.2259037"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356210"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/P3HPC.2018.00009"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/2259016.2259044"}],"event":{"name":"SC-W 2023: Workshops of The International Conference on High Performance Computing, Network, Storage, and Analysis","location":"Denver CO USA","acronym":"SC-W 2023"},"container-title":["Proceedings of the SC '23 Workshops of the International Conference on High Performance Computing, Network, Storage, and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3624062.3624177","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3624062.3624177","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T03:06:09Z","timestamp":1755745569000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3624062.3624177"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,12]]},"references-count":66,"alternative-id":["10.1145\/3624062.3624177","10.1145\/3624062"],"URL":"https:\/\/doi.org\/10.1145\/3624062.3624177","relation":{},"subject":[],"published":{"date-parts":[[2023,11,12]]},"assertion":[{"value":"2023-11-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}