{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,10]],"date-time":"2026-04-10T12:00:54Z","timestamp":1775822454195,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,2,17]],"date-time":"2021-02-17T00:00:00Z","timestamp":1613520000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the U.S. Department of Energy's Office of Science and National Nuclear Security Administration","award":["17-SC-20-SC"],"award-info":[{"award-number":["17-SC-20-SC"]}]},{"name":"the U.S. Department of Energy's Office of Science","award":["DE-AC05-00OR22725, DE-AC02-06CH11357"],"award-info":[{"award-number":["DE-AC05-00OR22725, DE-AC02-06CH11357"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,2,17]]},"DOI":"10.1145\/3437801.3441598","type":"proceedings-article","created":{"date-parts":[[2021,2,20]],"date-time":"2021-02-20T23:04:20Z","timestamp":1613862260000},"page":"304-317","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":18,"title":["Improving communication by optimizing on-node data movement with data layout"],"prefix":"10.1145","author":[{"given":"Tuowen","family":"Zhao","sequence":"first","affiliation":[{"name":"University of Utah"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mary","family":"Hall","sequence":"additional","affiliation":[{"name":"University of Utah"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hans","family":"Johansen","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Samuel","family":"Williams","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,2,17]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Summit Training Workshop. https:\/\/www.olcf.ornl.gov\/wp-content\/uploads\/2018\/12\/summit_workshop_CUDA-Aware-MPI.pdf","author":"Abbott Steve","year":"2018","unstructured":"Steve Abbott . 2018 . GPUDIRECT, CUDA Aware MPI, and CUDA IPC . In Summit Training Workshop. https:\/\/www.olcf.ornl.gov\/wp-content\/uploads\/2018\/12\/summit_workshop_CUDA-Aware-MPI.pdf Steve Abbott. 2018. GPUDIRECT, CUDA Aware MPI, and CUDA IPC. In Summit Training Workshop. https:\/\/www.olcf.ornl.gov\/wp-content\/uploads\/2018\/12\/summit_workshop_CUDA-Aware-MPI.pdf"},{"key":"e_1_3_2_1_2_1","volume-title":"Recent Advances in Parallel Virtual Machine and Message Passing Interface","author":"Balaji Pavan","unstructured":"Pavan Balaji , Anthony Chan , William Gropp , Rajeev Thakur , and Ewing Lusk . 2008. Non-data-communication Overheads in MPI: Analysis on Blue Gene\/P . In Recent Advances in Parallel Virtual Machine and Message Passing Interface , Alexey Lastovetsky, Tahar Kechadi, and Jack Dongarra (Eds.). Springer Berlin Heidelberg , Berlin, Heidelberg , 13--22. Pavan Balaji, Anthony Chan, William Gropp, Rajeev Thakur, and Ewing Lusk. 2008. Non-data-communication Overheads in MPI: Analysis on Blue Gene\/P. In Recent Advances in Parallel Virtual Machine and Message Passing Interface, Alexey Lastovetsky, Tahar Kechadi, and Jack Dongarra (Eds.). Springer Berlin Heidelberg, Berlin, Heidelberg, 13--22."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/HiPC.2013.6799131"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2017.08.006"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2008.4536305"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/2851141.2851157"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/582034.582084"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2019.00045"},{"key":"e_1_3_2_1_9_1","unstructured":"S. Hunold and A. Carpen-Amarie. 2019. MPI Benchmarking Revisited: Experimental Design and Reproducibility. https:\/\/arxiv.org\/pdf\/1505.07734.pdf  S. Hunold and A. Carpen-Amarie. 2019. MPI Benchmarking Revisited: Experimental Design and Reproducibility. https:\/\/arxiv.org\/pdf\/1505.07734.pdf"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/296806.296830"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER.2015.41"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1654059.1654096"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/1654059.1654096"},{"key":"e_1_3_2_1_16_1","volume-title":"Palmer and Jarek Nieplocha","author":"Bruce","year":"2002","unstructured":"Bruce J. Palmer and Jarek Nieplocha . 2002 . Efficient Algorithms for Ghost Cell Updates on Two Classes of MPP Architectures. In IASTED PDCS. Bruce J. Palmer and Jarek Nieplocha. 2002. Efficient Algorithms for Ghost Cell Updates on Two Classes of MPP Architectures. In IASTED PDCS."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10915-011-9531-1"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Sean Treichler Michael Bauer Ankit Bhagatwala etal 2017. S3D-Legion: An Exascale Software for Direct Numerical Simulation of Turbulent Combustion with Complex Multicomponent Chemistry. In Exascale Scientific Applications. Chapman and Hall\/CRC.  Sean Treichler Michael Bauer Ankit Bhagatwala et al. 2017. S3D-Legion: An Exascale Software for Direct Numerical Simulation of Turbulent Combustion with Complex Multicomponent Chemistry. In Exascale Scientific Applications. Chapman and Hall\/CRC.","DOI":"10.1201\/b21930-12"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cpc.2010.04.018"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10586-013-0309-0"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis","author":"Williams Samuel","unstructured":"Samuel Williams , Dhiraj D. Kalamkar , Amik Singh , Anand M. Deshpande , Brian Van Straalen, Mikhail Smelyanskiy, Ann Almgren, Pradeep Dubey, John Shalf, and Leonid Oliker. 2012. Optimization of Geometric Multigrid for Emerging Multi- and Manycore Processors . In Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis ( Salt Lake City, Utah) (SC '12). IEEE Computer Society Press, Washington, DC, USA, Article 96, 11 pages. Samuel Williams, Dhiraj D. Kalamkar, Amik Singh, Anand M. Deshpande, Brian Van Straalen, Mikhail Smelyanskiy, Ann Almgren, Pradeep Dubey, John Shalf, and Leonid Oliker. 2012. Optimization of Geometric Multigrid for Emerging Multi- and Manycore Processors. In Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis (Salt Lake City, Utah) (SC '12). IEEE Computer Society Press, Washington, DC, USA, Article 96, 11 pages."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498765.1498785"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2000.845979"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/P3HPC.2018.00005"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/WOLFHPC.2016.08"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.01370"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356210"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/P3HPC.2018.00009"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356166"}],"event":{"name":"PPoPP '21: 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","location":"Virtual Event Republic of Korea","acronym":"PPoPP '21","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3437801.3441598","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3437801.3441598","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:17:25Z","timestamp":1750191445000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3437801.3441598"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,2,17]]},"references-count":27,"alternative-id":["10.1145\/3437801.3441598","10.1145\/3437801"],"URL":"https:\/\/doi.org\/10.1145\/3437801.3441598","relation":{},"subject":[],"published":{"date-parts":[[2021,2,17]]},"assertion":[{"value":"2021-02-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}