{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T14:48:52Z","timestamp":1779202132213,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,11,12]],"date-time":"2023-11-12T00:00:00Z","timestamp":1699747200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"the Advanced Scientific Computing Research Program in the U.S. Department of Energy, Office of Science","award":["DE-AC02-05CH11231"],"award-info":[{"award-number":["DE-AC02-05CH11231"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,11,12]]},"DOI":"10.1145\/3624062.3624182","type":"proceedings-article","created":{"date-parts":[[2023,11,10]],"date-time":"2023-11-10T13:53:39Z","timestamp":1699624419000},"page":"1059-1069","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Evaluating the Performance of One-sided Communication on CPUs and GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9624-9449","authenticated-orcid":false,"given":"Nan","family":"Ding","sequence":"first","affiliation":[{"name":"Lawrence Berkeley National Laboratory, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0697-6894","authenticated-orcid":false,"given":"Muhammad","family":"Haseeb","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Laboratory (LBNL), United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7020-8881","authenticated-orcid":false,"given":"Taylor","family":"Groves","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Laboratory, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8327-5717","authenticated-orcid":false,"given":"Samuel","family":"Williams","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Laboratory, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,11,12]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"2020. ROC_SHMEM. https:\/\/github.com\/ROCm-Developer-Tools\/ROC_SHMEM."},{"key":"e_1_3_2_2_2_1","unstructured":"2023. Accelerating Distributed Deep Learning using HCCL: Habana Collective Communication Library. http:\/\/nowlab.cse.ohio-state.edu\/static\/media\/workshops\/presentations\/exacomm23\/Habana_ExaComm_2023.pdf."},{"key":"e_1_3_2_2_3_1","unstructured":"2023. An Introduction to CUDA-Aware MPI. https:\/\/developer.nvidia.com\/blog\/introduction-cuda-aware-mpi\/."},{"key":"e_1_3_2_2_4_1","unstructured":"2023. Frontier User Guide. https:\/\/docs.olcf.ornl.gov\/systems\/frontier_user_guide.html#frontier-compute-nodes."},{"key":"e_1_3_2_2_5_1","unstructured":"2023. NVIDIA NVSHMEM Documentation. https:\/\/docs.nvidia.com\/hpc-sdk\/nvshmem\/index.html."},{"key":"e_1_3_2_2_6_1","unstructured":"2023. Perlmutter Interconnect. https:\/\/docs.nersc.gov\/systems\/perlmutter\/architecture\/#interconnect."},{"key":"e_1_3_2_2_7_1","unstructured":"2023. Perlmutter User Guide - CPU Architecture. https:\/\/docs.nersc.gov\/systems\/perlmutter\/architecture\/#cpu-nodes."},{"key":"e_1_3_2_2_8_1","unstructured":"2023. Perlmutter User Guide - GPU Architecture. https:\/\/docs.nersc.gov\/systems\/perlmutter\/architecture\/#gpu-nodes."},{"key":"e_1_3_2_2_9_1","unstructured":"2023. RCCL 2.16.5 Documentation. https:\/\/rocm.docs.amd.com\/projects\/rccl\/en\/latest\/."},{"key":"e_1_3_2_2_10_1","unstructured":"2023. Summit User Guide. https:\/\/docs.olcf.ornl.gov\/systems\/summit_user_guide.html#summit-nodes."},{"key":"e_1_3_2_2_11_1","unstructured":"2023. Top500 Highlights. https:\/\/www.top500.org\/lists\/top500\/2023\/06\/highs\/."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/215399.215427"},{"key":"e_1_3_2_2_13_1","unstructured":"George Almasi. 2011. PGAS (Partitioned Global Address Space) Languages."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2015.30"},{"key":"e_1_3_2_2_15_1","unstructured":"Valeria Cardellini Alessandro Fanfarillo Salvatore Filippone 2016. Overlapping communication with computation in MPI applications. Universit\u00e0 degli Studi di Roma Tor Vergata Technical Reports (2016)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/2020373.2020375"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2012.72"},{"key":"e_1_3_2_2_18_1","volume-title":"Multi-GPU Parallel Sparse Triangular Solver. In SIAM Conference on Applied and Computational Discrete Algorithms (ACDA21)","author":"Ding Nan","year":"2021","unstructured":"Nan Ding, Yang Liu, Samuel Williams, and Xiaoye\u00a0S Li. 2021. A Message-Driven, Multi-GPU Parallel Sparse Triangular Solver. In SIAM Conference on Applied and Computational Discrete Algorithms (ACDA21). SIAM, 147\u2013159."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611976137.9"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/PMBS54543.2021.00009"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1006\/jpdc.1994.1085"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2503210.2503286"},{"key":"e_1_3_2_2_23_1","volume-title":"Performance trade-offs in GPU communication: A study of host and device-initiated approaches","author":"Groves Taylor","unstructured":"Taylor Groves, Ben Brock, Yuxin Chen, Khaled\u00a0Z Ibrahim, Lenny Oliker, Nicholas\u00a0J Wright, Samuel Williams, and Katherine Yelick. 2020. Performance trade-offs in GPU communication: A study of host and device-initiated approaches. IEEE."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3332466.3374544"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW50202.2020.00104"},{"key":"e_1_3_2_2_26_1","volume-title":"GPU Technology Conference (GTC), Vol.\u00a02.","author":"Jeaugey Sylvain","year":"2017","unstructured":"Sylvain Jeaugey. 2017. Nccl 2.0. In GPU Technology Conference (GTC), Vol.\u00a02."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2019.2928289"},{"key":"e_1_3_2_2_28_1","volume-title":"Simplifying Multi-GPU Communication with NVSHMEM. In GPU Technology Conference. NVIDIA.","author":"Potluri Sreeram","year":"2016","unstructured":"Sreeram Potluri, Nathan Luehr, and Nikolay Sakharnykh. 2016. Simplifying Multi-GPU Communication with NVSHMEM. In GPU Technology Conference. NVIDIA."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/1964179.1964194"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2018.00034"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498765.1498785"},{"key":"e_1_3_2_2_32_1","volume-title":"Fast and Scalable Sparse Triangular Solver for Multi-GPU Based HPC Architectures. arXiv preprint arXiv:2012.06959","author":"Xie Chenhao","year":"2020","unstructured":"Chenhao Xie, Jieyang Chen, Jesun\u00a0S Firoz, Jiajia Li, Shuaiwen\u00a0Leon Song, Kevin Barker, Mark Raugas, and Ang Li. 2020. Fast and Scalable Sparse Triangular Solver for Multi-GPU Based HPC Architectures. arXiv preprint arXiv:2012.06959 (2020)."}],"event":{"name":"SC-W 2023: Workshops of The International Conference on High Performance Computing, Network, Storage, and Analysis","location":"Denver CO USA","acronym":"SC-W 2023"},"container-title":["Proceedings of the SC '23 Workshops of the International Conference on High Performance Computing, Network, Storage, and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3624062.3624182","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3624062.3624182","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T03:05:00Z","timestamp":1755745500000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3624062.3624182"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,12]]},"references-count":32,"alternative-id":["10.1145\/3624062.3624182","10.1145\/3624062"],"URL":"https:\/\/doi.org\/10.1145\/3624062.3624182","relation":{},"subject":[],"published":{"date-parts":[[2023,11,12]]},"assertion":[{"value":"2023-11-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}