{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T05:12:14Z","timestamp":1783746734280,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T00:00:00Z","timestamp":1783900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["CCF-2113996"],"award-info":[{"award-number":["CCF-2113996"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,13]]},"DOI":"10.1145\/3806645.3807576","type":"proceedings-article","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:21:11Z","timestamp":1783743671000},"page":"45-57","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["GICC: A High-Performance Runtime for GPU-Initiated Communication and Coordination in Modern HPC Systems"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1022-3801","authenticated-orcid":false,"given":"Baodi","family":"Shan","sequence":"first","affiliation":[{"name":"Stony Brook University, Stony Brook, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5537-1840","authenticated-orcid":false,"given":"Mauricio","family":"Araya-Polo","sequence":"additional","affiliation":[{"name":"TotalEnergies EP Research &amp; Technology USA, Houston, Texas, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8449-8579","authenticated-orcid":false,"given":"Barbara","family":"Chapman","sequence":"additional","affiliation":[{"name":"Stony Brook University, Stony Brook, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,13]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2017.29"},{"key":"e_1_3_3_2_3_2","unstructured":"AMD. 2026. RCCL: ROCm Communication Collectives Library. https:\/\/rocm.docs.amd.com\/projects\/rccl\/en\/latest\/. Accessed: 2026-02."},{"key":"e_1_3_3_2_4_2","unstructured":"AMD. 2026. rocSHMEM: AMD ROCm OpenSHMEM Implementation. https:\/\/rocm.docs.amd.com\/projects\/rocSHMEM\/en\/latest\/index.html. Accessed: 2026-02."},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.2172\/1088065"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.1145\/3731599.3767506"},{"key":"e_1_3_3_2_7_2","unstructured":"Khaled Hamidouche John Bachan Pak Markthub Peter-Jan Gootzen Elena Agostini Sylvain Jeaugey Aamir Shafi Georgios Theodorakis and Manjunath\u00a0Gorentla Venkata. 2025. GPU-Initiated Networking for NCCL. arxiv:https:\/\/arXiv.org\/abs\/2511.15076\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2511.15076"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126970"},{"key":"e_1_3_3_2_9_2","unstructured":"HPE Cray. 2024. HPE Cray MPI: GPU-NIC Async Progress (Stream Triggered and Kernel Triggered). https:\/\/cpe.ext.hpe.com\/docs\/latest\/mpt\/mpich\/intro_mpi.html. Section \u201cGPU-NIC Async Progress\u201d. Accessed 2026-02-01."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-15922-0_2"},{"key":"e_1_3_3_2_11_2","unstructured":"Jie Meng Andreas Atle Henri Calandra and Mauricio Araya-Polo. 2020. Minimod: A Finite Difference solver for Seismic Modeling. arXiv (2020). arxiv:https:\/\/arXiv.org\/abs\/2007.06048\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2007.06048"},{"key":"e_1_3_3_2_12_2","unstructured":"MPI Forum. 2021. MPI: A Message-Passing Interface Standard Version 4.0. https:\/\/www.mpi-forum.org\/docs\/mpi-4.0\/mpi40-report.pdf"},{"key":"e_1_3_3_2_13_2","unstructured":"MPI Forum. 2025. MPI: A Message-Passing Interface Standard Version 5.0. https:\/\/www.mpi-forum.org\/docs\/mpi-5.0\/. Approved June 5 2025."},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","unstructured":"Naveen Namashivayam. 2025. GPU-centric Communication Schemes for HPC and ML Applications. arxiv:https:\/\/arXiv.org\/abs\/2503.24230\u00a0[cs.DC] 10.48550\/arXiv.2503.24230","DOI":"10.48550\/arXiv.2503.24230"},{"key":"e_1_3_3_2_15_2","unstructured":"NVIDIA. 2024. NCCL: NVIDIA Collective Communication Library. https:\/\/developer.nvidia.com\/nccl."},{"key":"e_1_3_3_2_16_2","unstructured":"NVIDIA. 2024. NVSHMEM: NVSHMEM Library Documentation. https:\/\/docs.nvidia.com\/nvshmem."},{"key":"e_1_3_3_2_17_2","unstructured":"NVIDIA Corporation. 2024. NVIDIA Multi-GPU Programming Models: jacobi_nvshmem. https:\/\/github.com\/NVIDIA\/multi-gpu-programming-models\/tree\/master\/jacobi_nvshmem. Accessed: 2026-02."},{"key":"e_1_3_3_2_18_2","volume-title":"NVSHMEM Performance","author":"Corporation NVIDIA","year":"2025","unstructured":"NVIDIA Corporation. 2025. NVSHMEM Performance. https:\/\/docs.nvidia.com\/nvshmem\/release-notes-install-guide\/best-practice-guide\/performance.html Last updated: Dec. 30, 2025."},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/E2SC.2014.14"},{"key":"e_1_3_3_2_20_2","unstructured":"OpenFabrics Interfaces Working Group. 2025. fi_cxi(7): Libfabric CXI Provider. https:\/\/ofiwg.github.io\/libfabric\/man\/fi_cxi.7.html. Accessed: 2025-11. Describes CXI provider features including triggered operations and FI_PROGRESS_MANUAL.."},{"key":"e_1_3_3_2_21_2","unstructured":"OpenFabrics Interfaces Working Group. 2025. Libfabric: OpenFabrics Interfaces. https:\/\/ofiwg.github.io\/libfabric\/. Accessed: 2025-11."},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2013.17"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW63119.2024.00198"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3731599.3767505"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72567-8_5"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-032-06343-4_1"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3582514.3582519"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","unstructured":"Mark Silberstein Sangman Kim Seonggu Huh Xinya Zhang Yige Hu Amir Wated and Emmett Witchel. 2016. GPUnet: Networking Abstractions for GPU Programs. ACM Trans. Comput. Syst. 34 3 Article 9 (Sept. 2016) 31\u00a0pages. 10.1145\/2963098","DOI":"10.1145\/2963098"},{"key":"e_1_3_3_2_29_2","unstructured":"TOP500.org. 2025. November 2025 Top 500. https:\/\/www.top500.org\/lists\/top500\/2025\/11\/. Accessed: 2025-11 list published during SC25."},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/3712285.3759774"}],"event":{"name":"HPDC '26: 35th International Symposium on High-Performance Parallel and Distributed Computing","location":"Cleveland USA","acronym":"HPDC '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 35th International Symposium on High-Performance Parallel and Distributed Computing"],"original-title":[],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:21:53Z","timestamp":1783743713000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3806645.3807576"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,13]]},"references-count":29,"alternative-id":["10.1145\/3806645.3807576","10.1145\/3806645"],"URL":"https:\/\/doi.org\/10.1145\/3806645.3807576","relation":{},"subject":[],"published":{"date-parts":[[2026,7,13]]},"assertion":[{"value":"2026-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}