{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T17:37:30Z","timestamp":1780335450727,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,4,4]],"date-time":"2019-04-04T00:00:00Z","timestamp":1554336000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["OCI-0725070"],"award-info":[{"award-number":["OCI-0725070"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"name":"State of Illinois"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,4,4]]},"DOI":"10.1145\/3297663.3310299","type":"proceedings-article","created":{"date-parts":[[2019,4,5]],"date-time":"2019-04-05T13:27:26Z","timestamp":1554470846000},"page":"209-218","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":26,"title":["Evaluating Characteristics of CUDA Communication Primitives on High-Bandwidth Interconnects"],"prefix":"10.1145","author":[{"given":"Carl","family":"Pearson","sequence":"first","affiliation":[{"name":"University of Illinois at Urbana-Champaign, Urbana, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abdul","family":"Dakkak","sequence":"additional","affiliation":[{"name":"University of Illinois at Urbana-Champaign, Urbana, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sarah","family":"Hashash","sequence":"additional","affiliation":[{"name":"University of Illinois at Urbana-Champaign, Urbana, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cheng","family":"Li","sequence":"additional","affiliation":[{"name":"University of Illinois at Urbana-Champaign, Urbana, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"I-Hsin","family":"Chung","sequence":"additional","affiliation":[{"name":"IBM T.J. Watson Research, Yorktown Heights, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinjun","family":"Xiong","sequence":"additional","affiliation":[{"name":"IBM T.J. Watson Research, Yorktown Heights, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wen-Mei","family":"Hwu","sequence":"additional","affiliation":[{"name":"University of Illinois at Urbana-Champaign, Urbana, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2019,4,4]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"{n. d.}. Sierra Supercomputer. https:\/\/computation.llnl.gov\/computers\/sierra. Accessed: 2018-10-11.  {n. d.}. Sierra Supercomputer. https:\/\/computation.llnl.gov\/computers\/sierra. Accessed: 2018-10-11."},{"key":"e_1_3_2_1_2_1","unstructured":"{n. d.}. Summit Supercomputer. https:\/\/www.olcf.ornl.gov\/summit\/. Accessed: 2018-10-11.  {n. d.}. Summit Supercomputer. https:\/\/www.olcf.ornl.gov\/summit\/. Accessed: 2018-10-11."},{"key":"e_1_3_2_1_5_1","unstructured":"2018. Benchmarking unified memory in CUDA 6.0. https:\/\/users.ices.utexas.edu\/~sreepai\/automem\/.  2018. Benchmarking unified memory in CUDA 6.0. https:\/\/users.ices.utexas.edu\/~sreepai\/automem\/."},{"key":"e_1_3_2_1_6_1","unstructured":"2018. CUDA 9.2 Toolkit Downloads.https:\/\/developer.nvidia.com\/cuda-92-download-archive.  2018. CUDA 9.2 Toolkit Downloads.https:\/\/developer.nvidia.com\/cuda-92-download-archive."},{"key":"e_1_3_2_1_7_1","unstructured":"Advanced Micro Devices 2018. AMD64 Architecture Programmer's Manual(3.26ed.). Advanced Micro Devices.  Advanced Micro Devices 2018. AMD64 Architecture Programmer's Manual(3.26ed.). Advanced Micro Devices."},{"key":"e_1_3_2_1_8_1","unstructured":"amazon {n. d.}. Amazon AWS. https:\/\/aws.amazon.com. Accessed: 2018-10-11.  amazon {n. d.}. Amazon AWS. https:\/\/aws.amazon.com. Accessed: 2018-10-11."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3018743.3018756"},{"key":"e_1_3_2_1_10_1","unstructured":"Tal Ben-Nuun. 2017. mgbench. https:\/\/github.com\/tbennun\/mgbench.  Tal Ben-Nuun. 2017. mgbench. https:\/\/github.com\/tbennun\/mgbench."},{"key":"e_1_3_2_1_11_1","unstructured":"Rajesh Bordawekar and Pidad Dasfar D'Souza. 2018. Evaluation of Hybrid Cache-Coherent Concurrent Hash Table on POWER9 System with NVLink 2.  Rajesh Bordawekar and Pidad Dasfar D'Souza. 2018. Evaluation of Hybrid Cache-Coherent Concurrent Hash Table on POWER9 System with NVLink 2."},{"key":"e_1_3_2_1_12_1","unstructured":"Alexandre B Caldeira. 2018. IBM Power System AC922 Introduction and Technical Overview(1 ed.). IBM.  Alexandre B Caldeira. 2018. IBM Power System AC922 Introduction and Technical Overview(1 ed.). IBM."},{"key":"e_1_3_2_1_13_1","unstructured":"Alexandre B Caldeira Volker Haug and Scott Vetter. 2016. IBM Power System S822LC for High Performance Computing Introduction and Technical Overview(1ed.).  Alexandre B Caldeira Volker Haug and Scott Vetter. 2016. IBM Power System S822LC for High Performance Computing Introduction and Technical Overview(1ed.)."},{"key":"e_1_3_2_1_14_1","volume-title":"Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXivpreprint arXiv: 1512.01274(2015).","author":"Chen Tianqi","year":"2015","unstructured":"Tianqi Chen , Mu Li , Yutian Li , Min Lin , Naiyan Wang , Minjie Wang , Tianjun Xiao , Bing Xu , Chiyuan Zhang , and Zheng Zhang . 2015 . Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXivpreprint arXiv: 1512.01274(2015). Tianqi Chen, Mu Li, Yutian Li, Min Lin, Naiyan Wang, Minjie Wang, Tianjun Xiao, Bing Xu, Chiyuan Zhang, and Zheng Zhang. 2015. Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXivpreprint arXiv: 1512.01274(2015)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/1735688.1735702"},{"key":"e_1_3_2_1_16_1","unstructured":"google {n. d.}. Google Cloud Platform. https:\/\/cloud.google.com. Accessed: 2018-10-11.  google {n. d.}. Google Cloud Platform. https:\/\/cloud.google.com. Accessed: 2018-10-11."},{"key":"e_1_3_2_1_17_1","unstructured":"Google. 2018. Benchmark -- A microbenchmark support library. https:\/\/github.com\/google\/benchmark.  Google. 2018. Benchmark -- A microbenchmark support library. https:\/\/github.com\/google\/benchmark."},{"key":"e_1_3_2_1_18_1","unstructured":"rk Harris. 2013. Unified Memory in CUDA 6. (2013). https:\/\/devblogs.nvidia.com\/parallelforall\/unified-memory-in-cuda-6\/  rk Harris. 2013. Unified Memory in CUDA 6. (2013). https:\/\/devblogs.nvidia.com\/parallelforall\/unified-memory-in-cuda-6\/"},{"key":"e_1_3_2_1_19_1","unstructured":"Richard Hayden and Oleg Rasskazov. 2018. Juicing up ye old Monte Carlo GPUcode.  Richard Hayden and Oleg Rasskazov. 2018. Juicing up ye old Monte Carlo GPUcode."},{"key":"e_1_3_2_1_20_1","unstructured":"IBM 2018. POWER ISA(2.07B ed.). IBM.  IBM 2018. POWER ISA(2.07B ed.). IBM."},{"key":"e_1_3_2_1_21_1","unstructured":"Jiri Kraus. 2016. High Performance and Productivity with Unified Memory and Open ACC: A LBM Case Study.  Jiri Kraus. 2016. High Performance and Productivity with Unified Memory and Open ACC: A LBM Case Study."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2014.7040988"},{"key":"e_1_3_2_1_23_1","unstructured":"Ang Li. 2018. Tartan. https:\/\/github.com\/uuudown\/Tartan.  Ang Li. 2018. Tartan. https:\/\/github.com\/uuudown\/Tartan."},{"key":"e_1_3_2_1_24_1","volume-title":"International Symposium on Workload Characterization, IEEE.","author":"Li Ang","year":"2017","unstructured":"Ang Li , Shuaiwen Leon Song , Jieyang Cheng , Xu Liu , Nathan Tallent , and KevinBarker. 2017 . Tartan: Evaluating Modern GPU Interconnect via a Multi-GPU Benchmark Suite . In International Symposium on Workload Characterization, IEEE. Ang Li, Shuaiwen Leon Song, Jieyang Cheng, Xu Liu, Nathan Tallent, and KevinBarker. 2017. Tartan: Evaluating Modern GPU Interconnect via a Multi-GPU Benchmark Suite. In International Symposium on Workload Characterization, IEEE."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2014.05.018"},{"key":"e_1_3_2_1_26_1","unstructured":"microsoft {n. d.}. Microsoft Azure Cloud Platform. https:\/\/azure.microsoft.com.Accessed: 2018-10-11.  microsoft {n. d.}. Microsoft Azure Cloud Platform. https:\/\/azure.microsoft.com.Accessed: 2018-10-11."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2016.7482093"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/1964179.1964194"},{"key":"e_1_3_2_1_29_1","unstructured":"Super Micro 2018. Super Server 4029GP-TVRT(1 ed.). Super Micro.  Super Micro 2018. Super Server 4029GP-TVRT(1 ed.). Super Micro."},{"key":"e_1_3_2_1_30_1","volume-title":"Evaluating On-Node GPU Interconnects for Deep Learning Workloads. In International Workshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems. Springer, 3--21","author":"Tallent Nathan R","year":"2017","unstructured":"Nathan R Tallent , Nitin A Gawande , Charles Siegel , Abhinav Vishnu , and Adolfy Hoisie . 2017 . Evaluating On-Node GPU Interconnects for Deep Learning Workloads. In International Workshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems. Springer, 3--21 . Nathan R Tallent, Nitin A Gawande, Charles Siegel, Abhinav Vishnu, and Adolfy Hoisie. 2017. Evaluating On-Node GPU Interconnects for Deep Learning Workloads. In International Workshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems. Springer, 3--21."},{"key":"e_1_3_2_1_31_1","unstructured":"Cliff Wickman Christoph Lameter and Lee Schermerhorn. 2015. numactl v2.0.11.https:\/\/github.com\/numactl\/numactl  Cliff Wickman Christoph Lameter and Lee Schermerhorn. 2015. numactl v2.0.11.https:\/\/github.com\/numactl\/numactl"}],"event":{"name":"ICPE '19: Tenth ACM\/SPEC International Conference on Performance Engineering","location":"Mumbai India","acronym":"ICPE '19","sponsor":["SIGMETRICS ACM Special Interest Group on Measurement and Evaluation","SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 2019 ACM\/SPEC International Conference on Performance Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3297663.3310299","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3297663.3310299","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3297663.3310299","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:54:10Z","timestamp":1750204450000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3297663.3310299"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,4,4]]},"references-count":29,"alternative-id":["10.1145\/3297663.3310299","10.1145\/3297663"],"URL":"https:\/\/doi.org\/10.1145\/3297663.3310299","relation":{},"subject":[],"published":{"date-parts":[[2019,4,4]]},"assertion":[{"value":"2019-04-04","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}