{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T12:09:45Z","timestamp":1783512585880,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,5,7]],"date-time":"2024-05-07T00:00:00Z","timestamp":1715040000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Department of Energy Office of Science","award":["66150"],"award-info":[{"award-number":["66150"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,5,7]]},"DOI":"10.1145\/3629527.3651428","type":"proceedings-article","created":{"date-parts":[[2024,5,7]],"date-time":"2024-05-07T12:19:06Z","timestamp":1715084346000},"page":"14-20","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["Evaluating Emerging AI\/ML Accelerators: IPU, RDU, and NVIDIA\/AMD GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2025-2195","authenticated-orcid":false,"given":"Hongwu","family":"Peng","sequence":"first","affiliation":[{"name":"University of Connecticut, Storrs, CT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0891-1231","authenticated-orcid":false,"given":"Caiwen","family":"Ding","sequence":"additional","affiliation":[{"name":"University of Connecticut, Storrs, CT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3644-2922","authenticated-orcid":false,"given":"Tong","family":"Geng","sequence":"additional","affiliation":[{"name":"University of Rochester, Rochester, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7352-2035","authenticated-orcid":false,"given":"Sutanay","family":"Choudhury","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4947-0559","authenticated-orcid":false,"given":"Kevin","family":"Barker","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3734-9137","authenticated-orcid":false,"given":"Ang","family":"Li","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,5,7]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"CUDA C Programming Guide. Retrived from https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html. Accessed","author":"Nvidia Corporation","year":"2022","unstructured":"Nvidia Corporation. [n.,d.] a. CUDA C Programming Guide. Retrived from https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_2_1","unstructured":"Nvidia Corporation. [n. d.] b. NVIDIA A100 Tensor Core GPU Architecture unprecedented acceleration at every scale. Retrived from https:\/\/www.nvidia.com\/content\/dam\/en-zz\/Solutions\/Data-Center\/nvidia-ampere-architecture-whitepaper.pdf. Accessed: 2022 Oct. 30th."},{"key":"e_1_3_2_1_3_1","volume-title":"Nvidia Tesla V100 GPU Architecture","author":"Nvidia Corporation","unstructured":"Nvidia Corporation. [n.,d.] c. Nvidia Tesla V100 GPU Architecture, The World's Most Advanced Data Center GPU. Retrived from https:\/\/images.nvidia.com\/content\/volta-architecture\/pdf\/volta-architecture-whitepaper.pdf. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_4_1","volume-title":"SambaFlow - SambaNova Systems. Retrived from https:\/\/sambanova.ai\/wp-content\/uploads\/2021\/04\/SambaNova_SambaFlow_Datasheet_English.pdf. Accessed","author":"Sambanova System Corporation","year":"2022","unstructured":"Sambanova System Corporation. [n.,d.] d. SambaFlow - SambaNova Systems. Retrived from https:\/\/sambanova.ai\/wp-content\/uploads\/2021\/04\/SambaNova_SambaFlow_Datasheet_English.pdf. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/2049662.2049663"},{"key":"e_1_3_2_1_6_1","volume-title":"Fast Graph Representation Learning with PyTorch Geometric. In ICLR Workshop on Representation Learning on Graphs and Manifolds.","author":"Fey Matthias","unstructured":"Matthias Fey and Jan E. Lenssen. 2019. Fast Graph Representation Learning with PyTorch Geometric. In ICLR Workshop on Representation Learning on Graphs and Manifolds."},{"key":"e_1_3_2_1_7_1","volume-title":"Graphcore: Dynamic Sparsity. Retrived from https:\/\/github.com\/graphcore\/examples\/tree\/master\/sparsity\/dynamic_sparsity\/tensorflow1. Accessed","year":"2022","unstructured":"Graphcore. [n.,d.] a. Graphcore: Dynamic Sparsity. Retrived from https:\/\/github.com\/graphcore\/examples\/tree\/master\/sparsity\/dynamic_sparsity\/tensorflow1. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_8_1","volume-title":"Introducing the Colossus$^\u2122$ MK2 GC200 IPU. Retrived from https:\/\/www.graphcore.ai\/products\/ipu. Accessed","year":"2022","unstructured":"Graphcore. [n.,d.] b. Introducing the Colossus$^\u2122$ MK2 GC200 IPU. Retrived from https:\/\/www.graphcore.ai\/products\/ipu. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_9_1","volume-title":"IPU-POD16 Direct Attach Datasheet. Retrived from https:\/\/docs.graphcore.ai\/projects\/ipu-pod16-datasheet\/en\/latest\/product-description.html. Accessed","year":"2022","unstructured":"Graphcore. [n.,d.] c. IPU-POD16 Direct Attach Datasheet. Retrived from https:\/\/docs.graphcore.ai\/projects\/ipu-pod16-datasheet\/en\/latest\/product-description.html. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_10_1","volume-title":"IPU Programming Model. Retrived from https:\/\/docs.graphcore.ai\/projects\/ipu-programmers-guide\/en\/latest\/programming_model.html. Accessed","year":"2022","unstructured":"Graphcore. [n.,d.] d. IPU Programming Model. Retrived from https:\/\/docs.graphcore.ai\/projects\/ipu-programmers-guide\/en\/latest\/programming_model.html. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_11_1","volume-title":"Poplar Graph Framework Software. Retrived from https:\/\/www.graphcore.ai\/products\/poplar. Accessed","year":"2022","unstructured":"Graphcore. [n.,d.] e. Poplar Graph Framework Software. Retrived from https:\/\/www.graphcore.ai\/products\/poplar. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_12_1","volume-title":"PopSparse Matrix Multiplication (Dynamic Pattern) on the IPU. Retrived from https:\/\/docs.graphcore.ai\/projects\/dynamic-sparsity\/en\/latest\/dynamic-sparsity.html. Accessed","year":"2022","unstructured":"Graphcore. [n.,d.] f. PopSparse Matrix Multiplication (Dynamic Pattern) on the IPU. Retrived from https:\/\/docs.graphcore.ai\/projects\/dynamic-sparsity\/en\/latest\/dynamic-sparsity.html. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASPDAC.2001.913368"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_15_1","volume-title":"AMD CDNA 2 Architecture. Retrived from https:\/\/www.amd.com\/system\/files\/documents\/amd-cdna2-white-paper.pdf. Accessed","author":"AMD Inc. [n.,d.] a.","year":"2022","unstructured":"AMD Inc. [n.,d.] a. AMD CDNA 2 Architecture. Retrived from https:\/\/www.amd.com\/system\/files\/documents\/amd-cdna2-white-paper.pdf. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_16_1","volume-title":"AMD CDNA Architecture. Retrived from https:\/\/www.amd.com\/system\/files\/documents\/amd-cdna-whitepaper.pdf. Accessed","author":"AMD Inc. [n.,d.] b.","year":"2022","unstructured":"AMD Inc. [n.,d.] b. AMD CDNA Architecture. Retrived from https:\/\/www.amd.com\/system\/files\/documents\/amd-cdna-whitepaper.pdf. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_17_1","volume-title":"AMD Graphics Cores Next (GCN) Architecture. Retrived from https:\/\/www.techpowerup.com\/gpu-specs\/docs\/amd-gcn1-architecture.pdf. Accessed","author":"AMD Inc. [n.,d.] c.","year":"2022","unstructured":"AMD Inc. [n.,d.] c. AMD Graphics Cores Next (GCN) Architecture. Retrived from https:\/\/www.techpowerup.com\/gpu-specs\/docs\/amd-gcn1-architecture.pdf. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_18_1","volume-title":"HIP: C Heterogeneous-Compute Interface for Portability. Retrived from https:\/\/github.com\/rocm-developer-tools\/hip. Accessed","author":"AMD Inc. [n.,d.] d.","year":"2022","unstructured":"AMD Inc. [n.,d.] d. HIP: C Heterogeneous-Compute Interface for Portability. Retrived from https:\/\/github.com\/rocm-developer-tools\/hip. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_19_1","volume-title":"Retrived from https:\/\/rocmdocs.amd.com\/_\/downloads\/en\/latest\/pdf\/. Accessed","author":"AMD Inc. [n.,d.] e. ROCm Documentation.","year":"2022","unstructured":"AMD Inc. [n.,d.] e. ROCm Documentation. Retrived from https:\/\/rocmdocs.amd.com\/_\/downloads\/en\/latest\/pdf\/. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_20_1","volume-title":"Dissecting the graphcore ipu architecture via microbenchmarking. arXiv preprint arXiv:1912.03413","author":"Jia Zhe","year":"2019","unstructured":"Zhe Jia, Blake Tillman, Marco Maggioni, and Daniele Paolo Scarpazza. 2019. Dissecting the graphcore ipu architecture via microbenchmarking. arXiv preprint arXiv:1912.03413 (2019)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3192366.3192379"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"e_1_3_2_1_23_1","unstructured":"Kunle Olukotun. 2020. Plasticine-A Universal Data Analytics Accelerator. Technical Report. Leland Stanford Junior University Stanford United States."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC42614.2022.9731612"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2018.032271058"},{"key":"e_1_3_2_1_26_1","volume-title":"Retrived from https:\/\/pytorch.org\/docs\/stable\/sparse.html. Accessed","author":"SPARSE.","year":"2022","unstructured":"Pytorch. [n.,d.]. TORCH.SPARSE. Retrived from https:\/\/pytorch.org\/docs\/stable\/sparse.html. Accessed: 2022, Oct. 30th."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00474"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/223982.224449"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/79173.79181"},{"key":"e_1_3_2_1_30_1","volume-title":"Accel-GCN: High-Performance GPU Accelerator Design for Graph Convolution Networks. arXiv preprint arXiv:2308.11825","author":"Xie Xi","year":"2023","unstructured":"Xi Xie, Hongwu Peng, Amit Hasan, Shaoyi Huang, Jiahui Zhao, Haowen Fang, Wei Zhang, Tong Geng, Omer Khan, and Caiwen Ding. 2023. Accel-GCN: High-Performance GPU Accelerator Design for Graph Convolution Networks. arXiv preprint arXiv:2308.11825 (2023). io"}],"event":{"name":"ICPE '24: 15th ACM\/SPEC International Conference on Performance Engineering","location":"London United Kingdom","acronym":"ICPE '24","sponsor":["SIGMETRICS ACM Special Interest Group on Measurement and Evaluation","SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Companion of the 15th ACM\/SPEC International Conference on Performance Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3629527.3651428","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3629527.3651428","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T19:50:30Z","timestamp":1755892230000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3629527.3651428"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,7]]},"references-count":30,"alternative-id":["10.1145\/3629527.3651428","10.1145\/3629527"],"URL":"https:\/\/doi.org\/10.1145\/3629527.3651428","relation":{},"subject":[],"published":{"date-parts":[[2024,5,7]]},"assertion":[{"value":"2024-05-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}