{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T20:31:40Z","timestamp":1774384300436,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":31,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,6,21]],"date-time":"2021-06-21T00:00:00Z","timestamp":1624233600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Swedish Research Council","award":["Grant Agreement No. 2018-0597."],"award-info":[{"award-number":["Grant Agreement No. 2018-0597."]}]},{"DOI":"10.13039\/501100000780","name":"European Commission","doi-asserted-by":"publisher","award":["Grant Agreement No. 801039"],"award-info":[{"award-number":["Grant Agreement No. 801039"]}],"id":[{"id":"10.13039\/501100000780","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,6,21]]},"DOI":"10.1145\/3468044.3468053","type":"proceedings-article","created":{"date-parts":[[2021,6,21]],"date-time":"2021-06-21T10:43:03Z","timestamp":1624272183000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":19,"title":["Benchmarking the Nvidia GPU Lineage"],"prefix":"10.1145","author":[{"given":"Martin","family":"Svedin","sequence":"first","affiliation":[{"name":"KTH Royal Institute of Technology, Stockholm, Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Steven W. D.","family":"Chien","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology, Stockholm, Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gibson","family":"Chikafa","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology, Stockholm, Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Niclas","family":"Jansson","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology, Stockholm, Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Artur","family":"Podobas","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology, Stockholm, Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,6,21]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"A 30 Year Retrospective on Dennard's MOSFET Scaling Paper","author":"Bohr Mark","year":"2007"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/HCS49909.2020.9220622"},{"key":"e_1_3_2_1_3_1","unstructured":"Asanovic et al. 2006. The Landscape of Parallel Computing Research: A View from Berkeley 2006. (2006).  Asanovic et al. 2006. The Landscape of Parallel Computing Research: A View from Berkeley 2006. (2006)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Anzt et al. 2020. Evaluating the Performance of NVIDIA's A100 Ampere GPU for Sparse and Batched Computations. In 2020 IEEE\/ACM Performance Modeling Benchmarking and Simulation of High Performance Computer Systems (PMBS). IEEE.  Anzt et al. 2020. Evaluating the Performance of NVIDIA's A100 Ampere GPU for Sparse and Batched Computations. In 2020 IEEE\/ACM Performance Modeling Benchmarking and Simulation of High Performance Computer Systems (PMBS). IEEE.","DOI":"10.1109\/PMBS51919.2020.00009"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33518-1_16"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2514740"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2019.00019"},{"key":"e_1_3_2_1_9_1","unstructured":"Domke et al. 2020. Matrix Engines for High Performance Computing:A Paragon of Performance or Grasping at Straws? arXiv preprint arXiv:2010.14373 (2020).  Domke et al. 2020. Matrix Engines for High Performance Computing:A Paragon of Performance or Grasping at Straws? arXiv preprint arXiv:2010.14373 (2020)."},{"key":"e_1_3_2_1_10_1","unstructured":"Jia et al. 2018. Dissecting the NVIDIA Volta GPU Architecture via Microbench-marking. arXiv preprint arXiv:1804.06826 (2018).  Jia et al. 2018. Dissecting the NVIDIA Volta GPU Architecture via Microbench-marking. arXiv preprint arXiv:1804.06826 (2018)."},{"key":"e_1_3_2_1_11_1","unstructured":"Karp et al. 2020. High-Performance Spectral Element Methods on Field-Programmable Gate Arrays. arXiv preprint arXiv:2010.13463 (2020).  Karp et al. 2020. High-Performance Spectral Element Methods on Field-Programmable Gate Arrays. arXiv preprint arXiv:2010.13463 (2020)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2018.8573483"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2903150.2906830"},{"key":"e_1_3_2_1_14_1","volume-title":"European Conference on Parallel Processing. Springer.","author":"Martineau","year":"2018"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2018.00091"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.23919\/FPL.2017.8056760"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Podobas et al. 2020. A Survey on Coarse-Grained Reconfigurable Architectures From a Performance Perspective. IEEE Access (2020).  Podobas et al. 2020. A Survey on Coarse-Grained Reconfigurable Architectures From a Performance Perspective. IEEE Access (2020).","DOI":"10.1109\/ACCESS.2020.3012084"},{"key":"e_1_3_2_1_18_1","unstructured":"Schuman et al. 2017. A Survey of Neuromorphic Computing and Neural Networks in Hardware. arXiv preprint arXiv:1705.06963 (2017).  Schuman et al. 2017. A Survey of Neuromorphic Computing and Neural Networks in Hardware. arXiv preprint arXiv:1705.06963 (2017)."},{"key":"e_1_3_2_1_19_1","unstructured":"Tsai et al. 2020. Evaluating the Performance of NVIDIA's A100 Ampere GPU for Sparse Linear Algebra Computations. arXiv preprint arXiv:2008.08478 (2020).  Tsai et al. 2020. Evaluating the Performance of NVIDIA's A100 Ampere GPU for Sparse Linear Algebra Computations. arXiv preprint arXiv:2008.08478 (2020)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2668930.2688046"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498765.1498785"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2010.5452013"},{"key":"e_1_3_2_1_23_1","unstructured":"Wang et al. 2019. Benchmarking tpu gpu and cpu platforms for deep learning. arXiv preprint arXiv:1907.10701 (2019).  Wang et al. 2019. Benchmarking tpu gpu and cpu platforms for deep learning. arXiv preprint arXiv:1907.10701 (2019)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid49817.2020.00-15"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.1972.5009071"},{"key":"e_1_3_2_1_26_1","volume-title":"A survey on quantum computing technology. Computer Science Review","author":"Gyongyosi Laszlo","year":"2019"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2016.2549523"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2010.5470394"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/SAMOS.2016.7818354"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/6.591665"},{"key":"e_1_3_2_1_31_1","unstructured":"Toshio Yoshida. 2018. Fujitsu High Performance CPU for the Post-K Computer. In Hot Chips.  Toshio Yoshida. 2018. Fujitsu High Performance CPU for the Post-K Computer. In Hot Chips."}],"event":{"name":"HEART '21: International Symposium on Highly Efficient Accelerators and Reconfigurable Technologies","location":"Online Germany","acronym":"HEART '21","sponsor":["German Research Foundation German Research Foundation"]},"container-title":["Proceedings of the 11th International Symposium on Highly Efficient Accelerators and Reconfigurable Technologies"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3468044.3468053","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3468044.3468053","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:28:06Z","timestamp":1750195686000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3468044.3468053"}},"subtitle":["From Early K80 to Modern A100 with Asynchronous Memory Transfers"],"short-title":[],"issued":{"date-parts":[[2021,6,21]]},"references-count":31,"alternative-id":["10.1145\/3468044.3468053","10.1145\/3468044"],"URL":"https:\/\/doi.org\/10.1145\/3468044.3468053","relation":{},"subject":[],"published":{"date-parts":[[2021,6,21]]},"assertion":[{"value":"2021-06-21","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}