{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T14:13:52Z","timestamp":1760710432677,"version":"3.28.0"},"reference-count":24,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,9,22]],"date-time":"2020-09-22T00:00:00Z","timestamp":1600732800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,9,22]],"date-time":"2020-09-22T00:00:00Z","timestamp":1600732800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,9,22]],"date-time":"2020-09-22T00:00:00Z","timestamp":1600732800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,9,22]]},"DOI":"10.1109\/hpec43674.2020.9286206","type":"proceedings-article","created":{"date-parts":[[2020,12,22]],"date-time":"2020-12-22T21:07:15Z","timestamp":1608671235000},"page":"1-7","source":"Crossref","is-referenced-by-count":13,"title":["At-Scale Sparse Deep Neural Network Inference With Efficient GPU Implementation"],"prefix":"10.1109","author":[{"given":"Mert","family":"Hidayetoglu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Carl","family":"Pearson","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vikram Sharma","family":"Mailthody","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eiman","family":"Ebrahimi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinjun","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rakesh","family":"Nagi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wen-mei","family":"Hwu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"journal-title":"Graphchallenge org sparse deep neural network performance","year":"2020","author":"kepner","key":"ref10"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2019.8916336"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2018.8547517"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2019.8916547"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2019.8916285"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356220"},{"key":"ref16","doi-asserted-by":"crossref","DOI":"10.1109\/SC41405.2020.00041","article-title":"Petascale XCT: 3D image reconstruction with hierarchical communications on multi-GPU nodes","author":"hidayetoglu","year":"2020","journal-title":"International Conference for High Performance Computing Networking Storage and Analysis"},{"journal-title":"Summit user guide","year":"2020","key":"ref17"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2019.8916223"},{"journal-title":"HPEC Graph Challenge","year":"0","key":"ref19"},{"journal-title":"Language models are few-shot learners","year":"2020","author":"brown","key":"ref4"},{"journal-title":"Megatron-lm Training multi-billion parameter language models using model parallelism","year":"2019","author":"shoeybi","key":"ref3"},{"journal-title":"GPipe Efficient training of giant neural networks using pipeline parallelism","year":"2018","author":"huang","key":"ref6"},{"journal-title":"Exploring massively multilingual massive neural machine translation","year":"2019","key":"ref5"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2019.00051"},{"key":"ref7","article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and Huffman coding","author":"han","year":"2016","journal-title":"International Conference on Learning Representations (ICLR)"},{"journal-title":"Rethinking pre-training and self-training","year":"2020","author":"zoph","key":"ref2"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014780"},{"key":"ref9","first-page":"6105","article-title":"EfficientNet: Rethinking model scaling for convolutional neural networks","volume":"97","author":"tan","year":"2019","journal-title":"Proceedings of the 36th International Conference on Machine Learning"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2019.8916550"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2019.8916498"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2019.8916378"},{"journal-title":"NVIDIA Tesla A100 Tensor Core GPU Architecture","year":"2020","key":"ref24"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2019.8916506"}],"event":{"name":"2020 IEEE High Performance Extreme Computing Conference (HPEC)","start":{"date-parts":[[2020,9,22]]},"location":"Waltham, MA, USA","end":{"date-parts":[[2020,9,24]]}},"container-title":["2020 IEEE High Performance Extreme Computing Conference (HPEC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9285977\/9286137\/09286206.pdf?arnumber=9286206","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,27]],"date-time":"2022-06-27T15:54:00Z","timestamp":1656345240000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9286206\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,9,22]]},"references-count":24,"URL":"https:\/\/doi.org\/10.1109\/hpec43674.2020.9286206","relation":{},"subject":[],"published":{"date-parts":[[2020,9,22]]}}}