{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,15]],"date-time":"2026-08-15T03:23:29Z","timestamp":1786764209531,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":31,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,11,13]],"date-time":"2021-11-13T00:00:00Z","timestamp":1636761600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["1925717"],"award-info":[{"award-number":["1925717"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,11,14]]},"DOI":"10.1145\/3458817.3476182","type":"proceedings-article","created":{"date-parts":[[2021,10,21]],"date-time":"2021-10-21T05:10:34Z","timestamp":1634793034000},"page":"1-14","source":"Crossref","is-referenced-by-count":51,"title":["Efficient tensor core-based GPU kernels for structured sparsity under reduced precision"],"prefix":"10.1145","author":[{"given":"Zhaodong","family":"Chen","sequence":"first","affiliation":[{"name":"University of California"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zheng","family":"Qu","sequence":"additional","affiliation":[{"name":"University of California"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liu","family":"Liu","sequence":"additional","affiliation":[{"name":"University of California"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yufei","family":"Ding","sequence":"additional","affiliation":[{"name":"University of California"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuan","family":"Xie","sequence":"additional","affiliation":[{"name":"University of California"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2021,11,13]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"Mart\u00edn Abadi Ashish Agarwal Paul Barham Eugene Brevdo Zhifeng Chen Craig Citro Greg S. Corrado Andy Davis Jeffrey Dean Matthieu Devin Sanjay Ghemawat Ian Goodfellow Andrew Harp Geoffrey Irving Michael Isard Yangqing Jia Rafal Jozefowicz Lukasz Kaiser Manjunath Kudlur Josh Levenberg Dandelion Man\u00e9 Rajat Monga Sherry Moore Derek Murray Chris Olah Mike Schuster Jonathon Shlens Benoit Steiner Ilya Sutskever Kunal Talwar Paul Tucker Vincent Vanhoucke Vijay Vasudevan Fernanda Vi\u00e9gas Oriol Vinyals Pete Warden Martin Wattenberg Martin Wicke Yuan Yu and Xiaoqiang Zheng. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems. https:\/\/www.tensorflow.org\/ Software available from tensorflow.org.  Mart\u00edn Abadi Ashish Agarwal Paul Barham Eugene Brevdo Zhifeng Chen Craig Citro Greg S. Corrado Andy Davis Jeffrey Dean Matthieu Devin Sanjay Ghemawat Ian Goodfellow Andrew Harp Geoffrey Irving Michael Isard Yangqing Jia Rafal Jozefowicz Lukasz Kaiser Manjunath Kudlur Josh Levenberg Dandelion Man\u00e9 Rajat Monga Sherry Moore Derek Murray Chris Olah Mike Schuster Jonathon Shlens Benoit Steiner Ilya Sutskever Kunal Talwar Paul Tucker Vincent Vanhoucke Vijay Vasudevan Fernanda Vi\u00e9gas Oriol Vinyals Pete Warden Martin Wattenberg Martin Wicke Yuan Yu and Xiaoqiang Zheng. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems. https:\/\/www.tensorflow.org\/ Software available from tensorflow.org."},{"key":"e_1_3_2_2_2_1","volume-title":"SnaPEA: Predictive Early Activation for Reducing Computation in Deep Convolutional Neural Networks. In 2018 ACM\/IEEE 45th Annual International Symposium on Computer Architecture (ISCA). 662--673","author":"Akhlaghi V.","year":"2018"},{"key":"e_1_3_2_2_3_1","volume-title":"2020 IEEE\/ACM International Conference On Computer Aided Design (ICCAD). IEEE, 1--9.","author":"Chen Zhaodong","year":"2020"},{"key":"e_1_3_2_2_4_1","volume-title":"Generating Long Sequences with Sparse Transformers. CoRR abs\/1904.10509","author":"Child Rewon","year":"2019"},{"key":"e_1_3_2_2_5_1","volume-title":"Proceedings of the ACM International Conference on Supercomputing. 46--57","author":"Dakkak Abdul","year":"2019"},{"key":"e_1_3_2_2_6_1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, SC","author":"Gale Trevor","year":"2020"},{"key":"e_1_3_2_2_7_1","volume-title":"Herbordt","author":"Geng Tong","year":"2019"},{"key":"e_1_3_2_2_8_1","unstructured":"Kai Han Yunhe Wang Hanting Chen Xinghao Chen Jianyuan Guo Zhenhua Liu Yehui Tang An Xiao Chunjing Xu Yixing Xu etal 2020. A Survey on Visual Transformer. arXiv preprint arXiv:2012.12556 (2020).  Kai Han Yunhe Wang Hanting Chen Xinghao Chen Jianyuan Guo Zhenhua Liu Yehui Tang An Xiao Chunjing Xu Yixing Xu et al. 2020. A Survey on Visual Transformer. arXiv preprint arXiv:2012.12556 (2020)."},{"key":"e_1_3_2_2_9_1","volume-title":"Deep Compression: Compressing Deep Neural Networks with Pruning, Trained Quantization and Huffman Coding.","author":"Han Song","year":"2016"},{"key":"e_1_3_2_2_10_1","volume":"201","author":"Han Song","journal-title":"William J. Dally."},{"key":"e_1_3_2_2_11_1","volume-title":"Dissecting the NVIDIA volta GPU architecture via microbenchmarking. arXiv preprint arXiv:1804.06826","author":"Jia Zhe","year":"2018"},{"key":"e_1_3_2_2_12_1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis. 1--22","author":"Kwasniewski Grzegorz","year":"2019"},{"key":"e_1_3_2_2_13_1","volume-title":"2018 23rd Asia and South Pacific Design Automation Conference (ASP-DAC). IEEE, 534--539","author":"Li Bing","year":"2018"},{"key":"e_1_3_2_2_14_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research","author":"Liu Liu","year":"2020"},{"key":"e_1_3_2_2_15_1","volume":"201","author":"Mao Huizi","journal-title":"William J Dally."},{"key":"e_1_3_2_2_16_1","volume-title":"Exploring Sparsity in Recurrent Neural Networks. CoRR abs\/1704.05119","author":"Narang Sharan","year":"2017"},{"key":"e_1_3_2_2_17_1","volume-title":"GPU Technology Conference.","author":"Naumov M","year":"2010"},{"key":"e_1_3_2_2_18_1","volume-title":"V100 GPU Architecture: The world's most advanced datacenter GPU","author":"Tesla NVIDIA.","year":"2017"},{"key":"e_1_3_2_2_19_1","volume-title":"Squantizer: Simultaneous learning for both sparse and low-precision neural networks. arXiv preprint arXiv:1812.08301","author":"Park Mi Sun","year":"2018"},{"key":"e_1_3_2_2_20_1","volume-title":"PyTorch: An Imperative Style","author":"Paszke Adam"},{"key":"e_1_3_2_2_21_1","volume-title":"2019 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS). IEEE, 79--92","author":"Raihan Md Aamir","year":"2019"},{"key":"e_1_3_2_2_22_1","unstructured":"Google Research. [n.d.]. Deep Learning Matrix Collection. https:\/\/github.com\/google-research\/google-research\/tree\/master\/sgk.  Google Research. [n.d.]. Deep Learning Matrix Collection. https:\/\/github.com\/google-research\/google-research\/tree\/master\/sgk."},{"key":"e_1_3_2_2_23_1","volume-title":"Adaptive Attention Span in Transformers. CoRR abs\/1905.07799","author":"Sukhbaatar Sainbayar","year":"2019"},{"key":"e_1_3_2_2_24_1","volume-title":"Long Range Arena: A Benchmark for Efficient Transformers. arXiv preprint arXiv:2011.04006","author":"Tay Yi","year":"2020"},{"key":"e_1_3_2_2_25_1","volume-title":"Efficient transformers: A survey. arXiv preprint arXiv:2009.06732","author":"Tay Yi","year":"2020"},{"key":"e_1_3_2_2_26_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In NIPS. 6000--6010. http:\/\/papers.nips.cc\/paper\/7181-attention-is-all-you-need  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In NIPS. 6000--6010. http:\/\/papers.nips.cc\/paper\/7181-attention-is-all-you-need"},{"key":"e_1_3_2_2_27_1","volume-title":"2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2861--2865","author":"Venkatesh Ganesh","year":"2017"},{"key":"e_1_3_2_2_28_1","volume-title":"Optimizing Batched Winograd Convolution on GPUs. In 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (PPoPP '20)","author":"Yan Da","year":"2020"},{"key":"e_1_3_2_2_29_1","volume-title":"Big Bird: Transformers for Longer Sequences. arXiv:2007.14062 [cs.LG]","author":"Zaheer Manzil","year":"2021"},{"key":"e_1_3_2_2_30_1","volume-title":"Joshua Ainslie","author":"Zaheer Manzil","year":"2020"},{"key":"e_1_3_2_2_31_1","volume-title":"Proceedings of the 52nd Annual IEEE\/ACM International Symposium on Microarchitecture","author":"Zhu Maohua","year":"2019"}],"event":{"name":"SC '21: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St. Louis Missouri","acronym":"SC '21","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","IEEE CS"]},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458817.3476182","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3458817.3476182","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3458817.3476182","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:12:21Z","timestamp":1750191141000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458817.3476182"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,11,13]]},"references-count":31,"alternative-id":["10.1145\/3458817.3476182","10.1145\/3458817"],"URL":"https:\/\/doi.org\/10.1145\/3458817.3476182","relation":{},"subject":[],"published":{"date-parts":[[2021,11,13]]}}}