{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T01:43:24Z","timestamp":1787017404872,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,2,21]],"date-time":"2023-02-21T00:00:00Z","timestamp":1676937600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,2,25]]},"DOI":"10.1145\/3572848.3577500","type":"proceedings-article","created":{"date-parts":[[2023,2,21]],"date-time":"2023-02-21T11:02:30Z","timestamp":1676977350000},"page":"369-379","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":28,"title":["Dynamic N:M Fine-Grained Structured Sparse Attention Mechanism"],"prefix":"10.1145","author":[{"given":"Zhaodong","family":"Chen","sequence":"first","affiliation":[{"name":"UC Santa Barbara, Santa Barbara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zheng","family":"Qu","sequence":"additional","affiliation":[{"name":"UC Santa Barbara, Santa Barbara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuying","family":"Quan","sequence":"additional","affiliation":[{"name":"UC Santa Barbara, Santa Barbara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liu","family":"Liu","sequence":"additional","affiliation":[{"name":"Rensselaer Polytechnic Institute, Santa Barbara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yufei","family":"Ding","sequence":"additional","affiliation":[{"name":"UC Santa Barbara, Santa Barbara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuan","family":"Xie","sequence":"additional","affiliation":[{"name":"ALIBABA DAMO Academy, Santa Barbara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,2,21]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Longformer: The long-document transformer. arXiv preprint arXiv:2004.05150","author":"Beltagy Iz","year":"2020","unstructured":"Iz Beltagy, Matthew E Peters, and Arman Cohan. 2020. Longformer: The long-document transformer. arXiv preprint arXiv:2004.05150 (2020)."},{"key":"e_1_3_2_2_2_1","volume-title":"International Conference on Machine Learning. PMLR, 1695--1706","author":"Chen Tianlong","year":"2021","unstructured":"Tianlong Chen, Yongduo Sui, Xuxi Chen, Aston Zhang, and Zhangyang Wang. 2021. A unified lottery ticket hypothesis for graph neural networks. In International Conference on Machine Learning. PMLR, 1695--1706."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Zhaodong Chen Zheng Qu Liu Liu Yufei Ding and Yuan Xie. 2021. Efficient Tensor Core-Based GPU Kernels for Structured Sparsity under Reduced Precision. (2021).","DOI":"10.1145\/3458817.3476182"},{"key":"e_1_3_2_2_4_1","volume-title":"Generating long sequences with sparse transformers. arXiv preprint arXiv:1904.10509","author":"Child Rewon","year":"2019","unstructured":"Rewon Child, Scott Gray, Alec Radford, and Ilya Sutskever. 2019. Generating long sequences with sparse transformers. arXiv preprint arXiv:1904.10509 (2019)."},{"key":"e_1_3_2_2_5_1","volume-title":"Rethinking Attention with Performers. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Ua6zuk0WRH","author":"Choromanski Krzysztof Marcin","year":"2021","unstructured":"Krzysztof Marcin Choromanski, Valerii Likhosherstov, David Dohan, Xingyou Song, Andreea Gane, Tamas Sarlos, Peter Hawkins, Jared Quincy Davis, Afroz Mohiuddin, Lukasz Kaiser, David Benjamin Belanger, Lucy J Colwell, and Adrian Weller. 2021. Rethinking Attention with Performers. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Ua6zuk0WRH"},{"key":"e_1_3_2_2_6_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=YicbFdNTTy","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=YicbFdNTTy"},{"key":"e_1_3_2_2_7_1","volume-title":"Trainable Neural Networks. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rJl-b3RcF7","author":"Frankle Jonathan","year":"2019","unstructured":"Jonathan Frankle and Michael Carbin. 2019. The Lottery Ticket Hypothesis: Finding Sparse, Trainable Neural Networks. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rJl-b3RcF7"},{"key":"e_1_3_2_2_8_1","volume-title":"Lightweight Self-Attention Mechanism in Neural Networks. In 2021 ACM\/IEEE 48th Annual International Symposium on Computer Architecture (ISCA). IEEE, 692--705","author":"Ham Tae Jun","year":"2021","unstructured":"Tae Jun Ham, Yejin Lee, Seong Hoon Seo, Soosung Kim, Hyunji Choi, Sung Jun Jung, and Jae W Lee. 2021. ELSA: Hardware-Software Co-design for Efficient, Lightweight Self-Attention Mechanism in Neural Networks. In 2021 ACM\/IEEE 48th Annual International Symposium on Computer Architecture (ISCA). IEEE, 692--705."},{"key":"e_1_3_2_2_9_1","volume-title":"International Conference on Machine Learning. PMLR, 5156--5165","author":"Katharopoulos Angelos","year":"2020","unstructured":"Angelos Katharopoulos, Apoorv Vyas, Nikolaos Pappas, and Fran\u00e7ois Fleuret. 2020. Transformers are rnns: Fast autoregressive transformers with linear attention. In International Conference on Machine Learning. PMLR, 5156--5165."},{"key":"e_1_3_2_2_10_1","unstructured":"Andrew Kerr. 2021. CUTLASS. https:\/\/github.com\/NVIDIA\/cutlass."},{"key":"e_1_3_2_2_11_1","volume-title":"Reformer: The Efficient Transformer. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkgNKkHtvB","author":"Kitaev Nikita","year":"2020","unstructured":"Nikita Kitaev, Lukasz Kaiser, and Anselm Levskaya. 2020. Reformer: The Efficient Transformer. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkgNKkHtvB"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2020.2979965"},{"key":"e_1_3_2_2_13_1","volume-title":"SOFT: Softmax-free Transformer with Linear Complexity. arXiv preprint arXiv:2110.11945","author":"Lu Jiachen","year":"2021","unstructured":"Jiachen Lu, Jinghan Yao, Junge Zhang, Xiatian Zhu, Hang Xu, Weiguo Gao, Chunjing Xu, Tao Xiang, and Li Zhang. 2021. SOFT: Softmax-free Transformer with Linear Complexity. arXiv preprint arXiv:2110.11945 (2021)."},{"key":"e_1_3_2_2_14_1","volume-title":"Jeff Pool, Darko Stosic, Dusan Stosic, Ganesh Venkatesh, Chong Yu, and Paulius Micikevicius.","author":"Mishra Asit","year":"2021","unstructured":"Asit Mishra, Jorge Albericio Latorre, Jeff Pool, Darko Stosic, Dusan Stosic, Ganesh Venkatesh, Chong Yu, and Paulius Micikevicius. 2021. Accelerating sparse deep neural networks. arXiv preprint arXiv:2104.08378 (2021)."},{"key":"e_1_3_2_2_15_1","unstructured":"NVIDIA. 2020. NVIDIA A100 Tensor Core GPU Architecture. https:\/\/images.nvidia.com\/aem-dam\/en-zz\/Solutions\/data-center\/nvidia-ampere-architecture-whitepaper.pdf."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-6301"},{"key":"e_1_3_2_2_17_1","volume-title":"International Conference on Machine Learning. PMLR, 4055--4064","author":"Parmar Niki","year":"2018","unstructured":"Niki Parmar, Ashish Vaswani, Jakob Uszkoreit, Lukasz Kaiser, Noam Shazeer, Alexander Ku, and Dustin Tran. 2018. Image transformer. In International Conference on Machine Learning. PMLR, 4055--4064."},{"key":"e_1_3_2_2_18_1","volume-title":"Channel Permutations for N: M Sparsity. Advances in Neural Information Processing Systems 34","author":"Pool Jeff","year":"2021","unstructured":"Jeff Pool and Chong Yu. 2021. Channel Permutations for N: M Sparsity. Advances in Neural Information Processing Systems 34 (2021)."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00353"},{"key":"e_1_3_2_2_20_1","volume-title":"International Conference on Machine Learning. PMLR, 9438--9447","author":"Tay Yi","year":"2020","unstructured":"Yi Tay, Dara Bahri, Liu Yang, Donald Metzler, and Da-Cheng Juan. 2020. Sparse sinkhorn attention. In International Conference on Machine Learning. PMLR, 9438--9447."},{"key":"e_1_3_2_2_21_1","volume-title":"Long Range Arena: A Benchmark for Efficient Transformers. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=qVyeW-grC2k","author":"Tay Yi","year":"2021","unstructured":"Yi Tay, Mostafa Dehghani, Samira Abnar, Yikang Shen, Dara Bahri, Philip Pham, Jinfeng Rao, Liu Yang, Sebastian Ruder, and Donald Metzler. 2021. Long Range Arena: A Benchmark for Efficient Transformers. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=qVyeW-grC2k"},{"key":"e_1_3_2_2_22_1","volume-title":"Efficient transformers: A survey. arXiv preprint arXiv:2009.06732","author":"Tay Yi","year":"2020","unstructured":"Yi Tay, Mostafa Dehghani, Dara Bahri, and Donald Metzler. 2020. Efficient transformers: A survey. arXiv preprint arXiv:2009.06732 (2020)."},{"key":"e_1_3_2_2_23_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008."},{"key":"e_1_3_2_2_24_1","volume-title":"SpAtten: Efficient Sparse Attention Architecture with Cascade Token and Head Pruning. In 2021 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, 97--110","author":"Wang Hanrui","year":"2021","unstructured":"Hanrui Wang, Zhekai Zhang, and Song Han. 2021. SpAtten: Efficient Sparse Attention Architecture with Cascade Token and Head Pruning. In 2021 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, 97--110."},{"key":"e_1_3_2_2_25_1","volume-title":"Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768","author":"Wang Sinong","year":"2020","unstructured":"Sinong Wang, Belinda Z Li, Madian Khabsa, Han Fang, and Hao Ma. 2020. Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768 (2020)."},{"key":"e_1_3_2_2_26_1","volume-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations. Association for Computational Linguistics, Online, 38--45","author":"Wolf Thomas","year":"2020","unstructured":"Thomas Wolf, Lysandre Debut, Victor Sanh, Julien Chaumond, Clement Delangue, Anthony Moi, Pierric Cistac, Tim Rault, R\u00e9mi Louf, Morgan Funtowicz, Joe Davison, Sam Shleifer, Patrick von Platen, Clara Ma, Yacine Jernite, Julien Plu, Canwen Xu, Teven Le Scao, Sylvain Gugger, Mariama Drame, Quentin Lhoest, and Alexander M. Rush. 2020. Transformers: State-of-the-Art Natural Language Processing. In Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations. Association for Computational Linguistics, Online, 38--45. https:\/\/www.aclweb.org\/anthology\/2020.emnlp-demos.6"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i16.17664"},{"key":"e_1_3_2_2_28_1","volume-title":"Joshua Ainslie, Chris Alberti, Santiago Ontanon, Philip Pham, Anirudh Ravula, Qifan Wang, Li Yang, et al.","author":"Zaheer Manzil","year":"2020","unstructured":"Manzil Zaheer, Guru Guruganesh, Kumar Avinava Dubey, Joshua Ainslie, Chris Alberti, Santiago Ontanon, Philip Pham, Anirudh Ravula, Qifan Wang, Li Yang, et al. 2020. Big Bird: Transformers for Longer Sequences.. In NeurIPS."},{"key":"e_1_3_2_2_29_1","volume-title":"Dropattention: A regularization method for fully-connected self-attention networks. arXiv preprint arXiv:1907.11065","author":"Zehui Lin","year":"2019","unstructured":"Lin Zehui, Pengfei Liu, Luyao Huang, Junkun Chen, Xipeng Qiu, and Xuanjing Huang. 2019. Dropattention: A regularization method for fully-connected self-attention networks. arXiv preprint arXiv:1907.11065 (2019)."},{"key":"e_1_3_2_2_30_1","volume-title":"Explicit sparse transformer: Concentrated attention through explicit selection. arXiv preprint arXiv:1912.11637","author":"Zhao Guangxiang","year":"2019","unstructured":"Guangxiang Zhao, Junyang Lin, Zhiyuan Zhang, Xuancheng Ren, Qi Su, and Xu Sun. 2019. Explicit sparse transformer: Concentrated attention through explicit selection. arXiv preprint arXiv:1912.11637 (2019)."},{"key":"e_1_3_2_2_31_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=K9bw7vqp_s","author":"Zhou Aojun","year":"2021","unstructured":"Aojun Zhou, Yukun Ma, Junnan Zhu, Jianbo Liu, Zhijie Zhang, Kun Yuan, Wenxiu Sun, and Hongsheng Li. 2021. Learning N:M Finegrained Structured Sparse Neural Networks From Scratch. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=K9bw7vqp_s"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358269"}],"event":{"name":"PPoPP '23: The 28th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming","location":"Montreal QC Canada","acronym":"PPoPP '23","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the 28th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3572848.3577500","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3572848.3577500","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T14:08:09Z","timestamp":1750169289000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3572848.3577500"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,21]]},"references-count":32,"alternative-id":["10.1145\/3572848.3577500","10.1145\/3572848"],"URL":"https:\/\/doi.org\/10.1145\/3572848.3577500","relation":{},"subject":[],"published":{"date-parts":[[2023,2,21]]},"assertion":[{"value":"2023-02-21","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}