{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T16:10:35Z","timestamp":1781885435027,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":14,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,5]],"date-time":"2024-08-05T00:00:00Z","timestamp":1722816000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"SRC\/DARPA, Samsung Electronics","award":["JUMP 2.0 CoCoSys"],"award-info":[{"award-number":["JUMP 2.0 CoCoSys"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,5]]},"DOI":"10.1145\/3665314.3670818","type":"proceedings-article","created":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T19:31:18Z","timestamp":1725910278000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["A 28nm Scalable and Flexible Accelerator for Sparse Transformer Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-5040-2381","authenticated-orcid":false,"given":"Yuan","family":"Liao","sequence":"first","affiliation":[{"name":"Cornell Tech, Cornell University, New York, NY, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7703-5020","authenticated-orcid":false,"given":"Jian","family":"Meng","sequence":"additional","affiliation":[{"name":"Cornell Tech, Cornell University, New York City, NY, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4551-7789","authenticated-orcid":false,"given":"Jae-sun","family":"Seo","sequence":"additional","affiliation":[{"name":"Cornell Tech, Cornell University, New York, NY, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,9,9]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Ashish Vaswani et al. 2017. Attention is All you Need. In NeurIPS."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Eunji Yoo et al. 2023. TF-MVP: Novel Sparsity-Aware Transformer Accelerator with Mixed-Length Vector Pruning. In DAC. 1--6.","DOI":"10.1109\/DAC56929.2023.10247799"},{"key":"e_1_3_2_1_3_1","volume-title":"SALO: An Efficient Spatial Accelerator Enabling Hybrid Sparse Attention Mechanisms for Long Sequences. In DAC. 571--576.","author":"Guan Shen","year":"2022","unstructured":"Guan Shen et al. 2022. SALO: An Efficient Spatial Accelerator Enabling Hybrid Sparse Attention Mechanisms for Long Sequences. In DAC. 571--576."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Hanrui Wang et al. 2021. SpAtten: Efficient Sparse Attention Architecture with Cascade Token and Head Pruning. In HPCA. 97--110.","DOI":"10.1109\/HPCA51647.2021.00018"},{"key":"e_1_3_2_1_5_1","volume-title":"Rethinking the Self-Attention in Vision Transformers. In CVPR Workshops. 3065--3069","author":"Kyungmin","unstructured":"Kyungmin Kim et al. 2021. Rethinking the Self-Attention in Vision Transformers. In CVPR Workshops. 3065--3069."},{"key":"e_1_3_2_1_6_1","volume-title":"Sanger: A Co-Design Framework for Enabling Sparse Attention Using Reconfigurable Architecture. In MICRO. 977--991.","author":"Liqiang Lu","year":"2021","unstructured":"Liqiang Lu et al. 2021. Sanger: A Co-Design Framework for Enabling Sparse Attention Using Reconfigurable Architecture. In MICRO. 977--991."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Meiqi Wang et al. 2018. A High-Speed and Low-Complexity Architecture for Softmax Function in Deep Learning. In APCCAS. 223--226.","DOI":"10.1109\/APCCAS.2018.8605654"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Nitish Srivastava et al. 2020. MatRaptor: A Sparse-Sparse Matrix Multiplication Accelerator Based on Row-Wise Product. In MICRO. 766--780.","DOI":"10.1109\/MICRO50266.2020.00068"},{"key":"e_1_3_2_1_9_1","volume-title":"Platon: Pruning large transformer models with upper confidence bound of weight importance. In ICLR. 26809--26823.","author":"Qingru Zhang","year":"2022","unstructured":"Qingru Zhang et al. 2022. Platon: Pruning large transformer models with upper confidence bound of weight importance. In ICLR. 26809--26823."},{"key":"e_1_3_2_1_10_1","unstructured":"Tae Jun Ham et al. 2020. A3: Accelerating Attention Mechanisms in Neural Networks with Approximation. In HPCA. 328--341."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Thierry Tambe et al. 2023. 22.9 A 12nm 18.1TFLOPs\/W Sparse Transformer Processor with Entropy-Based Early Exit Mixed-Precision Predication and FineGrained Power Management. In ISSCC. 342--344.","DOI":"10.1109\/ISSCC42615.2023.10067817"},{"key":"e_1_3_2_1_12_1","first-page":"1","article-title":"A 28nm 27.5TOPS\/W Approximate-Computing-Based Transformer Processor with Asymptotic Sparsity Speculating and Out-of-Order Computing","volume":"65","author":"Yang Wang","year":"2022","unstructured":"Yang Wang et al. 2022. A 28nm 27.5TOPS\/W Approximate-Computing-Based Transformer Processor with Asymptotic Sparsity Speculating and Out-of-Order Computing. In ISSCC, Vol. 65. 1--3.","journal-title":"ISSCC"},{"key":"e_1_3_2_1_13_1","article-title":"Two Fast Algorithms for Sparse Matrices: Multiplication and Permuted Transposition","volume":"4","author":"Fred G.","year":"1978","unstructured":"Fred G. Gustavson. 1978. Two Fast Algorithms for Sparse Matrices: Multiplication and Permuted Transposition. ACM Trans. Math. Software 4, 3 (1978).","journal-title":"ACM Trans. Math. Software"},{"key":"e_1_3_2_1_14_1","unstructured":"Shiwei Liu et al. 2021. Sparse Training via Boosting Pruning Plasticity with Neuroregeneration. Advances in Neural Information Processing Systems (2021)."}],"event":{"name":"ISLPED '24: 29th ACM\/IEEE International Symposium on Low Power Electronics and Design","location":"Newport Beach CA USA","acronym":"ISLPED '24","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEEE CAS","IEEE EDA"]},"container-title":["Proceedings of the 29th ACM\/IEEE International Symposium on Low Power Electronics and Design"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3665314.3670818","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3665314.3670818","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:51Z","timestamp":1750294671000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3665314.3670818"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,5]]},"references-count":14,"alternative-id":["10.1145\/3665314.3670818","10.1145\/3665314"],"URL":"https:\/\/doi.org\/10.1145\/3665314.3670818","relation":{},"subject":[],"published":{"date-parts":[[2024,8,5]]},"assertion":[{"value":"2024-09-09","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}