{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T16:32:16Z","timestamp":1781886736394,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,7,10]],"date-time":"2022-07-10T00:00:00Z","timestamp":1657411200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"NSF CRII Award","award":["2000722"],"award-info":[{"award-number":["2000722"]}]},{"name":"U.S. DOE Office of Science, Office of Advanced Scientific Computing Research","award":["66150"],"award-info":[{"award-number":["66150"]}]},{"name":"NSF CAREER Award","award":["2011236; 2006748; 2046102"],"award-info":[{"award-number":["2011236; 2006748; 2046102"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,7,10]]},"DOI":"10.1145\/3489517.3530585","type":"proceedings-article","created":{"date-parts":[[2022,8,23]],"date-time":"2022-08-23T23:19:29Z","timestamp":1661296769000},"page":"1135-1140","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":47,"title":["A length adaptive algorithm-hardware co-design of transformer on FPGA through sparse attention and dynamic pipelining"],"prefix":"10.1145","author":[{"given":"Hongwu","family":"Peng","sequence":"first","affiliation":[{"name":"University of Connecticut"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shaoyi","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Connecticut"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiyang","family":"Chen","sequence":"additional","affiliation":[{"name":"Stevens Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bingbing","family":"Li","sequence":"additional","affiliation":[{"name":"University of Connecticut"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tong","family":"Geng","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ang","family":"Li","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weiwen","family":"Jiang","sequence":"additional","affiliation":[{"name":"George Mason University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wujie","family":"Wen","sequence":"additional","affiliation":[{"name":"Lehigh University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinbo","family":"Bi","sequence":"additional","affiliation":[{"name":"University of Connecticut"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hang","family":"Liu","sequence":"additional","affiliation":[{"name":"Stevens Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Caiwen","family":"Ding","sequence":"additional","affiliation":[{"name":"University of Connecticut"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,8,23]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Why Multi-head Self Attention Works: Math, Intuitions and 10+1 Hidden Insights. https:\/\/theaisummer.com\/self-attention\/","author":"Adaloglou Nikolas","year":"2021","unstructured":"Nikolas Adaloglou. Why Multi-head Self Attention Works: Math, Intuitions and 10+1 Hidden Insights. https:\/\/theaisummer.com\/self-attention\/, 2021. [Online; accessed August 30, 2021]."},{"key":"e_1_3_2_1_2_1","volume-title":"On the Relationship Between Self-Attention and Convolutional Layers. In ICLR","author":"Jean-Baptiste","year":"2019","unstructured":"Jean-Baptiste Cordonnier et al. On the Relationship Between Self-Attention and Convolutional Layers. In ICLR, 2019."},{"key":"e_1_3_2_1_3_1","first-page":"5998","volume-title":"Advances in neural information processing systems","author":"Ashish Vaswani","year":"2017","unstructured":"Ashish Vaswani et al. Attention is All You Need. In Advances in neural information processing systems, pages 5998--6008, 2017."},{"key":"e_1_3_2_1_4_1","volume-title":"Deep Residual Learning for Image Recognition. In CVPR","author":"Kaiming","year":"2016","unstructured":"Kaiming He et al. Deep Residual Learning for Image Recognition. In CVPR, 2016."},{"key":"e_1_3_2_1_5_1","volume-title":"ICLR","author":"Alexey","year":"2021","unstructured":"Alexey Dosovitskiy et al. An Image is Worth 16\u00d7 16 Words: Transformers for Image Recognition at Scale. In ICLR, 2021."},{"key":"e_1_3_2_1_6_1","volume-title":"CVPR","author":"Ze","year":"2021","unstructured":"Ze Liu et al. Swin Transformer: Hierarchical Vision Transformer using Shifted Windows. In CVPR, 2021."},{"key":"e_1_3_2_1_7_1","volume-title":"End-to-End Human Pose and Mesh Reconstruction with Transformers. In CVPR","author":"Kevin","year":"2021","unstructured":"Kevin Lin et al. End-to-End Human Pose and Mesh Reconstruction with Transformers. In CVPR, 2021."},{"key":"e_1_3_2_1_8_1","first-page":"784","volume-title":"ACL","author":"Pranav","year":"2018","unstructured":"Pranav Rajpurkar et al. Know What You Don't Know: Unanswerable Questions for SQuAD. In ACL, pages 784--789, 2018."},{"key":"e_1_3_2_1_9_1","first-page":"17283","volume-title":"NeurIPS","volume":"33","author":"Manzil","year":"2020","unstructured":"Manzil Zaheer et al. Big Bird: Transformers for Longer Sequences. In H. Larochelle, M. Ranzato, R. Hadsell, M. F. Balcan, and H. Lin, editors, NeurIPS, volume 33, pages 17283--17297. Curran Associates, Inc., 2020."},{"key":"e_1_3_2_1_10_1","volume-title":"BP-Transformer: Modelling Long-range Context via Binary Partitioning. arXiv preprint arXiv:1911.04070","author":"Zihao Ye","year":"2019","unstructured":"Zihao Ye et al. BP-Transformer: Modelling Long-range Context via Binary Partitioning. arXiv preprint arXiv:1911.04070, 2019."},{"key":"e_1_3_2_1_11_1","volume-title":"Reformer: The Efficient Transformer. In ICLR","author":"Nikita","year":"2019","unstructured":"Nikita Kitaev et al. Reformer: The Efficient Transformer. In ICLR, 2019."},{"key":"e_1_3_2_1_12_1","first-page":"328","volume-title":"2020 HPCA","author":"Jun Tae","year":"2020","unstructured":"Tae Jun Ham et al. A^ 3: Accelerating Attention Mechanisms in Neural Networks with Approximation. In 2020 HPCA, pages 328--341. IEEE, 2020."},{"key":"e_1_3_2_1_13_1","first-page":"97","volume-title":"SpAtten: Efficient Sparse Attention Architecture with Cascade Token and Head Pruning. In 2021 HPCA","author":"Hanrui","year":"2021","unstructured":"Hanrui Wang et al. SpAtten: Efficient Sparse Attention Architecture with Cascade Token and Head Pruning. In 2021 HPCA, pages 97--110. IEEE, 2021."},{"key":"e_1_3_2_1_14_1","volume-title":"Lightweight Self-Attention Mechanism in Neural Networks. In ISCA. IEEE","author":"Jun Tae","year":"2021","unstructured":"Tae Jun Ham et al. ELSA: Hardware-Software Co-design for Efficient, Lightweight Self-Attention Mechanism in Neural Networks. In ISCA. IEEE, 2021."},{"key":"e_1_3_2_1_15_1","volume-title":"Retrived fromhttps:\/\/developer.nvidia.com\/tensorrt. Online","author":"NVIDIA.","year":"2021","unstructured":"NVIDIA. TensorRT. Retrived fromhttps:\/\/developer.nvidia.com\/tensorrt. Online; accessed: October 6, 2021."},{"key":"e_1_3_2_1_16_1","volume-title":"Pointer Sentinel Mixture Models. ICLR","author":"Merity","year":"2016","unstructured":"Merity Stephen et al. Pointer Sentinel Mixture Models. ICLR, 2016."},{"key":"e_1_3_2_1_17_1","first-page":"38","volume-title":"EMNLP","author":"Thomas","year":"2020","unstructured":"Thomas Wolf et al. Transformers: State-of-the-art natural language processing. In EMNLP, pages 38--45, 2020."},{"key":"e_1_3_2_1_18_1","first-page":"1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Shiyang","year":"2021","unstructured":"Shiyang Chen et al. Et: re-thinking self-attention for transformer models on gpus. In Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, pages 1--18, 2021."},{"key":"e_1_3_2_1_19_1","first-page":"169","volume-title":"Proceedings of the 2021 on Great Lakes Symposium on VLSI","author":"Shaoyi","year":"2021","unstructured":"Shaoyi Huang et al. Hmc-tran: A tensor-core inspired hierarchical model compression for transformer-based dnns on gpu. In Proceedings of the 2021 on Great Lakes Symposium on VLSI, pages 169--174, 2021."},{"key":"e_1_3_2_1_20_1","first-page":"389","volume-title":"Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","author":"Jiarui","year":"2021","unstructured":"Jiarui Fang et al. Turbotransformers: an efficient gpu serving system for transformer models. In Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pages 389--402, 2021."},{"key":"e_1_3_2_1_21_1","volume-title":"Benchmarking High Bandwidth Memory on FPGAs. arXiv preprint arXiv:2005.04324","author":"Zeke Wang","year":"2020","unstructured":"Zeke Wang et al. Benchmarking High Bandwidth Memory on FPGAs. arXiv preprint arXiv:2005.04324, 2020."},{"key":"e_1_3_2_1_22_1","volume-title":"GLSVLSI","author":"Panjie","year":"2021","unstructured":"Panjie Qi et al. Accommodating transformer onto fpga: Coupling the balanced model compression and fpga-implementation optimization. In GLSVLSI, 2021."},{"key":"e_1_3_2_1_23_1","first-page":"85","volume-title":"2021 IEEE 32nd ASAP","author":"Hongwu","year":"2021","unstructured":"Hongwu Peng et al. Binary complex neural network acceleration on fpga. In 2021 IEEE 32nd ASAP, pages 85--92. IEEE, 2021."},{"key":"e_1_3_2_1_24_1","volume-title":"Optimizing FPGA-based Accelerator Design for Large-Scale Molecular Similarity Search. In ICCAD","author":"Hongwu","year":"2021","unstructured":"Hongwu Peng et al. Optimizing FPGA-based Accelerator Design for Large-Scale Molecular Similarity Search. In ICCAD, 2021."},{"key":"e_1_3_2_1_25_1","first-page":"4171","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In 2019 ACL","author":"Jacob","year":"2019","unstructured":"Jacob Devlin et al. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In 2019 ACL, pages 4171--4186, 2019."},{"key":"e_1_3_2_1_26_1","volume-title":"Faster, Cheaper and Lighter. arXiv preprint arXiv:1910.01108","author":"Victor Sanh","year":"2019","unstructured":"Victor Sanh et al. DistilBERT, a Distilled Version of BERT: Smaller, Faster, Cheaper and Lighter. arXiv preprint arXiv:1910.01108, 2019."},{"key":"e_1_3_2_1_27_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Yinhan Liu","year":"2019","unstructured":"Yinhan Liu et al. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692, 2019."},{"key":"e_1_3_2_1_28_1","first-page":"2383","volume-title":"EMNLP","author":"Pranav","year":"2016","unstructured":"Pranav Rajpurkar et al. SQuAD: 100,000+ Questions for Machine Comprehension of Text. In EMNLP, pages 2383--2392, Austin, Texas, November 2016. Association for Computational Linguistics."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1017\/S1351324909990234"},{"key":"e_1_3_2_1_30_1","volume-title":"Third International Workshop on Paraphrasing. AFNLP","author":"Bill","year":"2005","unstructured":"Bill Dolan et al. Automatically constructing a corpus of sentential paraphrases. In Third International Workshop on Paraphrasing. AFNLP, 2005."},{"key":"e_1_3_2_1_31_1","first-page":"509","volume-title":"TernaryBERT: Distillation-aware Ultra-low Bit BERT. In 2020 EMNLP","author":"Wei","year":"2020","unstructured":"Wei Zhang et al. TernaryBERT: Distillation-aware Ultra-low Bit BERT. In 2020 EMNLP, pages 509--521, 2020."},{"key":"e_1_3_2_1_32_1","first-page":"1","volume-title":"2021 ICCAD","author":"Panjie","year":"2021","unstructured":"Panjie Qi et al. Accelerating framework of transformer by hardware design and model compression co-optimization. In 2021 ICCAD, pages 1--9. IEEE, 2021."}],"event":{"name":"DAC '22: 59th ACM\/IEEE Design Automation Conference","location":"San Francisco California","acronym":"DAC '22","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEEE CEDA"]},"container-title":["Proceedings of the 59th ACM\/IEEE Design Automation Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3489517.3530585","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3489517.3530585","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:18Z","timestamp":1750186938000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3489517.3530585"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,7,10]]},"references-count":32,"alternative-id":["10.1145\/3489517.3530585","10.1145\/3489517"],"URL":"https:\/\/doi.org\/10.1145\/3489517.3530585","relation":{},"subject":[],"published":{"date-parts":[[2022,7,10]]},"assertion":[{"value":"2022-08-23","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}