{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T18:01:40Z","timestamp":1775671300595,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,6,22]],"date-time":"2021-06-22T00:00:00Z","timestamp":1624320000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Shanghai Science and Technology Commission Project","award":["20511101600"],"award-info":[{"award-number":["20511101600"]}]},{"name":"National Nature Science Foundation of China (NSFC)","award":["61972154"],"award-info":[{"award-number":["61972154"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,6,22]]},"DOI":"10.1145\/3453688.3461739","type":"proceedings-article","created":{"date-parts":[[2021,6,18]],"date-time":"2021-06-18T23:13:45Z","timestamp":1624058025000},"page":"163-168","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":35,"title":["Accommodating Transformer onto FPGA"],"prefix":"10.1145","author":[{"given":"Panjie","family":"Qi","sequence":"first","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuhong","family":"Song","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongwu","family":"Peng","sequence":"additional","affiliation":[{"name":"University of Connecticut, Storrs, CT, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shaoyi","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Connecticut, Storrs, CT, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingfeng","family":"Zhuge","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Edwin Hsing-Mean","family":"Sha","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,6,22]]},"reference":[{"key":"e_1_3_2_3_1_1","volume-title":"Templates for Solution of Linear Systems: Building Blocks for Iterative Methods. Siam","author":"Rb Et Al.","year":"1995","unstructured":"Rb Et Al. 1995. Templates for Solution of Linear Systems: Building Blocks for Iterative Methods. Siam (1995)."},{"key":"e_1_3_2_3_2_1","doi-asserted-by":"crossref","unstructured":"Mia Xu Chen Orhan Firat Ankur Bapna Melvin Johnson Wolfgang Macherey George Foster Llion Jones Niki Parmar Mike Schuster Zhifeng Chen et al. 2018. The best of both worlds: Combining recent advances in neural machine translation. arXiv preprint arXiv:1804.09849 (2018).","DOI":"10.18653\/v1\/P18-1008"},{"key":"e_1_3_2_3_3_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_3_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3124552"},{"key":"e_1_3_2_3_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3194554.3194625"},{"key":"e_1_3_2_3_6_1","volume-title":"Compressing BERT: Studying the effects of weight pruning on transfer learning. arXiv preprint arXiv:2002.08307","author":"Gordon Mitchell A","year":"2020","unstructured":"Mitchell A Gordon, Kevin Duh, and Nicholas Andrews. 2020. Compressing BERT: Studying the effects of weight pruning on transfer learning. arXiv preprint arXiv:2002.08307 (2020)."},{"key":"e_1_3_2_3_7_1","volume-title":"Learning both weights and connections for efficient neural networks. arXiv preprint arXiv:1506.02626","author":"Han Song","year":"2015","unstructured":"Song Han, Jeff Pool, John Tran, and William J Dally. 2015. Learning both weights and connections for efficient neural networks. arXiv preprint arXiv:1506.02626 (2015)."},{"key":"e_1_3_2_3_8_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3358192","article-title":"Achieving super-linear speedup across multi-fpga for real-time dnn inference","volume":"18","author":"Jiang Weiwen","year":"2019","unstructured":"Weiwen Jiang, Edwin H-M Sha, Xinyi Zhang, Lei Yang, Qingfeng Zhuge, Yiyu Shi, and Jingtong Hu. 2019. Achieving super-linear speedup across multi-fpga for real-time dnn inference. ACM Transactions on Embedded Computing Systems (TECS) 18, 5s (2019), 1--23.","journal-title":"ACM Transactions on Embedded Computing Systems (TECS)"},{"key":"e_1_3_2_3_9_1","first-page":"1","article-title":"Heterogeneous FPGA-Based Cost-Optimal Design for Timing-Constrained CNNs","volume":"11","author":"Jiang W.","year":"2018","unstructured":"W. Jiang, H. M. Sha, Q. Zhuge, Y. Lei, X. Chen, and J. Hu. 2018. Heterogeneous FPGA-Based Cost-Optimal Design for Timing-Constrained CNNs. IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems PP, 11 (2018), 1--1.","journal-title":"IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems PP"},{"key":"e_1_3_2_3_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2020.2986127"},{"key":"e_1_3_2_3_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3289602.3293988"},{"key":"e_1_3_2_3_12_1","volume-title":"2019 56th ACM\/IEEE Design Automation Conference (DAC).","author":"Jiang W.","unstructured":"W. Jiang, X. Zhang, H. M Sha, L. Yang, Q. Zhuge, Y. Shi, and J. Hu. 2019. Accuracy vs. Efficiency: Achieving Both through FPGA-Implementation Aware Neural Architecture Search. In 2019 56th ACM\/IEEE Design Automation Conference (DAC)."},{"key":"e_1_3_2_3_13_1","volume-title":"Efficient Transformer-based Large Scale Language Representations using Hardware-friendly Block Structured Pruning. arXiv preprint arXiv:2009.08065","author":"Li Bingbing","year":"2020","unstructured":"Bingbing Li, Zhenglun Kong, Tianyun Zhang, Ji Li, Zhengang Li, Hang Liu, and Caiwen Ding. 2020. Efficient Transformer-based Large Scale Language Representations using Hardware-friendly Block Structured Pruning. arXiv preprint arXiv:2009.08065 (2020)."},{"key":"e_1_3_2_3_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3370748.3406567"},{"key":"e_1_3_2_3_15_1","volume-title":"The 29th ACM International Conference on Supercomputing (ICS '15)","author":"Liu W.","unstructured":"W. Liu and B. Vinter. 2015. CSR5: An Efficient Storage Format for Cross-Platform Sparse Matrix-Vector Multiplication. In The 29th ACM International Conference on Supercomputing (ICS '15)."},{"key":"e_1_3_2_3_16_1","volume-title":"Pointer sentinel mixture models. arXiv preprint arXiv:1609.07843","author":"Merity Stephen","year":"2016","unstructured":"Stephen Merity, Caiming Xiong, James Bradbury, and Richard Socher. 2016. Pointer sentinel mixture models. arXiv preprint arXiv:1609.07843 (2016)."},{"key":"e_1_3_2_3_17_1","volume-title":"Block-sparse recurrent neural networks. arXiv preprint arXiv:1711.02782","author":"Narang Sharan","year":"2017","unstructured":"Sharan Narang, Eric Undersander, and Gregory Diamos. 2017. Block-sparse recurrent neural networks. arXiv preprint arXiv:1711.02782 (2017)."},{"key":"e_1_3_2_3_18_1","volume-title":"a distilled version of BERT: smaller, faster, cheaper and lighter. arXiv preprint arXiv:1910.01108","author":"Sanh Victor","year":"2019","unstructured":"Victor Sanh, Lysandre Debut, Julien Chaumond, and Thomas Wolf. 2019. DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter. arXiv preprint arXiv:1910.01108 (2019)."},{"key":"e_1_3_2_3_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18072.2020.9218581"},{"key":"e_1_3_2_3_20_1","doi-asserted-by":"crossref","unstructured":"R. Shi P. Dong T. Geng Y. Ding and Y. Wang. 2020. CSB-RNN: A Faster-than- Realtime RNN Acceleration Framework with Compressed Structured Blocks. (2020).","DOI":"10.1145\/3392717.3392749"},{"key":"e_1_3_2_3_21_1","volume-title":"International Conference on Machine Learning. PMLR, 5877--5886","author":"So David","year":"2019","unstructured":"David So, Quoc Le, and Chen Liang. 2019. The evolved transformer. In International Conference on Machine Learning. PMLR, 5877--5886."},{"key":"e_1_3_2_3_22_1","volume-title":"Sakyasingha Dasgupta, Yiyu Shi, and Caiwen Ding.","author":"Song Yuhong","year":"2021","unstructured":"Yuhong Song, Weiwen Jiang, Bingbing Li, Panjie Qi, Qingfeng Zhuge, Edwin Hsing-Mean Sha, Sakyasingha Dasgupta, Yiyu Shi, and Caiwen Ding. 2021. Dancing along Battery: Enabling Transformer with Run-time Reconfigurability on Mobile Devices. arXiv preprint arXiv:2102.06336 (2021)."},{"key":"e_1_3_2_3_23_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008."},{"key":"e_1_3_2_3_24_1","volume-title":"Hat: Hardware-aware transformers for efficient natural language processing. arXiv preprint arXiv:2005.14187","author":"Wang Hanrui","year":"2020","unstructured":"Hanrui Wang, Zhanghao Wu, Zhijian Liu, Han Cai, Ligeng Zhu, Chuang Gan, and Song Han. 2020. Hat: Hardware-aware transformers for efficient natural language processing. arXiv preprint arXiv:2005.14187 (2020)."},{"key":"e_1_3_2_3_25_1","volume-title":"Accelerating neural transformer via an average attention network. arXiv preprint arXiv:1805.00631","author":"Zhang Biao","year":"2018","unstructured":"Biao Zhang, Deyi Xiong, and Jinsong Su. 2018. Accelerating neural transformer via an average attention network. arXiv preprint arXiv:1805.00631 (2018)."}],"event":{"name":"GLSVLSI '21: Great Lakes Symposium on VLSI 2021","location":"Virtual Event USA","acronym":"GLSVLSI '21","sponsor":["SIGDA ACM Special Interest Group on Design Automation"]},"container-title":["Proceedings of the 2021 Great Lakes Symposium on VLSI"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3453688.3461739","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3453688.3461739","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:28:47Z","timestamp":1750195727000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3453688.3461739"}},"subtitle":["Coupling the Balanced Model Compression and FPGA-Implementation Optimization"],"short-title":[],"issued":{"date-parts":[[2021,6,22]]},"references-count":25,"alternative-id":["10.1145\/3453688.3461739","10.1145\/3453688"],"URL":"https:\/\/doi.org\/10.1145\/3453688.3461739","relation":{},"subject":[],"published":{"date-parts":[[2021,6,22]]},"assertion":[{"value":"2021-06-22","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}