{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T16:16:25Z","timestamp":1759335385903,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":19,"publisher":"ACM","funder":[{"name":"NSFC","award":["62372442"],"award-info":[{"award-number":["62372442"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3716368.3735201","type":"proceedings-article","created":{"date-parts":[[2025,6,27]],"date-time":"2025-06-27T13:58:23Z","timestamp":1751032703000},"page":"733-739","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["AttenPU: An Area Efficient Attention Processor with Reconfigurable FP8 Precision and Dataflow"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-6944-2548","authenticated-orcid":false,"given":"Qiawei","family":"Zheng","sequence":"first","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China and University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7672-1299","authenticated-orcid":false,"given":"Pu","family":"Zhou","sequence":"additional","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2855-9570","authenticated-orcid":false,"given":"Zheng","family":"Wang","sequence":"additional","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9075-7626","authenticated-orcid":false,"given":"Zhuoyu","family":"Wu","sequence":"additional","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1140-1194","authenticated-orcid":false,"given":"Yike","family":"Li","sequence":"additional","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4918-619X","authenticated-orcid":false,"given":"Zhihao","family":"Du","sequence":"additional","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6488-224X","authenticated-orcid":false,"given":"Chao","family":"Chen","sequence":"additional","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1159-3115","authenticated-orcid":false,"given":"Yongkui","family":"Yang","sequence":"additional","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3347-0511","authenticated-orcid":false,"given":"Wenqi","family":"Fang","sequence":"additional","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8818-6983","authenticated-orcid":false,"given":"Anupam","family":"Chattopadhyay","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,29]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Jacob Devlin. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1810.04805 (2018)."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00311"},{"key":"e_1_3_3_1_4_2","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems (2017)."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Sneha Chaudhari Varun Mithal Gungor Polatkan and Rohan Ramanath. 2021. An attentive survey of attention models. ACM Transactions on Intelligent Systems and Technology (TIST) 12 5 (2021) 1\u201332.","DOI":"10.1145\/3465055"},{"key":"e_1_3_3_1_6_2","unstructured":"Andrey Kuzmin Mart Van\u00a0Baalen Yuwei Ren Markus Nagel Jorn Peters and Tijmen Blankevoort. 2022. Fp8 quantization: The power of the exponent. Advances in Neural Information Processing Systems 35 (2022) 14651\u201314662."},{"key":"e_1_3_3_1_7_2","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et\u00a0al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.12948 (2025)."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Wenzhe Zhao Qiwei Dang Tian Xia Jingming Zhang Nanning Zheng and Pengju Ren. 2023. Optimizing FPGA-Based DNN accelerator with shared exponential floating-point format. IEEE Transactions on Circuits and Systems I: Regular Papers (2023).","DOI":"10.1109\/TCSI.2023.3300657"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"crossref","unstructured":"Chen Wu Mingyu Wang Xinyuan Chu Kun Wang and Lei He. 2021. Low-precision floating-point arithmetic for high-performance fpga-based cnn acceleration. ACM Transactions on Reconfigurable Technology and Systems (TRETS) 15 1 (2021) 1\u201321.","DOI":"10.1145\/3474597"},{"key":"e_1_3_3_1_10_2","unstructured":"Mart van Baalen Andrey Kuzmin Suparna\u00a0S Nair Yuwei Ren Eric Mahurin Chirag Patel Sundar Subramanian Sanghyuk Lee Markus Nagel Joseph Soriaga et\u00a0al. 2023. FP8 versus INT8 for efficient deep learning inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.17951 (2023)."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/DAC56929.2023.10247981"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Fengbin Tu Zihan Wu Yiqi Wang Ling Liang Liu Liu Yufei Ding Leibo Liu Shaojun Wei Yuan Xie and Shouyi Yin. 2022. TranCIM: Full-digital bitline-transpose CIM-based sparse transformer accelerator with pipeline\/parallel reconfigurable modes. IEEE Journal of Solid-State Circuits 58 6 (2022) 1798\u20131809.","DOI":"10.1109\/JSSC.2022.3213542"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS46773.2023.10181465"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICFPT56656.2022.9974324"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"crossref","unstructured":"Zbigniew Hajduk. 2017. High accuracy FPGA activation function implementation for neural networks. Neurocomputing 247 (2017) 59\u201361.","DOI":"10.1016\/j.neucom.2017.03.044"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575747"},{"key":"e_1_3_3_1_17_2","unstructured":"Jay Shah Ganesh Bikshandi Ying Zhang Vijay Thakkar Pradeep Ramani and Tri Dao. 2024. Flashattention-3: Fast and accurate attention with asynchrony and low-precision. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.08608 (2024)."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Jeongwoo Park Sunwoo Lee and Dongsuk Jeon. 2021. A neural network training processor with 8-bit shared exponent bias floating point and multiple-way fused multiply-add trees. IEEE Journal of Solid-State Circuits 57 3 (2021) 965\u2013977.","DOI":"10.1109\/JSSC.2021.3103603"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"crossref","unstructured":"Shreyas\u00a0Kolala Venkataramanaiah Jian Meng Han-Sok Suh Injune Yeo Jyotishman Saikia Sai\u00a0Kiran Cherupally Yichi Zhang Zhiru Zhang and Jae-Sun Seo. 2023. A 28-nm 8-bit Floating-Point Tensor Core-Based Programmable CNN Training Processor With Dynamic Structured Sparsity. IEEE Journal of Solid-State Circuits 58 7 (2023) 1885\u20131897.","DOI":"10.1109\/JSSC.2023.3269148"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS58744.2024.10558053"}],"event":{"name":"GLSVLSI '25: Great Lakes Symposium on VLSI 2025","sponsor":["SIGDA ACM Special Interest Group on Design Automation"],"location":"New Orleans LA USA","acronym":"GLSVLSI '25"},"container-title":["Proceedings of the Great Lakes Symposium on VLSI 2025"],"original-title":[],"deposited":{"date-parts":[[2025,6,27]],"date-time":"2025-06-27T14:39:05Z","timestamp":1751035145000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3716368.3735201"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,29]]},"references-count":19,"alternative-id":["10.1145\/3716368.3735201","10.1145\/3716368"],"URL":"https:\/\/doi.org\/10.1145\/3716368.3735201","relation":{},"subject":[],"published":{"date-parts":[[2025,6,29]]},"assertion":[{"value":"2025-06-29","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}