{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,10]],"date-time":"2026-02-10T12:34:35Z","timestamp":1770726875661,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":67,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China","award":["32571284"],"award-info":[{"award-number":["32571284"]}]},{"name":"State Key Laboratory of Brain Cognition and Brain-inspired Intelligence Technology","award":["JS202401"],"award-info":[{"award-number":["JS202401"]}]},{"name":"Beijing Municipal Bureau of Economy and Information Technology","award":["N&#x5c;&#x2f;A"],"award-info":[{"award-number":["N&#x5c;&#x2f;A"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,2,22]]},"DOI":"10.1145\/3748173.3779189","type":"proceedings-article","created":{"date-parts":[[2026,2,5]],"date-time":"2026-02-05T21:17:35Z","timestamp":1770326255000},"page":"235-246","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Hummingbird+: Advancing FPGA-based LLM Deployment from Research Prototype to Edge Product"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4009-916X","authenticated-orcid":false,"given":"Jindong","family":"Li","sequence":"first","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3266-2075","authenticated-orcid":false,"given":"Tenglong","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4069-2107","authenticated-orcid":false,"given":"Guobin","family":"Shen","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0593-8650","authenticated-orcid":false,"given":"Dongcheng","family":"Zhao","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5314-4233","authenticated-orcid":false,"given":"Qian","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9595-9091","authenticated-orcid":false,"given":"Yi","family":"Zeng","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China, Center for Long-term Artificial Intelligence, Beijing, China, and Institute of AI Safety and Governance, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,2,21]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"[n.d.]. SpinalHDL: Scala based HDL. https:\/\/github.com\/SpinalHDL\/SpinalHDL"},{"key":"e_1_3_2_1_2_1","volume-title":"Julian Quevedo, and Daya Khudia.","author":"Agarwal Megha","year":"2023","unstructured":"Megha Agarwal, Asfandyar Qureshi, Linden Li Nikhil Sardana, Julian Quevedo, and Daya Khudia. 2023. Llm inference performance engineering: Best practices."},{"key":"e_1_3_2_1_3_1","volume-title":"Sarathi: Efficient llm inference by piggybacking decodes with chunked prefills. arXiv preprint arXiv:2308.16369","author":"Agrawal Amey","year":"2023","unstructured":"Amey Agrawal, Ashish Panwar, Jayashree Mohan, Nipun Kwatra, Bhargav S Gulavani, and Ramachandran Ramjee. 2023. Sarathi: Efficient llm inference by piggybacking decodes with chunked prefills. arXiv preprint arXiv:2308.16369 (2023)."},{"key":"e_1_3_2_1_4_1","volume-title":"Yury Zemlyanskiy, Federico Lebr\u00f3n, and Sumit Sanghai.","author":"Ainslie Joshua","year":"2023","unstructured":"Joshua Ainslie, James Lee-Thorp, Michiel De Jong, Yury Zemlyanskiy, Federico Lebr\u00f3n, and Sumit Sanghai. 2023. Gqa: Training generalized multi-query transformer models from multi-head checkpoints. arXiv preprint arXiv:2305.13245 (2023)."},{"key":"e_1_3_2_1_5_1","unstructured":"airockchip. 2025. Benchmark \u2014 RKNN-LLM. https:\/\/github.com\/airockchip\/rknnllm\/blob\/main\/benchmark.md. Accessed: 2025-10-02."},{"key":"e_1_3_2_1_6_1","unstructured":"AMD Inc. 2025. UltraScale Device Packaging and Pinouts: Product Specification User Guide (UG575). Technical Report UG575. AMD. https:\/\/docs.amd.com\/v\/u\/en-US\/ug575-ultrascale-pkg-pinout Accessed: 2025-10-02."},{"key":"e_1_3_2_1_7_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877-1901."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3656177"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3656401"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3524108"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Eric Chung Jeremy Fowers Kalin Ovtcharov Michael Papamichael Adrian Caulfield Todd Massengill Ming Liu Daniel Lo Shlomi Alkalay Michael Haselman et al. 2018. Serving dnns in real time at datacenter scale with project brainwave. iEEE Micro 38 2 (2018) 8-20.","DOI":"10.1109\/MM.2018.022071131"},{"key":"e_1_3_2_1_12_1","first-page":"4171","volume-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies","volume":"1","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers). 4171-4186."},{"key":"e_1_3_2_1_13_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/DAC56929.2023.10247793"},{"key":"e_1_3_2_1_15_1","volume-title":"Gptq: Accurate post-training quantization for generative pre-trained transformers. arXiv preprint arXiv:2210.17323","author":"Frantar Elias","year":"2022","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2022. Gptq: Accurate post-training quantization for generative pre-trained transformers. arXiv preprint arXiv:2210.17323 (2022)."},{"key":"e_1_3_2_1_16_1","volume-title":"8-bit dot-product acceleration","author":"Fu Yao","year":"2017","unstructured":"Yao Fu, Ephrem Wu, and Ashish Sirasao. 2017. 8-bit dot-product acceleration. Xilinx Inc.: San Jose, CA, USA (2017), 20."},{"key":"e_1_3_2_1_17_1","unstructured":"Georgi Gerganov and contributors. 2025. llama.cpp: LLM inference in C\/C. https:\/\/github.com\/ggml-org\/llama.cpp. Accessed: 2025-10-02."},{"key":"e_1_3_2_1_18_1","volume-title":"Angel-eye: A complete design flow for mapping CNN onto embedded FPGA","author":"Guo Kaiyuan","year":"2017","unstructured":"Kaiyuan Guo, Lingzhi Sui, Jiantao Qiu, Jincheng Yu, JunbinWang, Song Yao, Song Han, Yu Wang, and Huazhong Yang. 2017. Angel-eye: A complete design flow for mapping CNN onto embedded FPGA. IEEE transactions on computer-aided design of integrated circuits and systems 37, 1 (2017), 35-47."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISEDA62518.2024.10617499"},{"key":"e_1_3_2_1_21_1","volume-title":"Shubham Pawar, and Yuxuan Chen.","author":"Henry Alex","year":"2020","unstructured":"Alex Henry, Prudhvi Raj Dachapally, Shubham Pawar, and Yuxuan Chen. 2020. Query-key normalization for transformers. arXiv preprint arXiv:2010.04245 (2020)."},{"key":"e_1_3_2_1_22_1","first-page":"148","article-title":"Flashdecoding: Faster large language model inference with asynchronization, flat gemm optimization, and heuristics","volume":"6","author":"Hong Ke","year":"2024","unstructured":"Ke Hong, Guohao Dai, Jiaming Xu, Qiuli Mao, Xiuhong Li, Jun Liu, Kangdi Chen, Yuhan Dong, and Yu Wang. 2024. Flashdecoding: Faster large language model inference with asynchronization, flat gemm optimization, and heuristics. Proceedings of Machine Learning and Systems 6 (2024), 148-161.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_23_1","volume-title":"Edgellm: A highly efficient cpu-fpga heterogeneous edge accelerator for large language models","author":"Huang Mingqiang","year":"2025","unstructured":"Mingqiang Huang, Ao Shen, Kai Li, Haoxiang Peng, Boyu Li, Yupeng Su, and Hao Yu. 2025. Edgellm: A highly efficient cpu-fpga heterogeneous edge accelerator for large language models. IEEE Transactions on Circuits and Systems I: Regular Papers (2025)."},{"key":"e_1_3_2_1_24_1","volume-title":"TENET: An Efficient Sparsity-Aware LUT-Centric Architecture for Ternary LLM Inference On Edge. arXiv preprint arXiv:2509.13765","author":"Huang Zhirui","year":"2025","unstructured":"Zhirui Huang, Rui Ma, Shijie Cao, Ran Shu, Ian Wang, Ting Cao, Chixiao Chen, and Yongqiang Xiong. 2025. TENET: An Efficient Sparsity-Aware LUT-Centric Architecture for Ternary LLM Inference On Edge. arXiv preprint arXiv:2509.13765 (2025)."},{"key":"e_1_3_2_1_25_1","volume-title":"Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al.","author":"Jiang Albert Q","year":"2024","unstructured":"Albert Q Jiang, Alexandre Sablayrolles, Antoine Roux, Arthur Mensch, Blanche Savary, Chris Bamford, Devendra Singh Chaplot, Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al. 2024. Mixtral of experts. arXiv preprint arXiv:2401.04088 (2024)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589350"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373087.3375887"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731008"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCAD66269.2025.11241002"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPL64840.2024.00035"},{"key":"e_1_3_2_1_31_1","volume-title":"Automation & Test in Europe Conference (DATE). IEEE, 1-7.","author":"Li Jindong","year":"2025","unstructured":"Jindong Li, Tenglong Li, Guobin Shen, Dongcheng Zhao, Qian Zhang, and Yi Zeng. 2025. Pushing up to the limit of memory bandwidth and capacity utilization for efficient llm decoding on embedded fpga. In 2025 Design, Automation & Test in Europe Conference (DATE). IEEE, 1-7."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVLSI.2023.3279349"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2024.3380550"},{"key":"e_1_3_2_1_34_1","volume-title":"Firefly-s: Exploiting dual-side sparsity for spiking neural networks acceleration with reconfigurable spatial architecture","author":"Li Tenglong","year":"2024","unstructured":"Tenglong Li, Jindong Li, Guobin Shen, Dongcheng Zhao, Qian Zhang, and Yi Zeng. 2024. Firefly-s: Exploiting dual-side sparsity for spiking neural networks acceleration with reconfigurable spatial architecture. IEEE Transactions on Circuits and Systems I: Regular Papers (2024)."},{"key":"e_1_3_2_1_35_1","volume-title":"AccLLM: Accelerating Long-Context LLM Inference Via Algorithm-Hardware Co-Design. arXiv preprint arXiv:2505.03745","author":"Liang Yanbiao","year":"2025","unstructured":"Yanbiao Liang, Huihong Shi, Haikuo Shao, and Zhongfeng Wang. 2025. AccLLM: Accelerating Long-Context LLM Inference Via Algorithm-Hardware Co-Design. arXiv preprint arXiv:2505.03745 (2025)."},{"key":"e_1_3_2_1_36_1","unstructured":"Aixin Liu Bei Feng Bin Wang Bingxuan Wang Bo Liu Chenggang Zhao Chengqi Dengr Chong Ruan Damai Dai Daya Guo et al. 2024. Deepseekv2: A strong economical and efficient mixture-of-experts language model. arXiv preprint arXiv:2405.04434 (2024)."},{"key":"e_1_3_2_1_37_1","unstructured":"Aixin Liu Bei Feng Bing Xue BingxuanWang BochaoWu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan et al. 2024. Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437 (2024)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3706628.3708864"},{"key":"e_1_3_2_1_39_1","volume-title":"MambaOPU: An FPGA Overlay Processor for State-space-duality-based Mamba Models. In 2025 62nd ACM\/IEEE Design Automation Conference (DAC). IEEE, 1-7.","author":"Lu Shaoqiang","year":"2025","unstructured":"Shaoqiang Lu, Xuliang Yu, Tiandong Zhao, Siyuan Miao, Xinsong Sheng, Chen Wu, Liang Zhao, Ting-Jung Lin, and Lei He. 2025. MambaOPU: An FPGA Overlay Processor for State-space-duality-based Mamba Models. In 2025 62nd ACM\/IEEE Design Automation Conference (DAC). IEEE, 1-7."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3706628.3708863"},{"key":"e_1_3_2_1_41_1","volume-title":"Online normalizer calculation for softmax. arXiv preprint arXiv:1805.02867","author":"Milakov Maxim","year":"2018","unstructured":"Maxim Milakov and Natalia Gimelshein. 2018. Online normalizer calculation for softmax. arXiv preprint arXiv:1805.02867 (2018)."},{"key":"e_1_3_2_1_42_1","volume-title":"Lpu: A latency-optimized and highly scalable processor for large language model inference","author":"Moon Seungjae","year":"2024","unstructured":"Seungjae Moon, Jung-Hoon Kim, Junsoo Kim, Seongmin Hong, Junseo Cha, Minsu Kim, Sukbin Lim, Gyubin Choi, Dongjin Seo, Jongho Kim, et al. 2024. Lpu: A latency-optimized and highly scalable processor for large language model inference. IEEE Micro (2024)."},{"key":"e_1_3_2_1_43_1","volume-title":"TeLLMe: An Energy-Efficient Ternary LLM Accelerator for Prefilling and Decoding on Edge FPGAs. arXiv preprint arXiv:2504.16266","author":"Qiao Ye","year":"2025","unstructured":"Ye Qiao, Zhiheng Chen, Yifan Zhang, YianWang, and Sitao Huang. 2025. TeLLMe: An Energy-Efficient Ternary LLM Accelerator for Prefilling and Decoding on Edge FPGAs. arXiv preprint arXiv:2504.16266 (2025)."},{"key":"e_1_3_2_1_44_1","unstructured":"Qwen Team. 2025. Qwen3-Next-80B-A3B-Instruct. https:\/\/huggingface.co\/Qwen\/Qwen3-Next-80B-A3B-Instruct. Accessed: 2025-10-02."},{"key":"e_1_3_2_1_45_1","unstructured":"Alec Radford JeffreyWu Rewon Child David Luan Dario Amodei Ilya Sutskever et al. 2019. Language models are unsupervised multitask learners. OpenAI blog 1 8 (2019) 9."},{"key":"e_1_3_2_1_46_1","volume-title":"Rockchip Electronics Co","year":"2022","unstructured":"Ltd. Rockchip Electronics Co. 2022. RK3588 Brief Datasheet. https:\/\/www.rockchips. com\/uploads\/pdf\/2022.8.26\/192\/RK3588%20Brief%20Datasheet.pdf"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPL.2019.00061"},{"key":"e_1_3_2_1_48_1","volume-title":"2022 32nd International Conference on Field-Programmable Logic and Applications (FPL). IEEE, 160-166","author":"Sommer Jan","year":"2022","unstructured":"Jan Sommer, M Akif \u00d6zkan, Oliver Keszocze, and J\u00fcrgen Teich. 2022. Dsppacking: Squeezing low-precision arithmetic into fpga dsp blocks. In 2022 32nd International Conference on Field-Programmable Logic and Applications (FPL). IEEE, 160-166."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643681"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/HCS61935.2024.10665247"},{"key":"e_1_3_2_1_52_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al. 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_53_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_54_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPL64840.2024.00054"},{"key":"e_1_3_2_1_56_1","volume-title":"VEDA: Efficient LLM Generation Through Voting-based KV Cache Eviction and Dataflow-flexible Accelerator. arXiv preprint arXiv:2507.00797","author":"Wang Zhican","year":"2025","unstructured":"Zhican Wang, Hongxiang Fan, Haroon Waris, Gang Wang, Zhenyu Li, Jianfei Jiang, Yanan Sun, and Guanghui He. 2025. VEDA: Efficient LLM Generation Through Voting-based KV Cache Eviction and Dataflow-flexible Accelerator. arXiv preprint arXiv:2507.00797 (2025)."},{"key":"e_1_3_2_1_57_1","volume-title":"Automation & Test in Europe Conference (DATE). IEEE, 1-7.","author":"Wei Renjie","year":"2025","unstructured":"Renjie Wei, Songqiang Xu, Linfeng Zhong, Zebin Yang, Qingyu Guo, Yuan Wang, Runsheng Wang, and Meng Li. 2025. Lightmamba: Efficient mamba acceleration on fpga with quantization and hardware co-design. In 2025 Design, Automation & Test in Europe Conference (DATE). IEEE, 1-7."},{"key":"e_1_3_2_1_58_1","volume-title":"2017 27th International Conference on Field Programmable Logic and Applications (FPL). IEEE, 1-4.","author":"Wu Ephrem","year":"2017","unstructured":"Ephrem Wu, Xiaoqian Zhang, David Berman, and Inkeun Cho. 2017. A highthroughput reconfigurable processing array for neural networks. In 2017 27th International Conference on Field Programmable Logic and Applications (FPL). IEEE, 1-4."},{"key":"e_1_3_2_1_59_1","volume-title":"5: Efficient LLM Inference with Model Compression and Hardware Acceleration. arXiv preprint arXiv:2504.17376","author":"Xiang Maoyang","year":"2025","unstructured":"Maoyang Xiang, Ramesh Fernando, and Bo Wang. 2025. On-Device Qwen2. 5: Efficient LLM Inference with Model Compression and Hardware Acceleration. arXiv preprint arXiv:2504.17376 (2025)."},{"key":"e_1_3_2_1_60_1","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chang Gao Chengen Huang Chenxu Lv et al. 2025. Qwen3 technical report. arXiv preprint arXiv:2505.09388 (2025)."},{"key":"e_1_3_2_1_61_1","volume-title":"StreamTensor: Make Tensors Stream in Dataflow Accelerators for LLMs. arXiv preprint arXiv:2509.13694","author":"Ye Hanchen","year":"2025","unstructured":"Hanchen Ye and Deming Chen. 2025. StreamTensor: Make Tensors Stream in Dataflow Accelerators for LLMs. arXiv preprint arXiv:2509.13694 (2025)."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626202.3637562"},{"key":"e_1_3_2_1_63_1","volume-title":"Root mean square layer normalization. Advances in neural information processing systems 32","author":"Zhang Biao","year":"2019","unstructured":"Biao Zhang and Rico Sennrich. 2019. Root mean square layer normalization. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/2684746.2689060"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/DAC56929.2023.10247773"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3676536.3676761"},{"key":"e_1_3_2_1_67_1","volume-title":"Automation & Test in Europe Conference (DATE). IEEE, 1-7.","author":"Zheng Jianing","year":"2025","unstructured":"Jianing Zheng and Gang Chen. 2025. LoopLynx: A Scalable Dataflow Architecture for Efficient LLM Inference. In 2025 Design, Automation & Test in Europe Conference (DATE). IEEE, 1-7."}],"event":{"name":"FPGA '26:The 2026 ACM\/SIGDA International Symposium on Field Programmable Gate Arrays","location":"Seaside CA USA","sponsor":["SIGDA ACM Special Interest Group on Design Automation"]},"container-title":["Proceedings of the 2026 ACM\/SIGDA International Symposium on Field Programmable Gate Arrays"],"original-title":[],"deposited":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T16:17:25Z","timestamp":1770653845000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3748173.3779189"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,21]]},"references-count":67,"alternative-id":["10.1145\/3748173.3779189","10.1145\/3748173"],"URL":"https:\/\/doi.org\/10.1145\/3748173.3779189","relation":{},"subject":[],"published":{"date-parts":[[2026,2,21]]},"assertion":[{"value":"2026-02-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}