{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T04:53:29Z","timestamp":1750913609289,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T00:00:00Z","timestamp":1737331200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,1,20]]},"DOI":"10.1145\/3658617.3697562","type":"proceedings-article","created":{"date-parts":[[2025,3,4]],"date-time":"2025-03-04T14:32:21Z","timestamp":1741098741000},"page":"258-264","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["End-to-end Compilation is All FPGAs Need: A Unified Overlay-based FPGA Compiler for Deep Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9522-9216","authenticated-orcid":false,"given":"Kai","family":"Qian","sequence":"first","affiliation":[{"name":"Fudan Univ., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8628-2664","authenticated-orcid":false,"given":"Haodong","family":"Lu","sequence":"additional","affiliation":[{"name":"Fudan Univ., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6198-3712","authenticated-orcid":false,"given":"Yinqiu","family":"Liu","sequence":"additional","affiliation":[{"name":"Nanyang Technological Univ., Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0931-2414","authenticated-orcid":false,"given":"Zexu","family":"Zhang","sequence":"additional","affiliation":[{"name":"Fudan Univ., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7288-1789","authenticated-orcid":false,"given":"Kun","family":"Wang","sequence":"additional","affiliation":[{"name":"Fudan Univ., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,3,4]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACPR.2015.7486599"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.2979670"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3587440"},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 26--35","author":"Qiu Jiantao","year":"2016","unstructured":"Jiantao Qiu, Jie Wang, Song Yao, Kaiyuan Guo, Boxun Li, Erjin Zhou, Jincheng Yu, Tianqi Tang, Ningyi Xu, Sen Song, Yu Wang, and Huazhong Yang. 2016. Going Deeper with Embedded FPGA Platform for Convolutional Neural Network. In Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 26--35."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240765.3240850"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240765.3240838"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 147","author":"Sohrabizadeh Atefeh","year":"2021","unstructured":"Atefeh Sohrabizadeh, Cody Hao Yu, Min Gao, and Jason Cong. 2021. AutoDSE: Enabling Software Programmers Design Efficient FPGA Accelerators. In Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 147."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPL.2016.7577356"},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the ACM\/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA). 15--25","author":"Du Linfeng","year":"2023","unstructured":"Linfeng Du, Tingyuan Liang, Sharad Sinha, Zhiyao Xie, and Wei Zhang. 2023. FADO: Floorplan-Aware Directive Optimization for High-Level Synthesis Designs on Multi-Die FPGAs. In Proceedings of the ACM\/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA). 15--25."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 242--251","author":"Lai Yi-Hsiang","year":"2019","unstructured":"Yi-Hsiang Lai, Yuze Chi, Yuwei Hu, Jie Wang, Cody Hao Yu, Yuan Zhou, Jason Cong, and Zhiru Zhang. 2019. HeteroCL: A Multi-Paradigm Programming Infrastructure for Software-Defined Reconfigurable Computing. In Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 242--251."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00060"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 57th ACM\/EDAC\/IEEE Design Automation Conference (DAC). Article 248","author":"Vivancos Isak Edo","year":"2020","unstructured":"Isak Edo Vivancos, Sayeh Sharify, Milos Nikolic, Ciaran Bannon, Mostafa Mahmoud, Alberto Delm\u00e1s Lascorz, and Andreas Moshovos. 2020. Building an On-Chip Deep Learning Memory Hierarchy Brick by Brick: Late Breaking Results. In Proceedings of the 57th ACM\/EDAC\/IEEE Design Automation Conference (DAC). Article 248, 2 pages."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the USENIX Conference on Operating Systems Design and Implementation (OSDI). 579--594","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Meghan Cowan, Haichen Shen, Leyuan Wang, Yuwei Hu, Luis Ceze, Carlos Guestrin, and Arvind Krishnamurthy. 2018. TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In Proceedings of the USENIX Conference on Operating Systems Design and Implementation (OSDI). 579--594."},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the IEEE\/ACM International Conference on Computer-Aided Design (ICCAD). Article 6, 9 pages.","author":"Agostini Nicolas Bohm","year":"2022","unstructured":"Nicolas Bohm Agostini, Serena Curzel, Vinay Amatya, Cheng Tan, Marco Minutoli, Vito Giovanni Castellana, Joseph Manzano, David Kaeli, and Antonino Tumeo. 2022. An MLIR-Based Compiler Flow for System-Level Design and Hardware Acceleration. In Proceedings of the IEEE\/ACM International Conference on Computer-Aided Design (ICCAD). Article 6, 9 pages."},{"key":"e_1_3_2_1_16_1","volume-title":"CIRCT: Circt IR Compilers and Tools, https:\/\/github.com\/llvm\/circt. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/programmable\/683456\/21-4\/pro-edition-user-guide.html","author":"Community CIRCT","year":"2023","unstructured":"CIRCT Community. 2023. CIRCT: Circt IR Compilers and Tools, https:\/\/github.com\/llvm\/circt. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/programmable\/683456\/21-4\/pro-edition-user-guide.html"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 65--74","author":"Umuroglu Yaman","year":"2017","unstructured":"Yaman Umuroglu, Nicholas J. Fraser, Giulio Gambardella, Michaela Blott, Philip Leong, Magnus Jahre, and Kees Vissers. 2017. FINN: A Framework for Fast, Scalable Binarized Neural Network Inference. In Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 65--74."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 40--50","author":"Xu Pengfei","year":"2020","unstructured":"Pengfei Xu, Xiaofan Zhang, Cong Hao, Yang Zhao, Yongan Zhang, Yue Wang, Chaojian Li, Zetong Guan, Deming Chen, and Yingyan Lin. 2020. AutoDNNchip: An Automated DNN Chip Predictor and Builder for Both FPGAs and ASICs. In Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 40--50."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPL57034.2022.00015"},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 133--139","author":"Sohrabizadeh Atefeh","year":"2020","unstructured":"Atefeh Sohrabizadeh, Jie Wang, and Jason Cong. 2020. End-to-End Optimization of Deep Learning Applications. In Proceedings of the ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays (FPGA). 133--139."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS46773.2023.10181655"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","unstructured":"Haodong Lu Qichang Mei and Kun Wang. 2023. An Efficient Piecewise Linear Approximation of Non-linear Operations for Transformer Inference. In 2023 IEEE 31st Annual International Symposium on Field-Programmable Custom Computing Machines (FCCM). 206--206. 10.1109\/FCCM57271.2023.00034","DOI":"10.1109\/FCCM57271.2023.00034"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVLSI.2019.2939726"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2018.00023"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICTA50426.2020.9332090"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 32nd International Conference on International Conference on Machine Learning (ICML)","author":"Ioffe Sergey","year":"2015","unstructured":"Sergey Ioffe and Christian Szegedy. 2015. Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift. In Proceedings of the 32nd International Conference on International Conference on Machine Learning (ICML) (Lille, France). 448--456."}],"event":{"name":"ASPDAC '25: 30th Asia and South Pacific Design Automation Conference","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEICE","IPSJ","IEEE CAS","IEEE CEDA"],"location":"Tokyo Japan","acronym":"ASPDAC '25"},"container-title":["Proceedings of the 30th Asia and South Pacific Design Automation Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658617.3697562","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3658617.3697562","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T23:44:19Z","timestamp":1750290259000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658617.3697562"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,20]]},"references-count":26,"alternative-id":["10.1145\/3658617.3697562","10.1145\/3658617"],"URL":"https:\/\/doi.org\/10.1145\/3658617.3697562","relation":{},"subject":[],"published":{"date-parts":[[2025,1,20]]},"assertion":[{"value":"2025-03-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}