{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T14:54:32Z","timestamp":1781794472969,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T00:00:00Z","timestamp":1782086400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/100000015","name":"DOE U.S. Department of Energy","doi-asserted-by":"publisher","award":["DE-SC0026344"],"award-info":[{"award-number":["DE-SC0026344"]}],"id":[{"id":"10.13039\/100000015","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2348306"],"award-info":[{"award-number":["2348306"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2511445"],"award-info":[{"award-number":["2511445"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2518375"],"award-info":[{"award-number":["2518375"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2536952"],"award-info":[{"award-number":["2536952"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2544032"],"award-info":[{"award-number":["2544032"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,22]]},"DOI":"10.1145\/3787109.3815309","type":"proceedings-article","created":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T14:17:19Z","timestamp":1781792239000},"page":"898-905","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["\u03bc-ORCA: Optimizing Acceleration for Microsecond-Scale Deep Neural Network Inference on ACAP"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-3429-4692","authenticated-orcid":false,"given":"Shixin","family":"Ji","sequence":"first","affiliation":[{"name":"Brown University, Providence, RI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3659-339X","authenticated-orcid":false,"given":"Jinming","family":"Zhuang","sequence":"additional","affiliation":[{"name":"Brown University, Providence, RI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7655-4080","authenticated-orcid":false,"given":"Zhuoping","family":"Yang","sequence":"additional","affiliation":[{"name":"Brown University, Providence, RI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4865-3708","authenticated-orcid":false,"given":"Xingzhen","family":"Chen","sequence":"additional","affiliation":[{"name":"Brown University, Providence, RI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4590-6559","authenticated-orcid":false,"given":"Wei","family":"Zhang","sequence":"additional","affiliation":[{"name":"Brown University, Providence, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0493-1844","authenticated-orcid":false,"given":"Peipei","family":"Zhou","sequence":"additional","affiliation":[{"name":"Brown University, Providence, RI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,22]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"cern. Real time analysis with the CMS Level-1 Trigger 2025."},{"key":"e_1_3_3_1_3_2","unstructured":"CERN. The next-generation triggers for CERN detectors 2025."},{"key":"e_1_3_3_1_4_2","unstructured":"LHC. Taking a closer look at LHC 2025. Last accessed January 10 2026."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Abdelkhalik H. et\u00a0al. Demystifying the Nvidia Ampere Architecture through Microbenchmarking and Instruction-level Analysis. In HPEC pages 1\u20138 2022.","DOI":"10.1109\/HPEC55821.2022.9926299"},{"key":"e_1_3_3_1_6_2","unstructured":"Zhuang J. et\u00a0al. CHARM: Composing Heterogeneous AcceleRators for Matrix Multiply on Versal ACAP Architecture. In FPGA New York NY USA 2023."},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Zhuang J. et\u00a0al. High Performance Low Power Matrix Multiply Design on ACAP: from Architecture Design Challenges and DSE Perspectives. In DAC DAC \u201923 page 1\u20136. IEEE Press 2025.","DOI":"10.1109\/DAC56929.2023.10247981"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Zhuang J. et\u00a0al. CHARM 2.0: Composing Heterogeneous Accelerators for Deep Learning on Versal ACAP Architecture. TRETS 17(3) September 2024.","DOI":"10.1145\/3686163"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"crossref","unstructured":"Zhuang J. et\u00a0al. SSR: Spatial Sequential Hybrid Architecture for Latency Throughput Tradeoff in Transformer Acceleration. FPGA \u201924 New York NY USA 2024.","DOI":"10.1145\/3626202.3637569"},{"key":"e_1_3_3_1_10_2","unstructured":"Dong P. et\u00a0al. EQ-ViT: Algorithm-hardware co-design for end-to-end acceleration of real-time vision transformer inference on Versal ACAP architecture. IEEE TCAD."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"crossref","unstructured":"Zhuang J. et\u00a0al. ARIES: An Agile MLIR-Based Compilation Flow for Reconfigurable Devices with AI Engines. FPGA \u201925 2024.","DOI":"10.1145\/3706628.3708870"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Taka E. et\u00a0al. MaxEVA: Maximizing the Efficiency of Matrix Multiplication on Versal AI Engine. In ICFPT pages 96\u2013105 2023.","DOI":"10.1109\/ICFPT59805.2023.00016"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"crossref","unstructured":"Deng X. et\u00a0al. Ama: An analytical approach to maximizing the efficiency of deep learning on versal ai engine. In FPL pages 227\u2013235. IEEE 2024.","DOI":"10.1109\/FPL64840.2024.00039"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"crossref","unstructured":"Mhatre K. et\u00a0al. GAMA: High-Performance GEMM Acceleration on AMD Versal ML-Optimized AI Engines. In FPL. IEEE 2025.","DOI":"10.1109\/FPL68686.2025.00051"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"crossref","unstructured":"Danopoulos D. et\u00a0al. AIE4ML: An End-to-End Framework for Compiling Neural Networks for the Next Generation of AMD AI Engines. arXiv:https:\/\/arXiv.org\/abs\/2512.15946 2025.","DOI":"10.1109\/FCCM68464.2026.00035"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Hunhoff E. et\u00a0al. Efficiency Expressivity and Extensibility in a Close-to-Metal NPU Programming Interface. In FCCM pages 85\u201394 2025.","DOI":"10.1109\/FCCM62733.2025.00043"},{"key":"e_1_3_3_1_17_2","unstructured":"Ma Z. et\u00a0al. Design Rules for Extreme-Edge Scientific Computing on AI Engines. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2604.19106 2026."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Wang C. et\u00a0al. Reconfigurable Stream Network Architecture. ISCA \u201925 page 1848\u20131866 New York NY USA 2025. ACM.","DOI":"10.1145\/3695053.3731088"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"crossref","unstructured":"Nguyen T. et\u00a0al. SPADES: A Productive Design Flow for Versal Programmable Logic. In FPL pages 65\u201371 2023.","DOI":"10.1109\/FPL60245.2023.00017"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"Bouaziz M. et\u00a0al. A Dataflow Overlay for Monte Carlo Multi-Asset Option Pricing on AMD Versal AI Engines. In ISC pages 1\u201312 2025.","DOI":"10.23919\/ISC.2025.11020612"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Ji S. et\u00a0al. DERCA: DetERministic Cycle-Level Accelerator on Reconfigurable Platforms in DNN-Enabled Real-Time Safety-Critical Systems. In RTSS pages 392\u2013405 2025.","DOI":"10.1109\/RTSS66672.2025.00039"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Yang Z. et\u00a0al. AIM: Accelerating Arbitrary-Precision Integer Multiplication on Heterogeneous Reconfigurable Computing Platform Versal ACAP. In ICCAD pages 1\u20139 2023.","DOI":"10.1109\/ICCAD57390.2023.10323754"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Yan H. et\u00a0al. The Optimal The Fast and The Hybrid: Automatic Placement and Routing for AIE Arrays. In FCCM pages 1\u20139. IEEE Computer Society May 2026.","DOI":"10.1109\/FCCM68464.2026.00015"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"crossref","unstructured":"Mhatre K.M. et\u00a0al. Performance analysis of gemm workloads on the amd versal platform. In ISPASS pages 150\u2013161. IEEE 2025.","DOI":"10.1109\/ISPASS64960.2025.00023"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"crossref","unstructured":"Nag S. et\u00a0al. LL-ViT: Edge Deployable Vision Transformers with Look Up Table Neurons. In ICFPT pages 19\u201329 2025.","DOI":"10.1109\/ICFPT67023.2025.00013"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"crossref","unstructured":"Wang E. et\u00a0al. LUTNet: Rethinking Inference in FPGA Soft Logic. In FCCM pages 26\u201334 Los Alamitos CA USA May 2019. IEEE Computer Society.","DOI":"10.1109\/FCCM.2019.00014"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"crossref","unstructured":"Umuroglu Y. et\u00a0al. LogicNets: Co-Designed Neural Networks and Circuits for Extreme-Throughput Applications. In FPL pages 291\u2013297 2020.","DOI":"10.1109\/FPL50879.2020.00055"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"crossref","unstructured":"Andronic M. et\u00a0al. PolyLUT: Learning Piecewise Polynomials for Ultra-Low Latency FPGA LUT-based Inference. In ICFPT pages 60\u201368 2023.","DOI":"10.1109\/ICFPT59805.2023.00012"},{"key":"e_1_3_3_1_29_2","unstructured":"Schulte J.F. et\u00a0al. hls4ml: A Flexible Open-Source Platform for Deep Learning Acceleration on Reconfigurable Hardware. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2512.01463 2025."},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Sun C. et\u00a0al. da4ml: Distributed Arithmetic for Real-time Neural Networks on FPGAs. ACM Trans. Reconfigurable Technol. Syst. 19(1) March 2026.","DOI":"10.1145\/3777387"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","unstructured":"Cassidy O. et\u00a0al. ReducedLUT: Table Decomposition with\" Don\u2019t Care\" Conditions. In FPGA pages 36\u201342 2025.","DOI":"10.1145\/3706628.3708823"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"crossref","unstructured":"Ramanathan A.K. et\u00a0al. Look-Up Table based Energy Efficient Processing in Cache Support for Neural Network Acceleration. In MICRO pages 88\u2013101 2020.","DOI":"10.1109\/MICRO50266.2020.00020"},{"key":"e_1_3_3_1_33_2","unstructured":"Bacellar A.T.L. et\u00a0al. Differentiable Weightless Neural Networks. In Salakhutdinov R. et\u00a0al editors ICML pages 2277\u20132295 2024."},{"key":"e_1_3_3_1_34_2","unstructured":"Sun C. et\u00a0al. Gradient-based automatic mixed precision quantization for neural networks on-chip. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.00645 2024."},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"crossref","unstructured":"Odagiu P. et\u00a0al. Ultrafast jet classification at the HL-LHC. Machine Learning: Science and Technology 5(3):035017 2024.","DOI":"10.1088\/2632-2153\/ad5f10"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"crossref","unstructured":"Sun C. et\u00a0al. Fast Jet Tagging with MLP-Mixers on FPGAs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.03103 2025.","DOI":"10.1088\/2632-2153\/adf596"},{"key":"e_1_3_3_1_37_2","unstructured":"Laatu L. et\u00a0al. Sub-microsecond Transformers for Jet Tagging on FPGAs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.24784 2025."},{"key":"e_1_3_3_1_38_2","unstructured":"AMD. AI Engine API User Guide. https:\/\/download.amd.com\/docnav\/aiengine\/xilinx2024_1\/aiengine_api\/aie_api\/doc\/index.html 2025."},{"key":"e_1_3_3_1_39_2","unstructured":"AMD. AI Engine-ML Intrinsics User Guide. https:\/\/download.amd.com\/docnav\/aiengine\/xilinx2024_1\/aiengine_ml_intrinsics\/intrinsics\/index.html 2025."}],"event":{"name":"GLSVLSI '26: Great Lakes Symposium on VLSI 2026","location":"Canandaigua , NY , USA","acronym":"GLSVLSI '26","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEEE CEDA"]},"container-title":["Proceedings of the Great Lakes Symposium on VLSI 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3787109.3815309","content-type":"text\/html","content-version":"vor","intended-application":"syndication"}],"deposited":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T14:23:08Z","timestamp":1781792588000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3787109.3815309"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,22]]},"references-count":38,"alternative-id":["10.1145\/3787109.3815309","10.1145\/3787109"],"URL":"https:\/\/doi.org\/10.1145\/3787109.3815309","relation":{},"subject":[],"published":{"date-parts":[[2026,6,22]]},"assertion":[{"value":"2026-06-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}