{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,14]],"date-time":"2026-06-14T22:05:40Z","timestamp":1781474740799,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,6,23]],"date-time":"2024-06-23T00:00:00Z","timestamp":1719100800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"IBM-Illinois Discovery Accelerator Institute,"},{"name":"AMD Center of Excellence at UIUC"},{"name":"AMD Heterogeneous Adaptive Compute Cluster (HACC) initiative"},{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2117997"],"award-info":[{"award-number":["2117997"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Semiconductor Research Corporation (SRC)","award":["2023-CT-3175"],"award-info":[{"award-number":["2023-CT-3175"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,6,23]]},"DOI":"10.1145\/3649329.3663517","type":"proceedings-article","created":{"date-parts":[[2024,11,7]],"date-time":"2024-11-07T19:27:22Z","timestamp":1731007642000},"page":"1-4","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":25,"title":["Invited: New Solutions on LLM Acceleration, Optimization, and Application"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-8244-2917","authenticated-orcid":false,"given":"Yingbing","family":"Huang","sequence":"first","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8006-0019","authenticated-orcid":false,"given":"Lily Jiaxin","family":"Wan","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6646-8146","authenticated-orcid":false,"given":"Hanchen","family":"Ye","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5613-7039","authenticated-orcid":false,"given":"Manvi","family":"Jha","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5290-5773","authenticated-orcid":false,"given":"Jinghua","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3769-6772","authenticated-orcid":false,"given":"Yuhong","family":"Li","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5081-3972","authenticated-orcid":false,"given":"Xiaofan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Google, San Jose, CA, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3016-0270","authenticated-orcid":false,"given":"Deming","family":"Chen","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Champaign, IL, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,11,7]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Medusa: Simple llm inference acceleration framework with multiple decoding heads. arXiv:2401.10774.","author":"Tianle Cai","year":"2024","unstructured":"Tianle Cai et al. 2024. Medusa: Simple llm inference acceleration framework with multiple decoding heads. arXiv:2401.10774."},{"key":"e_1_3_2_1_2_1","unstructured":"Charlie Chen et al. 2023. Accelerating large language model decoding with speculative sampling. arXiv:2302.01318."},{"key":"e_1_3_2_1_3_1","unstructured":"Deming Chen et al. 2005. xPilot: A Platform-Based Behavioral Synthesis System. In SRC Techcon."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Jason Cong et al. 2022. FPGA HLS Today: Successes Challenges and Opportunities. In TRETS.","DOI":"10.1145\/3530775"},{"key":"e_1_3_2_1_5_1","volume-title":"Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in Neural Information Processing Systems.","author":"Tri Dao","year":"2022","unstructured":"Tri Dao et al. 2022. Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_6_1","unstructured":"Tim Dettmers et al. 2022. Llm. int8 (): 8-bit matrix multiplication for transformers at scale. arXiv preprint arXiv:2208.07339."},{"key":"e_1_3_2_1_7_1","unstructured":"Yichao Fu et al. 2024. Break the sequential dependency of llm inference using lookahead decoding. arXiv:2402.02057."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Cong Hao et al. 2018. Deep neural network model and FPGA accelerator co-design: Opportunities and challenges. In ICSICT.","DOI":"10.1109\/ICSICT.2018.8564956"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Cong Hao et al. 2019. FPGA\/DNN co-design: An efficient design methodology for IoT intelligence on the edge. In DAC.","DOI":"10.1145\/3316781.3317829"},{"key":"e_1_3_2_1_10_1","volume-title":"DFX: A low-latency multi-FPGA appliance for accelerating transformer-based text generation. In MICRO.","author":"Seongmin Hong","year":"2022","unstructured":"Seongmin Hong et al. 2022. DFX: A low-latency multi-FPGA appliance for accelerating transformer-based text generation. In MICRO."},{"key":"e_1_3_2_1_11_1","unstructured":"Hyegang Jun et al. 2023. AutoScaleDSE: A scalable design space exploration engine for high-level synthesis. In TRETS."},{"key":"e_1_3_2_1_12_1","volume-title":"Hydragen: High-Throughput LLM Inference with Shared Prefixes. arXiv:2402.05099.","author":"Jordan Juravsky","year":"2024","unstructured":"Jordan Juravsky et al. 2024. Hydragen: High-Throughput LLM Inference with Shared Prefixes. arXiv:2402.05099."},{"key":"e_1_3_2_1_13_1","unstructured":"Rahul Kande et al. 2023. Llm-assisted generation of hardware assertions. arXiv:2306.14027."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Achintya Kundu et al. 2024. Efficiently Distilling LLMs for Edge Applications. arXiv preprint arXiv:2404.01353.","DOI":"10.18653\/v1\/2024.naacl-industry.5"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Woosuk Kwon et al. 2023. Efficient Memory Management for Large Language Model Serving with PagedAttention. In SOSP.","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_16_1","unstructured":"Yaniv Leviathan et al. 2023. Fast inference from transformers via speculative decoding. In ICML."},{"key":"e_1_3_2_1_17_1","unstructured":"Patrick Lewis et al. 2020. Retrieval-augmented generation for knowledge-intensive nlp tasks. Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_18_1","volume-title":"Chipnemo: Domain-adapted llms for chip design. arXiv:2311.00176.","author":"Mingjie Liu","year":"2023","unstructured":"Mingjie Liu et al. 2023. Chipnemo: Domain-adapted llms for chip design. arXiv:2311.00176."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"James M Ortega and Werner C Rheinboldt. 2000. Iterative solution of nonlinear equations in several variables. Classics in Applied Mathematics.","DOI":"10.1137\/1.9780898719468"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Andrea Santilli et al. 2023. Accelerating transformer inference for translation via parallel decoding. arXiv:2305.10427.","DOI":"10.18653\/v1\/2023.acl-long.689"},{"key":"e_1_3_2_1_21_1","unstructured":"Mitchell Stern et al. 2018. Blockwise parallel decoding for deep autoregressive models. Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_22_1","volume-title":"Autochip: Automating hdl generation using llm feedback. arXiv:2311.04887.","author":"Shailja Thakur","year":"2023","unstructured":"Shailja Thakur et al. 2023. Autochip: Automating hdl generation using llm feedback. arXiv:2311.04887."},{"key":"e_1_3_2_1_23_1","volume-title":"Verigen: A large language model for verilog code generation. In TRETS.","author":"Shailja Thakur","year":"2023","unstructured":"Shailja Thakur et al. 2023. Verigen: A large language model for verilog code generation. In TRETS."},{"key":"e_1_3_2_1_24_1","volume-title":"Rtlfixer: Automatically fixing rtl syntax errors with large language models. arXiv:2311.16543.","author":"Tsai YunDa","year":"2023","unstructured":"YunDa Tsai et al. 2023. Rtlfixer: Automatically fixing rtl syntax errors with large language models. arXiv:2311.16543."},{"key":"e_1_3_2_1_25_1","unstructured":"Lily Jiaxin Wan et al. 2024. Software\/Hardware Co-design for LLM and Its Application for Design Verification. In ASP-DAC."},{"key":"e_1_3_2_1_26_1","unstructured":"Jason Wei et al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems."},{"key":"e_1_3_2_1_27_1","volume-title":"Smoothquant: Accurate and efficient post-training quantization for large language models. In ICML.","author":"Guangxuan Xiao","year":"2023","unstructured":"Guangxuan Xiao et al. 2023. Smoothquant: Accurate and efficient post-training quantization for large language models. In ICML."},{"key":"e_1_3_2_1_28_1","unstructured":"Hanchen Ye et al. 2022. ScaleHLS: A new scalable high-level synthesis framework on multi-level intermediate representation. In HPCA."},{"key":"e_1_3_2_1_29_1","unstructured":"Hanchen Ye et al. 2022. ScaleHLS: a scalable high-level synthesis framework with multi-level transformations and optimizations. In DAC."},{"key":"e_1_3_2_1_30_1","unstructured":"Hanchen Ye et al. 2023. High-level Synthesis for Domain Specific Computing. In ISPD."},{"key":"e_1_3_2_1_31_1","volume-title":"HIDA: A Hierarchical Dataflow Compiler for High-Level Synthesis. In ASPLOS.","author":"Hanchen Ye","year":"2024","unstructured":"Hanchen Ye et al. 2024. HIDA: A Hierarchical Dataflow Compiler for High-Level Synthesis. In ASPLOS."},{"key":"e_1_3_2_1_32_1","volume-title":"Vitcod: Vision transformer acceleration via dedicated algorithm and accelerator co-design. In HPCA.","author":"Haoran You","year":"2023","unstructured":"Haoran You et al. 2023. Vitcod: Vision transformer acceleration via dedicated algorithm and accelerator co-design. In HPCA."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Shulin Zeng et al. 2024. FlightLLM: Efficient Large Language Model Inference with a Complete Mapping Flow on FPGA. arXiv:2401.03868.","DOI":"10.1145\/3626202.3637562"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Xiaofan Zhang et al. 2018. DNNBuilder: An automated tool for building highperformance DNN hardware accelerators for FPGAs. In ICCAD.","DOI":"10.1145\/3240765.3240801"},{"key":"e_1_3_2_1_35_1","unstructured":"Xiaofan Zhang et al. 2022. AutoDistill: An end-to-end framework to explore and distill hardware-efficient language models. arXiv:2201.08539."},{"key":"e_1_3_2_1_36_1","unstructured":"Zhenyu Zhang et al. 2023. H2o: Heavy-hitter oracle for efficient generative inference of large language models. arXiv:2306.14048."},{"key":"e_1_3_2_1_37_1","unstructured":"Ruizhe Zhong et al. 2023. LLM4EDA: Emerging Progress in Large Language Models for Electronic Design Automation. arXiv:2401.12224."}],"event":{"name":"DAC '24: 61st ACM\/IEEE Design Automation Conference","location":"San Francisco CA USA","acronym":"DAC '24","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEEE-CEDA","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 61st ACM\/IEEE Design Automation Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3649329.3663517","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3649329.3663517","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3649329.3663517","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:02Z","timestamp":1750295882000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3649329.3663517"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,23]]},"references-count":37,"alternative-id":["10.1145\/3649329.3663517","10.1145\/3649329"],"URL":"https:\/\/doi.org\/10.1145\/3649329.3663517","relation":{},"subject":[],"published":{"date-parts":[[2024,6,23]]},"assertion":[{"value":"2024-11-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}