{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T16:14:25Z","timestamp":1784650465476,"version":"3.55.0"},"reference-count":47,"publisher":"Tsinghua University Press","issue":"4","funder":[{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"publisher","award":["2021ZD0110400"],"award-info":[{"award-number":["2021ZD0110400"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62406114,62306117"],"award-info":[{"award-number":["62406114,62306117"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["2024ZYGXZR074,2023ZYGXZR023"],"award-info":[{"award-number":["2024ZYGXZR074,2023ZYGXZR023"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Big Data Min. Anal."],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.26599\/bdma.2024.9020074","type":"journal-article","created":{"date-parts":[[2025,5,12]],"date-time":"2025-05-12T17:46:07Z","timestamp":1747071967000},"page":"779-793","source":"Crossref","is-referenced-by-count":5,"title":["ResDecode: Accelerating Large Language Models Inference via Residual Decoding Heads"],"prefix":"10.26599","volume":"8","author":[{"given":"Ziqian","family":"Zeng","sequence":"first","affiliation":[{"name":"Shien Ming Wu School of Intelligent Engineering, South China University of Technology,Guangzhou,China,511442"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiahong","family":"Yu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, South China University of Technology,Guangzhou,China,510006"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qianshi","family":"Pang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, South China University of Technology,Guangzhou,China,510006"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zihao","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Engineering, The Hong Kong University of Science and Technology,Department of Computer Science and Engineering,Hong Kong,China,999077"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huiping","family":"Zhuang","sequence":"additional","affiliation":[{"name":"Shien Ming Wu School of Intelligent Engineering, South China University of Technology,Guangzhou,China,511442"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fan","family":"Yu","sequence":"additional","affiliation":[{"name":"Huawei Technologies Co. Ltd.,Hangzhou,China,310000"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongen","family":"Shao","sequence":"additional","affiliation":[{"name":"School of Future Technology, South China University of Technology,Guangzhou,China,511442"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaofeng","family":"Zou","sequence":"additional","affiliation":[{"name":"School of Future Technology, South China University of Technology,Guangzhou,China,511442"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"11138","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-13-7311-4"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2023.123753"},{"key":"ref3","volume-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/s13042-024-02323-z"},{"key":"ref5","volume-title":"Fast transformer decoding: One write-head is all you need","author":"Shazeer","year":"2019"},{"key":"ref6","first-page":"10107","article-title":"Blockwise parallel decoding for deep autoregressive models","volume-title":"Proc. 32nd Int. Conf. Neural Information Processing Systems","author":"Stern"},{"key":"ref7","first-page":"19274","article-title":"Fast inference from transformers via speculative decoding","volume-title":"Proc. 40th Int. Conf. Machine Learning","author":"Leviathan"},{"key":"ref8","volume-title":"Accelerating large language model decoding with speculative sampling","author":"Chen","year":"2023"},{"key":"ref9","article-title":"Medusa: Simple LLM inference acceleration framework with multiple decoding heads","volume-title":"Proc. 41st Int. Conf. Machine Learning","author":"Cai"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1356"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00349"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.257"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3389714"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00039"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3090167"},{"key":"ref16","first-page":"1","article-title":"Designing neural network architectures using reinforcement learning","volume-title":"Proc. 5th Int. Conf. Learning Representations","author":"Baker"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1016\/j.dcan.2022.06.014"},{"key":"ref18","volume-title":"D2RL: Deep dense architectures in reinforcement learning","author":"Sinha","year":"2020"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1057\/s41599-024-03611-3"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.437"},{"key":"ref21","first-page":"1314","article-title":"Spectr: Fast speculative decoding via optimal transport","volume-title":"Proc. 37th Int. Conf. Neural Information Processing Systems","author":"Sun"},{"key":"ref22","volume-title":"CTIBench: A benchmark for evaluating LLMs in cyber threat intelligence","author":"Alam","year":"2024"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2018.2794343"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/s13042-024-02446-3"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3140238"},{"key":"ref26","volume-title":"Evaluating and modeling social intelligence: A comparative study of human and AI capabilities","author":"Wang","year":"2024"},{"key":"ref27","volume-title":"A comprehensive overview of large language models (LLMs) for cyber defences: Opportunities and directions","author":"Hassanin","year":"2024"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3363469"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TSG.2024.3373256"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3066410"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651335"},{"key":"ref32","article-title":"Distillspec: Improving speculative decoding via knowledge distillation","volume-title":"Proc. 12th International Conference on Learning Representations","author":"Zhou"},{"key":"ref33","first-page":"1705","article-title":"Speculative decoding with big little decoder","volume-title":"Proc. 37th Int. Conf. Neural Information Processing Systems","author":"Kim"},{"key":"ref34","article-title":"Online speculative decoding","volume-title":"Proc. 41st Int. Conf. Machine Learning","author":"Liu"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.689"},{"key":"ref36","volume-title":"SPEED: Speculative pipelined execution for efficient decoding","author":"Hooper","year":"2024"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.607"},{"key":"ref38","volume-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref39","volume-title":"Cascade speculative drafting for even faster LLM inference","author":"Chen","year":"2023"},{"key":"ref40","volume-title":"Vicuna:An open-source chatbot impressing GPT-4 with 90%* ChatGPT quality","year":"2023"},{"key":"ref41","volume-title":"Sharegpt_vicuna_unfiltered","year":"2022"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.semeval-1.210"},{"key":"ref43","volume-title":"Training verifiers to solve math word problems","author":"Cobbe","year":"2021"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.456"},{"key":"ref45","article-title":"Break the sequential dependency of LLM inference using lookahead decoding","volume-title":"Proc. 41st Int. Conf. Machine Learning","author":"Fu"},{"key":"ref46","volume-title":"Hydra:Sequentially-dependent draft heads for medusa decoding","author":"Ankner","year":"2024"},{"key":"ref47","volume-title":"Prompt lookup decoding","author":"Saxena","year":"2023"}],"container-title":["Big Data Mining and Analytics"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/8254253\/11002434\/11002449.pdf?arnumber=11002449","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,13]],"date-time":"2025-05-13T06:16:52Z","timestamp":1747117012000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11002449\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8]]},"references-count":47,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.26599\/bdma.2024.9020074","relation":{},"ISSN":["2096-0654","2097-406X"],"issn-type":[{"value":"2096-0654","type":"print"},{"value":"2097-406X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,8]]}}}