{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T05:01:23Z","timestamp":1784782883001,"version":"3.55.0"},"reference-count":60,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100004826","name":"Natural Science Foundation of Beijing Municipality","doi-asserted-by":"publisher","award":["L234078"],"award-info":[{"award-number":["L234078"]}],"id":[{"id":"10.13039\/501100004826","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62502498"],"award-info":[{"award-number":["62502498"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFB4503500"],"award-info":[{"award-number":["2023YFB4503500"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Future Generation Computer Systems"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.future.2026.108646","type":"journal-article","created":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T15:38:19Z","timestamp":1780501099000},"page":"108646","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["HALO: A heterogeneous accelerator for low-latency and energy-efficient edge LLM inference"],"prefix":"10.1016","volume":"185","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-6000-3869","authenticated-orcid":false,"given":"Kunming","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5950-7370","authenticated-orcid":false,"given":"Zhihua","family":"Fan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanhuan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lexin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuqun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haibin","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaochun","family":"Ye","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenming","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.future.2026.108646_b1","series-title":"GPT-4 technical report","author":"Achiam","year":"2023"},{"issue":"8","key":"10.1016\/j.future.2026.108646_b2","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"10.1016\/j.future.2026.108646_b3","series-title":"The LLaMA 3 herd of models","first-page":"arXiv","author":"Dubey","year":"2024"},{"key":"10.1016\/j.future.2026.108646_b4","series-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"10.1016\/j.future.2026.108646_b5","series-title":"DeepSeek-V3 technical report","author":"Liu","year":"2024"},{"key":"10.1016\/j.future.2026.108646_b6","series-title":"DeepSeek-R1: Incentivizing reasoning capability in LLMs via reinforcement learning","author":"Guo","year":"2025"},{"issue":"1","key":"10.1016\/j.future.2026.108646_b7","doi-asserted-by":"crossref","first-page":"1","DOI":"10.61969\/jai.1311271","article-title":"Google Bard generated literature review: metaverse","volume":"7","author":"Ayd\u0131n","year":"2023","journal-title":"J. AI"},{"key":"10.1016\/j.future.2026.108646_b8","series-title":"ChatGPT","author":"OpenAI","year":"2022"},{"key":"10.1016\/j.future.2026.108646_b9","series-title":"Evaluating large language models trained on code","author":"Chen","year":"2021"},{"key":"10.1016\/j.future.2026.108646_b10","series-title":"GitHub copilot (GitHub repository)","author":"GitHub","year":"2025"},{"key":"10.1016\/j.future.2026.108646_b11","series-title":"Codegen: An open large language model for code with multi-turn program synthesis","author":"Nijkamp","year":"2022"},{"issue":"240","key":"10.1016\/j.future.2026.108646_b12","first-page":"1","article-title":"PaLM: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2023","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.future.2026.108646_b13","series-title":"BLOOM: A 176B-Parameter open-access multilingual language model","author":"Workshop","year":"2022"},{"key":"10.1016\/j.future.2026.108646_b14","series-title":"OPT: Open Pre-trained transformer language models","author":"Zhang","year":"2022"},{"key":"10.1016\/j.future.2026.108646_b15","series-title":"Agent AI: Surveying the horizons of multimodal interaction","author":"Durante","year":"2024"},{"key":"10.1016\/j.future.2026.108646_b16","doi-asserted-by":"crossref","unstructured":"L.B. Hensel, N. Yongsatianchot, P. Torshizi, E. Minucci, S. Marsella, Large language models in textual analysis for gesture selection, in: Proceedings of the 25th International Conference on Multimodal Interaction, 2023, pp. 378\u2013387.","DOI":"10.1145\/3577190.3614158"},{"key":"10.1016\/j.future.2026.108646_b17","series-title":"International Conference on Machine Learning","first-page":"9118","article-title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","author":"Huang","year":"2022"},{"key":"10.1016\/j.future.2026.108646_b18","series-title":"Inner monologue: Embodied reasoning through planning with language models","author":"Huang","year":"2022"},{"key":"10.1016\/j.future.2026.108646_b19","doi-asserted-by":"crossref","first-page":"247","DOI":"10.1162\/tacl_a_00646","article-title":"Retrieve what you need: A mutual learning framework for Open-Domain question answering","volume":"12","author":"Wang","year":"2024","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"10.1016\/j.future.2026.108646_b20","doi-asserted-by":"crossref","unstructured":"G. Heo, S. Lee, J. Cho, H. Choi, S. Lee, H. Ham, G. Kim, D. Mahajan, J. Park, NeuPIMS: NPU-PIM Heterogeneous Acceleration for Batched LLM Inferencing, in: Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 3, 2024, pp. 722\u2013737.","DOI":"10.1145\/3620666.3651380"},{"key":"10.1016\/j.future.2026.108646_b21","doi-asserted-by":"crossref","unstructured":"J. Park, J. Choi, K. Kyung, M.J. Kim, Y. Kwon, N.S. Kim, J.H. Ahn, Attacc! unleashing the power of pim for batched transformer-based generative model inference, in: Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2, 2024, pp. 103\u2013119.","DOI":"10.1145\/3620665.3640422"},{"key":"10.1016\/j.future.2026.108646_b22","doi-asserted-by":"crossref","unstructured":"C. Guo, J. Tang, W. Hu, J. Leng, C. Zhang, F. Yang, Y. Liu, M. Guo, Y. Zhu, Olive: Accelerating large language models via hardware-friendly outlier-victim pair quantization, in: Proceedings of the 50th Annual International Symposium on Computer Architecture, 2023, pp. 1\u201315.","DOI":"10.1145\/3579371.3589038"},{"key":"10.1016\/j.future.2026.108646_b23","series-title":"2022 55th IEEE\/ACM International Symposium on Microarchitecture","first-page":"1414","article-title":"Ant: Exploiting adaptive numerical data type for low-bit deep neural network quantization","author":"Guo","year":"2022"},{"key":"10.1016\/j.future.2026.108646_b24","series-title":"2024 ACM\/IEEE 51st Annual International Symposium on Computer Architecture","first-page":"1048","article-title":"Tender: Accelerating large language models via tensor decomposition and runtime requantization","author":"Lee","year":"2024"},{"key":"10.1016\/j.future.2026.108646_b25","series-title":"2018 ACM\/IEEE 45th Annual International Symposium on Computer Architecture","first-page":"764","article-title":"Bit fusion: Bit-level dynamically composable architecture for accelerating deep neural network","author":"Sharma","year":"2018"},{"key":"10.1016\/j.future.2026.108646_b26","series-title":"2020 IEEE International Symposium on High Performance Computer Architecture","first-page":"328","article-title":"\u00c2 3: Accelerating attention mechanisms in neural networks with approximation","author":"Ham","year":"2020"},{"key":"10.1016\/j.future.2026.108646_b27","doi-asserted-by":"crossref","unstructured":"L. Lu, Y. Jin, H. Bi, Z. Luo, P. Li, T. Wang, Y. Liang, Sanger: A co-design framework for enabling sparse attention using reconfigurable architecture, in: MICRO-54: 54th Annual IEEE\/ACM International Symposium on Microarchitecture, 2021, pp. 977\u2013991.","DOI":"10.1145\/3466752.3480125"},{"key":"10.1016\/j.future.2026.108646_b28","doi-asserted-by":"crossref","unstructured":"Y. Qin, Y. Wang, D. Deng, Z. Zhao, X. Yang, L. Liu, S. Wei, Y. Hu, S. Yin, Fact: Ffn-attention co-optimized transformer architecture with eager correlation prediction, in: Proceedings of the 50th Annual International Symposium on Computer Architecture, 2023, pp. 1\u201314.","DOI":"10.1145\/3579371.3589057"},{"key":"10.1016\/j.future.2026.108646_b29","doi-asserted-by":"crossref","unstructured":"Z. Qu, L. Liu, F. Tu, Z. Chen, Y. Ding, Y. Xie, Dota: detect and omit weak attentions for scalable transformer acceleration, in: Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, 2022, pp. 14\u201326.","DOI":"10.1145\/3503222.3507738"},{"key":"10.1016\/j.future.2026.108646_b30","first-page":"87","article-title":"AWQ: Activation-aware weight quantization for On-Device LLM compression and acceleration","volume":"6","author":"Lin","year":"2024","journal-title":"Proc. Mach. Learn. Syst."},{"key":"10.1016\/j.future.2026.108646_b31","series-title":"International Conference on Machine Learning","first-page":"38087","article-title":"SmoothQuant: Accurate and efficient post-training quantization for large language models","author":"Xiao","year":"2023"},{"key":"10.1016\/j.future.2026.108646_b32","series-title":"2020 53rd Annual IEEE\/ACM International Symposium on Microarchitecture","first-page":"811","article-title":"Gobo: Quantizing attention-based nlp models for low latency and energy efficient inference","author":"Zadeh","year":"2020"},{"key":"10.1016\/j.future.2026.108646_b33","article-title":"DFU-E: A dataflow architecture for edge DSP and AI applications","author":"Li","year":"2025","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"10.1016\/j.future.2026.108646_b34","series-title":"2021 ACM\/IEEE 48th Annual International Symposium on Computer Architecture","first-page":"692","article-title":"ELSA: Hardware-software co-design for efficient, lightweight self-attention mechanism in neural networks","author":"Ham","year":"2021"},{"key":"10.1016\/j.future.2026.108646_b35","series-title":"2022 55th IEEE\/ACM International Symposium on Microarchitecture","first-page":"616","article-title":"Dfx: A low-latency multi-fpga appliance for accelerating transformer-based text generation","author":"Hong","year":"2022"},{"key":"10.1016\/j.future.2026.108646_b36","doi-asserted-by":"crossref","unstructured":"T. Tambe, C. Hooper, L. Pentecost, T. Jia, E.-Y. Yang, M. Donato, V. Sanh, P. Whatmough, A.M. Rush, D. Brooks, et al., EdgeBERT: Sentence-Level Energy Optimizations for Latency-Aware Multi-task NLP Inference, in: MICRO-54: 54th Annual IEEE\/ACM International Symposium on Microarchitecture, 2021, pp. 830\u2013844.","DOI":"10.1145\/3466752.3480095"},{"issue":"6","key":"10.1016\/j.future.2026.108646_b37","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3549937","article-title":"Accelerating attention mechanism on FPGAs based on efficient reconfigurable systolic array","volume":"22","author":"Ye","year":"2023","journal-title":"ACM Trans. Embed. Comput. Syst."},{"key":"10.1016\/j.future.2026.108646_b38","series-title":"SqueezeLLM: Dense-and-Sparse quantization","author":"Kim","year":"2023"},{"key":"10.1016\/j.future.2026.108646_b39","series-title":"2025 IEEE Hot Chips 37 Symposium","first-page":"1","article-title":"Adelia: A 4nm LLM processor for efficient generative AI inference","author":"Moon","year":"2025"},{"key":"10.1016\/j.future.2026.108646_b40","doi-asserted-by":"crossref","DOI":"10.1109\/MM.2024.3420728","article-title":"LPU: A latency-optimized and highly scalable processor for large language model inference","author":"Moon","year":"2024","journal-title":"IEEE Micro"},{"key":"10.1016\/j.future.2026.108646_b41","doi-asserted-by":"crossref","unstructured":"S. Zeng, J. Liu, G. Dai, X. Yang, T. Fu, H. Wang, W. Ma, H. Sun, S. Li, Z. Huang, et al., Flightllm: Efficient large language model inference with a complete mapping flow on fpgas, in: Proceedings of the 2024 ACM\/SIGDA International Symposium on Field Programmable Gate Arrays, 2024, pp. 223\u2013234.","DOI":"10.1145\/3626202.3637562"},{"key":"10.1016\/j.future.2026.108646_b42","doi-asserted-by":"crossref","unstructured":"J. Qin, T. Xia, C. Tan, J. Zhang, S.Q. Zhang, PICACHU: Plug-In CGRA Handling Upcoming Nonlinear Operations in LLMs, in: Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2, 2025, pp. 845\u2013861.","DOI":"10.1145\/3676641.3716013"},{"key":"10.1016\/j.future.2026.108646_b43","doi-asserted-by":"crossref","unstructured":"J. Li, S. Huang, J. Xu, J. Liu, L. Ding, N. Xu, G. Dai, Marca: Mamba accelerator with reconfigurable architecture, in: Proceedings of the 43rd IEEE\/ACM International Conference on Computer-Aided Design, 2024, pp. 1\u20139.","DOI":"10.1145\/3676536.3676798"},{"key":"10.1016\/j.future.2026.108646_b44","doi-asserted-by":"crossref","unstructured":"S. Liu, G. Tao, Y. Zou, D. Chow, Z. Fan, K. Lei, B. Pan, D. Sylvester, G. Kielian, M. Saligane, Consmax: Hardware-friendly alternative softmax with learnable parameters, in: Proceedings of the 43rd IEEE\/ACM International Conference on Computer-Aided Design, 2024, pp. 1\u20139.","DOI":"10.1145\/3676536.3676766"},{"key":"10.1016\/j.future.2026.108646_b45","doi-asserted-by":"crossref","unstructured":"Z. Fan, Q. Zhang, P. Abillama, S. Shoouri, C. Lee, D. Blaauw, H.-S. Kim, D. Sylvester, Taskfusion: An efficient transfer learning architecture with dual delta sparsity for multi-task natural language processing, in: Proceedings of the 50th Annual International Symposium on Computer Architecture, 2023, pp. 1\u201314.","DOI":"10.1145\/3579371.3589040"},{"key":"10.1016\/j.future.2026.108646_b46","series-title":"2024 57th IEEE\/ACM International Symposium on Microarchitecture","first-page":"1458","article-title":"FuseMax: Leveraging extended einsums to optimize attention accelerator design","author":"Nayak","year":"2024"},{"key":"10.1016\/j.future.2026.108646_b47","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.future.2026.108646_b48","series-title":"Towards efficient and reliable LLM serving: A real-world workload study","author":"Wang","year":"2024"},{"key":"10.1016\/j.future.2026.108646_b49","series-title":"PFID: Privacy-First inference delegation framework for LLMs","author":"Yang","year":"2024"},{"key":"10.1016\/j.future.2026.108646_b50","series-title":"ShareGPT vicuna unfiltered","author":"ShareGPT Team","year":"2023"},{"key":"10.1016\/j.future.2026.108646_b51","doi-asserted-by":"crossref","unstructured":"Y. Bai, X. Lv, J. Zhang, H. Lyu, J. Tang, Z. Huang, Z. Du, X. Liu, A. Zeng, L. Hou, et al., Longbench: A bilingual, multitask benchmark for long context understanding, in: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2024, pp. 3119\u20133137.","DOI":"10.18653\/v1\/2024.acl-long.172"},{"key":"10.1016\/j.future.2026.108646_b52","doi-asserted-by":"crossref","unstructured":"J. Li, M. Wang, Z. Zheng, M. Zhang, Loogle: Can long-context language models understand long contexts?, in: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2024, pp. 16304\u201316333.","DOI":"10.18653\/v1\/2024.acl-long.859"},{"key":"10.1016\/j.future.2026.108646_b53","doi-asserted-by":"crossref","first-page":"16344","DOI":"10.52202\/068431-1189","article-title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","volume":"35","author":"Dao","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.future.2026.108646_b54","series-title":"Flashattention-2: Faster attention with better parallelism and work partitioning","author":"Dao","year":"2023"},{"issue":"1","key":"10.1016\/j.future.2026.108646_b55","doi-asserted-by":"crossref","first-page":"112","DOI":"10.1109\/LCA.2023.3333759","article-title":"Ramulator 2.0: A modern, modular, and extensible dram simulator","volume":"23","author":"Luo","year":"2023","journal-title":"IEEE Comput. Archit. Lett."},{"key":"10.1016\/j.future.2026.108646_b56","series-title":"2022 IEEE Hot Chips 34 Symposium","first-page":"1","article-title":"NVIDIA orin system-on-chip","author":"Ditty","year":"2022"},{"key":"10.1016\/j.future.2026.108646_b57","series-title":"TensorRT-LLM: A TensorRT toolbox for large language models","author":"NVIDIA Corporation","year":"2024"},{"key":"10.1016\/j.future.2026.108646_b58","doi-asserted-by":"crossref","unstructured":"W. Kwon, Z. Li, S. Zhuang, Y. Sheng, L. Zheng, C.H. Yu, J. Gonzalez, H. Zhang, I. Stoica, Efficient memory management for large language model serving with pagedattention, in: Proceedings of the 29th Symposium on Operating Systems Principles, 2023, pp. 611\u2013626.","DOI":"10.1145\/3600006.3613165"},{"key":"10.1016\/j.future.2026.108646_b59","unstructured":"NVIDIA Corporation, tegrastats Utility, https:\/\/docs.nvidia.com\/jetson\/archives\/r36.4.4\/DeveloperGuide\/AT\/JetsonLinuxDevelopmentTools\/TegrastatsUtility.html."},{"key":"10.1016\/j.future.2026.108646_b60","series-title":"Pointer sentinel mixture models","author":"Merity","year":"2016"}],"container-title":["Future Generation Computer Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167739X26002803?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167739X26002803?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T04:19:16Z","timestamp":1784780356000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167739X26002803"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":60,"alternative-id":["S0167739X26002803"],"URL":"https:\/\/doi.org\/10.1016\/j.future.2026.108646","relation":{},"ISSN":["0167-739X"],"issn-type":[{"value":"0167-739X","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"HALO: A heterogeneous accelerator for low-latency and energy-efficient edge LLM inference","name":"articletitle","label":"Article Title"},{"value":"Future Generation Computer Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.future.2026.108646","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"108646"}}