{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T06:01:03Z","timestamp":1780639263804,"version":"3.54.1"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,4,20]],"date-time":"2026-04-20T00:00:00Z","timestamp":1776643200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,20]],"date-time":"2026-04-20T00:00:00Z","timestamp":1776643200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,4,20]]},"DOI":"10.23919\/date69613.2026.11539471","type":"proceedings-article","created":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T19:53:10Z","timestamp":1780602790000},"page":"1-7","source":"Crossref","is-referenced-by-count":0,"title":["Efficient LLM Decoding on Ryzen AI NPUs"],"prefix":"10.23919","author":[{"given":"Zhenyu","family":"Xu","sequence":"first","affiliation":[{"name":"Clemson University,Electrical and Computer Engineering,Clemson,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Miaoxiang","family":"Yu","sequence":"additional","affiliation":[{"name":"Clemson University,Electrical and Computer Engineering,Clemson,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jillian","family":"Cai","sequence":"additional","affiliation":[{"name":"Clemson University,Electrical and Computer Engineering,Clemson,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qing","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Rhode Island,Electrical and Computer Engineering,Kingston,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Wei","sequence":"additional","affiliation":[{"name":"Clemson University,Electrical and Computer Engineering,Clemson,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Why Edge AI Inferencing is Crucial","year":"2025"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3669940.3707239"},{"key":"ref3","volume-title":"Best AI PCs in 2024","year":"2024"},{"key":"ref4","volume-title":"Copilot+ PCs developer guide","year":"2024"},{"issue":"6","key":"ref5","first-page":"12","article-title":"AMD XDNA NPU in Ryzen AI Processors","volume-title":"IEEE Micro","volume":"44","author":"Rico","year":"2024"},{"key":"ref6","doi-asserted-by":"crossref","DOI":"10.1109\/FCCM62733.2025.00031","article-title":"Unlocking the AMD Neural Processing Unit for ML Training on the Client Using Bare-Metal-Programming Tools","author":"Rosti","year":"2025"},{"key":"ref7","volume-title":"Riallto Video Overview","year":"2024"},{"key":"ref8","article-title":"mlir-aie: An MLIR-based toolchain for AMD AI Engine-enabled devices","year":"2025"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/fccm62733.2025.00043"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3706628.3708870"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3706628.3708822"},{"key":"ref12","volume-title":"Release Notes \u2014 Ryzen AI Software 1.4 Documentation","year":"2025"},{"key":"ref13","volume-title":"LLaMA 3: Open Foundation and Instruction Models","author":"Touvron","year":"2024"},{"key":"ref14","volume-title":"AMD Ryzen AI 5 340 Processor","year":"2025"},{"key":"ref15","volume-title":"RyzenAI-1.4 LLM NPU Models","year":"2024"},{"key":"ref16","volume-title":"LLaMA 3: Open Foundation and Instruction Models","year":"2024"},{"key":"ref17","article-title":"Online normalizer calculation for softmax","volume-title":"CoRR","author":"Milakov","year":"2018"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1189"},{"key":"ref19","article-title":"ASRock 4X4 BOX-AI340 Mini-PC","year":"2025"},{"key":"ref20","first-page":"87","article-title":"Awq: Activation-aware weight quantization for on-device llm compression and acceleration","volume-title":"Proceedings of Machine Learning and Systems","volume":"6","author":"Lin"},{"key":"ref21","article-title":"AMD Quark: Cross-Platform Deep Learning Model Quantization Toolkit","year":"2025"},{"key":"ref22","volume-title":"GAIA: Generative AI Is Awesome","year":"2025"},{"key":"ref23","volume-title":"Lemonade: Local llm server with npu acceleration","year":"2025"},{"key":"ref24","article-title":"LM Studio: Your Local AI Toolkit","year":"2025"},{"key":"ref25","article-title":"HWiNFO: System Information and Diagnostics Tool","year":"2024"},{"key":"ref26","article-title":"Qwen3 technical report","author":"Yang","year":"2025"}],"event":{"name":"2026 Design, Automation &amp; Test in Europe Conference (DATE)","location":"Verona, Italy","start":{"date-parts":[[2026,4,20]]},"end":{"date-parts":[[2026,4,22]]}},"container-title":["2026 Design, Automation &amp;amp; Test in Europe Conference (DATE)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11539023\/11539024\/11539471.pdf?arnumber=11539471","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T05:10:12Z","timestamp":1780636212000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11539471\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,20]]},"references-count":26,"URL":"https:\/\/doi.org\/10.23919\/date69613.2026.11539471","relation":{},"subject":[],"published":{"date-parts":[[2026,4,20]]}}}