{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T00:27:39Z","timestamp":1787012859363,"version":"3.56.0"},"reference-count":69,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022YFB4400600"],"award-info":[{"award-number":["2022YFB4400600"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013064","name":"Postgraduate Research and Practice Innovation Program of Jiangsu Province","doi-asserted-by":"publisher","award":["KYCX24_0149"],"award-info":[{"award-number":["KYCX24_0149"]}],"id":[{"id":"10.13039\/501100013064","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Comput.-Aided Des. Integr. Circuits Syst."],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1109\/tcad.2025.3604321","type":"journal-article","created":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T17:42:29Z","timestamp":1756489349000},"page":"1935-1948","source":"Crossref","is-referenced-by-count":2,"title":["APT-LLM: Exploiting Arbitrary-Precision Tensor Core Computing for LLM Acceleration"],"prefix":"10.1109","volume":"45","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-6173-3983","authenticated-orcid":false,"given":"Shaobo","family":"Ma","sequence":"first","affiliation":[{"name":"School of Electronic Science and Engineering, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3430-1189","authenticated-orcid":false,"given":"Chao","family":"Fang","sequence":"additional","affiliation":[{"name":"School of Electronic Science and Engineering, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6965-3436","authenticated-orcid":false,"given":"Haikuo","family":"Shao","sequence":"additional","affiliation":[{"name":"School of Electronic Science and Engineering, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7227-4786","authenticated-orcid":false,"given":"Zhongfeng","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Electronic Science and Engineering, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Sparks of artificial general intelligence: Early experiments with GPT-4","author":"Bubeck","year":"2023","journal-title":"arXiv:2303.12712"},{"key":"ref2","article-title":"The Llama 3 herd of models","author":"Grattafiori","year":"2024","journal-title":"arXiv:2407.21783"},{"key":"ref3","article-title":"DeepSeek-V3 technical report","volume-title":"arXiv:2412.19437","author":"Liu","year":"2024"},{"key":"ref4","article-title":"DeepSeek-r1: Incentivizing reasoning capability in LLMs via reinforcement learning","author":"Guo","year":"2025","journal-title":"arXiv:2501.12948"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/MCI.2014.2307227"},{"key":"ref6","article-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020","journal-title":"arXiv:2001.08361"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2024.3383347"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2022.3181541"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0441"},{"key":"ref10","first-page":"1","article-title":"OPTQ: Accurate quantization for generative pre-trained transformers","volume-title":"Proc. 11th Int. Conf. Learn. Represent. (ICLR)","author":"Frantar"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ASP-DAC58780.2024.10473817"},{"key":"ref12","first-page":"42097","article-title":"Token-scaled logit distillation for ternary weight generative language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Kim"},{"key":"ref13","article-title":"OneBit: Towards extremely low-bit large language models","author":"Xu","year":"2024","journal-title":"arXiv:2402.11295"},{"key":"ref14","first-page":"38087","article-title":"SmoothQuant: Accurate and efficient post-training quantization for large language models","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Xiao"},{"key":"ref15","first-page":"1","article-title":"OmniQuant: Omnidirectionally calibrated quantization for large language models","volume-title":"Proc. 12th Int. Conf. Learn. Represent. (ICLR)","author":"Shao"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD63220.2024.00039"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3300309"},{"key":"ref18","first-page":"196","article-title":"Atom: Low-bit quantization for efficient and accurate LLM serving","volume-title":"Proc. Mach. Learn. Syst. (MLSys)","author":"Zhao"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2021.3061394"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2020.2971677"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3217824"},{"key":"ref22","article-title":"QServe: W4A8KV4 quantization and system co-design for efficient LLM serving","author":"Lin","year":"2024","journal-title":"arXiv:2405.04532"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1093\/bioinformatics\/btu047"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3589013.3596678"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3178487.3178491"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2198"},{"key":"ref27","first-page":"699","article-title":"Quant-LLM: Accelerating the serving of large language models via FP6-centric algorithm-system co-design on modern GPUs","volume-title":"Proc. USENIX Annu. Tech. Conf. (ATC)","author":"Xia"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2021.02.013"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3572848.3577479"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2022.3200528"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2024.3436521"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2024.3518413"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356169"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2020.3045828"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC55821.2022.9926299"},{"key":"ref36","article-title":"Dissecting the NVidia Turing T4 GPU via microbenchmarking","author":"Jia","year":"2019","journal-title":"arXiv:1903.07486"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS57955.2024.00064"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2023.3317169"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/3578178.3578238"},{"key":"ref40","first-page":"149","article-title":"TC-GNN: Bridging sparse GNN computation and dense tensor cores on GPUs","volume-title":"Proc. USENIX Annu. Tech. Conf. (ATC)","author":"Wang"},{"key":"ref41","article-title":"A survey on efficient inference for large language models","author":"Zhou","year":"2024","journal-title":"arXiv:2404.14294"},{"key":"ref42","article-title":"Transformer tricks: Precomputing the first layer","author":"Graef","year":"2024","journal-title":"arXiv:2402.13388"},{"key":"ref43","article-title":"Model tells you what to discard: Adaptive KV cache compression for LLMs","author":"Ge","year":"2023","journal-title":"arXiv:2310.01801"},{"key":"ref44","first-page":"606","article-title":"Efficiently scaling transformer inference","volume-title":"Proc. Mach. Learn. Syst. (MLSys)","author":"Pope"},{"key":"ref45","article-title":"LLM inference unveiled: Survey and roofline model insights","author":"Yuan","year":"2024","journal-title":"arXiv:2402.16363"},{"key":"ref46","first-page":"23901","article-title":"SqueezeLLM: Dense-and-sparse quantization","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Kim"},{"key":"ref47","first-page":"56838","article-title":"QuantSR: Accurate low-bit quantization for efficient image super-resolution","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Qin"},{"key":"ref48","first-page":"41498","article-title":"Accurate LoRA-finetuning quantization of LLMs via information retention","volume-title":"Proc. 41st Int. Conf. Mach. Learn. (ICML)","author":"Qin"},{"key":"ref49","first-page":"43307","article-title":"BiMatting: Efficient video matting via binarization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Qin"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICIESTR60916.2024.10798212"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096223"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00110"},{"key":"ref53","first-page":"3123","article-title":"BinaryConnect: Training deep neural networks with binary weights during propagations","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"28","author":"Courbariaux"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476157"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1145\/3330345.3330386"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2020.3013637"},{"key":"ref57","article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and Huffman coding","author":"Han","year":"2015","journal-title":"arXiv:1510.00149"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00063"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00881"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01237-3_23"},{"key":"ref61","article-title":"DoReFa-Net: Training low bitwidth convolutional neural networks with low bitwidth gradients","author":"Zhou","year":"2016","journal-title":"arXiv:1606.06160"},{"key":"ref62","article-title":"BitNet: Scaling 1-bit transformers for large language models","author":"Wang","year":"2023","journal-title":"arXiv:2310.11453"},{"key":"ref63","volume-title":"CUTLASS","author":"Thakkar","year":"2023"},{"key":"ref64","article-title":"Understanding GEMM performance and energy on NVIDIA Ada Lovelace: A machine learning-based analytical approach","author":"Halim","year":"2024","journal-title":"arXiv:2411.16954"},{"key":"ref65","article-title":"Qwen2.5 technical report","volume-title":"arXiv:2412.15115","author":"Qwen","year":"2024"},{"key":"ref66","article-title":"OPT: Open pre-trained transformer language models","author":"Zhang","year":"2022","journal-title":"arXiv:2205.01068"},{"key":"ref67","article-title":"Bloom: A 176B-parameter open-access multilingual language model","author":"Workshop","year":"2022","journal-title":"arXiv:2211.05100"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3180"},{"key":"ref69","article-title":"Pointer sentinel mixture models","author":"Merity","year":"2016","journal-title":"arXiv:1609.07843"}],"container-title":["IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/43\/11450487\/11145154.pdf?arnumber=11145154","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T22:30:11Z","timestamp":1780007411000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11145154\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":69,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tcad.2025.3604321","relation":{},"ISSN":["0278-0070","1937-4151"],"issn-type":[{"value":"0278-0070","type":"print"},{"value":"1937-4151","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]}}}