{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T16:30:32Z","timestamp":1781195432423,"version":"3.54.1"},"reference-count":48,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Science Foundation","award":["2339084"],"award-info":[{"award-number":["2339084"]}]},{"DOI":"10.13039\/501100000266","name":"Engineering and Physical Sciences Research Council","doi-asserted-by":"crossref","award":["EP\/S030069\/1"],"award-info":[{"award-number":["EP\/S030069\/1"]}],"id":[{"id":"10.13039\/501100000266","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Comput."],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1109\/tc.2025.3628193","type":"journal-article","created":{"date-parts":[[2025,11,5]],"date-time":"2025-11-05T18:39:02Z","timestamp":1762367942000},"page":"567-581","source":"Crossref","is-referenced-by-count":1,"title":["Bit-Serial Acceleration of LLM Inference With Mixture-of-Datatype Quantization"],"prefix":"10.1109","volume":"75","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6387-327X","authenticated-orcid":false,"given":"Yuzong","family":"Chen","sequence":"first","affiliation":[{"name":"Cornell University, New York, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chi-Chih","family":"Chang","sequence":"additional","affiliation":[{"name":"Cornell University, New York, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xilai","family":"Dai","sequence":"additional","affiliation":[{"name":"Cornell University, New York, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6381-2936","authenticated-orcid":false,"given":"Ahmed F.","family":"AbouElhamayed","sequence":"additional","affiliation":[{"name":"Cornell University, New York, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6153-0491","authenticated-orcid":false,"given":"Marta","family":"Andronic","sequence":"additional","affiliation":[{"name":"Imperial College London, London, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0201-310X","authenticated-orcid":false,"given":"George A.","family":"Constantinides","sequence":"additional","affiliation":[{"name":"Imperial College London, London, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4568-8932","authenticated-orcid":false,"given":"Mohamed S.","family":"Abdelfattah","sequence":"additional","affiliation":[{"name":"Cornell University, New York, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"OPT: Open pre-trained transformer language models","author":"Zhang","year":"2022"},{"key":"ref2","article-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2024.3373763"},{"key":"ref4","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brown","year":"2020"},{"key":"ref5","article-title":"GPTQ: Accurate post-training compression for generative pretrained transformers","author":"Frantar","year":"2022"},{"key":"ref6","article-title":"AWQ: Activation-aware weight quantization for LLM compression and acceleration","volume-title":"Proc. Mach. Learn. Syst. (MLSys)","author":"Lin","year":"2024"},{"key":"ref7","article-title":"Learning from students: Applying t-distributions to explore accurate and efficient formats for LLMs","author":"Dotzel","year":"2024"},{"key":"ref8","article-title":"OmniQuant: Omnidirectionally calibrated quantization for large language models","author":"Shao","year":"2024"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00084"},{"key":"ref10","article-title":"KIVI: A tuning-free asymmetric 2bit quantization for KV cache","author":"Liu","year":"2024"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0040"},{"key":"ref12","article-title":"GEAR: An efficient KV cache compression recipe for near-lossless generative inference of LLM","author":"Kang","year":"2024"},{"key":"ref13","first-page":"1","article-title":"FlexGen: High-throughput generative inference of large language models with a single GPU","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Sheng","year":"2023"},{"key":"ref14","first-page":"1","article-title":"Keyformer: KV cache reduction through key tokens selection for efficient generative inference","volume-title":"Proc. Mach. Learn. Syst. (MLSys)","author":"Adnan","year":"2024"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3180"},{"key":"ref16","first-page":"1","article-title":"Atom: Low-bit quantization for efficient and accurate LLM serving","volume-title":"Proc. Mach. Learn. Syst. (MLSys)","author":"Zhao","year":"2024"},{"key":"ref17","article-title":"LLM inference unveiled: Survey and roofline model insights","author":"Yuan","year":"2024"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00064"},{"key":"ref19","article-title":"OCP Microscaling Formats (MX) Specification.\u201d"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00095"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589038"},{"key":"ref22","article-title":"8-bit optimizers via block-wise quantization","author":"Dettmers","year":"2022"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589351"},{"key":"ref24","article-title":"Pointer sentinel mixture models","author":"Merity","year":"2016"},{"key":"ref25","article-title":"QLoRA: Efficient finetuning of quantized LLMs","author":"Dettmers","year":"2023"},{"key":"ref26","article-title":"ZeroQuant(4\u2009+\u20092): Redefining LLMs quantization with a new FP6-centric strategy for diverse generative tasks","author":"Wu","year":"2023"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.98"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2016.7783722"},{"key":"ref29","first-page":"382","article-title":"Bit-pragmatic deep neural network computing","volume-title":"Proc. IEEE\/ACM 50th Annu. Int. Symp. Microarchit. (MICRO),","author":"Albericio","year":"2017"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00062"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00048"},{"key":"ref32","first-page":"1","article-title":"Torch2Chip: An end-to-end customizable deep neural network compression and deployment toolkit for prototype hardware accelerator design","volume-title":"Proc. Mach. Learn. Syst. (MLSys),","author":"Meng","year":"2024"},{"key":"ref33","article-title":"SmoothQuant: Accurate and efficient post-training quantization for large language models","author":"Xiao","year":"2022"},{"key":"ref34","first-page":"1","article-title":"VS-Quant: Per-vector scaled quantization for accurate low-precision neural network inference","volume-title":"Proc. Mach. Learn. Syst. (MLSys)","author":"Dai","year":"2021"},{"key":"ref35","article-title":"ZeroQuant: Efficient and affordable post-training quantization for large-scale transformers","author":"Yao","year":"2022"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480106"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00086"},{"key":"ref38","article-title":"Gsm8k dataset.\u201d"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/p19-1472"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6399"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2020.2973991"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3085572"},{"key":"ref44","article-title":"Chain of thought prompting elicits reasoning in large language models","author":"Wei","year":"2022"},{"key":"ref45","first-page":"1","article-title":"QServe: W4A8KV4 quantization and system co-design for efficient LLM serving","volume-title":"Proc. Mach. Learn. Syst. (MLSys)","author":"Lin","year":"2025"},{"key":"ref46","article-title":"FP8 formats for deep learning","author":"Micikevicius","year":"2022"},{"key":"ref47","article-title":"Gpt-Oss-120b & Gpt-Oss-20b Model Card","year":"2025"},{"key":"ref48","article-title":"Introducing NVFP4 for Efficient and Accurate Low-Precision Inference.\u201d"}],"container-title":["IEEE Transactions on Computers"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/12\/11353113\/11230072-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/12\/11353113\/11230072.pdf?arnumber=11230072","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,15]],"date-time":"2026-01-15T20:49:33Z","timestamp":1768510173000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11230072\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2]]},"references-count":48,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tc.2025.3628193","relation":{},"ISSN":["0018-9340","1557-9956","2326-3814"],"issn-type":[{"value":"0018-9340","type":"print"},{"value":"1557-9956","type":"electronic"},{"value":"2326-3814","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2]]}}}