{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,17]],"date-time":"2025-09-17T06:11:04Z","timestamp":1758089464922,"version":"3.44.0"},"reference-count":41,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,22]],"date-time":"2025-06-22T00:00:00Z","timestamp":1750550400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,22]],"date-time":"2025-06-22T00:00:00Z","timestamp":1750550400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100006190","name":"Research and Development","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006190","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,22]]},"DOI":"10.1109\/dac63849.2025.11133363","type":"proceedings-article","created":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T17:35:41Z","timestamp":1757957741000},"page":"1-7","source":"Crossref","is-referenced-by-count":0,"title":["XShift: FPGA-efficient Binarized LLM with Joint Quantization and Sparsification"],"prefix":"10.1109","author":[{"given":"Shuai","family":"Zhou","sequence":"first","affiliation":[{"name":"Fudan University,State Key Lab of Integrated Chips &#x0026; Systems, and School of Microelectronics,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huinan","family":"Tian","sequence":"additional","affiliation":[{"name":"Fudan University,State Key Lab of Integrated Chips &#x0026; Systems, and School of Microelectronics,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sisi","family":"Meng","sequence":"additional","affiliation":[{"name":"Fudan University,State Key Lab of Integrated Chips &#x0026; Systems, and School of Microelectronics,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianli","family":"Chen","sequence":"additional","affiliation":[{"name":"Fudan University,State Key Lab of Integrated Chips &#x0026; Systems, and School of Microelectronics,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Yu","sequence":"additional","affiliation":[{"name":"Fudan University,State Key Lab of Integrated Chips &#x0026; Systems, and School of Microelectronics,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kun","family":"Wang","sequence":"additional","affiliation":[{"name":"Fudan University,State Key Lab of Integrated Chips &#x0026; Systems, and School of Microelectronics,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"article-title":"Improving language understanding by generative pretraining","year":"2018","author":"Radford","key":"ref1"},{"key":"ref2","article-title":"The llama 3 herd of models","author":"Dubey","year":"2024","journal-title":"arXiv preprint arXiv:2407.21783"},{"key":"ref3","article-title":"BiLLM: Pushing the limit of post-training quantization for LLMs","author":"Huang","year":"2024","journal-title":"arXiv preprint arXiv:2402.04291"},{"key":"ref4","article-title":"BitNet: Scaling 1-bit transformers for large language models","author":"Wang","year":"2023","journal-title":"arXiv preprint arXiv:2310.11453"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i21.34385"},{"article-title":"OmniQuant: Omnidirectionally calibrated quantization for large language models","volume-title":"The Twelfth International Conference on Learning Representations (ICLR)","author":"Shao","key":"ref6"},{"key":"ref7","first-page":"38087","article-title":"SmoothQuant: Accurate and efficient post-training quantization for large language models","volume-title":"International Conference on Machine Learning (ICML)","author":"Xiao"},{"article-title":"SpQR: A sparse-quantized representation for near-lossless LLM weight compression","volume-title":"The Twelfth International Conference on Learning Representations (ICLR)","author":"Dettmers","key":"ref8"},{"key":"ref9","article-title":"ILLM: Efficient integer-only inference for fully-quantized low-bit large language models","author":"Hu","year":"2024","journal-title":"arXiv preprint arXiv:2405.17849"},{"key":"ref10","first-page":"87","article-title":"AWQ: Activation-aware weight quantization for on-device LLM compression and acceleration","volume-title":"Proceedings of Machine Learning and Systems (MLSys)","volume":"6","author":"Lin"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3649329.3658498"},{"article-title":"A simple and effective pruning approach for large language models","volume-title":"The Twelfth International Conference on Learning Representations (ICLR)","author":"Sun","key":"ref12"},{"key":"ref13","first-page":"21702","article-title":"LLM-Pruner: On the structural pruning of large language models","volume":"36","author":"Ma","year":"2023","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"ref14","first-page":"10323","article-title":"SparseGPT: Massive language models can be accurately pruned in one-shot","volume-title":"International Conference on Machine Learning (ICML)","author":"Frantar"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3626202.3637562"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00051"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3649329.3656507"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589057"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00080"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589038"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3649329.3657323"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3649329.3656221"},{"key":"ref23","first-page":"14303","article-title":"BiT: Robustly binarized multi-distilled transformer","volume":"35","author":"Liu","year":"2022","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00492"},{"key":"ref25","first-page":"521","article-title":"Orca: A distributed serving system for Transformer-Based generative models","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Yu"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.jml.2019.104047"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00063"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO50266.2020.00071"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.39"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i10.28960"},{"article-title":"Compressing large language models by joint sparsification and quantization","volume-title":"Forty-first International Conference on Machine Learning (ICML)","author":"Guo","key":"ref32"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3431920.3439296"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/3676536.3676667"},{"key":"ref35","article-title":"Optimal brain damage","volume":"2","author":"LeCun","year":"1989","journal-title":"Advances in neural information processing systems (NeurIPS)"},{"key":"ref36","article-title":"GPTQ: Accurate post-training quantization for generative pre-trained transformers","author":"Frantar","year":"2022","journal-title":"arXiv preprint arXiv:2210.17323"},{"key":"ref37","first-page":"17283","article-title":"Big bird: Transformers for longer sequences","volume":"33","author":"Zaheer","year":"2020","journal-title":"Advances in neural information processing systems (NeurIPS)"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/3489517.3530505"},{"article-title":"Pointer sentinel mixture models","year":"2016","author":"Merity","key":"ref39"},{"key":"ref40","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","author":"Raffel","year":"2019","journal-title":"arXiv e-prints"},{"article-title":"A framework for few-shot language model evaluation","year":"2024","author":"Gao","key":"ref41"}],"event":{"name":"2025 62nd ACM\/IEEE Design Automation Conference (DAC)","start":{"date-parts":[[2025,6,22]]},"location":"San Francisco, CA, USA","end":{"date-parts":[[2025,6,25]]}},"container-title":["2025 62nd ACM\/IEEE Design Automation Conference (DAC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11132383\/11132091\/11133363.pdf?arnumber=11133363","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,16]],"date-time":"2025-09-16T05:25:25Z","timestamp":1758000325000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11133363\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,22]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/dac63849.2025.11133363","relation":{},"subject":[],"published":{"date-parts":[[2025,6,22]]}}}