{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,15]],"date-time":"2026-01-15T22:52:28Z","timestamp":1768517548999,"version":"3.49.0"},"reference-count":29,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,12,14]],"date-time":"2025-12-14T00:00:00Z","timestamp":1765670400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,12,14]],"date-time":"2025-12-14T00:00:00Z","timestamp":1765670400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,12,14]]},"DOI":"10.1109\/icpads67057.2025.11322900","type":"proceedings-article","created":{"date-parts":[[2026,1,14]],"date-time":"2026-01-14T20:36:54Z","timestamp":1768423014000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["PQC-LLM: Post-Quantization Delta Compression for LLMs"],"prefix":"10.1109","author":[{"given":"Yujin","family":"Zhong","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology, Nanjing University of Science and Technology,Nanjing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chao","family":"Wu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Nanjing University of Science and Technology,Nanjing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cheng","family":"Ji","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Nanjing University of Science and Technology,Nanjing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"OPT: open pre-trained transformer language models","author":"Zhang","year":"2022"},{"key":"ref2","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref3","author":"Touvron","year":"2023","journal-title":"Llama 2: open foundation and fine-tuned chat models"},{"key":"ref4","volume-title":"GPTQ: accurate post-training quantization for generative pre-trained transformers","author":"Frantar","year":"2023"},{"key":"ref5","author":"Kim","year":"2023","journal-title":"Squeezellm: Dense-and-sparse quantization"},{"key":"ref6","first-page":"87","article-title":"Awq: Activation-aware weight quantization for on-device llm compression and acceleration","volume-title":"Proceedings of Machine Learning and Systems","volume":"6","author":"Lin","year":"2024"},{"key":"ref7","author":"Liu","year":"2024","journal-title":"Spinquant: Llm quantization with learned rotations"},{"key":"ref8","author":"Shao","year":"2023","journal-title":"Omniquant: Omnidirectionally calibrated quantization for large language models"},{"key":"ref9","volume-title":"LLM-QAT: data-free quantization aware training for large language models","author":"Liu","year":"2023"},{"key":"ref10","first-page":"42097","article-title":"Token-scaled logit distillation for ternary weight generative language models","volume":"36","author":"Kim","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref11","first-page":"38087","article-title":"SmoothQuant: accurate and efficient post-training quantization for large language models","volume-title":"International Conference on Machine Learning","author":"Xiao","year":"2023"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.17487\/RFC8478"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.331"},{"key":"ref14","volume-title":"LLM-QAT: data-free quantization aware training for large language models","author":"Liu","year":"2023"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3617688"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/JETCAS.2019.2950093"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2018.00024"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3404397.3404408"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3230840"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TCSI.2024.3506999"},{"key":"ref21","first-page":"14139","article-title":"GACT: activation compressed training for generic network architectures","volume-title":"Proceedings of the 39th International Conference on Machine Learning","volume":"162","author":"Liu","year":"2022"},{"key":"ref22","volume-title":"When compression meets model compression: memory-efficient double compression for large language models","author":"Wang","year":"2025"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2014.2346458"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICECS.2015.7440278"},{"key":"ref25","article-title":"A hardware implementation of the snappy compression algorithm","volume-title":"UC Berkeley EECS Masters Thesis","author":"Kovacs","year":"2019"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/DSD51259.2020.00088"},{"key":"ref27","doi-asserted-by":"crossref","first-page":"377","DOI":"10.1145\/2370816.2370870","article-title":"Base-delta-immediate compression: practical data compression for on-chip caches","volume-title":"Proceedings of the 21st International Conference on Parallel Architectures and Compilation Techniques","author":"Pekhimenko","year":"2012"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2016.11"},{"key":"ref29","article-title":"Lossy compression for checkpointing: fallible or feasible?","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis (SC)","author":"Ni","year":"2014"}],"event":{"name":"2025 IEEE 31th International Conference on Parallel and Distributed Systems (ICPADS)","location":"Hefei, China","start":{"date-parts":[[2025,12,14]]},"end":{"date-parts":[[2025,12,18]]}},"container-title":["2025 IEEE 31th International Conference on Parallel and Distributed Systems (ICPADS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11322805\/11322871\/11322900.pdf?arnumber=11322900","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,15]],"date-time":"2026-01-15T07:08:57Z","timestamp":1768460937000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11322900\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/icpads67057.2025.11322900","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]}}}