{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:33:43Z","timestamp":1763192023836,"version":"3.45.0"},"reference-count":35,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100010822","name":"Chengdu Science and Technology Bureau","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100010822","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/ijcnn64981.2025.11229416","type":"proceedings-article","created":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T18:46:15Z","timestamp":1763145975000},"page":"1-9","source":"Crossref","is-referenced-by-count":0,"title":["Low-Rank Decomposition Assisted Quantization and Inference Compensation for Quality Large Language Model Inference"],"prefix":"10.1109","author":[{"given":"Jie","family":"Ou","sequence":"first","affiliation":[{"name":"University of Electronic Science and Technology of China,School of Information and Software Engineering,Chengdu,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinyu","family":"Guo","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China,School of Information and Software Engineering,Chengdu,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuaihong","family":"Jiang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China,School of Information and Software Engineering,Chengdu,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhaokun","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China,School of Information and Software Engineering,Chengdu,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yueming","family":"Chen","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China,School of Information and Software Engineering,Chengdu,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruini","family":"Xue","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China,School of Computer Science and Engineering,Chengdu,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenhong","family":"Tian","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China,School of Information and Software Engineering,Chengdu,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"article-title":"Sparks of artificial general intelligence: Early experiments with gpt-4","year":"2023","author":"Bubeck","key":"ref1"},{"article-title":"Llama: Open and efficient foundation language models","year":"2023","author":"Touvron","key":"ref2"},{"article-title":"Llama 2: Open foundation and fine-tuned chat models","year":"2023","author":"Touvron","key":"ref3"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.jss.2023.111734"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/healthcom56612.2023.10472387"},{"article-title":"Measuring massive multitask language understanding","year":"2020","author":"Hendrycks","key":"ref6"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1472"},{"key":"ref9","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"},{"article-title":"Hugging face accelerate","year":"2022","author":"Face","key":"ref11"},{"key":"ref12","first-page":"31094","article-title":"Flexgen: High-throughput generative inference of large language models with a single gpu","volume-title":"International Conference on Machine Learning","author":"Sheng"},{"article-title":"Gptq: Accurate post-training quantization for generative pre-trained transformers","year":"2022","author":"Frantar","key":"ref13"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3714983.3714987"},{"article-title":"Owq: Lessons learned from activation outliers for weight quantization in large language models","year":"2023","author":"Lee","key":"ref15"},{"article-title":"Spqr: A sparse-quantized representation for near-lossless llm weight compression","year":"2023","author":"Dettmers","key":"ref16"},{"article-title":"Omniquant: Omnidirectionally calibrated quantization for large language models","volume-title":"The Twelfth International Conference on Learning Representations","author":"Shao","key":"ref17"},{"key":"ref18","first-page":"38087","article-title":"Smoothquant: Accurate and efficient post-training quantization for large language models","volume-title":"International Conference on Machine Learning","author":"Xiao"},{"article-title":"Rptq: Reorder-based post-training quantization for large language models","year":"2023","author":"Yuan","key":"ref19"},{"article-title":"Qlora: Efficient finetuning of quantized llms","year":"2023","author":"Dettmers","key":"ref20"},{"article-title":"Quip: 2-bit quantization of large language models with guarantees","year":"2023","author":"Chee","key":"ref21"},{"key":"ref22","first-page":"27168","article-title":"Zeroquant: Efficient and affordable post-training quantization for large-scale transformers","volume":"35","author":"Yao","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref23","first-page":"30318","article-title":"Gpt3. int8 (): 8-bit matrix multiplication for transformers at scale","volume":"35","author":"Dettmers","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref24","first-page":"17402","article-title":"Outlier suppression: Pushing the limit of low-bit transformer language models","volume":"35","author":"Wei","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.102"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.26"},{"article-title":"Omniquant: Omnidirectionally calibrated quantization for large language models","year":"2023","author":"Shao","key":"ref27"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.173"},{"article-title":"Pointer sentinel mixture models","year":"2016","author":"Merity","key":"ref29"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.3115\/1075812.1075835"},{"issue":"140","key":"ref31","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"Journal of machine learning research"},{"article-title":"Think you have solved question answering? try arc, the ai2 reasoning challenge","year":"2018","author":"Clark","key":"ref32"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3474381"},{"key":"ref34","first-page":"8","article-title":"A framework for few-shot language model evaluation","author":"Gao","year":"2021","journal-title":"Version v0. 0.1"},{"article-title":"Opt: Open pre-trained transformer language models","year":"2022","author":"Zhang","key":"ref35"}],"event":{"name":"2025 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2025,6,30]]},"location":"Rome, Italy","end":{"date-parts":[[2025,7,5]]}},"container-title":["2025 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11227166\/11227148\/11229416.pdf?arnumber=11229416","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:31:07Z","timestamp":1763191867000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11229416\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":35,"URL":"https:\/\/doi.org\/10.1109\/ijcnn64981.2025.11229416","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}