{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:13:55Z","timestamp":1763190835595,"version":"3.45.0"},"reference-count":32,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100006190","name":"Research and Development","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006190","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/ijcnn64981.2025.11228461","type":"proceedings-article","created":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T18:46:15Z","timestamp":1763145975000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["LSAQ: Layer-Specific Adaptive Quantization for Large Language Model Deployment"],"prefix":"10.1109","author":[{"given":"Binrui","family":"Zeng","sequence":"first","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Ji","sequence":"additional","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaodong","family":"Liu","sequence":"additional","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Yu","sequence":"additional","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shasha","family":"Li","sequence":"additional","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Ma","sequence":"additional","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaopeng","family":"Li","sequence":"additional","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shangwen","family":"Wang","sequence":"additional","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinran","family":"Hong","sequence":"additional","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongtao","family":"Tang","sequence":"additional","affiliation":[{"name":"National University of Defense Technology,College of Computer Science and Technology,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"article-title":"Large language models meet nlp: A survey","year":"2024","author":"Qin","key":"ref1"},{"article-title":"A survey on large language models for code generation","year":"2024","author":"Jiang","key":"ref2"},{"article-title":"Model editing for llms4code: How far are we?","year":"2024","author":"Li","key":"ref3"},{"article-title":"Revolutionizing finance with llms: An overview of applications and insights","year":"2024","author":"Zhao","key":"ref4"},{"article-title":"Large language models for education: A survey and outlook","year":"2024","author":"Wang","key":"ref5"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.39"},{"article-title":"Llama 2: Open foundation and fine-tuned chat models","year":"2023","author":"Touvron","key":"ref7"},{"article-title":"Gptq: Accurate post-training quantization for generative pre-trained transformers","year":"2022","author":"Frantar","key":"ref8"},{"article-title":"Llm.int8(): 8-bit matrix multiplication for transformers at scale","year":"2022","author":"Dettmers","key":"ref9"},{"article-title":"Duquant: Distributing outliers via dual transformation makes stronger quantized llms","year":"2024","author":"Lin","key":"ref10"},{"key":"ref11","first-page":"21702","article-title":"Llm-pruner: On the structural pruning of large language models","volume":"36","author":"Ma","year":"2023","journal-title":"Advances in neural information processing systems"},{"key":"ref12","first-page":"10323","article-title":"Sparsegpt: Massive language models can be accurately pruned in one-shot","volume-title":"International Conference on Machine Learning","author":"Frantar"},{"key":"ref13","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.findings-emnlp.372","article-title":"Laco: Large language model pruning via layer collapse","author":"Yang","year":"2024"},{"article-title":"Minillm: Knowledge distillation of large language models","volume-title":"The Twelfth International Conference on Learning Representations","author":"Gu","key":"ref14"},{"article-title":"In-context learning distillation: Transferring few-shot learning ability of pre-trained language models","year":"2022","author":"Huang","key":"ref15"},{"article-title":"Tensorgpt: Efficient compression of the embedding layer in llms based on the tensor-train decomposition","year":"2023","author":"Xu","key":"ref16"},{"article-title":"Omniquant: Omnidirectionally calibrated quantization for large language models","year":"2024","author":"Shao","key":"ref17"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1035"},{"article-title":"Layer-wise quantization: A pragmatic and effective method for quantizing llms beyond integer bit-levels","year":"2024","author":"Dumitru","key":"ref19"},{"article-title":"Pointer sentinel mixture models","year":"2016","author":"Merity","key":"ref20"},{"key":"ref21","first-page":"27168","article-title":"Zeroquant: Efficient and affordable post-training quantization for large-scale transformers","volume":"35","author":"Yao","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref22","first-page":"38087","article-title":"Smoothquant: Accurate and efficient post-training quantization for large language models","volume-title":"International Conference on Machine Learning","author":"Xiao"},{"key":"ref23","first-page":"87","article-title":"Awq: Activation-aware weight quantization for on-device llm compression and acceleration","volume-title":"Proceedings of Machine Learning and Systems","volume":"6","author":"Lin"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.579"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29818"},{"issue":"6","key":"ref26","first-page":"380","article-title":"Using of jaccard coefficient for keywords similarity","volume-title":"Proceedings of the international multiconference of engineers and computer scientists","volume":"1","author":"Niwattanakul"},{"article-title":"The llama 3 herd of models","year":"2024","author":"Dubey","key":"ref27"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"article-title":"Think you have solved question answering? try arc, the ai2 reasoning challenge","year":"2018","author":"Clark","key":"ref29"},{"article-title":"Boolq: Exploring the surprising difficulty of natural yes\/no questions","year":"2019","author":"Clark","key":"ref30"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1472"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/3474381"}],"event":{"name":"2025 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2025,6,30]]},"location":"Rome, Italy","end":{"date-parts":[[2025,7,5]]}},"container-title":["2025 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11227166\/11227148\/11228461.pdf?arnumber=11228461","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:12:18Z","timestamp":1763190738000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11228461\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":32,"URL":"https:\/\/doi.org\/10.1109\/ijcnn64981.2025.11228461","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}