{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T20:27:52Z","timestamp":1783628872198,"version":"3.55.0"},"reference-count":39,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100013804","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013804","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004826","name":"Beijing Natural Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004826","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2027,1]]},"DOI":"10.1016\/j.eswa.2026.133502","type":"journal-article","created":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T15:39:01Z","timestamp":1783093141000},"page":"133502","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["DisQ: Distribution-aware quantization for ultra-low-bit large language models"],"prefix":"10.1016","volume":"332","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-6390-6561","authenticated-orcid":false,"given":"Jin","family":"Peng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0068-8824","authenticated-orcid":false,"given":"Dong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133502_bib0001","unstructured":"Bengio, Y., L\u00e9onard, N., & Courville, A. (2013). Estimating or propagating gradients through stochastic neurons for conditional computation. arXiv: 1308.3432."},{"key":"10.1016\/j.eswa.2026.133502_bib0002","series-title":"Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr)","first-page":"1","article-title":"LSQ+: Improving low-bit quantization through learnable gradient scales","author":"Bhalgat","year":"2020"},{"key":"10.1016\/j.eswa.2026.133502_bib0003","series-title":"Advances in neural information processing systems","first-page":"1877","article-title":"Language models are few-shot learners","author":"Brown","year":"2020"},{"key":"10.1016\/j.eswa.2026.133502_bib0004","series-title":"Proceedings of the ieee\/cvf international conference on computer vision workshops (iccvw)","article-title":"Low-bit quantization of neural networks for efficient inference","author":"Choukroun","year":"2019"},{"key":"10.1016\/j.eswa.2026.133502_bib0005","series-title":"Proceedings of the neural information processing systems (neurips)","first-page":"1","article-title":"LLM.int8: 8-bit matrix multiplication for transformers at scale","author":"Dettmers","year":"2022"},{"key":"10.1016\/j.eswa.2026.133502_bib0006","series-title":"Proceedings of the international conference on machine learning (icml)","first-page":"1","article-title":"Rtn: Rounding toward nearest for post-training quantization","author":"Dong","year":"2023"},{"key":"10.1016\/j.eswa.2026.133502_bib0007","series-title":"Proceedings of the international conference on machine learning (icml)","article-title":"Extreme compression of large language models via additive quantization","author":"Egiazarian","year":"2024"},{"key":"10.1016\/j.eswa.2026.133502_bib0008","series-title":"Proceedings of the international conference on learning representations (iclr)","first-page":"1","article-title":"Learned step size quantization","author":"Esser","year":"2020"},{"key":"10.1016\/j.eswa.2026.133502_bib0009","series-title":"Proceedings of the international conference on learning representations (iclr)","first-page":"1","article-title":"GPTQ: Accurate post-training quantization for generative pre-trained transformers","author":"Frantar","year":"2023"},{"key":"10.1016\/j.eswa.2026.133502_bib0010","series-title":"Proceedings of the international conference on learning representations (iclr)","first-page":"1","article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding","author":"Han","year":"2016"},{"key":"10.1016\/j.eswa.2026.133502_bib0011","unstructured":"Jiang, A. Q., Sablayrolles, A., Mensch, A., Bamford, C., Chaplot, D. S., Casas, D. D. L., Bressand, F., Lengyel, G., Lample, G., & Saulnier, L., et al., (2023). Mistral 7b. arXiv: 2310.06825."},{"key":"10.1016\/j.eswa.2026.133502_bib0012","series-title":"Proceedings of icml","article-title":"Squeezellm: Dense-and-sparse quantization","author":"Kim","year":"2024"},{"key":"10.1016\/j.eswa.2026.133502_bib0013","series-title":"Proceedings of the international conference on learning representations (iclr)","article-title":"Brecq: Pushing the limit of post-training quantization by block reconstruction","author":"Li","year":"2021"},{"key":"10.1016\/j.eswa.2026.133502_bib0014","unstructured":"Lin, Z., Tang, J., & Han, S., et al., (2023). Awq: Activation-aware weight quantization for large language models. arXiv: 2306.00978."},{"key":"10.1016\/j.eswa.2026.133502_bib0015","series-title":"Proceedings of the neural information processing systems (neurips)","first-page":"1","article-title":"Quarot: Outlier-free post-training quantization for large language models","author":"Lin","year":"2024"},{"key":"10.1016\/j.eswa.2026.133502_bib0016","series-title":"Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr)","first-page":"4942","article-title":"Nonuniform-to-uniform quantization: Towards accurate quantization via generalized straight-through estimation","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133502_bib0017","unstructured":"Liu, Z., Oguz, B., Pappu, A., Xiao, L., Brown, S., Sekhon, H., & Oguz, B. (2023). Llm-qat: Data-free quantization aware training for large language models. arXiv: 2305.17888."},{"key":"10.1016\/j.eswa.2026.133502_bib0018","series-title":"Proceedings of the international conference on machine learning (icml)","article-title":"Spinquant: Llm quantization with learned rotations","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133502_bib0019","series-title":"Proceedings of the international conference on learning representations (iclr)","first-page":"1","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2019"},{"key":"10.1016\/j.eswa.2026.133502_bib0020","unstructured":"Ma, S., Wang, H., Ma, L., Wang, L., Wang, W., Huang, S., Dong, L., Wang, R., Xue, J., & Wei, F. (2024a). The era of 1-bit llms: All large language models are in 1.58 bits. arXiv: 2402.17764."},{"key":"10.1016\/j.eswa.2026.133502_bib0021","unstructured":"Ma, X., Fang, G., & Wang, X. (2024b). Affinequant: Affine transformation quantization for large language models. arXiv: 2403.03640."},{"key":"10.1016\/j.eswa.2026.133502_bib0022","series-title":"Proceedings of the international conference on learning representations (iclr)","first-page":"1","article-title":"Pointer sentinel mixture models","author":"Merity","year":"2017"},{"key":"10.1016\/j.eswa.2026.133502_bib0023","unstructured":"Meta AI (2024). Introducing Meta Llama 3: The most capable openly available LLM to date. https:\/\/ai.meta.com\/blog\/meta-llama-3\/. (accessed on 18 April 2024)."},{"key":"10.1016\/j.eswa.2026.133502_bib0024","series-title":"Proceedings of the international conference on machine learning (icml)","first-page":"7197","article-title":"Up or down? adaptive rounding for post-training quantization","author":"Nagel","year":"2020"},{"key":"10.1016\/j.eswa.2026.133502_bib0025","series-title":"Advances in neural information processing systems","first-page":"8024","article-title":"PyTorch: An imperative style, high-performance deep learning library","author":"Paszke","year":"2019"},{"issue":"140","key":"10.1016\/j.eswa.2026.133502_bib0026","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.eswa.2026.133502_bib0027","unstructured":"Q. Team (2024). Qwen2.5: A party of foundation models. https:\/\/qwenlm.github.io\/blog\/qwen2.5\/."},{"key":"10.1016\/j.eswa.2026.133502_bib0028","unstructured":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y., Bashlykov, N., Batra, S., Bhargava, P., & Bhosale, S., et al., (2023a). Llama 2: Open foundation and fine-tuned chat models. arXiv: 2307.09288."},{"key":"10.1016\/j.eswa.2026.133502_bib0029","unstructured":"Touvron, H., Martin, L., & Stone, K. et al., (2023b). LLaMA: Open and efficient foundation language models. arXiv: 2302.13971."},{"key":"10.1016\/j.eswa.2026.133502_bib0030","unstructured":"Tseng, A., Chee, J., Sun, Q., Kuleshov, V., & De Sa, C. (2024). QuIP: Even better llm quantization with hadamard incoherence and lattice codebooks. arXiv: 2402.04396."},{"key":"10.1016\/j.eswa.2026.133502_bib0031","unstructured":"Wang, H., Ma, S., Dong, L., Huang, S., Wang, H., Ma, L., Yang, F., Wang, R., Wu, Y., & Wei, F. (2023). Bitnet: Scaling 1-bit transformers for large language models. arXiv: 2310.11453."},{"key":"10.1016\/j.eswa.2026.133502_bib0032","series-title":"Proceedings of the 2020 conference on empirical methods in natural language processing: System demonstrations","first-page":"38","article-title":"Transformers: State-of-the-art natural language processing","author":"Wolf","year":"2020"},{"key":"10.1016\/j.eswa.2026.133502_bib0033","series-title":"Proceedings of the international conference on learning representations (iclr)","first-page":"1","article-title":"SpQR: A sparse-quantized representation for efficient llm inference","author":"Xiao","year":"2024"},{"key":"10.1016\/j.eswa.2026.133502_bib0034","doi-asserted-by":"crossref","unstructured":"Xu, Y., Han, W., Zhang, Y., Wu, H., He, P., & Chen, W. (2024). Onebit: Towards extremely low-bit large language models. arXiv: 2402.11295.","DOI":"10.52202\/079017-2122"},{"key":"10.1016\/j.eswa.2026.133502_bib0035","series-title":"Proceedings of the international conference on machine learning (icml)","first-page":"1","article-title":"Smoothquant: Accurate and efficient post-training quantization for large language models","author":"Xu","year":"2023"},{"key":"10.1016\/j.eswa.2026.133502_bib0036","unstructured":"Zhang, S., Roller, S., Goyal, N., Artetxe, M., Chen, M., Chen, S., Dewan, C., Diab, M., Li, X., & Lin, X. V., et al., (2022). Opt: Open pre-trained transformer language models. arXiv: 2205.01068."},{"key":"10.1016\/j.eswa.2026.133502_bib0037","series-title":"Proceedings of the international conference on machine learning (icml)","first-page":"1","article-title":"Omniquant: Omnidirectionally calibrated quantization for large language models","author":"Zhang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133502_bib0038","series-title":"Proceedings of the 42nd international conference on machine learning (icml)","article-title":"GANQ: Gpu-adaptive non-uniform quantization for large language models","author":"Zhao","year":"2025"},{"key":"10.1016\/j.eswa.2026.133502_bib0039","unstructured":"Zhou, S., Wu, Y., Ni, Z., Zhou, X., Wen, H., & Zou, Y. (2016). Dorefa-net: Training low bitwidth convolutional neural networks with low bitwidth gradients. arXiv: 1606.06160."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426024115?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426024115?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T20:06:36Z","timestamp":1783627596000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426024115"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,1]]},"references-count":39,"alternative-id":["S0957417426024115"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133502","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2027,1]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"DisQ: Distribution-aware quantization for ultra-low-bit large language models","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133502","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133502"}}