{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T20:05:37Z","timestamp":1780949137658,"version":"3.54.1"},"reference-count":61,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"7","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1109\/tpami.2026.3664873","type":"journal-article","created":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T21:06:50Z","timestamp":1771276010000},"page":"7607-7621","source":"Crossref","is-referenced-by-count":0,"title":["MC#: Mixture Compressor for Mixture-of-Experts Large Models"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-9885-0028","authenticated-orcid":false,"given":"Wei","family":"Huang","sequence":"first","affiliation":[{"name":"Department of Electrical and Computer Engineering, The University of Hong Kong, Hong Kong SAR, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2671-0655","authenticated-orcid":false,"given":"Yue","family":"Liao","sequence":"additional","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yukang","family":"Chen","sequence":"additional","affiliation":[{"name":"NVIDIA Research, Santa Clara, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianhui","family":"Liu","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, The University of Hong Kong, Hong Kong SAR, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6721-2468","authenticated-orcid":false,"given":"Haoru","family":"Tan","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, The University of Hong Kong, Hong Kong SAR, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9180-2935","authenticated-orcid":false,"given":"Si","family":"Liu","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiming","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, The University of Hong Kong, Hong Kong SAR, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuicheng","family":"Yan","sequence":"additional","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4285-1626","authenticated-orcid":false,"given":"Xiaojuan","family":"Qi","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, The University of Hong Kong, Hong Kong SAR, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1","article-title":"OLMoE: Open mixture-of-experts language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Muennighoff","year":"2025"},{"key":"ref2","article-title":"Mixtral of experts","author":"Jiang","year":"2024"},{"key":"ref3","article-title":"DeepSeek-VL2: Mixture-of-experts vision-language models for advanced multimodal understanding","author":"Wu","year":"2024"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/tmm.2026.3654458"},{"key":"ref5","article-title":"Examining post-training quantization for mixture-of-experts: A benchmark","author":"Li","year":"2024"},{"key":"ref6","first-page":"1","article-title":"MC-MoE: Mixture compressor for mixture-of-experts LLMs gains more","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Huang","year":"2025"},{"key":"ref7","first-page":"34600","article-title":"On the representation collapse of sparse mixture of experts","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chi","year":"2022"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.334"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.198"},{"key":"ref10","article-title":"Scalable and efficient MoE training for multitask multilingual models","author":"Kim","year":"2021"},{"key":"ref11","first-page":"1","article-title":"GPTQ: Accurate post-training quantization for generative pre-trained transformers","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Frantar","year":"2023"},{"key":"ref12","first-page":"48630","article-title":"QuIP#: Even better LLM quantization with Hadamard incoherence and lattice codebooks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Tseng","year":"2014"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.498"},{"key":"ref14","first-page":"1","article-title":"OmniQuant: Omnidirectionally calibrated quantization for large language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Shao","year":"2024"},{"key":"ref15","first-page":"12284","article-title":"Extreme compression of large language models via additive quantization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Egiazarian","year":"2024"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.1168"},{"key":"ref17","first-page":"1","article-title":"Categorical reparameterization with Gumbel-softmax","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Jang","year":"2017"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3641289"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-026-60308-3"},{"key":"ref20","article-title":"Deepseek-VL: Towards real-world vision-language understanding","author":"Lu","year":"2024"},{"key":"ref21","first-page":"1","article-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Shazeer","year":"2017"},{"key":"ref22","article-title":"Toward inference-optimal mixture-of-expert large language models","author":"Yun","year":"2024"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-025-09422-z"},{"key":"ref24","article-title":"A survey on efficient inference for large language models","author":"Zhou","year":"2024"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00704"},{"key":"ref26","first-page":"87","article-title":"AWQ: Activation-aware weight quantization for on-device LLM compression and acceleration","volume-title":"Proc. Mach. Learn. Syst.","author":"Lin","year":"2024"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00394"},{"key":"ref28","first-page":"38087","article-title":"SmoothQuant: Accurate and efficient post-training quantization for large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Xiao","year":"2023"},{"key":"ref29","first-page":"18518","article-title":"HAWQ-V2: Hessian aware trace-weighted quantization of neural networks","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Dong","year":"2020"},{"key":"ref30","first-page":"1","article-title":"PB-LLM: Partially binarized large language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Shang","year":"2024"},{"key":"ref31","first-page":"20023","article-title":"BiLLM: Pushing the limit of post-training quantization for LLMs","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Huang","year":"2024"},{"key":"ref32","first-page":"1","article-title":"SPQR: A sparse-quantized representation for near-lossless LLM weight compression","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dettmers","year":"2024"},{"key":"ref33","first-page":"25672","article-title":"SLIM-LLM: Salience-driven mixed-precision quantization for large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Huang","year":"2024"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.26"},{"key":"ref35","first-page":"1","article-title":"LQ-LORA: Low-rank plus quantized matrix decomposition for efficient language model finetuning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Guo","year":"2024"},{"key":"ref36","article-title":"How good are low-bit quantized LLaMA3 models? An empirical study","author":"Huang","year":"2024"},{"key":"ref37","first-page":"24101","article-title":"A fast post-training pruning framework for transformers","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Kwon","year":"2022"},{"key":"ref38","first-page":"21099","article-title":"Accelerated sparse neural training: A provable and efficient method to find N:M transposable masks","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Hubara","year":"2021"},{"key":"ref39","first-page":"10323","article-title":"SparseGPT: Massive language models can be accurately pruned in one-shot","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Frantar","year":"2023"},{"key":"ref40","first-page":"1","article-title":"A simple and effective pruning approach for large language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Sun","year":"2024"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0248"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.334"},{"issue":"140","key":"ref43","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref44","article-title":"LLaVA-next-interleave: Tackling multi-image, video, and 3D in large multimodal models","author":"Li","year":"2024"},{"key":"ref45","first-page":"288","article-title":"MegaBlocks: Efficient sparse training with mixture-of-experts","volume-title":"Proc. Mach. Learn. Syst.","author":"Gale","year":"2023"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_32"},{"key":"ref47","article-title":"Towards MoE deployment: Mitigating inefficiencies in mixture-of-expert (MoE) inference","author":"Huang","year":"2023"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.155"},{"key":"ref49","article-title":"Measuring mathematical problem solving with the math dataset","author":"Hendrycks","year":"2021"},{"key":"ref50","article-title":"Gurobi optimizer reference manual","year":"2024"},{"key":"ref51","article-title":"Towards 1-bit machine learning models","author":"Badri","year":"2024"},{"key":"ref52","article-title":"A framework for few-shot language model evaluation","author":"Gao","year":"2013"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3685520"},{"key":"ref54","article-title":"ParetoQ: Scaling laws in extremely low-bit LLM quantization","author":"Liu","year":"2025"},{"key":"ref55","article-title":"Training verifiers to solve math word problems","author":"Cobbe","year":"2021"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.172"},{"key":"ref57","article-title":"Evaluating large language models trained on code","author":"Chen","year":"2021"},{"key":"ref58","first-page":"1","article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hu","year":"2022"},{"key":"ref59","first-page":"1","article-title":"G-LLaVA: Solving geometric problem with multi-modal large language model","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gao","year":"2024"},{"key":"ref60","article-title":"QeRL: Beyond efficiency\u2013quantization-enhanced reinforcement learning for LLMs","author":"Huang","year":"2025"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acllong.353"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11552636\/11397211.pdf?arnumber=11397211","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T19:52:04Z","timestamp":1780948324000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11397211\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":61,"journal-issue":{"issue":"7"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2026.3664873","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]}}}