{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T00:52:10Z","timestamp":1780447930774,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":60,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755433","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"1783-1792","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["MQuant: Unleashing the Inference Potential of Multimodal Large Language Models via Static Quantization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-1819-8513","authenticated-orcid":false,"given":"Jiangyong","family":"Yu","sequence":"first","affiliation":[{"name":"HOUMO AI, Nanjing, Jiangsu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3602-7566","authenticated-orcid":false,"given":"Sifan","family":"Zhou","sequence":"additional","affiliation":[{"name":"Southeast University, Nanjing, Jiangsu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0581-7015","authenticated-orcid":false,"given":"Dawei","family":"Yang","sequence":"additional","affiliation":[{"name":"HOUMO AI, Nanjing, Jiangsu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5194-4130","authenticated-orcid":false,"given":"Shuoyu","family":"Li","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xian, ShaanXi, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9907-6186","authenticated-orcid":false,"given":"Shuo","family":"Wang","sequence":"additional","affiliation":[{"name":"HOUMO AI, Nanjing, Jiangsu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4510-898X","authenticated-orcid":false,"given":"Xing","family":"Hu","sequence":"additional","affiliation":[{"name":"HOUMO AI, Nan Jing, Jiang Su, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0626-9434","authenticated-orcid":false,"given":"Chen","family":"Xu","sequence":"additional","affiliation":[{"name":"HOUMO AI, Nanjing, Jiangsu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5899-3865","authenticated-orcid":false,"given":"Zukang","family":"Xu","sequence":"additional","affiliation":[{"name":"HOUMO AI, Nanjin, Jiangsu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2555-6417","authenticated-orcid":false,"given":"Changyong","family":"Shu","sequence":"additional","affiliation":[{"name":"HOUMO AI, Shanghai, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7846-0240","authenticated-orcid":false,"given":"Zhihang","family":"Yuan","sequence":"additional","affiliation":[{"name":"HOUMO AI, Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023a. GPT-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023b. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_3_1","first-page":"23716","article-title":"Flamingo: A visual language model for few-shot learning","volume":"35","author":"Alayrac Jean-Baptiste","year":"2022","unstructured":"Jean-Baptiste Alayrac, Jeff Donahue, Pauline Luc, Antoine Miech, Iain Barr, Yana Hasson, Karel Lenc, Arthur Mensch, Katherine Millican, Malcolm Reynolds, et al., 2022. Flamingo: A visual language model for few-shot learning. NeurIPS, Vol. 35 (2022), 23716-23736.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_4_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Ashkboos Saleh","year":"2024","unstructured":"Saleh Ashkboos, Maximilian L Croci, Marcelo Gennari do Nascimento, Torsten Hoefler, and James Hensman. 2024a. Slicegpt: Compress large language models by deleting rows and columns. International Conference on Learning Representations (ICLR) (2024)."},{"key":"e_1_3_2_1_5_1","volume-title":"Quarot: Outlier-free 4-bit inference in rotated llms. arXiv preprint arXiv:2404.00456","author":"Ashkboos Saleh","year":"2024","unstructured":"Saleh Ashkboos, Amirkeivan Mohtashami, Maximilian L Croci, Bo Li, Martin Jaggi, Dan Alistarh, Torsten Hoefler, and James Hensman. 2024b. Quarot: Outlier-free 4-bit inference in rotated llms. arXiv preprint arXiv:2404.00456 (2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"Qwen-vl: A frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, Shijie Wang, Sinan Tan, Peng Wang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-vl: A frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966 (2023)."},{"key":"e_1_3_2_1_7_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in Neural Information Processing Systems (2020)."},{"key":"e_1_3_2_1_8_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Chee Jerry","year":"2024","unstructured":"Jerry Chee, Yaohui Cai, Volodymyr Kuleshov, and Christopher M De Sa. 2024. Quip: 2-bit quantization of large language models with guarantees. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_9_1","volume-title":"Prefixquant: Static quantization beats dynamic through prefixed outliers in llms. arXiv preprint arXiv:2410.05265","author":"Chen Mengzhao","year":"2024","unstructured":"Mengzhao Chen, Yi Liu, Jiahao Wang, Yi Bin, Wenqi Shao, and Ping Luo. 2024a. Prefixquant: Static quantization beats dynamic through prefixed outliers in llms. arXiv preprint arXiv:2410.05265 (2024)."},{"key":"e_1_3_2_1_10_1","volume-title":"Forty-second International Conference on Machine Learning.","author":"Chen Zhixuan","unstructured":"Zhixuan Chen, Xing Hu, Dawei Yang, Zukang Xu, Zhihang Yuan, Sifan Zhou, et al., [n.d.]. MoEQuant: Enhancing Quantization for Mixture-of-Experts Large Language Models via Expert-Balanced Sampling and Affinity Guidance. In Forty-second International Conference on Machine Learning."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Zhe Chen Weiyun Wang Hao Tian Shenglong Ye Zhangwei Gao Erfei Cui Wenwen Tong Kongzhi Hu Jiapeng Luo Zheng Ma et al. 2024b. How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites. arXiv preprint arXiv:2404.16821 (2024).","DOI":"10.1007\/s11432-024-4231-5"},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al., 2024c. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198."},{"key":"e_1_3_2_1_13_1","unstructured":"Xiangxiang Chu Limeng Qiao Xinyu Zhang Shuang Xu Fei Wei Yang Yang Xiaofei Sun Yiming Hu Xinyang Lin Bo Zhang et al. 2024. MobileVLM V2: Faster and Stronger Baseline for Vision Language Model. arXiv preprint arXiv:2402.03766 (2024)."},{"key":"e_1_3_2_1_14_1","first-page":"16344","article-title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","volume":"35","author":"Dao Tri","year":"2022","unstructured":"Tri Dao, Dan Fu, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. 2022. Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in Neural Information Processing Systems, Vol. 35 (2022), 16344-16359.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3685520"},{"key":"e_1_3_2_1_16_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers. arXiv preprint arXiv:2210.17323","author":"Frantar Elias","year":"2022","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2022. GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers. arXiv preprint arXiv:2210.17323 (2022)."},{"key":"e_1_3_2_1_18_1","volume-title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models. arXiv preprint arXiv:2306.13394","author":"Fu Chaoyou","year":"2023","unstructured":"Chaoyou Fu, Peixian Chen, Yunhang Shen, Yulei Qin, Mengdan Zhang, Xu Lin, Jinrui Yang, Xiawu Zheng, Ke Li, Xing Sun, Yunsheng Wu, and Rongrong Ji. 2023. MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models. arXiv preprint arXiv:2306.13394 (2023)."},{"key":"e_1_3_2_1_19_1","unstructured":"Wenyi Hong Weihan Wang Ming Ding Wenmeng Yu Qingsong Lv Yan Wang Yean Cheng Shiyu Huang Junhui Ji Zhao Xue et al. 2024. CogVLM2: Visual Language Models for Image and Video Understanding. arXiv preprint arXiv:2408.16500 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"I-LLM: Efficient Integer-Only Inference for Fully-Quantized Low-Bit Large Language Models. arXiv preprint arXiv:2405.17849","author":"Hu Xing","year":"2024","unstructured":"Xing Hu, Yuan Chen, Dawei Yang, Sifan Zhou, Zhihang Yuan, Jiangyong Yu, and Chen Xu. 2024. I-LLM: Efficient Integer-Only Inference for Fully-Quantized Low-Bit Large Language Models. arXiv preprint arXiv:2405.17849 (2024)."},{"key":"e_1_3_2_1_21_1","volume-title":"OstQuant: Refining Large Language Model Quantization with Orthogonal and Scaling Transformations for Better Distribution Fitting. arXiv preprint arXiv:2501.13987","author":"Hu Xing","year":"2025","unstructured":"Xing Hu, Yuan Cheng, Dawei Yang, Zukang Xu, Zhihang Yuan, Jiangyong Yu, Chen Xu, Zhe Jiang, and Sifan Zhou. 2025. OstQuant: Refining Large Language Model Quantization with Orthogonal and Scaling Transformations for Better Distribution Fitting. arXiv preprint arXiv:2501.13987 (2025)."},{"key":"e_1_3_2_1_22_1","volume-title":"Barun Patra, et al.","author":"Huang Shaohan","year":"2024","unstructured":"Shaohan Huang, Li Dong, Wenhui Wang, Yaru Hao, Saksham Singhal, Shuming Ma, Tengchao Lv, Lei Cui, Owais Khan Mohammed, Barun Patra, et al., 2024. Language is not all you need: Aligning perception with language models. NeurIPS, Vol. 36 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"Ptq4ris: Post-training quantization for referring image segmentation. arXiv preprint arXiv:2409.17020","author":"Jiang Xiaoyan","year":"2024","unstructured":"Xiaoyan Jiang, Hang Yang, Kaiying Zhu, Xihe Qiu, Shibo Zhao, and Sifan Zhou. 2024. Ptq4ris: Post-training quantization for referring image segmentation. arXiv preprint arXiv:2409.17020 (2024)."},{"key":"e_1_3_2_1_24_1","volume-title":"ICML (2023)","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023a. BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. ICML (2023), 19730-19742."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29815"},{"key":"e_1_3_2_1_26_1","volume-title":"Fptq: Fine-grained post-training quantization for large language models. arXiv preprint arXiv:2308.15987","author":"Li Qingyuan","year":"2023","unstructured":"Qingyuan Li, Yifan Zhang, Liang Li, Peng Yao, Bo Zhang, Xiangxiang Chu, Yerui Sun, Li Du, and Yuchen Xie. 2023b. Fptq: Fine-grained post-training quantization for large language models. arXiv preprint arXiv:2308.15987 (2023)."},{"key":"e_1_3_2_1_27_1","volume-title":"MBQ: Modality-Balanced Quantization for Large Vision-Language Models. arXiv:2412.19509 [cs.CV] https:\/\/arxiv.org\/abs\/2412.19509","author":"Li Shiyao","year":"2024","unstructured":"Shiyao Li, Yingchun Hu, Xuefei Ning, Xihui Liu, Ke Hong, Xiaotao Jia, Xiuhong Li, Yaqi Yan, Pei Ran, Guohao Dai, Shengen Yan, Huazhong Yang, and Yu Wang. 2024a. MBQ: Modality-Balanced Quantization for Large Vision-Language Models. arXiv:2412.19509 [cs.CV] https:\/\/arxiv.org\/abs\/2412.19509"},{"key":"e_1_3_2_1_28_1","volume-title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models. arXiv preprint arXiv:2403.18814","author":"Li Yanwei","year":"2024","unstructured":"Yanwei Li, Yuechen Zhang, Chengyao Wang, Zhisheng Zhong, Yixin Chen, Ruihang Chu, Shaoteng Liu, and Jiaya Jia. 2024c. Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models. arXiv preprint arXiv:2403.18814 (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration. arXiv preprint arXiv:2306.00978","author":"Lin Ji","year":"2023","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Xingyu Dang, and Song Han. 2023. AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration. arXiv preprint arXiv:2306.00978 (2023)."},{"key":"e_1_3_2_1_30_1","volume-title":"NeurIPS","volume":"36","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2024. Visual Instruction Tuning. NeurIPS, Vol. 36 (2024)."},{"key":"e_1_3_2_1_31_1","unstructured":"Yuliang Liu Zhang Li Hongliang Li Wenwen Yu Mingxin Huang Dezhi Peng Mingyu Liu Mingrui Chen Chunyuan Li Lianwen Jin et al. 2023. On the hidden mystery of OCR in large multimodal models. arXiv preprint arXiv:2305.07895 (2023)."},{"key":"e_1_3_2_1_32_1","first-page":"2200","article-title":"DocVQA: A dataset for VQA on document images","author":"Mathew Minesh","year":"2021","unstructured":"Minesh Mathew, Dimosthenis Karatzas, and CV Jawahar. 2021. DocVQA: A dataset for VQA on document images. In WACV. 2200-2209.","journal-title":"WACV."},{"key":"e_1_3_2_1_33_1","unstructured":"Machel Reid Nikolay Savinov Denis Teplyashin Dmitry Lepikhin Timothy Lillicrap Jean-baptiste Alayrac Radu Soricut Angeliki Lazaridou Orhan Firat Julian Schrittwieser et al. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv preprint arXiv:2403.05530 (2024)."},{"key":"e_1_3_2_1_34_1","volume-title":"Pb-llm: Partially binarized large language models. arXiv preprint arXiv:2310.00034","author":"Shang Yuzhang","year":"2023","unstructured":"Yuzhang Shang, Zhihang Yuan, Qiang Wu, and Zhen Dong. 2023. Pb-llm: Partially binarized large language models. arXiv preprint arXiv:2310.00034 (2023)."},{"key":"e_1_3_2_1_35_1","volume-title":"OmniQuant: Omnidirectionally Calibrated Quantization for Large Language Models. CoRR","author":"Shao Wenqi","year":"2023","unstructured":"Wenqi Shao, Mengzhao Chen, Zhaoyang Zhang, Peng Xu, Lirui Zhao, Zhiqian Li, Kaipeng Zhang, Peng Gao, Yu Qiao, and Ping Luo. 2023. OmniQuant: Omnidirectionally Calibrated Quantization for Large Language Models. CoRR, Vol. abs\/2308.13137 (2023)."},{"key":"e_1_3_2_1_36_1","volume-title":"Meet Shah, Yu Jiang, Xinlei Chen, Dhruv Batra, Devi Parikh, and Marcus Rohrbach.","author":"Singh Amanpreet","year":"2019","unstructured":"Amanpreet Singh, Vivek Natarajan, Meet Shah, Yu Jiang, Xinlei Chen, Dhruv Batra, Devi Parikh, and Marcus Rohrbach. 2019. Towards VQA models that can read. In CVPR. 8317-8326."},{"key":"e_1_3_2_1_37_1","volume-title":"Sourav Bhattacharya, Timothy Hospedales, Georgios Tzimiropoulos, and Brais Martinez.","author":"Tan Fuwen","year":"2024","unstructured":"Fuwen Tan, Royson Lee, \u0141ukasz Dudziak, Shell Xu Hu, Sourav Bhattacharya, Timothy Hospedales, Georgios Tzimiropoulos, and Brais Martinez. 2024a. Mobilequant: Mobile-friendly quantization for on-device language models. arXiv preprint arXiv:2408.13933 (2024)."},{"key":"e_1_3_2_1_38_1","volume-title":"MobileQuant: Mobile-friendly Quantization for On-device Language Models. In The 2024 Conference on Empirical Methods in Natural Language Processing. https:\/\/openreview.net\/forum?id=48ptWWA54E","author":"Tan Fuwen","year":"2024","unstructured":"Fuwen Tan, Royson Lee, Lukasz Dudziak, Shell Xu Hu, Sourav Bhattacharya, Timothy Hospedales, Georgios Tzimiropoulos, and Brais Martinezs. 2024b. MobileQuant: Mobile-friendly Quantization for On-device Language Models. In The 2024 Conference on Empirical Methods in Natural Language Processing. https:\/\/openreview.net\/forum?id=48ptWWA54E"},{"key":"e_1_3_2_1_39_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al., 2023a. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_40_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023b. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_41_1","volume-title":"Forty-first International Conference on Machine Learning","author":"Tseng Albert","year":"2024","unstructured":"Albert Tseng, Jerry Chee, Qingyao Sun, Volodymyr Kuleshov, and Christopher De Sa. 2024. Quip#: Even better LLM quantization with hadamard incoherence and lattice codebooks. Forty-first International Conference on Machine Learning (2024)."},{"key":"e_1_3_2_1_42_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems (2017)."},{"key":"e_1_3_2_1_43_1","volume-title":"Q-VLM: Post-training Quantization for Large Vision-Language Models. In The Thirty-eighth Annual Conference on Neural Information Processing Systems.","author":"Wang Changyuan","year":"2024","unstructured":"Changyuan Wang, Ziwei Wang, Xiuwei Xu, Yansong Tang, Jie Zhou, and Jiwen Lu. 2024b. Q-VLM: Post-training Quantization for Large Vision-Language Models. In The Thirty-eighth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_1_44_1","volume-title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191","author":"Wang Peng","year":"2024","unstructured":"Peng Wang, Shuai Bai, Sinan Tan, Shijie Wang, Zhihao Fan, Jinze Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Yang Fan, Kai Dang, Mengfei Du, Xuancheng Ren, Rui Men, Dayiheng Liu, Chang Zhou, Jingren Zhou, and Junyang Lin. 2024a. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_45_1","unstructured":"Weihan Wang Qingsong Lv Wenmeng Yu Wenyi Hong Ji Qi Yan Wang Junhui Ji Zhuoyi Yang Lei Zhao Xixuan Song et al. 2023. CogVLM: Visual expert for pretrained language models. arXiv preprint arXiv:2311.03079 (2023)."},{"key":"e_1_3_2_1_46_1","volume-title":"Outlier suppression: Pushing the limit of low-bit transformer language models. Advances in Neural Information Processing Systems","author":"Wei Xiuying","year":"2022","unstructured":"Xiuying Wei, Yunchen Zhang, Xiangguo Zhang, Ruihao Gong, Shanghang Zhang, Qi Zhang, Fengwei Yu, and Xianglong Liu. 2022. Outlier suppression: Pushing the limit of low-bit transformer language models. Advances in Neural Information Processing Systems (2022)."},{"key":"e_1_3_2_1_47_1","volume-title":"Smoothquant: Accurate and efficient post-training quantization for large language models. arXiv preprint arXiv:2211.10438","author":"Xiao Guangxuan","year":"2022","unstructured":"Guangxuan Xiao, Ji Lin, Mickael Seznec, Julien Demouth, and Song Han. 2022. Smoothquant: Accurate and efficient post-training quantization for large language models. arXiv preprint arXiv:2211.10438 (2022)."},{"key":"e_1_3_2_1_48_1","volume-title":"RWKVQuant: Quantizing the RWKV Family with Proxy Guided Hybrid of Scalar and Vector Quantization. In Forty-second International Conference on Machine Learning.","author":"Xu Chen","unstructured":"Chen Xu, Yuxuan Yue, Zukang Xu, Xing Hu, Zhixuan Chen, Sifan Zhou, Zhihang Yuan, Dawei Yang, et al., [n.,d.] b. RWKVQuant: Quantizing the RWKV Family with Proxy Guided Hybrid of Scalar and Vector Quantization. In Forty-second International Conference on Machine Learning."},{"key":"e_1_3_2_1_49_1","volume-title":"MambaQuant: Quantizing the Mamba Family with Variance Aligned Rotation Methods. In The Thirteenth International Conference on Learning Representations.","author":"Xu Zukang","unstructured":"Zukang Xu, Yuxuan Yue, Xing Hu, Dawei Yang, Zhihang Yuan, Zixu Jiang, Zhixuan Chen, Sifan Zhou, et al., [n.,d.] a. MambaQuant: Quantizing the Mamba Family with Variance Aligned Rotation Methods. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_50_1","volume-title":"Minicpm-v: A gpt-4v level mllm on your phone. arXiv preprint arXiv:2408.01800","author":"Yao Yuan","year":"2024","unstructured":"Yuan Yao, Tianyu Yu, Ao Zhang, Chongyi Wang, Junbo Cui, Hongji Zhu, Tianchi Cai, Haoyu Li, Weilin Zhao, Zhihui He, et al., 2024. Minicpm-v: A gpt-4v level mllm on your phone. arXiv preprint arXiv:2408.01800 (2024)."},{"key":"e_1_3_2_1_51_1","volume-title":"Q-PETR: Quant-aware Position Embedding Transformation for Multi-View 3D Object Detection. arXiv preprint arXiv:2502.15488","author":"Yu Jiangyong","year":"2025","unstructured":"Jiangyong Yu, Changyong Shu, Dawei Yang, Sifan Zhou, Zichen Yu, Xing Hu, and Yan Chen. 2025. Q-PETR: Quant-aware Position Embedding Transformation for Multi-View 3D Object Detection. arXiv preprint arXiv:2502.15488 (2025)."},{"key":"e_1_3_2_1_52_1","volume-title":"RPTQ: Reorder-based Post-training Quantization for Large Language Models. arXiv preprint arXiv:2304.01089","author":"Yuan Zhihang","year":"2023","unstructured":"Zhihang Yuan, Lin Niu, Jiawei Liu, Wenyu Liu, Xinggang Wang, Yuzhang Shang, Guangyu Sun, Qiang Wu, Jiaxiang Wu, and Bingzhe Wu. 2023a. RPTQ: Reorder-based Post-training Quantization for Large Language Models. arXiv preprint arXiv:2304.01089 (2023)."},{"key":"e_1_3_2_1_53_1","volume-title":"ASVD: Activation-aware Singular Value Decomposition for Compressing Large Language Models. arXiv preprint arXiv:2312.05821","author":"Yuan Zhihang","year":"2023","unstructured":"Zhihang Yuan, Yuzhang Shang, Yue Song, Qiang Wu, Yan Yan, and Guangyu Sun. 2023b. ASVD: Activation-aware Singular Value Decomposition for Compressing Large Language Models. arXiv preprint arXiv:2312.05821 (2023)."},{"key":"e_1_3_2_1_54_1","volume-title":"Yan Yan, et al.","author":"Yuan Zhihang","year":"2024","unstructured":"Zhihang Yuan, Yuzhang Shang, Yang Zhou, Zhen Dong, Chenhao Xue, Bingzhe Wu, Zhikai Li, Qingyi Gu, Yong Jae Lee, Yan Yan, et al., 2024. Llm inference unveiled: Survey and roofline model insights. arXiv preprint arXiv:2402.16363 (2024)."},{"key":"e_1_3_2_1_55_1","volume-title":"Wkvquant: Quantizing weight and key\/value cache for large language models gains more. arXiv preprint arXiv:2402.12065.","author":"Yue Yuxuan","year":"2024","unstructured":"Yuxuan Yue, Zhihang Yuan, Haojie Duanmu, Sifan Zhou, Jianlong Wu, and Liqiang Nie. 2024. Wkvquant: Quantizing weight and key\/value cache for large language models gains more. arXiv preprint arXiv:2402.12065."},{"key":"e_1_3_2_1_56_1","volume-title":"QQQ: Quality Quattuor-Bit Quantization for Large Language Models. arXiv preprint arXiv:2406.09904","author":"Zhang Ying","year":"2024","unstructured":"Ying Zhang, Peng Zhang, Mincong Huang, Jingyang Xiang, Yujie Wang, Chao Wang, Yineng Zhang, Lei Yu, Chuan Liu, and Wei Lin. 2024. QQQ: Quality Quattuor-Bit Quantization for Large Language Models. arXiv preprint arXiv:2406.09904 (2024)."},{"key":"e_1_3_2_1_57_1","unstructured":"Wayne Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong et al. 2023. A survey of large language models. arXiv preprint arXiv:2303.18223 (2023)."},{"key":"e_1_3_2_1_58_1","volume-title":"International Conference on Learning Representations","author":"Zhou Sifan","year":"2024","unstructured":"Sifan Zhou, Liang Li, Xinyu Zhang, Bo Zhang, Shipeng Bai, Miao Sun, Ziyu Zhao, Xiaobo Lu, and Xiangxiang Chu. 2024. LiDAR-PTQ: Post-Training Quantization for Point Cloud 3D Object Detection. International Conference on Learning Representations (2024)."},{"key":"e_1_3_2_1_59_1","first-page":"22971","volume-title":"GSQ-Tuning: Group-Shared Exponents Integer in Fully Quantized Training for LLMs On-Device Fine-tuning. In Findings of the Association for Computational Linguistics: ACL 2025","author":"Zhou Sifan","year":"2025","unstructured":"Sifan Zhou, Shuo Wang, Zhihang Yuan, Mingjia Shi, Yuzhang Shang, and Dawei Yang. 2025. GSQ-Tuning: Group-Shared Exponents Integer in Fully Quantized Training for LLMs On-Device Fine-tuning. In Findings of the Association for Computational Linguistics: ACL 2025. Association for Computational Linguistics, Vienna, Austria, 22971-22988. https:\/\/aclanthology.org\/2025.findings-acl.1178\/"},{"key":"e_1_3_2_1_60_1","volume-title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. arXiv preprint arXiv:2304.10592","author":"Zhu Deyao","year":"2023","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2023. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. arXiv preprint arXiv:2304.10592 (2023)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755433","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:13:37Z","timestamp":1765340017000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755433"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":60,"alternative-id":["10.1145\/3746027.3755433","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755433","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}