{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T17:01:53Z","timestamp":1780765313558,"version":"3.54.1"},"reference-count":205,"publisher":"China Science Publishing & Media Ltd.","issue":"1","content-domain":{"domain":["engine.scichina.com"],"crossmark-restriction":false},"short-container-title":["DI"],"published-print":{"date-parts":[[2026,3,1]]},"DOI":"10.3724\/2096-7004.di.2025.0077","type":"journal-article","created":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T05:49:18Z","timestamp":1758779358000},"page":"92-136","update-policy":"https:\/\/doi.org\/10.1360\/scp-crossmark-policy-page","source":"Crossref","is-referenced-by-count":1,"title":["Lightweighting Large Models: A Review of Model Compression Techniques"],"prefix":"10.3724","volume":"8","author":[{"given":"Haoran","family":"Guan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junhao","family":"Lv","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lei","family":"Qi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"2026","published-online":{"date-parts":[[2025,9,25]]},"reference":[{"key":"","unstructured":"Girshick R., Donahue J., Darrell T., and Malik J., \u201cRich feature hierarchies for accurate object detection and semantic segmentation,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 580\u2013587, 2014."},{"key":"","unstructured":"Cho K., van Merri\u00ebnboer B., Gulcehre C., Bahdanau D., Bougares F., Schwenk H., and Bengio Y., \u201cLearning phrase representations using RNN encoder-decoder for statistical machine translation,\u201d in Empirical Methods in Natural Language Processing (EMNLP), pp. 1724\u20131734, 2014."},{"key":"","unstructured":"He X., Liao L., Zhang H., Nie L., Hu X., and Chua T.-S., \u201cNeural collaborative filtering,\u201d in International Conference on World Wide Web (WWW), p. 173\u2013182, 2017."},{"key":"","unstructured":"Krizhevsky A., Sutskever I., and Hinton G. E., \u201cImagenet classification with deep convolutional neural networks,\u201d in Advances in neural information processing systems (NeurIPS), p. 1097\u20131105, 2012."},{"key":"","unstructured":"Costa-Juss\u00e0 M. R., Cross J., \u00c7elebi O., Elbayad M., Heafield K., Heffernan K., Kalbassi E., Lam J., Licht D., Maillard J., et al., \u201cNo language left behind: Scaling human-centered machine translation,\u201d arXiv preprint arXiv: 2207.04672, 2022."},{"key":"","unstructured":"Pratap V., Tjandra A., Shi B., Tomasello P., Babu A., Kundu S., Elkahky A., Ni Z., Vyas A., Fazel-Zarandi M., et al., \u201cScaling speech technology to 1,000+ languages,\u201d Journal of Machine Learning Research (JMLR), vol. 25, no. 97, pp. 1\u201352, 2024."},{"key":"","unstructured":"Barrault L., Chung Y.-A., Meglioli M. C., Dale D., Dong N., Duppenthaler M., Duquenne P.-A., Ellis B., Elsahar H., Haaheim J., et al., \u201cSeamless: Multilingual expressive and streaming speech translation,\u201d arXiv preprint arXiv: 2312.05187, 2023."},{"key":"","unstructured":"Goodfellow I. J., Pouget-Abadie J., Mirza M., Xu B., Warde-Farley D., Ozair S., Courville A., and Bengio Y., \u201cGenerative adversarial nets,\u201d Advances in neural information processing systems (NeurIPS), vol. 27, 2014."},{"key":"","unstructured":"Karras T., Laine S., and Aila T., \u201cA style-based generator architecture for generative adversarial networks,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4401\u20134410, 2019."},{"key":"","unstructured":"Ho J., Jain A., and Abbeel P., \u201cDenoising diffusion probabilistic models,\u201d Advances in neural information processing systems (NeurIPS), vol. 33, pp. 6840\u20136851, 2020."},{"key":"","unstructured":"He K., Zhang X., Ren S., and Sun J., \u201cDeep residual learning for image recognition,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778, 2016."},{"key":"","unstructured":"Dosovitskiy A., Beyer L., Kolesnikov A., Weissenborn D., Zhai X., Unterthiner T., Dehghani M., Minderer M., Heigold G., Gelly S., Uszkoreit J., and Houlsby N., \u201cAn image is worth 16x16 words: Transformers for image recognition at scale,\u201d in International Conference on Learning Representations (ICLR), 2021."},{"key":"","unstructured":"Iandola F. N., Han S., Moskewicz M. W., Ashraf K., Dally W. J., and Keutzer K., \u201cSqueezenet: Alexnet-level accuracy with 50x fewer parameters and\u00a1 0.5 mb model size,\u201d arXiv preprint arXiv: 1602.07360, 2016."},{"key":"","unstructured":"Redmon J., Divvala S., Girshick R., and Farhadi A., \u201cYou only look once: Unified, real-time object detection,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 779\u2013788, 2016."},{"key":"","unstructured":"Ren S., He K., Girshick R., and Sun J., \u201cFaster r-cnn: Towards real-time object detection with region proposal networks,\u201d Advances in neural information processing systems (NeurIPS), vol. 28, 2015."},{"key":"","unstructured":"Carion N., Massa F., Synnaeve G., Usunier N., Kirillov A., and Zagoruyko S., \u201cEnd-to-end object detection with transformers,\u201d in European conference on computer vision (ECCV), pp. 213\u2013229, 2020."},{"key":"","unstructured":"Hannun A., Case C., Casper J., Catanzaro B., Diamos G., Elsen E., Prenger R., Satheesh S., Sengupta S., Coates A., et al., \u201cDeep speech: Scaling up end-to-end speech recognition,\u201d arXiv preprint arXiv: 1412.5567, 2014."},{"key":"","unstructured":"Chan W., Jaitly N., Le Q. V., and Vinyals O., \u201cListen, attend and spell,\u201d arXiv preprint arXiv: 1508.01211, 2015."},{"key":"","unstructured":"Radford A., Kim J. W., Xu T., Brockman G., McLeavey C., and Sutskever I., \u201cRobust speech recognition via large-scale weak supervision,\u201d in International conference on machine learning (ICML), pp. 28492\u201328518, 2023."},{"key":"","unstructured":"Simonyan K. and Zisserman A., \u201cVery deep convolutional networks for large-scale image recognition,\u201d International Conference on Learning Representations (ICLR), 2015."},{"key":"","unstructured":"Huang G., Liu Z., Van Der Maaten L., and Weinberger K. Q., \u201cDensely connected convolutional networks,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2261\u20132269, 2017."},{"key":"","unstructured":"Devlin J., Chang M.-W., Lee K., and Toutanova K., \u201cBERT: Pre-training of deep bidirectional transformers for language understanding,\u201d in North American Chapter of the Association for Computational Linguistics (NAACL), vol. 1, pp. 4171\u20134186, 2019."},{"key":"","unstructured":"Brown T., Mann B., Ryder N., Subbiah M., Kaplan J. D., Dhariwal P., Neelakantan A., Shyam P., Sastry G., Askell A., et al., \u201cLanguage models are few-shot learners,\u201d Advances in neural information processing systems (NeurIPS), vol. 33, pp. 1877\u20131901, 2020."},{"key":"","unstructured":"Achiam J., Adler S., Agarwal S., Ahmad L., Akkaya I., Aleman F. L., Almeida D., Altenschmidt J., Altman S., Anadkat S., et al., \u201cGpt-4 technical report,\u201d arXiv preprint arXiv: 2303.08774, 2023."},{"key":"","unstructured":"Liu J., Yang M., Yu Y., Xu H., Li K., and Zhou X., \u201cLarge language models in bioinformatics: applications and perspectives,\u201d arXiv preprint arXiv: 2401.04155, 2024."},{"key":"","unstructured":"Chen H., \u201cLarge knowledge model: Perspectives and challenges,\u201d Data Intelligence, vol. 6, no. 3, pp. 587\u2013620, 2024."},{"key":"","unstructured":"Azaria A., Azoulay R., and Reches S., \u201cChatgpt is a remarkable tool\u2014for experts,\u201d Data Intelligence, vol. 6, no. 1, pp. 240\u2013296, 2024."},{"key":"","unstructured":"Liu Z., Lin Y., Cao Y., Hu H., Wei Y., Zhang Z., Lin S., and Guo B., \u201cSwin transformer: Hierarchical vision transformer using shifted windows,\u201d in IEEE International Conference on Computer Vision (ICCV), pp. 9992\u201310002, 2021."},{"key":"","unstructured":"Radford A., Kim J. W., Hallacy C., Ramesh A., Goh G., Agarwal S., Sastry G., Askell A., Mishkin P., Clark J., Krueger G., and Sutskever I., \u201cLearning transferable visual models from natural language supervision,\u201d in International Conference on Machine Learning (ICML), vol. 139, pp. 8748\u20138763, 2021."},{"key":"","unstructured":"Ramesh A., Pavlov M., Goh G., Gray S., Voss C., Radford A., Chen M., and Sutskever I., \u201cZero-shot text-to-image generation,\u201d in International Conference on Machine Learning (ICML), vol. 139, pp. 8821\u20138831, 2021."},{"key":"","unstructured":"Liu H., Li C., Wu Q., and Lee Y. J., \u201cVisual instruction tuning,\u201d in Advances in neural information processing systems (NeurIPS), 2023."},{"key":"","unstructured":"Xu M., Yin W., Cai D., Yi R., Xu D., Wang Q., Wu B., Zhao Y., Yang C., Wang S., et al., \u201cA survey of resourceefficient llm and multimodal foundation models,\u201d arXiv preprint arXiv: 2401.08092, 2024."},{"key":"","unstructured":"Menghani G., \u201cEfficient deep learning: A survey on making deep learning models smaller, faster, and better,\u201d ACM Computing Surveys, vol. 55, no. 12, 2023."},{"key":"","unstructured":"Lyu L., \u201cA pathway towards responsible ai generated content,\u201d in International Joint Conference on Artificial Intelligence (IJCAI), 2023."},{"key":"","unstructured":"Shen L., Sun Y., Yu Z., Ding L., Tian X., and Tao D., \u201cOn efficient training of large-scale deep learning models,\u201d ACM Computing Surveys, vol. 57, no. 3, 2024."},{"key":"","unstructured":"de La Torre J., \u201cTransformadores: Fundamentos te\u00b4oricos y aplicaciones,\u201d arXiv preprint arXiv: 2302.09327, 2023."},{"key":"","unstructured":"Li Y., Wen H., Wang W., Li X., Yuan Y., Liu G., Liu J., Xu W., Wang X., Sun Y., et al., \u201cPersonal llm agents: Insights and survey about the capability, efficiency and security,\u201d arXiv preprint arXiv: 2401.05459, 2024."},{"key":"","unstructured":"Li H., Zhang H., Qi X., Ruigang Y., and Huang G., \u201cImproved techniques for training adaptive deep networks,\u201d in IEEE International Conference on Computer Vision (ICCV), pp. 1891\u20131900, 2019."},{"key":"","unstructured":"Chitty-Venkata K. T., Mittal S., Emani M., Vishwanath V., and Somani A. K., \u201cA survey of techniques for optimizing transformer inference,\u201d Journal of Systems Architecture (JSA), vol. 144, no. C, 2023."},{"key":"","unstructured":"Movva R., Lei J., Longpre S., Gupta A., and DuBois C., \u201cCombining compressions for multiplicative size scaling on natural language tasks,\u201d in International Conference on Computational Linguistics (COLING), pp. 2861\u20132872, 2022."},{"key":"","unstructured":"Lagunas F., Charlaix E., Sanh V., and Rush A., \u201cBlock pruning for faster transformers,\u201d in Empirical Methods in Natural Language Processing (EMNLP), pp. 10619\u201310629, 2021."},{"key":"","unstructured":"Chen Y., Yan Y., Yang Q., Shu Y., He S., Shi Z., and Chen J., \u201cAccept: An acceleration scheme for speeding up edge pipeline-parallel training,\u201d IEEE Transactions on Mobile Computing (TMC), vol. 23, no. 12, pp. 10938\u201310951, 2024."},{"key":"","unstructured":"Dai W., Fan J., Miao Y., and Hwang K., \u201cDeep learning model compression with rank reduction in tensor decomposition,\u201d IEEE Transactions on Neural Networks and Learning Systems (TNNLS), vol. 36, no. 1, pp. 1315\u20131328, 2025."},{"key":"","unstructured":"Ma X., Fang G., and Wang X., \u201cLlm-pruner: On the structural pruning of large language models,\u201d in Advances in Neural Information Processing Systems (NeurIPS), vol. 36, pp. 21702\u201321720, 2023."},{"key":"","unstructured":"Hashemizadeh M., Ramirez J., Sukumaran R., Farnadi G., Lacoste-Julien S., and Gallego-Posada J., \u201cBalancing act: Constraining disparate impact in sparse models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Du M., Mukherjee S., Cheng Y., Shokouhi M., Hu X., and Awadallah A. H., \u201cRobustness challenges in model distillation and pruning for natural language understanding,\u201d in European Chapter of the Association for Computational Linguistics (EACL), pp. 1766\u20131778, 2023."},{"key":"","unstructured":"Liu H.-I., Galindo M., Xie H., Wong L.-K., Shuai H.-H., Li Y.-H., and Cheng W.-H., \u201cLightweight deep learning for resource-constrained environments: A survey,\u201d ACM Computing Surveys, vol. 56, no. 10, 2024."},{"key":"","unstructured":"Deng L., Li G., Han S., Shi L., and Xie Y., \u201cModel compression and hardware acceleration for neural networks: A comprehensive survey,\u201d Proceedings of the IEEE, vol. 108, no. 4, pp. 485\u2013532, 2020."},{"key":"","unstructured":"Yu S., Nguyen P., Anwar A., and Jannesari A., \u201cAdaptive dynamic pruning for non-iid federated learning,\u201d arXiv preprint arXiv: 2106.06921, vol. 2, 2021."},{"key":"","unstructured":"Wang Y., Zhang X., Hu X., Zhang B., and Su H., \u201cDynamic network pruning with interpretable layerwise channel selection,\u201d in AAAI Conference on Artificial Intelligence (AAAI), vol. 34, pp. 6299\u20136306, 2020."},{"key":"","unstructured":"Desai A. and Shrivastava A., \u201cIn defense of parameter sharing for model-compression,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Wang J., Chen Y.-G., Lin I.-C., Li B., and Zhang G. L., \u201cBasis sharing: Cross-layer parameter sharing for large language model compression,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Bae S., Fisch A., Harutyunyan H., Ji Z., Kim S., and Schuster T., \u201cRelaxed recursive transformers: Effective parameter sharing with layer-wise loRA,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Li J., Nie Q., Fu W., Lin Y., Tao G., Liu Y., and Wang C., \u201cLors: Low-rank residual structure for parameter-efficient network stacking,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 15866\u201315876, 2024."},{"key":"","unstructured":"Wallingford M., Li H., Achille A., Ravichandran A., Fowlkes C., Bhotika R., and Soatto S., \u201cTask adaptive parameter sharing for multi-task learning,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2022."},{"key":"","unstructured":"Park K. and Cho N. I., \u201cPartial filter-sharing: Improved parameter-sharing method for single image superresolution networks,\u201d in IEEE Winter Conference on Applications of Computer Vision (WACV), 2025."},{"key":"","unstructured":"Zhang S. and Papyan V., \u201cOATS: Outlier-aware pruning through sparse and low rank decomposition,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Bo\u017ea V. and Macko V., \u201cTwo sparse matrices are better than one: Sparsifying neural networks with double sparse factorization,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Qinsi W., Ke J., Tomizuka M., Keutzer K., and Xu C., \u201cDobi-SVD: Differentiable SVD for LLM compression and some new perspectives,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Liu X.-H., Du Y., Wang J., and Yu Y., \u201cOn the optimization landscape of low rank adaptation methods for large language models,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Zheng H., Lin X., Liang H., Zhou B., and Liang Y., \u201cRegion-aware mutual relational knowledge distillation for semantic segmentation,\u201d Pattern Recognition (PR), vol. 161, p. 111319, 2025."},{"key":"","unstructured":"Ni Z.-L., Yang F., Wen S., and Zhang G., \u201cDual relation knowledge distillation for object detection,\u201d in International Joint Conference on Artificial Intelligence (IJCAI), 2023."},{"key":"","unstructured":"Xu W., Wu Q., Liang Z., Han J., Ning X., Shi Y., Lin W., and Zhang Y., \u201cSLMRec: Distilling large language models into small for sequential recommendation,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Zhang Y., Ma X., Bai Y., Wang H., and Fu Y., \u201cAccessing vision foundation models via imagenet-1k,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Hao Z., Guo J., Han K., Tang Y., Hu H., Wang Y., and Xu C., \u201cOne-for-all: Bridge the gap between heterogeneous architectures in knowledge distillation,\u201d in Advances in Neural Information Processing Systems (NeurIPS), 2023."},{"key":"","unstructured":"Dasgupta S. and Cohn T., \u201cImproving language model distillation through hidden state matching,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Shu F., Liao Y., Zhang L., Zhuo L., Xu C., Zhang G., Shi H., Chan L., Zhong T., Yu Z., He W., Fu S., Li H., Liu S., Li H., and Jiang H., \u201cLLaVA-mod: Making LLaVA tiny via moe-knowledge distillation,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Xu W., Han R., Wang Z., Le L., Madeka D., Li L., Wang W. Y., Agarwal R., Lee C.-Y., and Pfister T., \u201cSpeculative knowledge distillation: Bridging the teacher-student gap through interleaved sampling,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Sun S., Ren W., Li J., Wang R., and Cao X., \u201cLogit standardization in knowledge distillation,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 15731\u201315740, 2024."},{"key":"","unstructured":"Wei Y., Hu Z., Shen L., Wang Z., Yuan C., and Tao D., \u201cOpen-vocabulary customization from CLIP via datafree knowledge distillationdasgupta2025improving,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Dettmers T., Svirschevski R. A., Egiazarian V., Kuznedelev D., Frantar E., Ashkboos S., Borzunov A., Hoefler T., and Alistarh D., \u201cSpQR: A sparse-quantized representation for near-lossless LLM weight compression,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Liu J., Gong R., Wei X., Dong Z., Cai J., and Zhuang B., \u201cQLLM: Accurate and efficient low-bitwidth quantization for large language models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Heo J. H., Kim J., Kwon B., Kim B., Kwon S. J., and Lee D., \u201cRethinking channel dimensions to isolate outliers for low-bit weight quantization of large language models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Ding X., Liu X., Tu Z., Zhang Y., Li W., Hu J., Chen H., Tang Y., Xiong Z., Yin B., and Wang Y., \u201cCBQ: Crossblock quantization for large language models,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Li M., Lin Y., Zhang Z., Cai T., Guo J., Li X., Xie E., Meng C., Zhu J.-Y., and Han S., \u201cSVDQuant: Absorbing outliers by low-rank component for 4-bit diffusion models,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Yuan Z., Shang Y., and Dong Z., \u201cPB-LLM: Partially binarized large language models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Fishman M., Chmiel B., Banner R., and Soudry D., \u201cScaling FP8 training to trillion-token LLMs,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"He Y., Liu J., Wu W., Zhou H., and Zhuang B., \u201cEfficientDM: Efficient quantization-aware fine-tuning of low-bit diffusion models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Liu Z., Oguz B., Zhao C., Chang E., Stock P., Mehdad Y., Shi Y., Krishnamoorthi R., and Chandra V., \u201cLLMQAT: Data-free quantization aware training for large language models,\u201d in Annual Meeting of the Association for Computational Linguistics (ACL), pp. 467\u2013484, 2024."},{"key":"","unstructured":"Le Q., Diao E., Wang Z., Wang X., Ding J., Yang L., and Anwar A., \u201cProbe pruning: Accelerating LLMs through dynamic pruning via model-probing,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Someki M., Peng Y., Arora S., M\u00fcller M., Mouchtaris A., Strimel G., Liu J., and Watanabe S., \u201cContext-aware dynamic pruning for speech foundation models,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Sung Y.-L., Yoon J., and Bansal M., \u201cECoFLap: Efficient coarse-to-fine layer-wise pruning for vision-language models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Sun M., Fang Z., Wang J., Jiang J., Kong D., Hu C., FANG Y., and Xu R., \u201cOptimal brain apoptosis,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Lucas R. and Mazumder R., \u201cPreserving deep representations in one-shot pruning: A hessian-free second-order optimization framework,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Wang D., \u0160iki\u0107 H., Thiele L., and Saukh O., \u201cForget the data and fine-tuning! just fold the network to compress,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Han R. and Tang J., \u201cStraightforward layer-wise pruning for more efficient visual adaptation,\u201d in European Conference on Computer Vision (ECCV), p. 236\u2013252, 2024."},{"key":"","unstructured":"Dong P., Li L., Tang Z., Liu X., Pan X., Wang Q., and Chu X., \u201cPruner-zero: Evolving symbolic pruning metric from scratch for large language models,\u201d in International Conference on Machine Learning (ICML), 2024."},{"key":"","unstructured":"Bair A., Yin H., Shen M., Molchanov P., and \u00c1lvarez J. M., \u201cAdaptive sharpness-aware pruning for robust sparse networks,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"van der Ouderaa T. F. A., Nagel M., Baalen M. V., and Blankevoort T., \u201cThe LLM surgeon,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Zhang Y., Zhao L., Lin M., Sun Y., Yao Y., Han X., Tanner J., Liu S., and Ji R., \u201cDynamic sparse no training: Training-free fine-tuning for sparse llms,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Li G., Yin L., Ji J., Niu W., Qin M., Ren B., Guo L., Liu S., and Ma X., \u201cNeurrev: Train better sparse neural network practically via neuron revitalization,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Cheng H., Zhang M., and Shi J. Q., \u201cA survey on deep neural network pruning: Taxonomy, comparison, analysis, and recommendations,\u201d IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), vol. 46, no. 12, pp. 10558\u201310578, 2024."},{"key":"","unstructured":"Mo Z., Shi H., and Pan S. J., \u201cProbabilistic neural pruning via sparsity evolutionary fokker-planck-kolmogorov equation,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Harma S. B., Chakraborty A., Kostenok E., Mishin D., Ha D., Falsafi B., Jaggi M., Liu M., Oh Y., Subramanian S., and Yazdanbakhsh A., \u201cEffective interplay between sparsity and quantization: From theory to practice,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Guo Y., Zhang S., Pan H., Liu J., Zhang Y., and Chen J., \u201cGap preserving distillation by building bidirectional mappings with a dynamic teacher,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Hossain M. I., Akhter S., Hong C. S., and Huh E.-N., \u201cSingle teacher, multiple perspectives: Teacher knowledge augmentation for enhanced knowledge distillation,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Jeong H. and Chung H. W., \u201cRethinking self-distillation: Label averaging and enhanced soft label refinement with partial labels,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Kim M., Kim J., and Kang U., \u201cSynq: Accurate zero-shot quantization by synthesis-aware fine-tuning,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Ding Y., Fan K., Wang Y., Sun X., and Fu Y., \u201cAdaptive pruning of pretrained transformer via differential inclusions,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Guo H., Greengard P., Xing E., and Kim Y., \u201cLQ-loRA: Low-rank plus quantized matrix decomposition for efficient language model finetuning,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Chen X., Hu Y., Zhang J., Wang Y., Li C., and Chen H., \u201cStreamlining redundant layers to compress large language models,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Sengupta A., Chaudhary S., and Chakraborty T., \u201cYou only prune once: Designing calibration-free model compression with policy learning,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Liu Z., Zhao C., Fedorov I., Soran B., Choudhary D., Krishnamoorthi R., Chandra V., Tian Y., and Blankevoort T., \u201cSpinquant: LLM quantization with learned rotations,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Ashkboos S., Croci M. L., do Nascimento M. G., Hoefler T., and Hensman J., \u201cSliceGPT: Compress large language models by deleting rows and columns,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Kaushal A., Vaidhya T., Mondal A. K., Pandey T., Bhagat A., and Rish I., \u201cSurprising effectiveness of pretraining ternary language model at scale,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Zhang T. and Shrivastava A., \u201cLeanquant: Accurate and scalable large language model quantization with losserror-aware grid,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Gromov A., Tirumala K., Shapourian H., Glorioso P., and Roberts D., \u201cThe unreasonable ineffectiveness of the deeper layers,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Li Z., Yan X., Zhang T., Qin H., Xie D., Tian J., Shi Z. C., Kong L., Zhang Y., and Yang X., \u201cARB-LLM: Alternating refined binarizations for large language models,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Dong P., Li L., Zhong Y., Du D., FAN R., Chen Y., Tang Z., Wang Q., Xue W., Guo Y., and Chu X., \u201cSTBLLM: Breaking the 1-bit barrier with structured binary LLMs,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Hu X., Cheng Y., Yang D., Chen Z., Xu Z., Yu J.Y., Chen X., Yuan Z., Jiang Z., and Zhou S., \u201cOSTQuant: Refining large language model quantization with orthogonal and scaling transformations for better distribution fitting,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Xu Y., Jie Z., Dong H., Wang L., Lu X., Zhou A., Saha A., Xiong C., and Sahoo D., \u201cThink: Thinner key cache by query-driven pruning,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Yi K., Liu Z., Zhang J.W., Li C., Zhang T., Lin J., and Zhou J., \u201cRotated runtime smooth: Training-free activation smoother for accurate INT4 inference,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Shi D., Tao C., Jin Y., Yang Z., Yuan C., and Wang J., \u201cUpop: unified and progressive pruning for compressing vision-language transformers,\u201d in International Conference on Machine Learning (ICML), 2023."},{"key":"","unstructured":"Bai G., Li Y., Ling C., Kim K., and Zhao L., \u201cSparseLLM: Towards global pruning of pre-trained language models,\u201d in Advances in Neural Information Processing Systems (NeurIPS), 2024."},{"key":"","unstructured":"Xia M., Gao T., Zeng Z., and Chen D., \u201cSheared LLaMA: Accelerating language model pre-training via structured pruning,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Park S., Choi H., and Kang U., \u201cAccurate retraining-free pruning for pretrained encoder-based language models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Liang Y., Long J., Shi Z., Song Z., and Zhou Y., \u201cBeyond linear approximations: A novel pruning approach for attention matrix,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Zimmer M., Spiegel C., and Pokutta S., \u201cSparse model soups: A recipe for improved pruning via model averaging,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Choi M., Lee H., Nam G., and Lee J., \u201cSparse weight averaging with multiple particles for iterative magnitude pruning,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Atashgahi Z., Zhang X., Kichler N., Liu S., Yin L., Pechenizkiy M., Veldhuis R., and Mocanu D. C., \u201cSupervised feature selection with neuron evolution in sparse neural networks,\u201d Transactions on Machine Learning Research (TMLR), 2023."},{"key":"","unstructured":"Sokar G., Mocanu E., Mocanu D. C., Pechenizkiy M., and Stone P., \u201cDynamic sparse training for deep reinforcement learning,\u201d in International Joint Conference on Artificial Intelligence (IJCAI), pp. 3437\u20133443, 2022."},{"key":"","unstructured":"Chen T., Cheng Y., Gan Z., Yuan L., Zhang L., and Wang Z., \u201cChasing sparsity in vision transformers: An end-toend exploration,\u201d in Advances in Neural Information Processing Systems (NeurIPS), vol. 34, pp. 19974\u201319988, 2021."},{"key":"","unstructured":"Shen J., Xu Q., Pan G., and Chen B., \u201cImproving the sparse structure learning of spiking neural networks from the view of compression efficiency,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Merity S., Xiong C., Bradbury J., and Socher R., \u201cPointer sentinel mixture models,\u201d in International Conference on Learning Representations (ICLR), 2017."},{"key":"","unstructured":"Yin L., Wu Y., Zhang Z., Hsieh C.-Y., Wang Y., Jia Y., Li G., JAISWAL A. K., Pechenizkiy M., Liang Y., Bendersky M., Wang Z., and Liu S., \u201cOutlier weighed layerwise sparsity (OWL): A missing secret sauce for pruning LLMs to high sparsity,\u201d in International Conference on Machine Learning (ICML), 2024."},{"key":"","unstructured":"Li J., Gao W., Lei Q., and Xu D., \u201cBreaking through deterministic barriers: Randomized pruning mask generation and selection,\u201d in Empirical Methods in Natural Language Processing (EMNLP), 2023."},{"key":"","unstructured":"Yu R., Li A., Chen C.-F., Lai J.-H., Morariu V. I., Han X., Gao M., Lin C.-Y., and Davis L. S., \u201cNisp: Pruning networks using neuron importance score propagation,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 9194\u20139203, 2018."},{"key":"","unstructured":"Yao Z., Ma L., Shen S., Keutzer K., and Mahoney M. W., \u201cMlpruning: A multilevel structured pruning framework for transformer-based models,\u201d arXiv preprint arXiv: 2105.14636, 2021."},{"key":"","unstructured":"Sun M., Liu Z., Bair A., and Kolter J. Z., \u201cA simple and effective pruning approach for large language models,\u201d in International Conference on Machine Learning (ICML), 2023."},{"key":"","unstructured":"Song Z., Wang R., Ru D., Peng Z., Huang H., Zhao H., Liang X., and Jiang L., \u201cApproximate random dropout for dnn training acceleration in gpgpu,\u201d in Design, Automation &amp; Test in Europe (DATE), pp. 108\u2013113, 2019."},{"key":"","unstructured":"Sharma A., Wolfe N., and Raj B., \u201cThe incredible shrinking neural network: New perspectives on learning representations through the lens of pruning,\u201d arXiv preprint arXiv: 1701.04465, 2017."},{"key":"","unstructured":"Ma X., Lin S., Ye S., He Z., Zhang L., Yuan G., Tan S. H., Li Z., Fan D., Qian X., Lin X., Ma K., and Wang Y., \u201cNon-structured dnn weight pruning\u2014is it beneficial in any platform?,\u201d IEEE Transactions on Neural Networks and Learning Systems (TNNLS), vol. 33, no. 9, pp. 4930\u20134944, 2022."},{"key":"","unstructured":"Wen W., Wu C., Wang Y., Chen Y., and Li H., \u201cLearning structured sparsity in deep neural networks,\u201d in Advances in Neural Information Processing Systems (NeurIPS), p. 2082\u20132090, 2016."},{"key":"","unstructured":"Zhang H., Zhou Y., and Wang G.-H., \u201cDense vision transformer compression with few samples,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 15825\u201315834, 2024."},{"key":"","unstructured":"Cheng H., Zhang M., and Shi J. Q., \u201cA survey on deep neural network pruning: Taxonomy, comparison, analysis, and recommendations,\u201d IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), vol. 46, no. 12, pp. 10558\u201310578, 2024."},{"key":"","unstructured":"Hua W., Zhou Y., De Sa C., Zhang Z., and Suh G. E., \u201cBoosting the performance of cnn accelerators with dynamic fine-grained channel gating,\u201d in International Symposium on Microarchitecture (MICRO), p. 139\u2013150, 2019."},{"key":"","unstructured":"Song J., Jo D., Kim Y., and Kim J.-J., \u201cReasoning path compression: Compressing generation trajectories for efficient llm reasoning,\u201d arXiv preprint arXiv: 2505.13866, 2025."},{"key":"","unstructured":"Zhang H., Su R., Yuan Z., Chen P., Shen M., Fan Y., Yan S., Dai G., and Wang Y., \u201cDitfastattnv2: Head-wise attention compression for multi-modality diffusion transformers,\u201d arXiv preprint arXiv: 2503.22796, 2025."},{"key":"","unstructured":"Wang H., Nie Y., Ye Y., Guan Y., Yan W., Li S., Yu H., Lu J., and Huang C., \u201cDynamic-vlm: Simple dynamic visual token compression for videollm,\u201d arXiv preprint arXiv: 2412.09530, 2024."},{"key":"","unstructured":"Shafipour R., Harrison D., Horton M., MARKER J., Bedayat H., Mehta S., Rastegari M., Najibi M., and Naderiparizi S., \u201cSeedLM: Compressing LLM weights into seeds of pseudo-random generators,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Zhou S., Li L., Zhang X., Zhang B., Bai S., Sun M., Zhao Z., Lu X., and Chu X., \u201cLiDAR-PTQ: Post-training quantization for point cloud 3d object detection,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Liu J., Niu L., Yuan Z., Yang D., Wang X., and Liu W., \u201cPd-quant: Post-training quantization based on prediction difference metric,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 24427\u201324437, 2023."},{"key":"","unstructured":"YVINEC E., Dapogny A., and Bailly K., \u201cNetwork memory footprint compression through jointly learnable codebooks and mappings,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Ma Y., Li H., Zheng X., Ling F., Xiao X., Wang R., Wen S., Chao F., and Ji R., \u201cAffinequant: Affine transformation quantization for large language models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Park G., park B., Kim M., Lee S., Kim J., Kwon B., Kwon S. J., Kim B., Lee Y., and Lee D., \u201cLUT-GEMM: Quantized matrix multiplication based on LUTs for efficient inference in large-scale generative language models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Zhang T., Hariri M., Zhong S., Chaudhary V., Sui Y., Hu X., and Shrivastava A., Sketch to adapt: fine-tunable sketches for efficient LLM adaptation. in Proceedings of the 42nd International Conference on Machine Learning (ICML'25), vol. 267. article 3052, pp. 76088\u201376110, 2026."},{"key":"","unstructured":"Gou J., Yu B., Maybank S. J., et al., \u201cKnowledge distillation: A survey,\u201d International Journal of Computer Vision (IJCV), vol. 129, pp. 1789\u20131819, 2021."},{"key":"","unstructured":"Jin Y., Wang J., and Lin D., \u201cMulti-level logit distillation,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 24276\u201324285, 2023."},{"key":"","unstructured":"Waheed A., Mitra C., Wang L. Z., Ramanan D., and Raj B., \u201cLess is more tokens: Efficient math reasoning via difficulty-aware chain-of-thought distillation,\u201d arXiv preprint arXiv: 2509.05226, 2025."},{"key":"","unstructured":"Bai Y., Wang Z., Xiao J., Wei C., Wang H., Yuille A., Zhou Y., and Xie C., \u201cMasked autoencoders enable efficient knowledge distillers,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 24256\u201324265, 2023."},{"key":"","unstructured":"Kim J., You J., Lee D., Kim H. Y., and Jung J.-H., \u201cDo topological characteristics help in knowledge distillation?,\u201d in International Conference on Machine Learning (ICML), 2024."},{"key":"","unstructured":"Park W., Kim D., Lu Y., and Cho M., \u201cRelational knowledge distillation,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3962\u20133971, 2019."},{"key":"","unstructured":"Yang C., Zhou H., An Z., Jiang X., Xu Y., and Zhang Q., \u201cCross-image relational knowledge distillation for semantic segmentation,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12309\u201312318, 2022."},{"key":"","unstructured":"Liu Y., Cao J., Li B., Yuan C., Hu W., Li Y., and Duan Y., \u201cKnowledge distillation via instance relationship graph,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7089\u20137097, 2019."},{"key":"","unstructured":"Lin S., Ji R., Chen C., Tao D., and Luo J., \u201cHolistic cnn compression via low-rank decomposition with knowledge transfer,\u201d IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), vol. 41, no. 12, pp. 2889\u20132905, 2019."},{"key":"","unstructured":"Hu E. J., Shen Y.L., Wallis P., Allen-Zhu Z., Li Y., Wang S., Wang L., and Chen W., \u201cLoRA: Low-rank adaptation of large language models,\u201d in International Conference on Learning Representations (ICLR), 2022."},{"key":"","unstructured":"Koohpayegani S. A., Navaneet N. K., Nooralinejad P., Kolouri S., and Pirsiavash H., \u201cNOLA: Compressing loRA using linear combination of random basis,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Lialin V., Muckatira S., Shivagunde N., and Rumshisky A., \u201cReloRA: High-rank training through low-rank updates,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Wang S., Chen L., CHEN P., Dong J., XUE B., Jiang J., Kong L., and Wu C., \u201cMos: Unleashing parameter efficiency of low-rank adaptation with mixture of shards,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Xu Y., Xie L., Gu X., Chen X., Chang H., Zhang H., Chen Z., ZHANG X., and Tian Q., \u201cQA-loRA: Quantizationaware low-rank adaptation of large language models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Li Y., Yu Y., Liang C., Karampatziakis N., He P., Chen W., and Zhao T., \u201cLoftq: LoRA-fine-tuning-aware quantization for large language models,\u201d in International Conference on Learning Representations (ICLR), 2024."},{"key":"","unstructured":"Mall S. and Henriques J. F., \u201cCram: Large-scale video continual learning with bootstrapped compression,\u201d arXiv preprint arXiv: 2508.05001, 2025."},{"key":"","unstructured":"Wang X., Ye P., Huang C., Zheng S., Zhang B., Bai L., Ouyang W., and Chen T., \u201cBreaking the compression ceiling: Data-free pipeline for ultra-efficient delta compression,\u201d arXiv preprint arXiv: 2505.13563, 2025."},{"key":"","unstructured":"Tang H., Lin Y., Lin J., Han Q., Ke D., Hong S., Yao Y., and Wang G., \u201cRazorattention: Efficient KV cache compression through retrieval heads,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Fu Y., Cai Z., Asi A., Xiong W., Dong Y., and Xiao W., \u201cNot all heads matter: A head-level KV cache compression method with integrated retrieval and reasoning,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Chang C.-C., Lin W.-C., Lin C.-Y., Chen C.-Y., Hu Y.-F., Wang P.-S., Huang N.-C., Ceze L., Abdelfattah M. S., and Wu K.-C., \u201cPalu: KV-cache compression with low-rank projection,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Lin B., Zeng Z., Xiao Z., Kou S., Hou T., Gao X., Zhang H., and Deng Z., \u201cMatryoshkaKV: Adaptive KV compression via trainable orthogonal projection,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Wang Z., CUI B., and Gan S., \u201cSqueezeattention: 2d management of KV-cache in LLM inference via layer-wise optimal budget,\u201d in International Conference on Learning Representations (ICLR), 2025."},{"key":"","unstructured":"Tong B., Lai B., Zhou Y., Luo G., Shen Y., Li K., Sun X., and Ji R., \u201cFlashsloth: Lightning multimodal large language models via embedded visual compression,\u201d IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2024."},{"key":"","unstructured":"Ye X., Gan Y., Huang X., Ge Y., and Tang Y., \u201cVoco-llama: Towards vision compression with large language models,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2025."},{"key":"","unstructured":"Li S., Tan Q., Dai Y., Kong Z., Wang T., Liu J., Li A., Liu N., Ding Y., Tang X., and Yuan G., \u201cMutual effort for efficiency: A similarity-based token pruning for vision transformers in self-supervised learning,\u201d in The Thirteenth International Conference on Learning Representations, 2025."},{"key":"","unstructured":"Yang C., Dong X., Zhu X., Su W., Wang J., Tian H., Chen Z., Wang W., Lu L., and Dai J., \u201cPvc: Progressive visual token compression for unified image and video processing in large vision-language models,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 24939\u201324949, 2025."},{"key":"","unstructured":"Tao K., Qin C., You H., Sui Y., and Wang H., \u201cDycoke: Dynamic compression of tokens for fast video large language models,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2025."},{"key":"","unstructured":"Kang S., Kim J., Kim J., and Hwang S. J., \u201cYour large vision-language model only needs a few attention heads for visual grounding,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2025."},{"key":"","unstructured":"Liu Z., Xie C.-W., Li P., Zhao L., Tang L., Zheng Y., Liu C., and Xie H., \u201cHybrid-level instruction injection for video token compression in multi-modal large language models,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2025."},{"key":"","unstructured":"Krizhevsky A., \u201cLearning multiple layers of features from tiny images,\u201d tech. rep., University of Toronto, 2009."},{"key":"","unstructured":"Deng J., Dong W., Socher R., Li L.-J., Li K., and Fei-Fei L., \u201cImagenet: A large-scale hierarchical image database,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 248\u2013255, 2009."},{"key":"","unstructured":"Deng L., \u201cThe mnist database of handwritten digit images for machine learning research,\u201d IEEE Signal Processing Magazine, vol. 29, no. 6, pp. 141\u2013142, 2012."},{"key":"","unstructured":"Yang L., Luo P., Loy C. C., and Tang X., \u201cA large-scale car dataset for fine-grained categorization and verification,\u201d in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3973\u20133981, 2015."},{"key":"","unstructured":"Kaur P., Sikka K., Wang W., Belongie S., and Divakaran A., \u201cFoodx-251: A dataset for fine-grained food classification,\u201d arXiv preprint arXiv: 1907.06167, 2019."},{"key":"","unstructured":"Puri R. and Catanzaro B., \u201cZero-shot text classification with generative language models,\u201d arXiv preprint arXiv: 1912.10165, 2019."},{"key":"","unstructured":"Zhu Y., Kiros R., Zemel R., Salakhutdinov R., Urtasun R., Torralba A., and Fidler S., \u201cAligning books and movies: Towards story-like visual explanations by watching movies and reading books,\u201d in IEEE International Conference on Computer Vision (ICCV), 2015."},{"key":"","unstructured":"Gao L., Biderman S., Black S., Golding L., Hoppe T., Foster C., Phang J., He H., Thite A., Nabeshima N., Presser S., and Leahy C., \u201cThe Pile: An 800gb dataset of diverse text for language modeling,\u201d arXiv preprint arXiv: 2101.00027, 2020."},{"key":"","unstructured":"Wang A., Singh A., Michael J., Hill F., Levy O., and Bowman S. R., \u201cGLUE: A multi-task benchmark and analysis platform for natural language understanding,\u201d in International Conference on Learning Representations (ICLR), 2019."},{"key":"","unstructured":"Wang A., Pruksachatkun Y., Nangia N., Singh A., Michael J., Hill F., Levy O., and Bowman S. R., \u201cSuperglue: A stickier benchmark for general-purpose language understanding systems,\u201d arXiv preprint arXiv: 1905.00537, 2019."},{"key":"","unstructured":"Rajpurkar P., Jia R., and Liang P., \u201cKnow what you don\u2019t know: Unanswerable questions for squad,\u201d in Annual Meeting of the Association for Computational Linguistics (ACL), pp. 784\u2013789, 2018."},{"key":"","unstructured":"Joshi M., Choi E., Weld D., and Zettlemoyer L., \u201cTriviaQA: A large scale distantly supervised challenge dataset for reading comprehension,\u201d in Annual Meeting of the Association for Computational Linguistics (ACL), pp. 1601\u20131611, 2017."},{"key":"","unstructured":"Kwiatkowski T., Palomaki J., Redfield O., Collins M., Parikh A., Alberti C., Epstein D., Polosukhin I., Devlin J., Lee K., Toutanova K., Jones L., Kelcey M., Chang M.-W., Dai A. M., Uszkoreit J., Le Q., and Petrov S., \u201cNatural questions: A benchmark for question answering research,\u201d Transactions of the Association for Computational Linguistics (TACL), vol. 7, pp. 452\u2013466, 2019."},{"key":"","unstructured":"Bisk Y., Zellers R., Le bras R., Gao J., and Choi Y., \u201cPiqa: Reasoning about physical commonsense in natural language,\u201d in AAAI Conference on Artificial Intelligence (AAAI), vol. 34, pp. 7432\u20137439, 2020."},{"key":"","unstructured":"Sap M., Rashkin H., Chen D., Le Bras R., and Choi Y., \u201cSocial IQa: Commonsense reasoning about social inter-actions,\u201d in Empirical Methods in Natural Language Processing and International Joint Conference on Natural Language Processing (EMNLP-IJCNLP), pp. 4463\u20134473, 2019."},{"key":"","unstructured":"Mihaylov T., Clark P., Khot T., and Sabharwal A., \u201cCan a suit of armor conduct electricity? a new dataset for open book question answering,\u201d in Empirical Methods in Natural Language Processing (EMNLP), 2018."},{"key":"","unstructured":"Hendrycks D., Burns C., Basart S., Zou A., Mazeika M., Song D., and Steinhardt J., \u201cMeasuring massive multitask language understanding,\u201d International Conference on Learning Representations (ICLR), 2021."},{"key":"","unstructured":"Hendrycks D., Burns C., Basart S., Critch A., Li J., Song D., and Steinhardt J., \u201cAligning ai with shared human values,\u201d International Conference on Learning Representations (ICLR), 2021."},{"key":"","unstructured":"Clark P., Cowhey I., Etzioni O., Khot T., Sabharwal A., Schoenick C., and Tafjord O., \u201cThink you have solved question answering? try arc, the ai2 reasoning challenge,\u201d arXiv preprint arXiv: 1803.05457, 2018."},{"key":"","unstructured":"Yu L., Jiang W., Shi H., Yu J., Liu Z., Zhang Y., Kwok J. T., Li Z., Weller A., and Liu W., \u201cMetamath: Bootstrap your own mathematical questions for large language models,\u201d arXiv preprint arXiv: 2309.12284, 2023."},{"key":"","unstructured":"Zellers R., Holtzman A., Bisk Y., Farhadi A., and Choi Y., \u201cHellaSwag: Can a machine really finish your sentence?,\u201d in Annual Meeting of the Association for Computational Linguistics (ACL), pp. 4791\u20134800, 2019."},{"key":"","unstructured":"Chen M., Tworek J., Jun H., Yuan Q., Pinto H. P. D. O., Kaplan J., Edwards H., Burda Y., Joseph N., Brockman G., et al., \u201cEvaluating large language models trained on code,\u201d arXiv preprint arXiv: 2107.03374, 2021."},{"key":"","unstructured":"Austin J., Odena A., Nye M., Bosma M., Michalewski H., Dohan D., Jiang E., Cai C., Terry M., Le Q., et al., \u201cProgram synthesis with large language models,\u201d arXiv preprint arXiv: 2108.07732, 2021."},{"key":"","unstructured":"Ajith A., Pan C., Xia M., Deshpande A., and Narasimhan K., \u201cInstructeval: Systematic evaluation of instruction selection methods,\u201d in North American Chapter of the Association for Computational Linguistics (NAACL), pp. 4336\u20134350, 2024."},{"key":"","unstructured":"Dubois Y., Liang P., and Hashimoto T., \u201cLength-controlled alpacaeval: A simple debiasing of automatic evaluators,\u201d in Conference on Language Modeling (COLM), 2024."},{"key":"","unstructured":"Choi J., Lee S., Ko B., Kim E., Kil J., and Kim H. J., \u201cRepresentation shift: Unifying token compression with flashattention,\u201d arXiv preprint arXiv: 2508.00367, 2025."},{"key":"","unstructured":"Liang F., Zhang Z., Lu H., Leung V., Guo Y., and Hu X., \u201cCommunication-efficient large-scale distributed deep learning: A comprehensive survey,\u201d arXiv preprint arXiv: 2404.06114, 2024."},{"key":"","unstructured":"Shen Y., Sun M., Lin J., Zhao J., and Zou A., \u201cOrder of compression: A systematic and optimal sequence to combinationally compress cnn,\u201d arXiv preprint arXiv: 2403.17447, 2024."},{"key":"","unstructured":"Zheng Y., Chen Y., Qian B., Shi X., Shu Y., and Chen J., \u201cA review on edge large language models: Design, execution, and applications,\u201d ACM Computing Surveys, vol. 57, no. 8, 2025."},{"key":"","unstructured":"Kim G. I., Hwang S., and Jang B., \u201cEfficient compressing and tuning methods for large language models: A systematic literature review,\u201d ACM Computing Surveys, 2025."}],"container-title":["Data Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.sciengine.com\/doi\/pdf\/CF3BC541E6E04D2697469D024A59BCB8","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0077","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/doi\/pdf\/CF3BC541E6E04D2697469D024A59BCB8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T09:04:39Z","timestamp":1774429479000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0077"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,25]]},"references-count":205,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,9,25]]},"published-print":{"date-parts":[[2026,3,1]]}},"URL":"https:\/\/doi.org\/10.3724\/2096-7004.di.2025.0077","relation":{},"ISSN":["2096-7004"],"issn-type":[{"value":"2096-7004","type":"print"}],"subject":[],"published":{"date-parts":[[2025,9,25]]}}}