{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T07:21:42Z","timestamp":1781076102817,"version":"3.54.1"},"publisher-location":"Cham","reference-count":57,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032128393","type":"print"},{"value":"9783032128409","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-12840-9_21","type":"book-chapter","created":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:23:11Z","timestamp":1767324191000},"page":"320-336","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Investigating Structural Pruning and\u00a0Recovery Techniques for\u00a0Compressing Multimodal Large Language Models: An Empirical Study"],"prefix":"10.1007","author":[{"given":"Yiran","family":"Huang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lukas","family":"Thede","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Massimiliano","family":"Mancini","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenjia","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zeynep","family":"Akata","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,1,2]]},"reference":[{"key":"21_CR1","unstructured":"Ashkboos, S., Croci, M.L., Nascimento, M.G.d., Hoefler, T., Hensman, J.: Slicegpt: compress large language models by deleting rows and columns. arXiv preprint arXiv:2401.15024 (2024)"},{"key":"21_CR2","unstructured":"Bai, H., et al.: Binarybert: pushing the limit of bert quantization (2021). https:\/\/arxiv.org\/abs\/2012.15701"},{"key":"21_CR3","unstructured":"Bo, L., et al.: Lmms-eval: accelerating the development of large multimoal models (2024). https:\/\/github.com\/EvolvingLMMs-Lab\/lmms-eval"},{"key":"21_CR4","doi-asserted-by":"crossref","unstructured":"Chen, Z., et\u00a0al.: Internvl: scaling up vision foundation models and aligning for generic visual-linguistic tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 24185\u201324198 (2024)","DOI":"10.1109\/CVPR52733.2024.02283"},{"key":"21_CR5","unstructured":"Chu, X., et\u00a0al.: Mobilevlm: a fast, reproducible and strong vision language assistant for mobile devices. arXiv preprint arXiv:2312.16886 (2023)"},{"key":"21_CR6","unstructured":"Dery, L., Kolawole, S., Kagy, J.F., Smith, V., Neubig, G., Talwalkar, A.: Everybody prune now: structured pruning of LLMs with only forward passes (2024). https:\/\/arxiv.org\/abs\/2402.05406"},{"key":"21_CR7","unstructured":"Dettmers, T., Lewis, M., Belkada, Y., Zettlemoyer, L.: Gpt3. int8 (): 8-bit matrix multiplication for transformers at scale. In: Advances in Neural Information Processing Systems, vol. 35, pp. 30318\u201330332 (2022)"},{"key":"21_CR8","doi-asserted-by":"crossref","unstructured":"Ding, X., Ding, G., Guo, Y., Han, J.: Centripetal SGD for pruning very deep convolutional networks with complicated structure (2019). https:\/\/arxiv.org\/abs\/1904.03837","DOI":"10.1109\/CVPR.2019.00508"},{"key":"21_CR9","unstructured":"Dong, X., Chen, S., Pan, S.J.: Learning to prune deep neural networks via layer-wise optimal brain surgeon (2017). https:\/\/arxiv.org\/abs\/1705.07565"},{"key":"21_CR10","unstructured":"Fan, A., Grave, E., Joulin, A.: Reducing transformer depth on demand with structured dropout. arXiv preprint arXiv:1909.11556 (2019)"},{"key":"21_CR11","doi-asserted-by":"crossref","unstructured":"Fang, G., Ma, X., Song, M., Mi, M.B., Wang, X.: Depgraph: towards any structural pruning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16091\u201316101 (2023)","DOI":"10.1109\/CVPR52729.2023.01544"},{"key":"21_CR12","doi-asserted-by":"crossref","unstructured":"Farina, M., Mancini, M., Cunegatti, E., Liu, G., Iacca, G., Ricci, E.: Multiflow: shifting towards task-agnostic vision-language pruning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16185\u201316195 (2024)","DOI":"10.1109\/CVPR52733.2024.01532"},{"key":"21_CR13","unstructured":"Frankle, J., Carbin, M.: The lottery ticket hypothesis: finding sparse, trainable neural networks (2019). https:\/\/arxiv.org\/abs\/1803.03635"},{"issue":"6","key":"21_CR14","doi-asserted-by":"publisher","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","volume":"129","author":"J Gou","year":"2021","unstructured":"Gou, J., Yu, B., Maybank, S.J., Tao, D.: Knowledge distillation: a survey. Int. J. Comput. Vis. 129(6), 1789\u20131819 (2021). https:\/\/doi.org\/10.1007\/s11263-021-01453-z","journal-title":"Int. J. Comput. Vis."},{"key":"21_CR15","unstructured":"Gu, Y., Dong, L., Wei, F., Huang, M.: Knowledge distillation of large language models. arXiv preprint arXiv:2306.08543 (2023)"},{"key":"21_CR16","unstructured":"Gu, Y., Dong, L., Wei, F., Huang, M.: Minillm: knowledge distillation of large language models (2024). https:\/\/arxiv.org\/abs\/2306.08543"},{"key":"21_CR17","unstructured":"He, M., et al.: Efficient multimodal learning from data-centric perspective. arXiv preprint arXiv:2402.11530 (2024)"},{"key":"21_CR18","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network (2015). https:\/\/arxiv.org\/abs\/1503.02531"},{"key":"21_CR19","unstructured":"Holtzman, A., Buys, J., Du, L., Forbes, M., Choi, Y.: The curious case of neural text degeneration. arXiv preprint arXiv:1904.09751 (2019)"},{"key":"21_CR20","unstructured":"Hsu, Y.C., Hua, T., Chang, S., Lou, Q., Shen, Y., Jin, H.: Language model compression with weighted low-rank factorization (2022). https:\/\/arxiv.org\/abs\/2207.00112"},{"key":"21_CR21","unstructured":"Hu, E.J., et al.: Lora: low-rank adaptation of large language models (2021). https:\/\/arxiv.org\/abs\/2106.09685"},{"key":"21_CR22","doi-asserted-by":"crossref","unstructured":"Hudson, D.A., Manning, C.D.: GQA: a new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6700\u20136709 (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"21_CR23","doi-asserted-by":"crossref","unstructured":"Jiang, F.: Identifying and mitigating vulnerabilities in LLM-integrated applications. Master\u2019s thesis, University of Washington (2024)","DOI":"10.1145\/3634737.3659433"},{"key":"21_CR24","doi-asserted-by":"crossref","unstructured":"Jiao, X., et al.: Tinybert: distilling BERT for natural language understanding (2020). https:\/\/arxiv.org\/abs\/1909.10351","DOI":"10.18653\/v1\/2020.findings-emnlp.372"},{"key":"21_CR25","unstructured":"Karamcheti, S., Nair, S., Balakrishna, A., Liang, P., Kollar, T., Sadigh, D.: Prismatic vlms: investigating the design space of visually-conditioned language models. arXiv preprint arXiv:2402.07865 (2024)"},{"key":"21_CR26","unstructured":"Lan, Z., Chen, M., Goodman, S., Gimpel, K., Sharma, P., Soricut, R.: Albert: a lite bert for self-supervised learning of language representations (2020). https:\/\/arxiv.org\/abs\/1909.11942"},{"key":"21_CR27","unstructured":"Lee, N., Ajanthan, T., Gould, S., Torr, P.H.S.: A signal propagation perspective for pruning neural networks at initialization (2020). https:\/\/arxiv.org\/abs\/1906.06307"},{"key":"21_CR28","unstructured":"Li, H., Kadav, A., Durdanovic, I., Samet, H., Graf, H.P.: Pruning filters for efficient convnets (2017). https:\/\/arxiv.org\/abs\/1608.08710"},{"key":"21_CR29","doi-asserted-by":"crossref","unstructured":"Li, Y., Du, Y., Zhou, K., Wang, J., Zhao, W.X., Wen, J.R.: Evaluating object hallucination in large vision-language models. arXiv preprint arXiv:2305.10355 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"21_CR30","unstructured":"Liang, K.J., et al.: Mixkd: towards efficient distillation of large-scale language models (2021). https:\/\/arxiv.org\/abs\/2011.00593"},{"key":"21_CR31","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26296\u201326306 (2024)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"21_CR32","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning (2023). https:\/\/arxiv.org\/abs\/2304.08485"},{"key":"21_CR33","unstructured":"Liu, L., et al.: Group fisher pruning for practical network compression (2021). https:\/\/arxiv.org\/abs\/2108.00708"},{"key":"21_CR34","first-page":"2507","volume":"35","author":"P Lu","year":"2022","unstructured":"Lu, P., et al.: Learn to explain: multimodal reasoning via thought chains for science question answering. Adv. Neural. Inf. Process. Syst. 35, 2507\u20132521 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"21_CR35","first-page":"21702","volume":"36","author":"X Ma","year":"2023","unstructured":"Ma, X., Fang, G., Wang, X.: Llm-pruner: on the structural pruning of large language models. Adv. Neural. Inf. Process. Syst. 36, 21702\u201321720 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"21_CR36","unstructured":"McCarley, J., Chakravarti, R., Sil, A.: Structured pruning of a bert-based question answering model. arXiv preprint arXiv:1910.06360 (2019)"},{"key":"21_CR37","doi-asserted-by":"crossref","unstructured":"Men, X., et al.: Shortgpt: layers in large language models are more redundant than you expect. arXiv preprint arXiv:2403.03853 (2024)","DOI":"10.18653\/v1\/2025.findings-acl.1035"},{"key":"21_CR38","unstructured":"Michel, P., Levy, O., Neubig, G.: Are sixteen heads really better than one? In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"21_CR39","unstructured":"Park, S., Lee, J., Mo, S., Shin, J.: Lookahead: a far-sighted alternative of magnitude-based pruning (2020). https:\/\/arxiv.org\/abs\/2002.04809"},{"key":"21_CR40","unstructured":"Popp, N., Metzen, J.H., Hein, M.: Zero-shot distillation for image encoders: how to make effective use of synthetic data. arXiv preprint arXiv:2404.16637 (2024)"},{"key":"21_CR41","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2022.101429","volume":"77","author":"H Sajjad","year":"2023","unstructured":"Sajjad, H., Dalvi, F., Durrani, N., Nakov, P.: On the effect of dropping layers of pre-trained transformer models. Comput. Speech Lang. 77, 101429 (2023)","journal-title":"Comput. Speech Lang."},{"key":"21_CR42","unstructured":"Sanh, V., Debut, L., Chaumond, J., Wolf, T.: Distilbert, a distilled version of bert: smaller, faster, cheaper and lighter (2020). https:\/\/arxiv.org\/abs\/1910.01108"},{"key":"21_CR43","unstructured":"Sanh, V., Wolf, T., Rush, A.M.: Movement pruning: adaptive sparsity by fine-tuning (2020). https:\/\/arxiv.org\/abs\/2005.07683"},{"key":"21_CR44","doi-asserted-by":"crossref","unstructured":"Sun, S., Cheng, Y., Gan, Z., Liu, J.: Patient knowledge distillation for bert model compression (2019). https:\/\/arxiv.org\/abs\/1908.09355","DOI":"10.18653\/v1\/D19-1441"},{"key":"21_CR45","unstructured":"Touvron, H., et\u00a0al.: Llama: open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"21_CR46","doi-asserted-by":"crossref","unstructured":"Voita, E., Talbot, D., Moiseev, F., Sennrich, R., Titov, I.: Analyzing multi-head self-attention: specialized heads do the heavy lifting, the rest can be pruned. arXiv preprint arXiv:1905.09418 (2019)","DOI":"10.18653\/v1\/P19-1580"},{"key":"21_CR47","unstructured":"Wang, W., Wei, F., Dong, L., Bao, H., Yang, N., Zhou, M.: Minilm: deep self-attention distillation for task-agnostic compression of pre-trained transformers (2020). https:\/\/arxiv.org\/abs\/2002.10957"},{"key":"21_CR48","unstructured":"Xia, M., Gao, T., Zeng, Z., Chen, D.: Sheared llama: accelerating language model pre-training via structured pruning (2024). https:\/\/arxiv.org\/abs\/2310.06694"},{"key":"21_CR49","unstructured":"Xu, X., et al.: A survey on knowledge distillation of large language models (2024). https:\/\/arxiv.org\/abs\/2402.13116"},{"key":"21_CR50","doi-asserted-by":"crossref","unstructured":"Yang, C., et al.: Clip-kd: an empirical study of clip model distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15952\u201315962 (2024)","DOI":"10.1109\/CVPR52733.2024.01510"},{"key":"21_CR51","unstructured":"Yao, Z., Aminabadi, R.Y., Zhang, M., Wu, X., Li, C., He, Y.: Zeroquant: efficient and affordable post-training quantization for large-scale transformers (2022). https:\/\/arxiv.org\/abs\/2206.01861"},{"key":"21_CR52","unstructured":"Yin, S., et al.: A survey on multimodal large language models. arXiv preprint arXiv:2306.13549 (2023)"},{"key":"21_CR53","unstructured":"You, Z., Yan, K., Ye, J., Ma, M., Wang, P.: Gate decorator: global filter pruning method for accelerating deep convolutional neural networks (2019). https:\/\/arxiv.org\/abs\/1909.08174"},{"key":"21_CR54","doi-asserted-by":"crossref","unstructured":"Yue, X., et al.: Mmmu: a massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI. In: Proceedings of CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"21_CR55","doi-asserted-by":"publisher","unstructured":"Zafrir, O., Boudoukh, G., Izsak, P., Wasserblat, M.: Q8bert: quantized 8bit bert. In: 2019 Fifth Workshop on Energy Efficient Machine Learning and Cognitive Computing - NeurIPS Edition (EMC2-NIPS). IEEE (2019). https:\/\/doi.org\/10.1109\/emc2-nips53020.2019.00016","DOI":"10.1109\/emc2-nips53020.2019.00016"},{"key":"21_CR56","unstructured":"Zhu, M., et al.: A comprehensive overhaul of multimodal assistant with small language models. arXiv preprint arXiv:2403.06199 (2024)"},{"key":"21_CR57","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Zhu, M., Liu, N., Xu, Z., Peng, Y.: Llava-phi: efficient multi-modal assistant with small language model. In: Proceedings of the 1st International Workshop on Efficient Multimedia Computing under Limited, pp. 18\u201322 (2024)","DOI":"10.1145\/3688863.3689575"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-12840-9_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:23:15Z","timestamp":1767324195000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-12840-9_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032128393","9783032128409"],"references-count":57,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-12840-9_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 January 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"DAGM GCPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"DAGM German Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Freiburg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"47","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"dagm2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.dagm-gcpr.de\/year\/2025","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}