{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,30]],"date-time":"2026-07-30T14:31:50Z","timestamp":1785421910962,"version":"3.56.0"},"publisher-location":"Cham","reference-count":49,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031730030","type":"print"},{"value":"9783031730047","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73004-7_2","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T17:02:14Z","timestamp":1730394134000},"page":"19-35","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":124,"title":["An Image is Worth 1\/2 Tokens After Layer 2: Plug-and-Play Inference Acceleration for\u00a0Large Vision-Language Models"],"prefix":"10.1007","author":[{"given":"Liang","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haozhe","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianyu","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuai","family":"Bai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junyang","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chang","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Baobao","family":"Chang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"2_CR1","doi-asserted-by":"crossref","unstructured":"Agrawal, H., et al.: Nocaps: novel object captioning at scale. In: 2019 IEEE\/CVF International Conference on Computer Vision, ICCV 2019, Seoul, Korea (South), 27 October\u20132 November 2019, pp. 8947\u20138956 (2019)","DOI":"10.1109\/ICCV.2019.00904"},{"key":"2_CR2","unstructured":"Bai, J., et al.: Qwen-VL: a frontier large vision-language model with versatile abilities. ArXiv preprint abs\/2308.12966 (2023)"},{"key":"2_CR3","unstructured":"Bavishi, R., et al.: Introducing our multimodal models (2023). https:\/\/www.adept.ai\/blog\/fuyu-8b"},{"key":"2_CR4","doi-asserted-by":"crossref","unstructured":"Cao, Q., Paranjape, B., Hajishirzi, H.: PuMer: pruning and merging tokens for efficient vision language models (2023). https:\/\/arxiv.org\/abs\/2305.17530","DOI":"10.18653\/v1\/2023.acl-long.721"},{"key":"2_CR5","unstructured":"Chen, L., et al.: Towards end-to-end embodied decision making via multi-modal large language model: explorations with gpt4-vision and beyond. ArXiv (2023)"},{"key":"2_CR6","doi-asserted-by":"crossref","unstructured":"Chen, L., et al.: PCA-bench: evaluating multimodal large language models in perception-cognition-action chain (2024)","DOI":"10.18653\/v1\/2024.findings-acl.64"},{"key":"2_CR7","unstructured":"Dao, T.: FlashAttention-2: faster attention with better parallelism and work partitioning (2023)"},{"key":"2_CR8","unstructured":"Dao, T., Fu, D.Y., Ermon, S., Rudra, A., R\u00e9, C.: FlashAttention: fast and memory-efficient exact attention with IO-awareness (2022)"},{"key":"2_CR9","unstructured":"Driess, D., Xia, F., et al.: PaLM-E: an embodied multimodal language model. vol. abs\/2303.03378 (2023)"},{"key":"2_CR10","unstructured":"Fu, C., et\u00a0al.: MME: a comprehensive evaluation benchmark for multimodal large language models. arXiv preprint arXiv:2306.13394 (2023)"},{"key":"2_CR11","unstructured":"Ge, S., Zhang, Y., Liu, L., Zhang, M., Han, J., Gao, J.: Model tells you what to discard: adaptive KV cache compression for LLMs (2024)"},{"key":"2_CR12","doi-asserted-by":"crossref","unstructured":"Jang, Y., Song, Y., Yu, Y., Kim, Y., Kim, G.: TGIF-QA: toward spatio-temporal reasoning in visual question answering (2017)","DOI":"10.1109\/CVPR.2017.149"},{"key":"2_CR13","unstructured":"Kondratyuk, D., et al.: VideoPoet: a large language model for zero-shot video generation (2023)"},{"key":"2_CR14","doi-asserted-by":"crossref","unstructured":"Kong, Z., et al.: SPViT: enabling faster vision transformers via soft token pruning (2022). https:\/\/arxiv.org\/abs\/2112.13890","DOI":"10.1007\/978-3-031-20083-0_37"},{"key":"2_CR15","doi-asserted-by":"crossref","unstructured":"Kwon, W., et al.: Efficient memory management for large language model serving with pagedattention (2023)","DOI":"10.1145\/3600006.3613165"},{"key":"2_CR16","unstructured":"Li, B., Wang, R., Wang, G., Ge, Y., Ge, Y., Shan, Y.: Seed-bench: benchmarking multimodal LLMs with generative comprehension (2023). https:\/\/arxiv.org\/abs\/2307.16125"},{"key":"2_CR17","unstructured":"Li, J., et al.: Empowering vision-language models to follow interleaved vision-language instructions. arXiv preprint arXiv:2308.04152 (2023)"},{"key":"2_CR18","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. ArXiv preprint abs\/2301.12597 (2023)"},{"key":"2_CR19","doi-asserted-by":"crossref","unstructured":"Li, Y., Wang, C., Jia, J.: LLaMA-VID: an image is worth 2 tokens in large language models (2023)","DOI":"10.1007\/978-3-031-72952-2_19"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Li, Z., et al.: Monkey: image resolution and text label are important things for large multi-modal models. arXiv preprint arXiv:2311.06607 (2023)","DOI":"10.1109\/CVPR52733.2024.02527"},{"key":"2_CR21","unstructured":"Liang, Y., Ge, C., Tong, Z., Song, Y., Wang, J., Xie, P.: Not all patches are what you need: expediting vision transformers via token reorganizations (2022). https:\/\/arxiv.org\/abs\/2202.07800"},{"key":"2_CR22","doi-asserted-by":"crossref","unstructured":"Lin, B., Zhu, B., Ye, Y., Ning, M., Jin, P., Yuan, L.: Video-LLaVA: learning united visual representation by alignment before projection. arXiv preprint arXiv:2311.10122 (2023)","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"2_CR23","unstructured":"Liu, H., Yan, W., Zaharia, M., Abbeel, P.: World model on million-length video and language with ringattention (2024)"},{"key":"2_CR24","unstructured":"Liu, H., Zaharia, M., Abbeel, P.: Ring attention with blockwise transformers for near-infinite context (2023)"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning (2023)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"2_CR26","unstructured":"Liu, H., et al.: LLaVA-next: improved reasoning, OCR, and world knowledge (2024). https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/"},{"key":"2_CR27","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. ArXiv preprint abs\/2304.08485 (2023)"},{"key":"2_CR28","doi-asserted-by":"crossref","unstructured":"Lu, J., et al.: Unified-IO 2: scaling autoregressive multimodal models with vision, language, audio, and action (2023)","DOI":"10.1109\/CVPR52733.2024.02497"},{"key":"2_CR29","unstructured":"Lu, P., et al.: Learn to explain: multimodal reasoning via thought chains for science question answering. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems, vol.\u00a035, pp. 2507\u20132521. Curran Associates, Inc. (2022). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/11332b6b6cf4485b84afadb1352d3a9a-Paper-Conference.pdf"},{"key":"2_CR30","doi-asserted-by":"crossref","unstructured":"Maaz, M., Rasheed, H., Khan, S., Khan, F.S.: Video-ChatGPT: towards detailed video understanding via large vision and language models. arXiv:2306.05424 (2023)","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"2_CR31","doi-asserted-by":"crossref","unstructured":"Mishra, A., Shekhar, S., Singh, A.K., Chakraborty, A.: OCR-VQA: visual question answering by reading text in images. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 947\u2013952. IEEE (2019)","DOI":"10.1109\/ICDAR.2019.00156"},{"key":"2_CR32","unstructured":"OpenAI: GPT-4V(ision) system card (2023)"},{"key":"2_CR33","doi-asserted-by":"crossref","unstructured":"Plummer, B.A., Wang, L., Cervantes, C.M., Caicedo, J.C., Hockenmaier, J., Lazebnik, S.: Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2641\u20132649 (2015)","DOI":"10.1109\/ICCV.2015.303"},{"key":"2_CR34","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18\u201324 July 2021, Virtual Event. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 8748\u20138763 (2021)"},{"key":"2_CR35","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"146","DOI":"10.1007\/978-3-031-20074-8_9","volume-title":"Computer Vision \u2013 ECCV 2022","author":"D Schwenk","year":"2022","unstructured":"Schwenk, D., Khandelwal, A., Clark, C., Marino, K., Mottaghi, R.: A-OKVQA: a benchmark for visual question answering using world knowledge. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022, Part VIII. LNCS, vol. 13668, pp. 146\u2013162. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20074-8_9"},{"key":"2_CR36","unstructured":"Team, G., et\u00a0al.: Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)"},{"key":"2_CR37","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Guyon, I., et al. (eds.) Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, 4\u20139 December 2017, Long Beach, CA, USA, pp. 5998\u20136008 (2017)"},{"key":"2_CR38","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Zitnick, C.L., Parikh, D.: CIDEr: consensus-based image description evaluation (2015)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"2_CR39","unstructured":"Wang, J., et al.: Mobile-agent: autonomous multi-modal mobile device agent with visual perception (2024)"},{"key":"2_CR40","doi-asserted-by":"publisher","unstructured":"Wang, L., et al.: Label words are anchors: An information flow perspective for understanding in-context learning. In: Bouamor, H., Pino, J., Bali, K. (eds.) Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 9840\u20139855. Association for Computational Linguistics, Singapore (2023). https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.609, https:\/\/aclanthology.org\/2023.emnlp-main.609","DOI":"10.18653\/v1\/2023.emnlp-main.609"},{"key":"2_CR41","unstructured":"Xiao, G., Tian, Y., Chen, B., Han, S., Lewis, M.: Efficient streaming language models with attention sinks. arXiv (2023)"},{"key":"2_CR42","doi-asserted-by":"crossref","unstructured":"Xiong, Y., et al.: PYRA: parallel yielding re-activation for training-inference efficient task adaptation (2024). https:\/\/arxiv.org\/abs\/2403.09192","DOI":"10.1007\/978-3-031-72673-6_25"},{"key":"2_CR43","unstructured":"Xu, D., et al.: Video question answering via gradually refined attention over appearance and motion. In: Proceedings of the 2017 ACM on Multimedia Conference, MM 2017, Mountain View, CA, USA, 23\u201327 October 2017, pp. 1645\u20131653 (2017)"},{"key":"2_CR44","doi-asserted-by":"crossref","unstructured":"Xu, D., et al.: Video question answering via gradually refined attention over appearance and motion. In: ACM Multimedia (2017)","DOI":"10.1145\/3123266.3123427"},{"key":"2_CR45","unstructured":"Yu, W., et al.: MM-vet: evaluating large multimodal models for integrated capabilities (2023). https:\/\/arxiv.org\/abs\/2308.02490"},{"key":"2_CR46","doi-asserted-by":"crossref","unstructured":"Yue, X., et al.: MMMU: a massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI. arXiv preprint arXiv:2311.16502 (2023)","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"2_CR47","unstructured":"Zhao, H., et al.: MMICL: empowering vision-language model with multi-modal in-context learning. ArXiv preprint abs\/2309.07915 (2023)"},{"key":"2_CR48","unstructured":"Zheng, B., Gou, B., Kil, J., Sun, H., Su, Y.: GPT-4V(ision) is a generalist web agent, if grounded (2024)"},{"key":"2_CR49","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X., Elhoseiny, M.: MiniGPT-4: enhancing vision-language understanding with advanced large language models. ArXiv preprint abs\/2304.10592 (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73004-7_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T16:15:48Z","timestamp":1732983348000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73004-7_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9783031730030","9783031730047"],"references-count":49,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73004-7_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}