{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,25]],"date-time":"2026-02-25T21:05:17Z","timestamp":1772053517415,"version":"3.50.1"},"reference-count":41,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,12,1]]},"DOI":"10.1109\/vcip67698.2025.11396917","type":"proceedings-article","created":{"date-parts":[[2026,2,24]],"date-time":"2026-02-24T20:55:27Z","timestamp":1771966527000},"page":"1-5","source":"Crossref","is-referenced-by-count":0,"title":["Quadtree Partitioning-based Visual Token Pruning for MLLMs Considering Information Density"],"prefix":"10.1109","author":[{"given":"Yuntao","family":"Wei","sequence":"first","affiliation":[{"name":"The Hong Kong Polytechnic University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinming","family":"Liu","sequence":"additional","affiliation":[{"name":"Eastern Institute of Technology,Ningbo Institute of Digital Twin,Ningbo,P.R. China,315200"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shengyang","family":"Zhao","sequence":"additional","affiliation":[{"name":"Eastern Institute of Technology,Ningbo Institute of Digital Twin,Ningbo,P.R. China,315200"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhibo","family":"Chen","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenjun","family":"Zeng","sequence":"additional","affiliation":[{"name":"Eastern Institute of Technology,Ningbo Institute of Digital Twin,Ningbo,P.R. China,315200"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Jin","sequence":"additional","affiliation":[{"name":"Eastern Institute of Technology,Ningbo Institute of Digital Twin,Ningbo,P.R. China,315200"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"34 892","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023","journal-title":"Advances in neural information processing systems"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"ref3","article-title":"Llavanext: Improved reasoning, ocr, and world knowledge","author":"Liu","year":"2024"},{"key":"ref4","article-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"issue":"3","key":"ref5","first-page":"6","article-title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality","volume":"2","author":"Chiang","year":"2023"},{"key":"ref6","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref8","article-title":"Rethinking attention with performers","author":"Choromanski","year":"2020"},{"key":"ref9","first-page":"5156","article-title":"Transformers are rnns: Fast autoregressive transformers with linear attention","volume-title":"International conference on machine learning","author":"Katharopoulos"},{"key":"ref10","first-page":"16 344","article-title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","volume":"35","author":"Dao","year":"2022","journal-title":"Advances in neural information processing systems"},{"key":"ref11","article-title":"Minicpm-v: A gpt-4v level mllm on your phone","author":"Yao","year":"2024"},{"key":"ref12","article-title":"Matryoshka multimodal models","author":"Cai","year":"2024"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-025-02491-7"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i5.32567"},{"key":"ref15","article-title":"Token merging: Your vit but faster","author":"Bolya","year":"2022"},{"key":"ref16","article-title":"Not all patches are what you need: Expediting vision transformers via token reorganizations","author":"Liang","year":"2022"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73004-7_2"},{"key":"ref18","article-title":"Llava-prumerge: Adaptive token reduction for efficient large multimodal models","author":"Shang","year":"2024"},{"key":"ref19","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-06895-4"},{"key":"ref21","article-title":"Qwen2. 5-vl technical report","author":"Bai","year":"2025"},{"key":"ref22","article-title":"Kimi-vl technical report","author":"Team","year":"2025"},{"key":"ref23","article-title":"Internvl3: Exploring advanced training and test-time recipes for open-source multimodal models","author":"Zhu","year":"2025"},{"key":"ref24","article-title":"Qwen technical report","author":"Bai","year":"2023"},{"key":"ref25","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref26","first-page":"19 730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"International conference on machine learning","author":"Li"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2025.3637265"},{"issue":"2","key":"ref28","first-page":"3","article-title":"Qwen-vl: A frontier large vision-language model with versatile abilities","volume":"1","author":"Bai","year":"2023"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"ref30","article-title":"Llava-onevision: Easy visual task transfer","author":"Li","year":"2024"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i21.34366"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73004-7_2"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i5.32567"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.1178"},{"key":"ref35","first-page":"2507","article-title":"Learn to explain: Multimodal reasoning via thought chains for science question answering","volume":"35","author":"Lu","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00851"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00380"},{"key":"ref39","article-title":"Mme: A comprehensive evaluation benchmark for multimodal large language models","author":"Fu","year":"2023"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01263"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"}],"event":{"name":"2025 International Conference on Visual Communications and Image Processing (VCIP)","location":"Klagenfurt, Austria","start":{"date-parts":[[2025,12,1]]},"end":{"date-parts":[[2025,12,4]]}},"container-title":["2025 International Conference on Visual Communications and Image Processing (VCIP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11396756\/11396790\/11396917.pdf?arnumber=11396917","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,25]],"date-time":"2026-02-25T20:54:03Z","timestamp":1772052843000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11396917\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,1]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/vcip67698.2025.11396917","relation":{},"subject":[],"published":{"date-parts":[[2025,12,1]]}}}