{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,27]],"date-time":"2026-08-27T17:17:36Z","timestamp":1787851056596,"version":"build-2784847793"},"reference-count":58,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61301253"],"award-info":[{"award-number":["61301253"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012659","name":"Foundation for Innovative Research Groups of the National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012659","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.neucom.2026.133887","type":"journal-article","created":{"date-parts":[[2026,5,7]],"date-time":"2026-05-07T23:30:36Z","timestamp":1778196636000},"page":"133887","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["GridPrune: From \u201cWhere to look\u201d to \u201cWhat to select\u201d in visual token pruning for MLLMs"],"prefix":"10.1016","volume":"694","author":[{"given":"Yuxiang","family":"Duan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ao","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingqin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Luyu","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6470-140X","authenticated-orcid":false,"given":"Pengwei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.133887_bib0005","doi-asserted-by":"crossref","first-page":"34892","DOI":"10.52202\/075280-1516","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133887_bib0010","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"26296","article-title":"Improved baselines with visual instruction tuning","author":"Liu","year":"2024"},{"key":"10.1016\/j.neucom.2026.133887_bib0015","author":"Bai"},{"key":"10.1016\/j.neucom.2026.133887_bib0020","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"24185","article-title":"InternVL: scaling up vision foundation models and aligning for generic visual-linguistic tasks","author":"Chen","year":"2024"},{"key":"10.1016\/j.neucom.2026.133887_bib0025","doi-asserted-by":"crossref","first-page":"49250","DOI":"10.52202\/075280-2142","article-title":"InstructBLIP: towards general-purpose vision-language models with instruction tuning","volume":"36","author":"Dai","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133887_bib0030","author":"Zhu"},{"issue":"8","key":"10.1016\/j.neucom.2026.133887_bib0035","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3711680","article-title":"Natural language understanding and inference with MLLM in visual question answering: a survey","volume":"57","author":"Kuang","year":"2025","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.neucom.2026.133887_bib0040","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.130810","article-title":"EmoVerse: enhancing multimodal large language models for affective computing via multitask learning","volume":"650","author":"Li","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.133887_bib0045","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10867","article-title":"From images to textual prompts: zero-shot visual question answering with frozen large language models","author":"Guo","year":"2023"},{"key":"10.1016\/j.neucom.2026.133887_bib0050","series-title":"Proceedings of the Thirty-Fourth International Joint Conference on Artificial Intelligence","first-page":"6227","article-title":"Connecting giants: synergistic knowledge transfer of large multimodal models for few-shot learning","author":"Tang","year":"2025"},{"key":"10.1016\/j.neucom.2026.133887_bib0055","author":"Li"},{"key":"10.1016\/j.neucom.2026.133887_bib0060","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.neucom.2026.133887_bib0065","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"11975","article-title":"Sigmoid loss for language image pre-training","author":"Zhai","year":"2023"},{"key":"10.1016\/j.neucom.2026.133887_bib0070","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.neucom.2026.133887_bib0075","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"13817","article-title":"Honeybee: locality-enhanced projector for multimodal LLM","author":"Cha","year":"2024"},{"key":"10.1016\/j.neucom.2026.133887_bib0080","author":"Touvron"},{"key":"10.1016\/j.neucom.2026.133887_bib0085","author":"Bai"},{"key":"10.1016\/j.neucom.2026.133887_bib0090","author":"Cai"},{"key":"10.1016\/j.neucom.2026.133887_bib0095","unstructured":"H. Liu, C. Li, Y. Li, B. Li, Y. Zhang, S. Shen, Y.J. Lee, LLaVA-NeXT: improved reasoning, OCR, and world knowledge, January 2024, Available at: https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/"},{"issue":"12","key":"10.1016\/j.neucom.2026.133887_bib0100","doi-asserted-by":"crossref","DOI":"10.1007\/s11432-024-4231-5","article-title":"How far are we to GPT-4V? Closing the gap to commercial multimodal models with open-source suites","volume":"67","author":"Chen","year":"2024","journal-title":"Sci. China Inf. Sci."},{"key":"10.1016\/j.neucom.2026.133887_bib0105","author":"Li"},{"key":"10.1016\/j.neucom.2026.133887_bib0110","author":"Lin"},{"key":"10.1016\/j.neucom.2026.133887_bib0115","author":"Liu"},{"key":"10.1016\/j.neucom.2026.133887_bib0120","author":"Wang"},{"key":"10.1016\/j.neucom.2026.133887_bib0125","author":"Li"},{"key":"10.1016\/j.neucom.2026.133887_bib0130","series-title":"European Conference on Computer Vision","first-page":"19","article-title":"An image is worth 1\/2 tokens after layer 2: plug-and-play inference acceleration for large vision-language models","author":"Chen","year":"2024"},{"key":"10.1016\/j.neucom.2026.133887_bib0135","author":"Zhang"},{"key":"10.1016\/j.neucom.2026.133887_bib0140","author":"Peng"},{"key":"10.1016\/j.neucom.2026.133887_bib0145","author":"Xing"},{"key":"10.1016\/j.neucom.2026.133887_bib0150","author":"Bolya"},{"key":"10.1016\/j.neucom.2026.133887_bib0155","author":"Wen"},{"key":"10.1016\/j.neucom.2026.133887_bib0160","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"9392","article-title":"DivPrune: diversity-based visual token pruning for large multimodal models","author":"Alvar","year":"2025"},{"key":"10.1016\/j.neucom.2026.133887_bib0165","author":"Wen"},{"key":"10.1016\/j.neucom.2026.133887_bib0170","first-page":"39800","article-title":"Don\u2019t just chase \u201chighlighted tokens\u201d in MLLMs: revisiting visual holistic context retention","volume":"38","author":"Zou","year":"2026","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133887_bib0175","author":"Zhang"},{"key":"10.1016\/j.neucom.2026.133887_bib0180","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"19792","article-title":"VisionZip: longer is better but not necessary in vision language models","author":"Yang","year":"2025"},{"issue":"1","key":"10.1016\/j.neucom.2026.133887_bib0185","doi-asserted-by":"crossref","first-page":"97","DOI":"10.1016\/0010-0285(80)90005-5","article-title":"A feature-integration theory of attention","volume":"12","author":"Treisman","year":"1980","journal-title":"Cogn. Psychol."},{"key":"10.1016\/j.neucom.2026.133887_bib0190","series-title":"Proceedings of the 33rd ACM International Conference on Multimedia","first-page":"5903","article-title":"From individuals to crowds: dual-level public response prediction in social media","author":"Zhang","year":"2025"},{"issue":"3","key":"10.1016\/j.neucom.2026.133887_bib0195","doi-asserted-by":"crossref","first-page":"1958","DOI":"10.1109\/TPAMI.2024.3511621","article-title":"Divide-and-conquer: confluent triple-flow network for RGB-T salient object detection","volume":"47","author":"Tang","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.133887_bib0200","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.108792","article-title":"Learning attention-guided pyramidal features for few-shot fine-grained recognition","volume":"130","author":"Tang","year":"2022","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.neucom.2026.133887_bib0205","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"1719","article-title":"M3Net: multi-view encoding, matching, and fusion for few-shot fine-grained action recognition","author":"Tang","year":"2023"},{"key":"10.1016\/j.neucom.2026.133887_bib0210","author":"Wang"},{"key":"10.1016\/j.neucom.2026.133887_bib0215","author":"Sun"},{"key":"10.1016\/j.neucom.2026.133887_bib0220","author":"Dosovitskiy"},{"key":"10.1016\/j.neucom.2026.133887_bib0225","author":"Yue"},{"key":"10.1016\/j.neucom.2026.133887_bib0230","author":"Wang"},{"key":"10.1016\/j.neucom.2026.133887_bib0235","author":"Li"},{"key":"10.1016\/j.neucom.2026.133887_bib0240","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"22857","article-title":"LLaVA-PruMerge: adaptive token reduction for efficient large multimodal models","author":"Shang","year":"2025"},{"key":"10.1016\/j.neucom.2026.133887_bib0245","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"6904","article-title":"Making the V in VQA matter: elevating the role of image understanding in visual question answering","author":"Goyal","year":"2017"},{"key":"10.1016\/j.neucom.2026.133887_bib0250","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6700","article-title":"GQA: a new dataset for real-world visual reasoning and compositional question answering","author":"Hudson","year":"2019"},{"key":"10.1016\/j.neucom.2026.133887_bib0255","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"3608","article-title":"Vizwiz grand challenge: answering visual questions from blind people","author":"Gurari","year":"2018"},{"key":"10.1016\/j.neucom.2026.133887_bib0260","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8317","article-title":"Towards VQA models that can read","author":"Singh","year":"2019"},{"key":"10.1016\/j.neucom.2026.133887_bib0265","doi-asserted-by":"crossref","first-page":"2507","DOI":"10.52202\/068431-0182","article-title":"Learn to explain: multimodal reasoning via thought chains for science question answering","volume":"35","author":"Lu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133887_bib0270","author":"Fu"},{"key":"10.1016\/j.neucom.2026.133887_bib0275","author":"Li"},{"key":"10.1016\/j.neucom.2026.133887_bib0280","author":"Li"},{"key":"10.1016\/j.neucom.2026.133887_bib0285","series-title":"European Conference on Computer Vision","first-page":"216","article-title":"MMBench: is your multi-modal model an all-around player?","author":"Liu","year":"2024"},{"key":"10.1016\/j.neucom.2026.133887_bib0290","author":"Yu"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226012841?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226012841?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,27]],"date-time":"2026-08-27T16:22:03Z","timestamp":1787847723000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226012841"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":58,"alternative-id":["S0925231226012841"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.133887","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"GridPrune: From \u201cWhere to look\u201d to \u201cWhat to select\u201d in visual token pruning for MLLMs","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.133887","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133887"}}