{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T00:25:49Z","timestamp":1783124749780,"version":"3.54.6"},"reference-count":47,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.eswa.2026.132839","type":"journal-article","created":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T17:31:57Z","timestamp":1778779917000},"page":"132839","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Spatial -aware efficient projector for MLLMs via multi-layer feature aggregation"],"prefix":"10.1016","volume":"328","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2443-2954","authenticated-orcid":false,"given":"Shun","family":"Qian","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6352-2170","authenticated-orcid":false,"given":"Bingquan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9081-1410","authenticated-orcid":false,"given":"Chengjie","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3021-4576","authenticated-orcid":false,"given":"Peijin","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9597-7476","authenticated-orcid":false,"given":"Yunhe","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9688-6958","authenticated-orcid":false,"given":"Zhen","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Baoxun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132839_bib0001","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F. L., Almeida, D., Altenschmidt, J., Altman, S., & Anadkat, S. et al. (2023). GPT-4 technical report. arXiv preprint arXiv: 2303.08774."},{"key":"10.1016\/j.eswa.2026.132839_bib0002","series-title":"Advances in neural information processing systems","first-page":"23716","article-title":"Flamingo: A visual language model for few-shot learning","volume":"vol. 35","author":"Alayrac","year":"2022"},{"key":"10.1016\/j.eswa.2026.132839_bib0003","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"9392","article-title":"DivPrune: Diversity-based visual token pruning for large multimodal models","author":"Alvar","year":"2025"},{"key":"10.1016\/j.eswa.2026.132839_bib0004","series-title":"Proceedings of the IEEE International conference on computer vision","first-page":"2425","article-title":"VQA: Visual question answering","author":"Antol","year":"2015"},{"key":"10.1016\/j.eswa.2026.132839_bib0005","unstructured":"Bai, J., Bai, S., Yang, S., Wang, S., Tan, S., Wang, P., Lin, J., Zhou, C., & Zhou, J. (2023b). Qwen-VL: A frontier large vision-language model with versatile abilities. arXiv preprint arXiv: 2308.12966."},{"key":"10.1016\/j.eswa.2026.132839_bib0006","unstructured":"Bai, S., Chen, K., Liu, X., Wang, J., Ge, W., Song, S., Dang, K., Wang, P., Wang, S., Tang, J., & et al. (2025). Qwen2.5-VL technical report. arXiv preprint arXiv: 2502.13923."},{"key":"10.1016\/j.eswa.2026.132839_bib0007","series-title":"Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition","first-page":"22861","article-title":"Sequential modeling enables scalable learning for large vision models","author":"Bai","year":"2024"},{"key":"10.1016\/j.eswa.2026.132839_bib0008","unstructured":"Cai, M., Yang, J., Gao, J., & Lee, Y. J. (2024). Matryoshka multimodal models. arXiv preprint arXiv: 2405.17430."},{"key":"10.1016\/j.eswa.2026.132839_bib0009","series-title":"Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition","first-page":"13","article-title":"Honeybee: Locality-enhanced projector for multimodal llm","author":"Cha","year":"2024"},{"issue":"12","key":"10.1016\/j.eswa.2026.132839_bib0010","doi-asserted-by":"crossref","DOI":"10.1007\/s11432-024-4231-5","article-title":"How far are we to GPT-4V? Closing the gap to commercial multimodal models with open-source suites","volume":"67","author":"Chen","year":"2024","journal-title":"Science China Information Sciences"},{"key":"10.1016\/j.eswa.2026.132839_bib0011","unstructured":"Chiang, W.-L., Li, Z., Lin, Z., Sheng, Y., Wu, Z., Zhang, H., Zheng, L., Zhuang, S., Zhuang, Y., Gonzalez, J. E., Stoica, I., & Xing, E. P. (2023). Vicuna: An open-source chatbot impressing GPT-4 with 90%* ChatGPT quality. https:\/\/vicuna.lmsys.org (accessed 14 April 2023), 2(3) pages 6."},{"key":"10.1016\/j.eswa.2026.132839_bib0012","unstructured":"Chu, X., Qiao, L., Zhang, X., Xu, S., Wei, F., Yang, Y., Sun, X., Hu, Y., Lin, X., & Zhang, B. et al. (2024). MobileVLM v2: Faster and stronger baseline for vision language model. arXiv preprint arXiv: 2402.03766."},{"key":"10.1016\/j.eswa.2026.132839_bib0013","unstructured":"Chu, X., Tian, Z., Zhang, B., Wang, X., & Shen, C. (2021). Conditional positional encodings for vision transformers. arXiv preprint arXiv: 2102.10882."},{"key":"10.1016\/j.eswa.2026.132839_bib0014","series-title":"Proceedings of the 37th International conference on neural information processing systems","article-title":"InstructBLIP: Towards general-purpose vision-language models with instruction tuning","author":"Dai","year":"2024"},{"key":"10.1016\/j.eswa.2026.132839_bib0015","unstructured":"Dong, X., Zhang, P., Zang, Y., Cao, Y., Wang, B., Ouyang, L., Wei, X., Zhang, S., Duan, H., Cao, M. et al. (2024a). InternLM-XComposer2: Mastering free-form text-image composition and comprehension in vision-language large model. arXiv preprint arXiv: 2401.16420."},{"key":"10.1016\/j.eswa.2026.132839_bib0016","doi-asserted-by":"crossref","unstructured":"Dong, X., Zhang, P., Zang, Y., Cao, Y., Wang, B., Ouyang, L., Zhang, S., Duan, H., Zhang, W., Li, Y. et al. (2024b). InternLM-XComposer2-4KHD: A pioneering large vision-language model handling resolutions from 336 pixels to 4k hd. arXiv preprint arXiv: 2404.06512.","DOI":"10.52202\/079017-1348"},{"key":"10.1016\/j.eswa.2026.132839_bib0017","series-title":"Proceedings of the IEEE\/CVF International conference on computer vision","first-page":"4019","article-title":"Semantic equitable clustering: A simple and effective strategy for clustering vision tokens","author":"Fan","year":"2025"},{"key":"10.1016\/j.eswa.2026.132839_bib0018","unstructured":"Fu, C., Chen, P., Shen, Y., Qin, Y., Zhang, M., Lin, X., Yang, J., Zheng, X., Li, K., & Sun, X. et al. (2023). Mme: A comprehensive evaluation benchmark for multimodal large language models. arXiv preprint arXiv: 2306.13394."},{"key":"10.1016\/j.eswa.2026.132839_bib0019","unstructured":"Liu, D., Zhang, R., Qiu, L., Huang, S., Lin, W., Zhao, S., Geng, S., Lin, Z., Jin, P., Zhang, K., et al. (2024). Sphinx-X: Scaling data and parameters for a family of multi-modal large language models. In Proceedings of the 41st International Conference on Machine Learning (pp. 32400\u201332420)."},{"key":"10.1016\/j.eswa.2026.132839_bib0020","series-title":"Proceedings of the IEEE Conference on computer vision and pattern recognition","first-page":"3608","article-title":"Vizwiz grand challenge: Answering visual questions from blind people","author":"Gurari","year":"2018"},{"key":"10.1016\/j.eswa.2026.132839_bib0021","series-title":"Proceedings of the IEEE Conference on computer vision and pattern recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.eswa.2026.132839_bib0022","series-title":"Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition","first-page":"6700","article-title":"GQA: A new dataset for real-world visual reasoning and compositional question answering","author":"Hudson","year":"2019"},{"key":"10.1016\/j.eswa.2026.132839_bib0023","series-title":"International conference on machine learning","first-page":"4651","article-title":"Perceiver: General perception with iterative attention","author":"Jaegle","year":"2021"},{"key":"10.1016\/j.eswa.2026.132839_bib0024","unstructured":"Jiang, A. Q., Sablayrolles, A., Roux, A., Mensch, A., Savary, B., Bamford, C., Chaplot, D. S., Casas, D. d. l., Hanna, E. B., Bressand, F. et al. (2024). Mixtral of experts. arXiv preprint arXiv: 2401.04088."},{"key":"10.1016\/j.eswa.2026.132839_bib0025","unstructured":"Dongsheng, J., Yuchen, L., Songlin, L., Jin\u2019e, Z., Hao, Z., Zhen, G., & Xiaopeng, Z., Jin, L., Hongkai, X. (2023). From clip to dino: Visual encoders shout in multi-modal large language models. arXiv preprint arXiv: 2310.08825."},{"key":"10.1016\/j.eswa.2026.132839_bib0026","unstructured":"Li, B., Wang, R., Wang, G., Ge, Y., Ge, Y., & Shan, Y. (2023a). Seed-bench: Benchmarking multimodal llms with generative comprehension. arXiv preprint arXiv: 2307.16125."},{"key":"10.1016\/j.eswa.2026.132839_bib0027","unstructured":"Li, B., Zhang, P., Yang, J., Zhang, Y., Pu, F., & Liu, Z. (2023b). OtterHD: A high-resolution multi-modality model. arXiv preprint arXiv: 2311.04219."},{"key":"10.1016\/j.eswa.2026.132839_bib0028","unstructured":"Li, F., Zhang, R., Zhang, H., Zhang, Y., Li, B., Li, W., Ma, Z., & Li, C. (2024b). Llava-next-interleave: Tackling multi-image, video, and 3D in large multimodal models. arXiv preprint arXiv: 2407.07895."},{"key":"10.1016\/j.eswa.2026.132839_bib0029","unstructured":"Li, J., Li, D., Savarese, S., & Hoi, S. (2023c). BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv preprint arXiv: 2301.12597."},{"key":"10.1016\/j.eswa.2026.132839_bib0030","unstructured":"Li, W., Yuan, Y., Liu, J., Tang, D., Wang, S., Zhu, J., & Zhang, L. (2024c). TokenPacker: Efficient visual projector for multimodal LLM. arXiv preprint arXiv: 2407.02392."},{"key":"10.1016\/j.eswa.2026.132839_bib0031","series-title":"Proceedings of the 2023 Conference on empirical methods in natural language processing","first-page":"292","article-title":"Evaluating object hallucination in large vision-language models","author":"Li","year":"2023"},{"key":"10.1016\/j.eswa.2026.132839_bib0032","series-title":"Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition","first-page":"26763","article-title":"Monkey: Image resolution and text label are important things for large multi-modal models","author":"Li","year":"2024"},{"key":"10.1016\/j.eswa.2026.132839_bib0033","series-title":"International conference on learning representations (ICLR)","article-title":"Not all patches are what you need: Expediting vision transformers via token reorganizations","author":"Liang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132839_bib0034","doi-asserted-by":"crossref","first-page":"635","DOI":"10.1162\/tacl_a_00566","article-title":"Visual spatial reasoning","volume":"11","author":"Liu","year":"2023","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"10.1016\/j.eswa.2026.132839_bib0035","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"26296","article-title":"Improved baselines with visual instruction tuning","author":"Liu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132839_bib0036","unstructured":"Liu, H., Li, C., Li, Y., Li, B., Zhang, Y., Shen, S., & Lee, Y. J. (2024). Llava-next: Improved reasoning, ocr, and world knowledge, January 2024. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next, 1 (8) 2024."},{"key":"10.1016\/j.eswa.2026.132839_bib0037","series-title":"Advances in neural information processing systems","article-title":"Visual instruction tuning","volume":"vol. 36","author":"Liu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132839_bib0038","doi-asserted-by":"crossref","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Li, C., Yang, J., Su, H., Zhu, J. et al. (2023d). Grounding dino: Marrying dino with grounded pre-training for open-set object detection. arXiv preprint arXiv: 2303.05499.","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"10.1016\/j.eswa.2026.132839_bib0039","series-title":"European conference on computer vision","first-page":"216","article-title":"MMBench: Is your multi-modal model an all-around player?","author":"Liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132839_bib0040","unstructured":"Loshchilov, I. (2017). Decoupled weight decay regularization. arXiv preprint arXiv: 1711.05101."},{"key":"10.1016\/j.eswa.2026.132839_bib0050","unstructured":"Wang, P., Bai, S., Tan, S., Wang, S., Fan, Z., Bai, J., Chen, K., Liu, X., Wang, J., & Ge, W. et al. (2024). Qwen2-VL: Enhancing vision-language model\u2019s perception of the world at any resolution. arXiv preprint arXiv: 2409.12191."},{"key":"10.1016\/j.eswa.2026.132839_bib0051","unstructured":"Yang, A., Yang, B., Hui, B., Zheng, B., Yu, B., Zhou, C., Li, C., Li, C., Liu, D., & Huang, F. et al. (2024). Qwen2 technical report. arXiv preprint arXiv: 2407.10671."},{"key":"10.1016\/j.eswa.2026.132839_bib0052","series-title":"Computer vision\u2013ECCV 2016: 14th European conference, Amsterdam, the Netherlands, October 11\u201314, 2016, proceedings, part II 14","first-page":"69","article-title":"Modeling context in referring expressions","author":"Yu","year":"2016"},{"key":"10.1016\/j.eswa.2026.132839_bib0053","series-title":"Forty-first international conference on machine learning","article-title":"MM-VET: Evaluating large multimodal models for integrated capabilities","author":"Yu","year":"2024"},{"key":"10.1016\/j.eswa.2026.132839_bib0054","series-title":"Proceedings of the IEEE\/CVF International conference on computer vision","first-page":"11975","article-title":"Sigmoid loss for language image pre-training","author":"Zhai","year":"2023"},{"key":"10.1016\/j.eswa.2026.132839_bib0055","unstructured":"Zhang, P., Wang, X. D. B., Cao, Y., Xu, C., Ouyang, L., Zhao, Z., Ding, S., Zhang, S., Duan, H., Yan, H., & et al. (2023). InternLM-XComposer: A vision-language large model for advanced text-image comprehension and composition. arXiv preprint arXiv: 2309.15112."},{"key":"10.1016\/j.eswa.2026.132839_bib0056","unstructured":"Zhu, J., Wang, W., Chen, Z., Liu, Z., Ye, S., Gu, L., Tian, H., Duan, Y., Su, W., Shao, J. et al. (2025). Internvl3: Exploring advanced training and test-time recipes for open-source multimodal models. arXiv preprint arXiv: 2504.10479."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426017525?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426017525?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T00:05:40Z","timestamp":1783123540000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426017525"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":47,"alternative-id":["S0957417426017525"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132839","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Spatial -aware efficient projector for MLLMs via multi-layer feature aggregation","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132839","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132839"}}