{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:47:09Z","timestamp":1785649629827,"version":"3.56.0"},"publisher-location":"Cham","reference-count":45,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032316653","type":"print"},{"value":"9783032316660","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-31666-0_27","type":"book-chapter","created":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:44:57Z","timestamp":1785649497000},"page":"406-421","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MMFuser: Multimodal Multi-layer Feature Fuser for\u00a0Fine-Grained Vision-Language Understanding"],"prefix":"10.1007","author":[{"given":"Yue","family":"Cao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yong","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yangzhou","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhe","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guangchen","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yong","family":"Fa","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yujie","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Song","family":"Mei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7051-5347","authenticated-orcid":false,"given":"Tong","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,3]]},"reference":[{"key":"27_CR1","unstructured":"Achiam, J., et al.: GPT-4 Technical Report (2023). arXiv:2303.08774 arXiv preprint"},{"key":"27_CR2","unstructured":"Anthropic: The Claude 3 model family: Opus, sonnet, haiku (2024). https:\/\/www.anthropic.com"},{"key":"27_CR3","unstructured":"Bai, J., Bai, S., Yang, S., Wang, S., Tan, S., et\u00a0al.: Qwen-VL: a frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966 (2023)"},{"key":"27_CR4","unstructured":"Chen, J., et\u00a0al.: MiniGPT-V2: large language model as a unified interface for vision-language multi-task learning. arXiv preprint arXiv:2310.09478 (2023)"},{"key":"27_CR5","unstructured":"Chen, K., Zhang, Z., Zeng, W., Zhang, R., et\u00a0al.: Shikra: unleashing multimodal LLM\u2019s referential dialogue magic. arXiv preprint arXiv:2306.15195 (2023)"},{"key":"27_CR6","unstructured":"Chen, X., Wang, X., Changpinyo, S., Piergiovanni, A.: PaLI: a jointly-scaled multilingual language-image model. In: Proceedings of International Conference on Learning Representations (2023)"},{"key":"27_CR7","doi-asserted-by":"crossref","unstructured":"Chen, Z., Wu, J., Wang, W., Su, W., Chen, G., Xing, S., et\u00a0al.: InternVL: scaling up vision foundation models and aligning for generic visual-linguistic tasks. In: Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 24185\u201324198 (2024)","DOI":"10.1109\/CVPR52733.2024.02283"},{"key":"27_CR8","unstructured":"Chiang, W.L., Li, Z., Lin, Z., Sheng, Y., Wu, Z., et\u00a0al.: Vicuna: an open-source chatbot impressing GPT-4 with 90%* ChatGPT quality (2023). https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"27_CR9","doi-asserted-by":"crossref","unstructured":"Dai, W., Li, J., Li, D., Huat, A., Zhao, J., et\u00a0al.: InstructBLIP: towards general-purpose vision-language models with instruction tuning. In: Proceedings of Advances in Neural Information Processing Systems, vol.\u00a036 (2023)","DOI":"10.52202\/075280-2142"},{"key":"27_CR10","unstructured":"Dubey, A., Jauhri, A., Pandey, A., Kadian, A.: The llama 3 herd of models (2024). arXiv:2407.21783 arXiv preprint"},{"key":"27_CR11","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2024.105171","volume":"149","author":"Y Fang","year":"2024","unstructured":"Fang, Y., Sun, Q., Wang, X., Huang, T., Wang, X., Cao, Y.: Eva-02: A visual representation for neon genesis. Image Vision Comput. 149, 105171 (2024)","journal-title":"Image Vision Comput."},{"key":"27_CR12","unstructured":"Fu, C., Chen, P., Shen, Y.: MME: a comprehensive evaluation benchmark for multimodal large language models (2023). arXiv:2306.13394 arXiv preprint"},{"key":"27_CR13","unstructured":"Ge, C., Cheng, S., Wang, Z.: ConvLLaVa: hierarchical backbones as visual encoder for large multimodal models (2024). arXiv:2405.15738 arXiv preprint"},{"key":"27_CR14","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6325\u20136334 (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"27_CR15","doi-asserted-by":"crossref","unstructured":"Gurari, D., Li, Q., Stangl, A.J., Guo, A.: VizWiz grand challenge: answering visual questions from blind people. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3608\u20133617 (2018)","DOI":"10.1109\/CVPR.2018.00380"},{"key":"27_CR16","doi-asserted-by":"crossref","unstructured":"Hudson, D.A., Manning, C.D.: GQA: a new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6700\u20136709 (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"27_CR17","unstructured":"IDEFICS: Introducing IDEFICS: An Open Reproduction of State-of-the-Art Visual Language Model (2023). https:\/\/huggingface.co\/blog\/idefics"},{"key":"27_CR18","doi-asserted-by":"crossref","unstructured":"Kazemzadeh, S., Ordonez, V., Matten, M., Berg, T.: ReferitGame: referring to objects in photographs of natural scenes. In: Proceedings Empirical Methods Natural Langange Process, pp. 787\u2013798 (2014)","DOI":"10.3115\/v1\/D14-1086"},{"key":"27_CR19","doi-asserted-by":"crossref","unstructured":"Krishna, R., Zhu, Y., Groth, O., Johnson, J.: Visual Genome: connecting language and vision using crowdsourced dense image annotations. In: International Journal of Computer Vision, pp. 32\u201373 (2017)","DOI":"10.1007\/s11263-016-0981-7"},{"key":"27_CR20","unstructured":"Li, B., Wang, R., Wang, G., et\u00a0al.: Seed-Bench: benchmarking multimodal LLMs with generative comprehension. arXiv preprint arXiv:2307.16125 (2023)"},{"key":"27_CR21","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of Machine Learning, pp. 19730\u201319742 (2023)"},{"key":"27_CR22","doi-asserted-by":"crossref","unstructured":"Li, Y., Du, Y., Zhou, K., Wang, J.: Evaluating object hallucination in large vision-language models. In: Proceedings of Empirical Methods Natural Langage Process, pp. 292\u2013305 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"27_CR23","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Doll\u00e1r, P., Girshick, R., He, K.: Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2117\u20132125 (2017)","DOI":"10.1109\/CVPR.2017.106"},{"key":"27_CR24","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 26296\u201326306 (2024)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"27_CR25","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: Proceedings of Advances in Neural Information Processing Systems, vol 36 (2024)","DOI":"10.52202\/075280-1516"},{"key":"27_CR26","doi-asserted-by":"crossref","unstructured":"Liu, Y., Duan, H., Zhang, Y., Li, B., Zhang, S., et\u00a0al.: MMBench: is your multi-modal model an all-around player? arXiv preprint arXiv:2307.06281 (2023)","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"27_CR27","doi-asserted-by":"crossref","unstructured":"Liu, Y., Li, Z., Huang, M., Yang, B.: OCRBench: on the hidden mystery of OCR in large multimodal models (2024). arXiv:2305.07895 arXiv preprint","DOI":"10.1007\/s11432-024-4235-6"},{"key":"27_CR28","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.Y.: A convnet for the 2020s. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 11976\u201311986 (2022)","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"27_CR29","doi-asserted-by":"crossref","unstructured":"Lu, P., Mishra, S., Xia, T., Qiu, L.: Learn to explain: multimodal reasoning via thought chains for science question answering. In: Proceedings of Advances in Neural Information Processing Systems (2022)","DOI":"10.52202\/068431-0182"},{"key":"27_CR30","unstructured":"Luo, G., Zhou, Y., Zhang, Y., Zheng, X., Sun, X., Ji, R.: Feast your eyes: mixture-of-resolution adaptation for multimodal large language models (2024). arXiv:2403.03003 arXiv preprint"},{"key":"27_CR31","doi-asserted-by":"crossref","unstructured":"Mao, J., Huang, J., Toshev, A., Camburu, O., Yuille, A.L., Murphy, K.: Generation and comprehension of unambiguous object descriptions. In: Proceedings from the 2016 IEEE Conference on Computer Vision and Pattern Recognition, pp. 11\u201320 (2016)","DOI":"10.1109\/CVPR.2016.9"},{"key":"27_CR32","unstructured":"OpenAI: GPT-4V(ision) System Card (2023). https:\/\/api.semanticscholar.org\/CorpusID:263218031"},{"key":"27_CR33","unstructured":"Oquab, M., Darcet, T., Moutakanni, T.: DINOV2: learning robust visual features without supervision. Trans. Mach. Learn. Res., 1\u201331 (2024)"},{"key":"27_CR34","unstructured":"Radford, A., Kim, J.W., Hallacy, C.: Learning transferable visual models from natural language supervision. In: Proceedings of Machine Learning Research (2021)"},{"key":"27_CR35","unstructured":"Raghu, M., Unterthiner, T., Kornblith, S., Zhang, C., Dosovitskiy, A.: Do vision transformers see like convolutional neural networks?. Proc. Adv. Neural Inf. Process. Syst 34, 12116\u201312128 (2021)"},{"key":"27_CR36","doi-asserted-by":"crossref","unstructured":"Singh, A., et al.: Towards VQA models that can read. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8317\u20138326 (2019)","DOI":"10.1109\/CVPR.2019.00851"},{"key":"27_CR37","doi-asserted-by":"crossref","unstructured":"Tong, S., Liu, Z., Zhai, Y., Ma, Y., LeCun, Y., Xie, S.: Eyes wide shut? Exploring the visual shortcomings of multimodal LLMs. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9568\u20139578 (2024)","DOI":"10.1109\/CVPR52733.2024.00914"},{"key":"27_CR38","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N.: Attention is all you need. In: Advances in Neural Information Processing Systems (NeurIPS) (2017)"},{"key":"27_CR39","unstructured":"Wang, S., Lu, H., Deng, Z.: Fast object detection in compressed video. In: Advances in Neural Information Processing Systems (NeurIPS), pp. 7104\u20137113 (2019)"},{"key":"27_CR40","doi-asserted-by":"crossref","unstructured":"Wang, W., Lv, Q., Yu, W., Hong, W., Qi, J., Wang, Y., , et\u00a0al.: CogVLM: visual expert for pretrained language models. arXiv preprint arXiv:2311.03079 (2023)","DOI":"10.52202\/079017-3860"},{"issue":"3","key":"27_CR41","doi-asserted-by":"publisher","first-page":"415","DOI":"10.1007\/s41095-022-0274-8","volume":"8","author":"W Wang","year":"2022","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.P., et al.: PVTV 2: improved baselines with pyramid vision transformer. Comput. Vis. Media 8(3), 415\u2013424 (2022)","journal-title":"Comput. Vis. Media"},{"key":"27_CR42","doi-asserted-by":"crossref","unstructured":"Yao, H., et al.: Dense connector for MLLMS (2024). arXiv:2405.13800 arXiv preprint","DOI":"10.52202\/079017-1043"},{"key":"27_CR43","unstructured":"Yu, W., Yang, Z., Li, L., Wang, J.: MM-Vet: evaluating large multimodal models for integrated capabilities (2023). arXiv:2308.02490 arXiv preprint"},{"key":"27_CR44","doi-asserted-by":"crossref","unstructured":"Zhai, X., Mustafa, B., Kolesnikov, A., Beyer, L.: Sigmoid loss for language image pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11975\u201311986 (2023)","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"27_CR45","unstructured":"Zhu, X., Su, W., Lu, L., Li, B.: Deformable DETR: deformable transformers for end-to-end object detection. In: Proceedings of the International Conference on Learning Representation (2020)"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-31666-0_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:45:04Z","timestamp":1785649504000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-31666-0_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,3]]},"ISBN":["9783032316653","9783032316660"],"references-count":45,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-31666-0_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,3]]},"assertion":[{"value":"3 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lyon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 August 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}