{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T06:04:19Z","timestamp":1785305059332,"version":"3.55.0"},"reference-count":62,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62572093"],"award-info":[{"award-number":["62572093"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62531022"],"award-info":[{"award-number":["62531022"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Liaoning Provincial Science and Technology Joint Program Project","award":["2024011188-JH2\/1026"],"award-info":[{"award-number":["2024011188-JH2\/1026"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["DUT24RC(3)025"],"award-info":[{"award-number":["DUT24RC(3)025"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s11263-026-02930-z","type":"journal-article","created":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T14:31:21Z","timestamp":1782916281000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["One Aligned LLM to Serve Them All: A Transfer Recipe for Training VLMs without Visual-Language Re-Alignment"],"prefix":"10.1007","volume":"134","author":[{"given":"Jiazuo","family":"Yu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunzhi","family":"Zhuge","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4648-4437","authenticated-orcid":false,"given":"Lu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zichen","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huchuan","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"You","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Long","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,1]]},"reference":[{"key":"2930_CR1","doi-asserted-by":"publisher","first-page":"23716","DOI":"10.52202\/068431-1723","volume":"35","author":"J-B Alayrac","year":"2022","unstructured":"Alayrac, J.-B., Donahue, J., Luc, P., Miech, A., Barr, I., Hasson, Y., Lenc, K., Mensch, A., Millican, K., Reynolds, M., et al. (2022). Flamingo: a visual language model for few-shot learning. Advances in neural information processing systems, 35, 23716\u201323736.","journal-title":"Advances in neural information processing systems"},{"key":"2930_CR2","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J. D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al. (2020). Language models are few-shot learners. Advances in neural information processing systems, 33, 1877\u20131901.","journal-title":"Advances in neural information processing systems"},{"key":"2930_CR3","unstructured":"Chiang, W.-L., Li, Z., Lin, Z., Sheng, Y., Wu, Z., Zhang, H., Zheng, L., Zhuang, S., Zhuang, Y., Gonzalez, J. E., et al. (2023). Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality. Seelmsys. org (accessed 14 April 2023),2(3), 6. https:\/\/vicuna"},{"key":"2930_CR4","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., & Gelly, S. (2020). An image is worth 16x16 words: Transformers for image recognition at scale arXiv preprint. arXiv:2010.11929"},{"key":"2930_CR5","doi-asserted-by":"crossref","unstructured":"Dai, W., Li, J., Li, D., Tiong, A.M.H., Zhao, J., Wang, W., Li, B., Fung, P.N., Hoi, S.: Instructblip: Towards general-purpose vision-language models with instruction tuning. Advances in Neural Information Processing Systems36 (2024)","DOI":"10.52202\/075280-2142"},{"key":"2930_CR6","doi-asserted-by":"crossref","unstructured":"Deng, M., Wang, J., Hsieh, C.-P., Wang, Y., Guo, H., Shu, T., Song, M., Xing, E.P., Hu, Z. (2022) Rlprompt: Optimizing discrete text prompts with reinforcement learning. arXiv preprint arXiv:2205.12548","DOI":"10.18653\/v1\/2022.emnlp-main.222"},{"key":"2930_CR7","doi-asserted-by":"publisher","first-page":"2322","DOI":"10.1109\/TIP.2023.3266887","volume":"32","author":"H Diao","year":"2023","unstructured":"Diao, H., Zhang, Y., Liu, W., Ruan, X., & Lu, H. (2023). Plug-and-play regulators for image-text matching. IEEE Transactions on Image Processing, 32, 2322\u20132334.","journal-title":"IEEE Transactions on Image Processing"},{"key":"2930_CR8","doi-asserted-by":"crossref","unstructured":"Hudson, D. A., Manning, C. D. (2019). Gqa: A new dataset for real-world visual reasoning and compositional question answering. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 6700\u20136709)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"2930_CR9","doi-asserted-by":"crossref","unstructured":"He, J., Wang, Y., Wang, L., Lu, H., He, J.-Y., Lan, J.-P., Luo, B., Xie, X. (2024) Multi-modal instruction tuned llms with fine-grained visual perception. arXiv preprint arXiv:2403.02969","DOI":"10.1109\/CVPR52733.2024.01326"},{"key":"2930_CR10","unstructured":"Han, Y., Zhang, C., Chen, X., Yang, X., Wang, Z., Yu, G., Fu, B., Zhang, H. (2023) Chartllama: A multimodal llm for chart understanding and generation. arXiv preprint arXiv:2311.16483"},{"key":"2930_CR11","unstructured":"Jiang, A. Q., Sablayrolles, A., Mensch, A., Bamford, C., Chaplot, D. S., Casas, D., Bressand, F., Lengyel, G., Lample, G., Saulnier, L., Lavaud, L. R., Lachaux, M.-A., Stock, P., Scao, T. L., Lavril, T., Wang, T., Lacroix, T., & Sayed, W. E. (2023). Mistral 7b arXiv preprint. arXiv:2403.08295"},{"key":"2930_CR12","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A. C., & Lo, W.-Y. (2023). Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 4015\u20134026) Segment anything.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"2930_CR13","unstructured":"Kornblith, S., Norouzi, M., Lee, H., & Hinton, G. (2019). Similarity of neural network representations revisited. International Conference on Machine Learning (pp. 3519\u20133529) PMlR."},{"key":"2930_CR14","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R., & Constant, N. (2021). The power of scale for parameter-efficient prompt tuning arXiv preprint arXiv:2104.08691.","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"2930_CR15","doi-asserted-by":"crossref","unstructured":"Li, Y., Du, Y., Zhou, K., Wang, J., Zhao, W. X., & Wen, J.-R. (2023). Evaluating object hallucination in large vision-language models arXiv preprint. arXiv:2305.10355.","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"issue":"1","key":"2930_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/sdata.2018.251","volume":"5","author":"JJ Lau","year":"2018","unstructured":"Lau, J. J., Gayen, S., Ben Abacha, A., & Demner-Fushman, D. (2018). A dataset of clinically generated visual questions and answers about radiology images. Scientific data, 5(1), 1\u201310.","journal-title":"Scientific data"},{"key":"2930_CR17","unstructured":"Lee, K., Joshi, M., Turc, I. R., Hu, H., Liu, F., Eisenschlos, J. M., Khandelwal, U., Shaw, P., Chang, M.-W., & Toutanova, K. (2023). Pix2struct: Screenshot parsing as pretraining for visual language understanding. International Conference on Machine Learning (pp. 18893\u201318912) PMLR."},{"key":"2930_CR18","doi-asserted-by":"crossref","unstructured":"Li, D., Li, J., Hoi, S. (2024) Blip-diffusion: Pre-trained subject representation for controllable text-to-image generation and editing. Advances in Neural Information Processing Systems36.","DOI":"10.52202\/075280-1312"},{"key":"2930_CR19","unstructured":"Liu, H., Li, C., Li, Y., Li, B., Zhang, Y., Shen, S., Lee, Y.J. (2024) LLaVA-NeXT: Improved reasoning, OCR, and world knowledge https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/."},{"key":"2930_CR20","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S. (2023) Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, pp. 19730\u201319742 PMLR."},{"key":"2930_CR21","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Wu, Q., & Lee, Y. J. (2024). Visual instruction tuning. Advances in neural information processing systems (p. 36).","DOI":"10.52202\/075280-1516"},{"key":"2930_CR22","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S. (2022) Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900 PMLR."},{"key":"2930_CR23","doi-asserted-by":"crossref","unstructured":"Lai, X., Tian, Z., Chen, Y., Li, Y., Yuan, Y., Liu, S., Jia, J. (2024) Lisa: Reasoning segmentation via large language model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9579\u20139589.","DOI":"10.1109\/CVPR52733.2024.00915"},{"key":"2930_CR24","doi-asserted-by":"crossref","unstructured":"Li, C., Wong, C., Zhang, S., Usuyama, N., Liu, H., Yang, J., Naumann, T., Poon, H., Gao, J. (2024) Llava-med: Training a large language-and-vision assistant for biomedicine in one day. Advances in Neural Information Processing Systems36.","DOI":"10.52202\/075280-1240"},{"key":"2930_CR25","doi-asserted-by":"crossref","unstructured":"Li, W., Yuan, Y., Liu, J., Tang, D., Wang, S., Qin, J., Zhu, J., Zhang, L. (2025) Tokenpacker: Efficient visual projector for multimodal llm. International Journal of Computer Vision, 1\u201319.","DOI":"10.1007\/s11263-025-02491-7"},{"key":"2930_CR26","unstructured":"Lester, B., Yurtsever, J., Shakeri, S., & Constant, N. (2022). Reducing retraining by recycling parameter-efficient prompts arXiv preprint. arXiv:2208.05577"},{"key":"2930_CR27","doi-asserted-by":"crossref","unstructured":"Liu, B., Zhan, L.-M., Xu, L., Ma, L., Yang, Y., & Wu, X.-M. (2021). Slake: A semantically-labeled knowledge-enhanced dataset for medical visual question answering. 2021 IEEE 18th International Symposium on Biomedical Imaging (ISBI) (pp. 1650\u20131654). IEEE.","DOI":"10.1109\/ISBI48211.2021.9434010"},{"key":"2930_CR28","doi-asserted-by":"crossref","unstructured":"Mathew, M., Karatzas, D., & Jawahar, C. (2021). Docvqa: A dataset for vqa on document images. Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (pp. 2200\u20132209)","DOI":"10.1109\/WACV48630.2021.00225"},{"key":"2930_CR29","doi-asserted-by":"crossref","unstructured":"Masry, A., Long, D. X., Tan, J. Q., Joty, S., & Hoque, E. (2022). Chartqa: A benchmark for question answering about charts with visual and logical reasoning arXiv preprint. arXiv:2203.10244.","DOI":"10.18653\/v1\/2022.findings-acl.177"},{"key":"2930_CR30","unstructured":"Oquab, M., Darcet, T., Moutakanni, T., Vo, H., Szafraniec, M., Khalidov, V., Fernandez, P., Haziza, D., Massa, F., & El-Nouby, A. (2023). Dinov2: Learning robust visual features without supervision arXiv preprint. arXiv:2304.07193."},{"key":"2930_CR31","doi-asserted-by":"crossref","unstructured":"Pei, B., Huang, Y., Chen, G., Xu, J., Wang, Y., Wang, L., Lu, T., Qiao, Y., Wu, F. (2025). Guiding audio-visual question answering with collective question reasoning. International Journal of Computer Vision, 1\u201318.","DOI":"10.1007\/s11263-025-02510-7"},{"key":"2930_CR32","doi-asserted-by":"crossref","unstructured":"Panagopoulou, A., Xue, L., Yu, N., Li, J., Li, D., Joty, S., Xu, R., Savarese, S., Xiong, C., Niebles, J.C. (2023). X-instructblip: A framework for aligning x-modal instruction-aware representations to llms and emergent cross-modal reasoning. arXiv preprint arXiv:2311.18799","DOI":"10.1007\/978-3-031-72995-9_11"},{"key":"2930_CR33","unstructured":"Qwen2 technical report (2024)"},{"key":"2930_CR34","doi-asserted-by":"crossref","unstructured":"Rohrbach, A., Hendricks, L. A., Burns, K., Darrell, T., Saenko, K. (2018). Object hallucination in image captioning. Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing (pp. 4035\u20134045)","DOI":"10.18653\/v1\/D18-1437"},{"key":"2930_CR35","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J. (2021). Learning transferable visual models from natural language supervision. International Conference on Machine Learning (pp. 8748\u20138763) PMLR."},{"issue":"140","key":"2930_CR36","first-page":"1","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., Shazeer, N., Roberts, A., Lee, K., Narang, S., Matena, M., Zhou, Y., Li, W., & Liu, P. J. (2020). Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of machine learning research, 21(140), 1\u201367.","journal-title":"Journal of machine learning research"},{"issue":"2","key":"2930_CR37","doi-asserted-by":"publisher","first-page":"742","DOI":"10.1007\/s11263-024-02171-y","volume":"133","author":"H Shi","year":"2025","unstructured":"Shi, H., Dao, S. D., & Cai, J. (2025). Llmformer: Large language model for open-vocabulary semantic segmentation. International Journal of Computer Vision, 133(2), 742\u2013759.","journal-title":"International Journal of Computer Vision"},{"key":"2930_CR38","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R. (2018). Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2556\u20132565.","DOI":"10.18653\/v1\/P18-1238"},{"key":"2930_CR39","unstructured":"Schuhmann, C., Vencu, R., Beaumont, R., Kaczmarczyk, R., Mullis, C., Katta, A., Coombes, T., Jitsev, J., Komatsuzaki, A. (2021). Laion-400m: Open dataset of clip-filtered 400 million image-text pairs arXiv preprint. arXiv:2111.02114"},{"key":"2930_CR40","doi-asserted-by":"crossref","unstructured":"Su, Y., Wang, X., Qin, Y., Chan, C.-M., Lin, Y., Wang, H., Wen, K., Liu, Z., Li, P., Li, J., et al. (2021) On transferability of prompt tuning for natural language processing. arXiv preprint arXiv:2111.06719","DOI":"10.18653\/v1\/2022.naacl-main.290"},{"key":"2930_CR41","doi-asserted-by":"crossref","unstructured":"Song, C. H., Wu, J., Washington, C., Sadler, B. M., Chao, W.-L., & Su, Y. (2023). Llm-planner: Few-shot grounded planning for embodied agents with large language models. Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 2998\u20133009).","DOI":"10.1109\/ICCV51070.2023.00280"},{"key":"2930_CR42","doi-asserted-by":"crossref","unstructured":"Tong, S., Brown, E., Wu, P., Woo, S., Middepogu, M., Akula, S.C., Yang, J., Yang, S., Iyer, A., Pan, X., et al. (2024). Cambrian-1: A fully open, vision-centric exploration of multimodal llms. arXiv preprint arXiv:2406.16860.","DOI":"10.52202\/079017-2771"},{"key":"2930_CR43","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., Lachaux, M.-A., Lacroix, T., Rozi\u00e8re, B., Goyal, N., Hambro, E., & Azhar, F. (2023). Llama: Open and efficient foundation language models arXiv preprint. arXiv:2302.13971."},{"key":"2930_CR44","unstructured":"Team, G., Mesnard, T., Hardin, C., Dadashi, R., Bhupatiraju, S., Pathak, S., Sifre, L., Rivi\u00e8re, M., Kale, M. S., & Love, J. (2024). Gemma: Open models based on gemini research and technology arXiv preprint. arXiv:2403.08295"},{"key":"2930_CR45","doi-asserted-by":"crossref","unstructured":"Truong, T.-D., Nguyen, H.-Q., Nguyen, X.-B., Dowling, A., Li, X., Luu, K. (2025). Insect-foundation: A foundation model and large multimodal dataset for vision-language insect understanding. International Journal of Computer Vision, 1\u201326.","DOI":"10.1007\/s11263-025-02521-4"},{"key":"2930_CR46","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I. (2017). Attention is all you need. Advances in neural information processing systems30."},{"key":"2930_CR47","unstructured":"Wang, H., Ge, S., Lipton, Z., Xing, E. P. (2019). Learning robust global representations by penalizing local predictive power. Advances in neural information processing systems, 32,"},{"key":"2930_CR48","doi-asserted-by":"crossref","unstructured":"Wang, J., Ke, L. (2024). Llm-seg: Bridging image segmentation and large language model reasoning. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 1765\u20131774).","DOI":"10.1109\/CVPRW63382.2024.00183"},{"key":"2930_CR49","doi-asserted-by":"crossref","unstructured":"Wang, Z., Rao, Y., Sun, S., Liu, X., Wei, Y., Yu, X., Liu, Z., Wang, Y., Liu, H., Zhou, J., et al. (2025). Vision generalist model: A survey: Z. wang et al. International Journal of Computer Vision, 1\u201329.","DOI":"10.1007\/s11263-025-02502-7"},{"issue":"1","key":"2930_CR50","doi-asserted-by":"publisher","first-page":"224","DOI":"10.1007\/s11263-023-01868-w","volume":"132","author":"Y Wang","year":"2024","unstructured":"Wang, Y., Yu, Z., Wang, J., Heng, Q., Chen, H., Ye, W., Xie, R., Xie, X., & Zhang, S. (2024). Exploring vision-language models for imbalanced learning. International Journal of Computer Vision, 132(1), 224\u2013237.","journal-title":"International Journal of Computer Vision"},{"key":"2930_CR51","doi-asserted-by":"crossref","unstructured":"Wang, Z., Zhao, L., Zhang, J., Song, R., Song, H., Meng, J., Wang, S. (2025). Multi-text guidance is important: Multi-modality image fusion via large generative vision-language model. International Journal of Computer Vision, 1\u201323.","DOI":"10.1007\/s11263-025-02409-3"},{"key":"2930_CR52","doi-asserted-by":"crossref","unstructured":"Yang, J., Chen, X., Qian, S., Madaan, N., Iyengar, M., Fouhey, D. F., Chai, J. (2023). Llm-grounder: Open-vocabulary 3d visual grounding with large language model as an agent arXiv preprint. arXiv:2309.12311","DOI":"10.1109\/ICRA57147.2024.10610443"},{"key":"2930_CR53","doi-asserted-by":"crossref","unstructured":"Yu, L., Poirson, P., Yang, S., Berg, A.C., Berg, T.L. (2016). Modeling context in referring expressions. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11-14, 2016, Proceedings, Part II 14, pp. 69\u201385 Springer.","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"2930_CR54","unstructured":"Yu, J., Xiong, H., Zhang, L., Diao, H., Zhuge, Y., Lanqing, H., Wang, D., Lu, H., He, Y., & Chen, L.Llms can evolve continually on modality for x-modal reasoning. The Thirty-eighth Annual Conference on Neural Information Processing Systems."},{"key":"2930_CR55","doi-asserted-by":"crossref","unstructured":"Zhang, A., Fei, H., Yao, Y., Ji, W., Li, L., Liu, Z., Chua, T.-S. (2024). Vpgtrans: Transfer visual prompt generator across llms. Advances in Neural Information Processing Systems36.","DOI":"10.52202\/075280-0891"},{"issue":"2","key":"2930_CR56","doi-asserted-by":"publisher","first-page":"825","DOI":"10.1007\/s11263-024-02214-4","volume":"133","author":"Y Zang","year":"2025","unstructured":"Zang, Y., Li, W., Han, J., Zhou, K., & Loy, C. C. (2025). Contextual object detection with multimodal large language models. International Journal of Computer Vision, 133(2), 825\u2013843.","journal-title":"International Journal of Computer Vision"},{"key":"2930_CR57","doi-asserted-by":"crossref","unstructured":"Zong, Z., Ma, B., Shen, D., Song, G., Shao, H., Jiang, D., Li, H., Liu, Y. (2024). Mova: Adapting mixture of vision experts to multimodal context. arXiv preprint arXiv:2404.13046.","DOI":"10.52202\/079017-3282"},{"issue":"2","key":"2930_CR58","doi-asserted-by":"publisher","first-page":"844","DOI":"10.1007\/s11263-024-02215-3","volume":"133","author":"W Zhang","year":"2025","unstructured":"Zhang, W., Shen, L., & Foo, C.-S. (2025). Source-free domain adaptation guided by vision and vision-language pre-training. International Journal of Computer Vision, 133(2), 844\u2013866.","journal-title":"International Journal of Computer Vision"},{"key":"2930_CR59","doi-asserted-by":"crossref","unstructured":"Zong, Z., Song, G., & Liu, Y. (2023). Detrs with collaborative hybrid assignments training. Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 6748\u20136758).","DOI":"10.1109\/ICCV51070.2023.00621"},{"key":"2930_CR60","unstructured":"Zhang, S., Xu, Y., Usuyama, N., Xu, H., Bagga, J., Tinn, R., Preston, S., Rao, R., Wei, M., Valluri, N., et al. (2023). Biomedclip: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs. arXiv preprint arXiv:2303.00915."},{"issue":"9","key":"2930_CR61","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C. C., & Liu, Z. (2022). Learning to prompt for vision-language models. International Journal of Computer Vision, 130(9), 2337\u20132348.","journal-title":"International Journal of Computer Vision"},{"issue":"7","key":"2930_CR62","doi-asserted-by":"publisher","first-page":"3891","DOI":"10.1007\/s11263-025-02353-2","volume":"133","author":"P Zhang","year":"2025","unstructured":"Zhang, P., Zhang, J., Cao, J., Li, H., & Jin, L. (2025). Smaller but better: Unifying layout generation with smaller large language models. International Journal of Computer Vision, 133(7), 3891\u20133917.","journal-title":"International Journal of Computer Vision"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02930-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02930-z","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02930-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T05:49:01Z","timestamp":1785304141000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02930-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":62,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["2930"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02930-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"14 August 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 June 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","label":"Competing interests","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"331"}}