{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T15:26:15Z","timestamp":1783610775488,"version":"3.55.0"},"reference-count":71,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s11432-024-4234-4","type":"journal-article","created":{"date-parts":[[2024,12,19]],"date-time":"2024-12-19T02:52:15Z","timestamp":1734576735000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Modality-experts coordinated adaptation for large multimodal models"],"prefix":"10.1007","volume":"67","author":[{"given":"Yan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhong","family":"Ji","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanwei","family":"Pang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jungong","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuelong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,13]]},"reference":[{"key":"4234_CR1","first-page":"4582","volume-title":"Proceedings of Annual Meeting of the Association for Computational Linguistics and International Joint Conference on Natural Language Processing","author":"X L Li","year":"2021","unstructured":"Li X L, Liang P. Prefix-tuning: optimizing continuous prompts for generation. In: Proceedings of Annual Meeting of the Association for Computational Linguistics and International Joint Conference on Natural Language Processing, Bangkok, 2021. 4582\u20134597"},{"key":"4234_CR2","first-page":"2790","volume-title":"Proceedings of International Conference on Machine Learning","author":"N Houlsby","year":"2019","unstructured":"Houlsby N, Giurgiu A, Jastrzebski S, et al. Parameter-efficient transfer learning for NLP. In: Proceedings of International Conference on Machine Learning, Los Angeles, 2019. 2790\u20132799"},{"key":"4234_CR3","volume-title":"Proceedings of International Conference on Learning Representations","author":"E J Hu","year":"2022","unstructured":"Hu E J, Shen Y L, Wallis P, et al. LoRA: low-rank adaptation of large language models. In: Proceedings of International Conference on Learning Representations, 2022"},{"key":"4234_CR4","first-page":"709","volume-title":"Proceedings of European Conference on Computer Vision","author":"M L Jia","year":"2022","unstructured":"Jia M L, Tang L M, Chen B C, et al. Visual prompt tuning. In: Proceedings of European Conference on Computer Vision, Tel Aviv, 2022. 709\u2013727"},{"key":"4234_CR5","volume-title":"Proceedings of International Conference on Learning Representations","author":"T Yang","year":"2023","unstructured":"Yang T, Zhu Y, Xie Y S, et al. AIM: adapting image models for efficient video understanding. In: Proceedings of International Conference on Learning Representations, Kigali, 2023"},{"key":"4234_CR6","volume-title":"Cross-Modal adapter for text-video retrieval","author":"H J Jiang","year":"2022","unstructured":"Jiang H J, Zhang J K, Huang R, et al. Cross-Modal adapter for text-video retrieval. 2022. ArXiv:2211.09623"},{"key":"4234_CR7","first-page":"6580","volume-title":"Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing","author":"Z Long","year":"2024","unstructured":"Long Z, Killick G, McCreadie R, et al. MultiWay-Adapter: adapting multimodal large language models for scalable imagetext retrieval. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, Seoul, 2024. 6580\u20136584"},{"key":"4234_CR8","volume-title":"Proceedings of International Conference on Learning Representations","author":"H Lu","year":"2024","unstructured":"Lu H, Huo Y, Yang G, et al. UniAdapter: unified parameter-efficient transfer learning for cross-modal modeling. In: Proceedings of International Conference on Learning Representations, Vienna, 2024"},{"key":"4234_CR9","first-page":"15752","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"H X Wang","year":"2023","unstructured":"Wang H X, Yang X L, Chang J L, et al. Parameter-efficient tuning of large-scale multimodal foundation model. In: Proceedings of Advances in Neural Information Processing Systems, New Orleans, 2023. 15752\u201315774"},{"key":"4234_CR10","first-page":"1","volume":"61","author":"Y Yuan","year":"2023","unstructured":"Yuan Y, Zhan Y, Xiong Z. Parameter-efficient transfer learning for remote sensing image-text retrieval. IEEE Trans Geosci Remote Sens, 2023, 61: 1\u201314","journal-title":"IEEE Trans Geosci Remote Sens"},{"key":"4234_CR11","volume-title":"Proceedings of International Conference on Learning Representations","author":"A Dosovitskiy","year":"2021","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, et al. An image is worth 16x16 words: transformers for image recognition at scale. In: Proceedings of International Conference on Learning Representations, 2021"},{"key":"4234_CR12","first-page":"4171","volume-title":"Proceedings of Annual Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","author":"J D M W C Kenton","year":"2019","unstructured":"Kenton J D M W C, Toutanova L K. BERT: pretraining of deep bidirectional transformers for language understanding. In: Proceedings of Annual Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Minneapolis, 2019. 4171\u20134186"},{"key":"4234_CR13","first-page":"12888","volume-title":"Proceedings of International Conference on Machine Learning","author":"J N Li","year":"2022","unstructured":"Li J N, Li D X, Xiong C M, et al. BLIP: bootstrapping language-image pretraining for unified vision-language understanding and generation. In: Proceedings of International Conference on Machine Learning, Baltimore, 2022. 12888\u201312900"},{"key":"4234_CR14","first-page":"19730","volume-title":"Proceedings of International Conference on Machine Learning","author":"J N Li","year":"2023","unstructured":"Li J N, Li D X, Savarese S, et al. BLIP-2: bootstrapping language-image pretraining with frozen image encoders and large language models. In: Proceedings of International Conference on Machine Learning, Honolulu, 2023. 19730\u201319742"},{"key":"4234_CR15","volume-title":"OPT: open pre-trained transformer language models","author":"S S Zhang","year":"2022","unstructured":"Zhang S S, Roller S, Goyal N, et al. OPT: open pre-trained transformer language models. 2022. ArXiv:2205.01068"},{"key":"4234_CR16","first-page":"19175","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"W Wang","year":"2023","unstructured":"Wang W, Bao H, Dong L, et al. Image as a foreign language: BEIT pretraining for vision and vision-language tasks. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Vancouver, 2023. 19175\u201319186"},{"key":"4234_CR17","first-page":"3558","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"S Changpinyo","year":"2021","unstructured":"Changpinyo S, Sharma P, Ding N, et al. Conceptual 12M: pushing web-scale image-text pretraining to recognize long-tail visual concepts. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, 2021. 3558\u20133568"},{"key":"4234_CR18","volume-title":"LAION-400M: open dataset of CLIP-filtered 400 million image-text pairs","author":"C Schuhmann","year":"2021","unstructured":"Schuhmann C, Vencu R, Beaumont R, et al. LAION-400M: open dataset of CLIP-filtered 400 million image-text pairs. 2021. ArXiv:2111.02114"},{"key":"4234_CR19","doi-asserted-by":"publisher","first-page":"172104","DOI":"10.1007\/s11432-021-3367-y","volume":"65","author":"Z Ji","year":"2022","unstructured":"Ji Z, Chen K X, He Y Q, et al. Heterogeneous memory enhanced graph reasoning network for cross-modal retrieval. Sci China Inf Sci, 2022, 65: 172104","journal-title":"Sci China Inf Sci"},{"key":"4234_CR20","doi-asserted-by":"publisher","first-page":"7900","DOI":"10.1109\/TCSVT.2023.3281507","volume":"33","author":"Y Zhang","year":"2023","unstructured":"Zhang Y, Ji Z, Pang Y W, et al. Consensus knowledge exploitation for partial query based image retrieval. IEEE Trans Circ Syst Video Technol, 2023, 33: 7900\u20137913","journal-title":"IEEE Trans Circ Syst Video Technol"},{"key":"4234_CR21","doi-asserted-by":"publisher","first-page":"595","DOI":"10.1109\/TIP.2023.3348297","volume":"33","author":"Y Zhang","year":"2024","unstructured":"Zhang Y, Ji Z, Wang D, et al. USER: unified semantic enhancement with momentum contrast for image-text retrieval. IEEE Trans Image Process, 2024, 33: 595\u2013609","journal-title":"IEEE Trans Image Process"},{"key":"4234_CR22","first-page":"1","volume":"61","author":"Z Ji","year":"2023","unstructured":"Ji Z, Meng C, Zhang Y, et al. Knowledge-aided momentum contrastive learning for remote-sensing image text retrieval. IEEE Trans Geosci Remote Sens, 2023, 61: 1\u201313","journal-title":"IEEE Trans Geosci Remote Sens"},{"key":"4234_CR23","doi-asserted-by":"publisher","first-page":"106200","DOI":"10.1016\/j.neunet.2024.106200","volume":"173","author":"Z Ji","year":"2024","unstructured":"Ji Z, Li Z, Zhang Y, et al. Hierarchical matching and reasoning for multi-query image retrieval. Neural Netws, 2024, 173: 106200","journal-title":"Neural Netws"},{"key":"4234_CR24","volume-title":"Proceedings of International Conference on Learning Representations","author":"Z R Wang","year":"2021","unstructured":"Wang Z R, Yu J H, Yu A W, et al. SimVLM: simple visual language model pretraining with weak supervision. In: Proceedings of International Conference on Learning Representations, 2021"},{"key":"4234_CR25","doi-asserted-by":"publisher","first-page":"222102","DOI":"10.1007\/s11432-021-3530-7","volume":"66","author":"Y Yang","year":"2023","unstructured":"Yang Y, Bao R, Guo W L, et al. Deep visual-linguistic fusion network considering cross-modal inconsistency for rumor detection. Sci China Inf Sci, 2023, 66: 222102","journal-title":"Sci China Inf Sci"},{"key":"4234_CR26","first-page":"2425","volume-title":"Proceedings of IEEE International Conference on Computer Vision","author":"S Antol","year":"2015","unstructured":"Antol S, Agrawal A, Lu J, et al. VQA: visual question answering. In: Proceedings of IEEE International Conference on Computer Vision, Santiago, 2015. 2425\u20132433"},{"key":"4234_CR27","first-page":"6418","volume-title":"Proceedings of Annual Meeting of the Association for Computational Linguistics, Bangkok","author":"A Suhr","year":"2019","unstructured":"Suhr A, Zhou S, Zhang A, et al. A corpus for reasoning about natural language grounded in photographs. In: Proceedings of Annual Meeting of the Association for Computational Linguistics, Bangkok, 2019. 6418\u20136428"},{"key":"4234_CR28","first-page":"8748","volume-title":"Proceedings of International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"Radford A, Kim J W, Hallacy C, et al. Learning transferable visual models from natural language supervision. In: Proceedings of International Conference on Machine Learning, 2021. 8748\u20138763"},{"key":"4234_CR29","first-page":"4904","volume-title":"Proceedings of International Conference on Machine Learning","author":"C Jia","year":"2021","unstructured":"Jia C, Yang Y F, Xia Y, et al. Scaling up visual and vision-language representation learning with noisy text supervision. In: Proceedings of International Conference on Machine Learning, 2021. 4904\u20134916"},{"key":"4234_CR30","first-page":"104","volume-title":"Proceedings of European Conference on Computer Vision","author":"Y C Chen","year":"2020","unstructured":"Chen Y C, Li L J, Yu L C, et al. UNITER: universal image-text representation learning. In: Proceedings of European Conference on Computer Vision, Glasgow, 2020. 104\u2013120"},{"key":"4234_CR31","first-page":"121","volume-title":"Proceedings of European Conference on Computer Vision","author":"X J Li","year":"2020","unstructured":"Li X J, Yin X, Li C Y, et al. Oscar: object-semantics aligned pretraining for vision-language tasks. In: Proceedings of European Conference on Computer Vision, Glasgow, 2020. 121\u2013137"},{"key":"4234_CR32","first-page":"5583","volume-title":"Proceedings of International Conference on Machine Learning","author":"W Kim","year":"2021","unstructured":"Kim W, Son B, Kim I. ViLT: vision-and-language transformer without convolution or region supervision. In: Proceedings of International Conference on Machine Learning, 2021. 5583\u20135594"},{"key":"4234_CR33","first-page":"9694","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"J N Li","year":"2021","unstructured":"Li J N, Selvaraju R, Gotmare A, et al. Align before fuse: vision and language representation learning with momentum distillation. In: Proceedings of Advances in Neural Information Processing Systems, 2021. 9694\u20139705"},{"key":"4234_CR34","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou K, Yang J, Loy C C, et al. Learning to prompt for vision-language models. Int J Comput Vis, 2022, 130: 2337\u20132348","journal-title":"Int J Comput Vis"},{"key":"4234_CR35","first-page":"16816","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"K Zhou","year":"2022","unstructured":"Zhou K, Yang J, Loy C C, et al. Conditional prompt learning for vision-language models. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, New Orleans, 2022. 16816\u201316825"},{"key":"4234_CR36","first-page":"19113","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"M U Khattak","year":"2023","unstructured":"Khattak M U, Rasheed H, Maaz M, et al. MaPLe: multimodal prompt learning. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Vancouver, 2023. 19113\u201319122"},{"key":"4234_CR37","volume-title":"Prompt tuning for generative multimodal pretrained models","author":"H Yang","year":"2022","unstructured":"Yang H, Lin J Y, Yang A, et al. Prompt tuning for generative multimodal pretrained models. 2022. ArXiv:2208.02532"},{"key":"4234_CR38","first-page":"3010","volume-title":"Proceedings of IEEE International Conference on Computer Vision","author":"Z Y Hu","year":"2023","unstructured":"Hu Z Y, Li Y Y, Lyu M R, et al. VL-PET: vision-and-language parameter-efficient tuning via granularity control. In: Proceedings of IEEE International Conference on Computer Vision, Paris, 2023. 3010\u20133020"},{"key":"4234_CR39","first-page":"5227","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"Y L Sung","year":"2022","unstructured":"Sung Y L, Cho J, Bansal M. Vl-Adapter: parameter-efficient transfer learning for vision-and-language tasks. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, New Orleans, 2022. 5227\u20135237"},{"key":"4234_CR40","first-page":"34892","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"H T Liu","year":"2023","unstructured":"Liu H T, Li C Y, Wu Q Y, et al. Visual instruction tuning. In: Proceedings of Advances in Neural Information Processing Systems, New Orleans, 2023. 34892\u201334916"},{"key":"4234_CR41","first-page":"1931","volume-title":"Proceedings of International Conference on Machine Learning","author":"J Cho","year":"2021","unstructured":"Cho J, Lei J, Tan H, et al. Unifying vision-and-language tasks via text generation. In: Proceedings of International Conference on Machine Learning, 2021. 1931\u20131942"},{"key":"4234_CR42","first-page":"29615","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"G Luo","year":"2023","unstructured":"Luo G, Zhou Y Y, Ren T H, et al. Cheap and quick: efficient vision-language instruction tuning for large language models. In: Proceedings of Advances in Neural Information Processing Systems, New Orleans, 2023. 29615\u201329627"},{"key":"4234_CR43","first-page":"16664","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"S F Chen","year":"2022","unstructured":"Chen S F, Ge C J, Tong Z, et al. AdaptFormer: adapting vision transformers for scalable visual recognition. In: Proceedings of Advances in Neural Information Processing Systems, New Orleans, 2022. 16664\u201316678"},{"key":"4234_CR44","volume-title":"Microsoft COCO captions: data collection and evaluation server","author":"X L Chen","year":"2015","unstructured":"Chen X L, Fang H, Lin T Y, et al. Microsoft COCO captions: data collection and evaluation server. 2015. ArXiv:1504.00325"},{"key":"4234_CR45","first-page":"2641","volume-title":"Proceedings of IEEE International Conference on Computer Vision","author":"B A Plummer","year":"2015","unstructured":"Plummer B A, Wang L W, Cervantes C M, et al. Flickr30K entities: collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of IEEE International Conference on Computer Vision, Santiago, 2015. 2641\u20132649"},{"key":"4234_CR46","volume-title":"Visual entailment: a novel task for fine-grained image understanding","author":"N Xie","year":"2019","unstructured":"Xie N, Lai F, Doran D, et al. Visual entailment: a novel task for fine-grained image understanding. 2019. ArXiv:1901.06706"},{"key":"4234_CR47","first-page":"6904","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"Y Goyal","year":"2017","unstructured":"Goyal Y, Khot T, Summers-Stay D, et al. Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Hawaii, 2017. 6904\u20136913"},{"key":"4234_CR48","first-page":"3128","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"K Andrej","year":"2015","unstructured":"Andrej K, Li F F. Deep visual-semantic alignments for generating image descriptions. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Boston, 2015. 3128\u20133137"},{"key":"4234_CR49","first-page":"8948","volume-title":"Proceedings of IEEE International Conference on Computer Vision","author":"H Agrawal","year":"2019","unstructured":"Agrawal H, Desai K, Wang Y F, et al. NoCaps: novel object captioning at scale. In: Proceedings of IEEE International Conference on Computer Vision, Seoul, 2019. 8948\u20138957"},{"key":"4234_CR50","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O, et al. Visual Genome: connecting language and vision using crowdsourced dense image annotations. Int J Comput Vis, 2017, 123: 32\u201373","journal-title":"Int J Comput Vis"},{"key":"4234_CR51","first-page":"6700","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"D A Hudson","year":"2019","unstructured":"Hudson D A, Manning C D. GQA: a new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Long Beach, 2019. 6700\u20136709"},{"key":"4234_CR52","first-page":"2507","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"P Lu","year":"2022","unstructured":"Lu P, Mishra S, Xia T, et al. Learn to explain: multimodal reasoning via thought chains for science question answering. In: Proceedings of Advances in Neural Information Processing Systems, New Orleans, 2022. 2507\u20132521"},{"key":"4234_CR53","volume-title":"MME: a comprehensive evaluation benchmark for multimodal large language models","author":"C Y Fu","year":"2023","unstructured":"Fu C Y, Chen P X, Shen Y H, et al. MME: a comprehensive evaluation benchmark for multimodal large language models. 2023. ArXiv:2306.13394"},{"key":"4234_CR54","volume-title":"Proceedings of International Conference on Machine Learning","author":"W Yu","year":"2024","unstructured":"Yu W, Yang Z, Li L, et al. MM-Vet: evaluating large multimodal models for integrated capabilities. In: Proceedings of International Conference on Machine Learning, Vienna, 2024"},{"key":"4234_CR55","first-page":"292","volume-title":"Proceedings of Empirical Methods in Natural Language Processing","author":"Y Li","year":"2023","unstructured":"Li Y, Du Y, Zhou K, et al. Evaluating object hallucination in large vision-language models. In: Proceedings of Empirical Methods in Natural Language Processing, Sentosa, 2023. 292\u2013305"},{"key":"4234_CR56","first-page":"3608","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"D Gurari","year":"2018","unstructured":"Gurari D, Li Q, Stangl A J, et al. Vizwiz grand challenge: answering visual questions from blind people. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Salt Lake City, 2018. 3608\u20133617"},{"key":"4234_CR57","volume-title":"MMBench: is your multi-modal model an all-around player?","author":"Y Liu","year":"2023","unstructured":"Liu Y, Duan H, Zhang Y, et al. MMBench: is your multi-modal model an all-around player? 2023. ArXiv:2307.06281"},{"key":"4234_CR58","first-page":"8317","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"A Singh","year":"2019","unstructured":"Singh A, Natarajan V, Shah M, et al. Towards VQA models that can read. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Long Beach, 2019. 8317\u20138326"},{"key":"4234_CR59","first-page":"26296","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"H T Liu","year":"2024","unstructured":"Liu H T, Li C Y, Li Y H, et al. Improved baselines with visual instruction tuning. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Seattle, 2024. 26296\u201326306"},{"key":"4234_CR60","first-page":"31","volume-title":"Proceedings of Annual Meeting of the Association for Computational Linguistics","author":"D X Li","year":"2023","unstructured":"Li D X, Li J N, Le H, et al. LAVIS: a one-stop library for language-vision intelligence. In: Proceedings of Annual Meeting of the Association for Computational Linguistics, Toronto, 2023. 31\u201341"},{"key":"4234_CR61","first-page":"1026","volume-title":"Proceedings of IEEE International Conference on Computer Vision","author":"K M He","year":"2015","unstructured":"He K M, Zhang X Y, Ren S Q, et al. Delving deep into rectifiers: surpassing human-level performance on imagenet classification. In: Proceedings of IEEE International Conference on Computer Vision, Santiago, 2015. 1026\u20131034"},{"key":"4234_CR62","first-page":"6616","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"Z Gan","year":"2020","unstructured":"Gan Z, Chen Y C, Li L J, et al. Large-scale adversarial training for vision-and-language representation learning. In: Proceedings of Advances in Neural Information Processing Systems, New Orleans, 2020. 6616\u20136628"},{"key":"4234_CR63","first-page":"61","volume-title":"Proceedings of Annual Meeting of the Association for Computational Linguistics","author":"X Liu","year":"2022","unstructured":"Liu X, Ji K X, Fu Y C, et al. P-Tuning v2: prompt tuning can be comparable to fine-tuning universally across scales and tasks. In: Proceedings of Annual Meeting of the Association for Computational Linguistics, Dublin, 2022. 61\u201368"},{"key":"4234_CR64","volume-title":"Proceedings of International Conference on Learning Representations","author":"R R Zhang","year":"2024","unstructured":"Zhang R R, Han J M, Liu C, et al. LLaMA-Adapter: efficient fine-tuning of language models with zero-init attention. In: Proceedings of International Conference on Learning Representations, Vienna, 2024"},{"key":"4234_CR65","first-page":"5579","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"P C Zhang","year":"2021","unstructured":"Zhang P C, Li X J, Hu X W, et al. VinVL: revisiting visual representations in vision-language models. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, 2021. 5579\u20135588"},{"key":"4234_CR66","first-page":"5100","volume-title":"Proceedings of Empirical Methods in Natural Language Processing and International Joint Conference on Natural Language Processing","author":"H Tan","year":"2019","unstructured":"Tan H, Bansal M. LXMERT: learning cross-modality encoder representations from transformers. In: Proceedings of Empirical Methods in Natural Language Processing and International Joint Conference on Natural Language Processing, 2019. 5100\u20135111"},{"key":"4234_CR67","volume-title":"Vicuna: an open-source chatbot impressing GPT-4 with 90%* ChatGPT quality","author":"W L Chiang","year":"2023","unstructured":"Chiang W L, Li Z H, Lin Z, et al. Vicuna: an open-source chatbot impressing GPT-4 with 90%* ChatGPT quality. 2023. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"4234_CR68","first-page":"2263","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"A Masry","year":"2022","unstructured":"Masry A, Do X L, Tan J Q, et al. ChartQA: a benchmark for question answering about charts with visual and logical reasoning. In: Proceedings of Findings of the Association for Computational Linguistics, 2022. 2263\u20132279"},{"key":"4234_CR69","first-page":"2200","volume-title":"Proceedings of IEEE Winter Conference on Applications of Computer Vision","author":"M Mathew","year":"2021","unstructured":"Mathew M, Karatzas D, Jawahar C V. DocVQA: a dataset for VQA on document images. In: Proceedings of IEEE Winter Conference on Applications of Computer Vision, 2021. 2200\u20132209"},{"key":"4234_CR70","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4231-5","volume-title":"How far are we to GPT-4V? Closing the gap to commercial multimodal models with open-source suites","author":"Z Chen","year":"2024","unstructured":"Chen Z, Wang W Y, Tian H, et al. How far are we to GPT-4V? Closing the gap to commercial multimodal models with open-source suites. 2024. ArXiv:2404.16821"},{"key":"4234_CR71","volume-title":"Qwen2-VL: enhancing vision-language model\u2019s perception of the world at any resolution","author":"P Wang","year":"2024","unstructured":"Wang P, Bai S, Tan S N, et al. Qwen2-VL: enhancing vision-language model\u2019s perception of the world at any resolution. 2024. ArXiv:2409.12191"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4234-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-024-4234-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4234-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,19]],"date-time":"2026-01-19T22:02:58Z","timestamp":1768860178000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-024-4234-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":71,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["4234"],"URL":"https:\/\/doi.org\/10.1007\/s11432-024-4234-4","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12]]},"assertion":[{"value":"8 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 September 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 November 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 December 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"220107"}}