{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:07:57Z","timestamp":1777655277695,"version":"3.51.4"},"reference-count":81,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s11432-024-4187-3","type":"journal-article","created":{"date-parts":[[2024,12,13]],"date-time":"2024-12-13T01:43:36Z","timestamp":1734054216000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["MMInstruct: a high-quality multi-modal instruction tuning dataset with extensive diversity"],"prefix":"10.1007","volume":"67","author":[{"given":"Yangzhou","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yue","family":"Cao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhangwei","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weiyun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhe","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenhai","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Tian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lewei","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xizhou","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tong","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jifeng","family":"Dai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,10]]},"reference":[{"key":"4187_CR1","volume-title":"LLaMA: open and efficient foundation language models","author":"H Touvron","year":"2023","unstructured":"Touvron H, Lavril T, Izacard G, et al. LLaMA: open and efficient foundation language models. 2023. ArXiv:2302.13971"},{"key":"4187_CR2","volume-title":"Internlm: A multilingual language model with progressively enhanced capabilities","author":"I Team","year":"2023","unstructured":"Team I. Internlm: A multilingual language model with progressively enhanced capabilities. 2023. https:\/\/static.aminer.cn\/upload\/pdf\/127\/1564\/656\/6481884993eaf7045294a0c4_0.pdf"},{"key":"4187_CR3","volume-title":"LLaMA 2: open foundation and fine-tuned chat models","author":"H Touvron","year":"2023","unstructured":"Touvron H, Martin L, Stone K, et al. LLaMA 2: open foundation and fine-tuned chat models. 2023. ArXiv:2307.09288"},{"key":"4187_CR4","volume-title":"Vicuna: an open-source chatbot impressing GPT-4 with 90% quality","author":"W L Chiang","year":"2023","unstructured":"Chiang W L, Li Z, Lin Z, et al. Vicuna: an open-source chatbot impressing GPT-4 with 90% quality. 2023. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"4187_CR5","first-page":"26296","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"H Liu","year":"2024","unstructured":"Liu H, Li C, Li Y, et al. Improved baselines with visual instruction tuning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2024. 26296\u201326306"},{"key":"4187_CR6","volume-title":"The all-seeing project V2: towards general relation comprehension of the open world","author":"W Wang","year":"2024","unstructured":"Wang W, Ren Y, Luo H, et al. The all-seeing project V2: towards general relation comprehension of the open world. 2024. ArXiv:2402.19474"},{"key":"4187_CR7","volume-title":"ShareGPT4V: improving large multi-modal models with better captions","author":"L Chen","year":"2023","unstructured":"Chen L, Li J, Dong X, et al. ShareGPT4V: improving large multi-modal models with better captions. 2023. ArXiv:2311.12793"},{"key":"4187_CR8","first-page":"740","volume-title":"Proceedings of the European Conference on Computer Vision","author":"T Y Lin","year":"2014","unstructured":"Lin T Y, Maire M, Belongie S, et al. Microsoft COCO: common objects in context. In: Proceedings of the European Conference on Computer Vision, 2014. 740\u2013755"},{"key":"4187_CR9","volume-title":"M3IT: a large-scale dataset towards multi-modal multilingual instruction tuning","author":"L Li","year":"2023","unstructured":"Li L, Yin Y, Li S, et al. M3IT: a large-scale dataset towards multi-modal multilingual instruction tuning. 2023. ArXiv:2306.04387"},{"key":"4187_CR10","volume-title":"Shikra: unleashing multimodal LLM\u2019s referential dialogue magic","author":"K Chen","year":"2023","unstructured":"Chen K, Zhang Z, Zeng W, et al. Shikra: unleashing multimodal LLM\u2019s referential dialogue magic. 2023. ArXiv:2306.15195"},{"key":"4187_CR11","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"W Dai","year":"2023","unstructured":"Dai W, Li J, Li D, et al. InstructBLIP: towards general-purpose vision-language models with instruction tuning. In: Proceedings of the Advances in Neural Information Processing Systems, 2023"},{"key":"4187_CR12","first-page":"11445","volume-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics","author":"Z Xu","year":"2023","unstructured":"Xu Z, Shen Y, Huang L. MultiInstruct: improving multi-modal zero-shot learning via instruction tuning. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, 2023. 11445\u201311465"},{"key":"4187_CR13","first-page":"15271","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"Z Xu","year":"2024","unstructured":"Xu Z, Feng C, Shao R, et al. Vision-flan: scaling human-labeled tasks in visual instruction tuning. In: Proceedings of Findings of the Association for Computational Linguistics, 2024. 15271\u201315342"},{"key":"4187_CR14","volume-title":"Qwen-VL: a frontier large vision-language model with versatile abilities","author":"J Bai","year":"2023","unstructured":"Bai J, Bai S, Yang S, et al. Qwen-VL: a frontier large vision-language model with versatile abilities. 2023. ArXiv:2308.12966"},{"key":"4187_CR15","volume-title":"MME: a comprehensive evaluation benchmark for multimodal large language models","author":"C Fu","year":"2023","unstructured":"Fu C, Chen P, Shen Y, et al. MME: a comprehensive evaluation benchmark for multimodal large language models. 2023. ArXiv:2306.13394"},{"key":"4187_CR16","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"H Liu","year":"2024","unstructured":"Liu H, Li C, Wu Q, et al. Visual instruction tuning. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4187_CR17","first-page":"8748","volume-title":"Proceedings of the International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"Radford A, Kim J W, Hallacy C, et al. Learning transferable visual models from natural language supervision. In: Proceedings of the International Conference on Machine Learning, 2021. 8748\u20138763"},{"key":"4187_CR18","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems","author":"C Jia","year":"2021","unstructured":"Jia C, Yang Y, Xia Y, et al. Scaling up visual and vision-language representation learning with noisy text supervision. In: Proceedings of the International Conference on Neural Information Processing Systems, 2021"},{"key":"4187_CR19","volume-title":"EVA-CLIP: improved training techniques for CLIP at scale","author":"Q Sun","year":"2023","unstructured":"Sun Q, Fang Y, Wu L, et al. EVA-CLIP: improved training techniques for CLIP at scale. 2023. ArXiv:2303.15389"},{"key":"4187_CR20","volume-title":"Proceedings of the International Conference on Learning Representations","author":"W Su","year":"2020","unstructured":"Su W, Zhu X, Cao Y, et al. VL-BERT: pre-training of generic visual-linguistic representations. In: Proceedings of the International Conference on Learning Representations, 2020"},{"key":"4187_CR21","volume-title":"VL-BEiT: generative vision-language pretraining","author":"H Bao","year":"2022","unstructured":"Bao H, Wang W, Dong L, et al. VL-BEiT: generative vision-language pretraining. 2022. ArXiv:2206.01127"},{"key":"4187_CR22","first-page":"9694","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"J Li","year":"2021","unstructured":"Li J, Selvaraju R, Gotmare A, et al. Align before fuse: vision and language representation learning with momentum distillation. In: Proceedings of the Advances in Neural Information Processing Systems, 2021. 9694\u20139705"},{"key":"4187_CR23","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"H Bao","year":"2022","unstructured":"Bao H, Wang W, Dong L, et al. VLMo: unified vision-language pre-training with mixture-of-modality-experts. In: Proceedings of the Advances in Neural Information Processing Systems, 2022"},{"key":"4187_CR24","first-page":"19175","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"W Wang","year":"2023","unstructured":"Wang W, Bao H, Dong L, et al. Image as a foreign language: Beit pretraining for vision and vision-language tasks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2023. 19175\u201319186"},{"key":"4187_CR25","volume-title":"CoCa: contrastive captioners are image-text foundation models","author":"J Yu","year":"2022","unstructured":"Yu J, Wang Z, Vasudevan V, et al. CoCa: contrastive captioners are image-text foundation models. 2022. ArXiv:2205.01917"},{"key":"4187_CR26","first-page":"16804","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"X Zhu","year":"2022","unstructured":"Zhu X, Zhu J, Li H, et al. Uni-Perceiver: pre-training unified architecture for generic perception for zero-shot and few-shot tasks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2022. 16804\u201316815"},{"key":"4187_CR27","first-page":"2664","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"J Zhu","year":"2022","unstructured":"Zhu J, Zhu X, Wang W, et al. Uni-Perceiver-MoE: learning sparse generalist models with conditional MoEs. In: Proceedings of the Advances in Neural Information Processing Systems, 2022. 2664\u20132678"},{"key":"4187_CR28","first-page":"2691","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"H Li","year":"2023","unstructured":"Li H, Zhu J, Jiang X, et al. Uni-Perceiver v2: a generalist model for large-scale vision and vision-language tasks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2023. 2691\u20132700"},{"key":"4187_CR29","first-page":"23716","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"J B Alayrac","year":"2022","unstructured":"Alayrac J B, Donahue J, Luc P, et al. Flamingo: a visual language model for few-shot learning. In: Proceedings of the Advances in Neural Information Processing Systems, 2022. 23716\u201323736"},{"key":"4187_CR30","first-page":"24185","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"Z Chen","year":"2024","unstructured":"Chen Z, Wu J, Wang W, et al. InternVL: scaling up vision foundation models and aligning for generic visual-linguistic tasks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2024. 24185\u201324198"},{"key":"4187_CR31","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4231-5","volume-title":"How far are we to GPT-4V? Closing the gap to commercial multimodal models with open-source suites","author":"Z Chen","year":"2024","unstructured":"Chen Z, Wang W, Tian H, et al. How far are we to GPT-4V? Closing the gap to commercial multimodal models with open-source suites. 2024. ArXiv:2404.16821"},{"key":"4187_CR32","volume-title":"InternGPT: solving vision-centric tasks by interacting with chatbots beyond language","author":"Z Liu","year":"2023","unstructured":"Liu Z, He Y, Wang W, et al. InternGPT: solving vision-centric tasks by interacting with chatbots beyond language. 2023. ArXiv:2305.05662"},{"key":"4187_CR33","volume-title":"Proceedings of the International Conference on Learning Representations","author":"X Chen","year":"2023","unstructured":"Chen X, Wang X, Changpinyo S, et al. PaLI: a jointly-scaled multilingual language-image model. In: Proceedings of the International Conference on Learning Representations, 2023"},{"key":"4187_CR34","first-page":"9579","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"X Lai","year":"2024","unstructured":"Lai X, Tian Z, Chen Y, et al. LISA: reasoning segmentation via large language model. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2024. 9579\u20139589"},{"key":"4187_CR35","volume-title":"Proceedings of the International Conference on Learning Representations","author":"W Wang","year":"2024","unstructured":"Wang W, Shi M, Li Q, et al. The all-seeing project: towards panoptic visual recognition and understanding of the open world. In: Proceedings of the International Conference on Learning Representations, 2024"},{"key":"4187_CR36","volume-title":"MM-interleaved: interleaved image-text generative modeling via multi-modal feature synchronizer","author":"C Tian","year":"2024","unstructured":"Tian C, Zhu X, Xiong Y, et al. MM-interleaved: interleaved image-text generative modeling via multi-modal feature synchronizer. 2024. ArXiv:2401.10208"},{"key":"4187_CR37","volume-title":"VisionLLM v2: an end-to-end generalist multimodal large language model for hundreds of vision-language tasks","author":"J Wu","year":"2024","unstructured":"Wu J, Zhong M, Xing S, et al. VisionLLM v2: an end-to-end generalist multimodal large language model for hundreds of vision-language tasks. 2024. ArXiv:2406.08394"},{"key":"4187_CR38","volume-title":"InternLM-XComposer2-4KHD: a pioneering large vision-language model handling resolutions from 336 pixels to 4K HD","author":"X Dong","year":"2024","unstructured":"Dong X, Zhang P, Zang Y, et al. InternLM-XComposer2-4KHD: a pioneering large vision-language model handling resolutions from 336 pixels to 4K HD. 2024. ArXiv:2404.06512"},{"key":"4187_CR39","first-page":"200","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"M Tsimpoukelli","year":"2021","unstructured":"Tsimpoukelli M, Menick J L, Cabi S, et al. Multimodal few-shot learning with frozen language models. In: Proceedings of the Advances in Neural Information Processing Systems, 2021. 200\u2013212"},{"key":"4187_CR40","first-page":"18030","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"J Chen","year":"2022","unstructured":"Chen J, Guo H, Yi K, et al. VisualGPT: data-efficient adaptation of pretrained language models for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2022. 18030\u201318040"},{"key":"4187_CR41","first-page":"19730","volume-title":"Proceedings of the International Conference on Machine Learning","author":"J Li","year":"2023","unstructured":"Li J, Li D, Savarese S, et al. BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of the International Conference on Machine Learning, 2023. 19730\u201319742"},{"key":"4187_CR42","volume-title":"Gemini: a family of highly capable multimodal models","author":"G Team","year":"2023","unstructured":"Team G, Anil R, Borgeaud S, et al. Gemini: a family of highly capable multimodal models. 2023. ArXiv:2312.11805"},{"key":"4187_CR43","volume-title":"Gemini 1.5: unlocking multimodal understanding across millions of tokens of context","author":"M Reid","year":"2024","unstructured":"Reid M, Savinov N, Teplyashin D, et al. Gemini 1.5: unlocking multimodal understanding across millions of tokens of context. 2024. ArXiv:2403.05530"},{"key":"4187_CR44","first-page":"27730","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"L Ouyang","year":"2022","unstructured":"Ouyang L, Wu J, Jiang X, et al. Training language models to follow instructions with human feedback. In: Proceedings of the Advances in Neural Information Processing Systems, 2022. 27730\u201327744"},{"key":"4187_CR45","first-page":"13484","volume-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics","author":"Y Wang","year":"2023","unstructured":"Wang Y, Kordi Y, Mishra S, et al. Self-instruct: aligning language models with self-generated instructions. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, 2023. 13484\u201313508"},{"key":"4187_CR46","volume-title":"Benchmarking generalization via in-context instructions on 1,600+ language tasks","author":"Y Wang","year":"2022","unstructured":"Wang Y, Mishra S, Alipoormolabashi P, et al. Benchmarking generalization via in-context instructions on 1,600+ language tasks. 2022. ArXiv:2204.07705"},{"key":"4187_CR47","volume-title":"GPT-4 technical report","author":"J Achiam","year":"2023","unstructured":"Achiam J, Adler S, Agarwal S, et al. GPT-4 technical report. 2023. ArXiv:2303.08774"},{"key":"4187_CR48","volume-title":"Proceedings of the International Conference on Learning Representations","author":"D Zhu","year":"2024","unstructured":"Zhu D, Chen J, Shen X, et al. MiniGPT-4: enhancing vision-language understanding with advanced large language models. In: Proceedings of the International Conference on Learning Representations, 2024"},{"key":"4187_CR49","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"Z Yin","year":"2024","unstructured":"Yin Z, Wang J, Cao J, et al. LAMM: language-assisted multi-modal instruction-tuning dataset, framework, and benchmark. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4187_CR50","volume-title":"MIMIC-IT: multi-modal in-context instruction tuning","author":"B Li","year":"2023","unstructured":"Li B, Zhang Y, Chen L, et al. MIMIC-IT: multi-modal in-context instruction tuning. 2023. ArXiv:2306.05425"},{"key":"4187_CR51","volume-title":"Macaw-LLM: multi-modal language modeling with image, audio, video, and text integration","author":"C Lyu","year":"2023","unstructured":"Lyu C, Wu M, Wang L, et al. Macaw-LLM: multi-modal language modeling with image, audio, video, and text integration. 2023. ArXiv:2306.09093"},{"key":"4187_CR52","volume-title":"VideoChat: chat-centric video understanding","author":"K Li","year":"2023","unstructured":"Li K, He Y, Wang Y, et al. VideoChat: chat-centric video understanding. 2023. ArXiv:2305.06355"},{"key":"4187_CR53","first-page":"14313","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"S Ren","year":"2024","unstructured":"Ren S, Yao L, Li S, et al. TimeChat: a time-sensitive multimodal large language model for long video understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2024. 14313\u201314323"},{"key":"4187_CR54","volume-title":"Valley: video assistant with large language model enhanced ability","author":"R Luo","year":"2023","unstructured":"Luo R, Zhao Z, Yang M, et al. Valley: video assistant with large language model enhanced ability. 2023. ArXiv:2306.07207"},{"key":"4187_CR55","first-page":"2507","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"P Lu","year":"2022","unstructured":"Lu P, Mishra S, Xia T, et al. Learn to explain: multimodal reasoning via thought chains for science question answering. In: Proceedings of the Advances in Neural Information Processing Systems, 2022. 2507\u20132521"},{"key":"4187_CR56","first-page":"9556","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"X Yue","year":"2024","unstructured":"Yue X, Ni Y, Zhang K, et al. MMMU: a massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2024. 9556\u20139567"},{"key":"4187_CR57","volume-title":"LLaVAR: enhanced visual instruction tuning for text-rich image understanding","author":"Y Zhang","year":"2023","unstructured":"Zhang Y, Zhang R, Gu J, et al. LLaVAR: enhanced visual instruction tuning for text-rich image understanding. 2023. ArXiv:2306.17107"},{"key":"4187_CR58","volume-title":"mPLUG-DocOwl: modularized multimodal large language model for document understanding","author":"J Ye","year":"2023","unstructured":"Ye J, Hu A, Xu H, et al. mPLUG-DocOwl: modularized multimodal large language model for document understanding. 2023. ArXiv:2307.02499"},{"key":"4187_CR59","first-page":"19071","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"R Tanaka","year":"2024","unstructured":"Tanaka R, Iki T, Nishida K, et al. InstructDoc: a dataset for zero-shot generalization of visual document understanding with instructions. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2024. 19071\u201319079"},{"key":"4187_CR60","volume-title":"Aligning large multi-modal model with robust instruction tuning","author":"F Liu","year":"2023","unstructured":"Liu F, Lin K, Li L, et al. Aligning large multi-modal model with robust instruction tuning. 2023. ArXiv:2306.14565"},{"key":"4187_CR61","first-page":"25278","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"C Schuhmann","year":"2022","unstructured":"Schuhmann C, Beaumont R, Vencu R, et al. LAION-5B: an open large-scale dataset for training next generation image-text models. In: Proceedings of the Advances in Neural Information Processing Systems, 2022. 25278\u201325294"},{"key":"4187_CR62","first-page":"1466","volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing","author":"M Seo","year":"2015","unstructured":"Seo M, Hajishirzi H, Farhadi A, et al. Solving geometry problems: combining text and diagram interpretation. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, 2015. 1466\u20131476"},{"key":"4187_CR63","first-page":"3313","volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing","author":"J Chen","year":"2022","unstructured":"Chen J, Li T, Qin J, et al. UniGeo: unifying geometry logical reasoning via reformulating mathematical expression. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, 2022. 3313\u20133323"},{"key":"4187_CR64","first-page":"1511","volume-title":"Proceedings of the 29th International Conference on Computational Linguistics","author":"J Cao","year":"2022","unstructured":"Cao J, Xiao J. An augmented benchmark dataset for geometric question answering through dual parallel text encoding. In: Proceedings of the 29th International Conference on Computational Linguistics, 2022. 1511\u20131520"},{"key":"4187_CR65","first-page":"6774","volume-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing","author":"P Lu","year":"2021","unstructured":"Lu P, Gong R, Jiang S, et al. Inter-GPS: interpretable geometry problem solving with formal language and symbolic reasoning. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, 2021. 6774\u20136786"},{"key":"4187_CR66","first-page":"155","volume-title":"Proceedings of the International Joint Conference on Learning & Reasoning","author":"A D Lindstr\u00f6m","year":"2022","unstructured":"Lindstr\u00f6m A D, Abraham S S. CLEVR-Math: a dataset for compositional language, visual and mathematical reasoning. In: Proceedings of the International Joint Conference on Learning & Reasoning, 2022. 155\u2013170"},{"key":"4187_CR67","first-page":"14963","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"Z Li","year":"2023","unstructured":"Li Z, Wang X, Stengel-Eskin E, et al. Super-CLEVR: a virtual benchmark to diagnose domain robustness in visual reasoning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2023. 14963\u201314973"},{"key":"4187_CR68","volume-title":"Proceedings of the International Conference on Learning Representations","author":"P Lu","year":"2023","unstructured":"Lu P, Qiu L, Weichang K, et al. Dynamic prompt learning via policy gradient for semi-structured mathematical reasoning. In: Proceedings of the International Conference on Learning Representations, 2023"},{"key":"4187_CR69","first-page":"5648","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"K Kafle","year":"2018","unstructured":"Kafle K, Price B, Cohen S, et al. DVQA: understanding data visualizations via question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018. 5648\u20135656"},{"key":"4187_CR70","volume-title":"Proceedings of the International Conference on Learning Representations","author":"S E Kahou","year":"2018","unstructured":"Kahou S E, Michalski V, Atkinson A, et al. FigureQA: an annotated figure dataset for visual reasoning. In: Proceedings of the International Conference on Learning Representations, 2018"},{"key":"4187_CR71","first-page":"4999","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"A Kembhavi","year":"2017","unstructured":"Kembhavi A, Seo M, Schwenk D, et al. Are you smarter than a sixth grader? Textbook question answering for multimodal machine comprehension. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017. 4999\u20135007"},{"key":"4187_CR72","volume-title":"Proceedings of Conference and Workshop on Neural Information Processing Systems","author":"S Chang","year":"2022","unstructured":"Chang S, Palzer D, Li J, et al. MapQA: a dataset for question answering on choropleth maps. In: Proceedings of Conference and Workshop on Neural Information Processing Systems, 2022"},{"key":"4187_CR73","first-page":"6904","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"Y Goyal","year":"2017","unstructured":"Goyal Y, Khot T, Summers-Stay D, et al. Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017. 6904\u20136913"},{"key":"4187_CR74","first-page":"6700","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"D A Hudson","year":"2019","unstructured":"Hudson D A, Manning C D. GQA: a new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2019. 6700\u20136709"},{"key":"4187_CR75","first-page":"3608","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"D Gurari","year":"2018","unstructured":"Gurari D, Li Q, Stangl A J, et al. VizWiz grand challenge: answering visual questions from blind people. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018. 3608\u20133617"},{"key":"4187_CR76","first-page":"8317","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"A Singh","year":"2019","unstructured":"Singh A, Natarajan V, Shah M, et al. Towards VQA models that can read. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2019. 8317\u20138326"},{"key":"4187_CR77","first-page":"292","volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing","author":"Y Li","year":"2023","unstructured":"Li Y, Du Y, Zhou K, et al. Evaluating object hallucination in large vision-language models. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, 2023. 292\u2013305"},{"key":"4187_CR78","volume-title":"MMBench: is your multi-modal model an all-around player?","author":"Y Liu","year":"2023","unstructured":"Liu Y, Duan H, Zhang Y, et al. MMBench: is your multi-modal model an all-around player? 2023. ArXiv:2307.06281"},{"key":"4187_CR79","volume-title":"SEED-Bench: benchmarking multimodal LLMs with generative comprehension","author":"B Li","year":"2023","unstructured":"Li B, Wang R, Wang G, et al. SEED-Bench: benchmarking multimodal LLMs with generative comprehension. 2023. ArXiv:2307.16125"},{"key":"4187_CR80","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"W Yu","year":"2024","unstructured":"Yu W, Yang Z, Li L, et al. MM-Vet: evaluating large multimodal models for integrated capabilities. In: Proceedings of the 41st International Conference on Machine Learning, 2024"},{"key":"4187_CR81","volume-title":"Introducing IDEFICS: an open reproduction of state-of-the-art visual language model","author":"H Laurencon","year":"2023","unstructured":"Laurencon H, van Strien D, Bekman S, et al. Introducing IDEFICS: an open reproduction of state-of-the-art visual language model. 2023. https:\/\/huggingface.co\/blog\/idefics"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4187-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-024-4187-3","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4187-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,19]],"date-time":"2026-01-19T22:03:01Z","timestamp":1768860181000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-024-4187-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":81,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["4187"],"URL":"https:\/\/doi.org\/10.1007\/s11432-024-4187-3","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12]]},"assertion":[{"value":"15 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 August 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 December 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"220103"}}