{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:18:54Z","timestamp":1782994734795,"version":"3.54.5"},"reference-count":64,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,9,28]],"date-time":"2025-09-28T00:00:00Z","timestamp":1759017600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,28]],"date-time":"2025-09-28T00:00:00Z","timestamp":1759017600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s11432-024-4602-x","type":"journal-article","created":{"date-parts":[[2025,10,7]],"date-time":"2025-10-07T06:34:10Z","timestamp":1759818850000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["MULTI: multimodal understanding leaderboard with text and images"],"prefix":"10.1007","volume":"68","author":[{"given":"Zichen","family":"Zhu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lu","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingkai","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yichuan","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiming","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hailin","family":"Wen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaqi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinyu","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingzi","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Situo","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zihan","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liangtai","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kai","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,28]]},"reference":[{"key":"4602_CR1","doi-asserted-by":"publisher","first-page":"888","DOI":"10.1007\/s11633-024-1502-8","volume":"21","author":"T Sun","year":"2024","unstructured":"Sun T, Zhang X, He Z, et al. MOSS: an open conversational large language model. Mach Intell Res, 2024, 21: 888\u2013905","journal-title":"Mach Intell Res"},{"key":"4602_CR2","doi-asserted-by":"publisher","first-page":"108","DOI":"10.1016\/j.aiopen.2025.04.001","volume":"6","author":"Z Chen","year":"2025","unstructured":"Chen Z, Ma D, Li H, et al. DFM: dialogue foundation model for universal large-scale dialogue-oriented task learning. AI Open, 2025, 6: 108\u2013117","journal-title":"AI Open"},{"key":"4602_CR3","volume-title":"Qwen2 technical report","author":"A Yang","year":"2024","unstructured":"Yang A, Yang B S, Hui B Y, et al. Qwen2 technical report. 2024. ArXiv:2407.10671"},{"key":"4602_CR4","volume-title":"InternLM2 technical report","author":"Z Cai","year":"2024","unstructured":"Cai Z, Cao M, Chen H, et al. InternLM2 technical report. 2024. ArXiv:2403.17297"},{"key":"4602_CR5","volume-title":"ChatGPT: optimizing language models for dialogue","author":"OpenAI.","year":"2022","unstructured":"OpenAI. ChatGPT: optimizing language models for dialogue. 2022. https:\/\/openai.com\/blog\/chatgpt"},{"key":"4602_CR6","volume-title":"GPT-4 technical report","author":"J Achiam","year":"2023","unstructured":"Achiam J, Adler S, Agarwal S, et al. GPT-4 technical report. 2023. ArXiv:2303.08774"},{"key":"4602_CR7","volume-title":"Gemini: a family of highly capable multimodal models","author":"G Team","year":"2023","unstructured":"Team G, Anil R, Borgeaud S, et al. Gemini: a family of highly capable multimodal models. 2023. ArXiv:2312.11805"},{"key":"4602_CR8","volume-title":"LLaMA: open and efficient foundation language models","author":"H Touvron","year":"2023","unstructured":"Touvron H, Lavril T, Izacard G, et al. LLaMA: open and efficient foundation language models. 2023. ArXiv:2302.13971"},{"key":"4602_CR9","volume-title":"Depression diagnosis dialogue simulation: self-improving psychiatrist with tertiary memory","author":"K Lan","year":"2024","unstructured":"Lan K, Jin B, Zhu Z, et al. Depression diagnosis dialogue simulation: self-improving psychiatrist with tertiary memory. 2024. ArXiv:2409.15084"},{"key":"4602_CR10","first-page":"1607","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics","author":"S Han","year":"2024","unstructured":"Han S, Chen L, Lin L M, et al. Ibsen: director-actor agent collaboration for controllable and interactive drama script generation. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics, 2024. 1607\u20131619"},{"key":"4602_CR11","doi-asserted-by":"publisher","first-page":"102523","DOI":"10.1016\/j.xcrp.2025.102523","volume":"6","author":"Z Zhao","year":"2025","unstructured":"Zhao Z, Ma D, Chen L, et al. Developing ChemDFM as a large language foundation model for chemistry. Cell Rep Phys Sci, 2025, 6: 102523","journal-title":"Cell Rep Phys Sci"},{"key":"4602_CR12","volume-title":"Proceedings of the 1st Conference on Language Modeling","author":"H Xu","year":"2024","unstructured":"Xu H, Zhu Z, Zhang S, et al. Rejection improves reliability: training LLMs to refuse unknown questions using RL from knowledge feedback. In: Proceedings of the 1st Conference on Language Modeling, 2024"},{"key":"4602_CR13","volume-title":"Proceedings of the 42nd International Conference on Machine Learning","author":"H Xu","year":"2025","unstructured":"Xu H, Zhu Z, Pan L, et al. Reducing tool hallucination via reliability alignment. In: Proceedings of the 42nd International Conference on Machine Learning, 2025"},{"key":"4602_CR14","first-page":"320","volume-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics","author":"Z Du","year":"2022","unstructured":"Du Z, Qian Y, Liu X, et al. GLM: general language model pretraining with autoregressive blank infilling. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics, 2022. 320\u2013335"},{"key":"4602_CR15","volume-title":"Proceedings of The 12th International Conference on Learning Representations","author":"J Hu","year":"2023","unstructured":"Hu J, Yao Y, Wang C, et al. Large multilingual models pivot zero-shot multimodal learning across languages. In: Proceedings of The 12th International Conference on Learning Representations, 2023"},{"key":"4602_CR16","volume-title":"Chinese LLaVA","author":"LinkSoul-AI.","year":"2023","unstructured":"LinkSoul-AI. Chinese LLaVA. 2023. https:\/\/github.com\/LinkSoul-AI\/Chinese-LLaVA"},{"key":"4602_CR17","volume-title":"Qwen-VL: a versatile vision-language model for understanding, localization, text reading, and beyond","author":"J Bai","year":"2023","unstructured":"Bai J, Bai S, Yang S, et al. Qwen-VL: a versatile vision-language model for understanding, localization, text reading, and beyond. 2023. ArXiv:2308.12966"},{"key":"4602_CR18","unstructured":"01.AI. Yi-VL. 2023. https:\/\/github.com\/01-ai\/Yi"},{"key":"4602_CR19","first-page":"24185","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Chen","year":"2024","unstructured":"Chen Z, Wu J, Wang W, et al. InternVL: scaling up vision foundation models and aligning for generic visual-linguistic tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 24185\u201324198"},{"key":"4602_CR20","doi-asserted-by":"publisher","first-page":"220101","DOI":"10.1007\/s11432-024-4231-5","volume":"67","author":"Z Chen","year":"2024","unstructured":"Chen Z, Wang W Y, Tian H, et al. How far are we to GPT-4V? Closing the gap to commercial multimodal models with open-source suites. Sci China Inf Sci, 2024, 67: 220101","journal-title":"Sci China Inf Sci"},{"key":"4602_CR21","volume-title":"MiniCPM-V: a GPT-4V level MLLM on your phone","author":"Y Yao","year":"2024","unstructured":"Yao Y, Yu T, Zhang A, et al. MiniCPM-V: a GPT-4V level MLLM on your phone. 2024. ArXiv:2408.01800"},{"key":"4602_CR22","volume-title":"Qwen2-VL: enhancing vision-language model\u2019s perception of the world at any resolution","author":"P Wang","year":"2024","unstructured":"Wang P, Bai S, Tan S, et al. Qwen2-VL: enhancing vision-language model\u2019s perception of the world at any resolution. 2024. ArXiv:2409.12191"},{"key":"4602_CR23","volume-title":"GPT-4V(ision) system card","author":"OpenAI.","year":"2023","unstructured":"OpenAI. GPT-4V(ision) system card. 2023. https:\/\/openai.com\/research\/gpt-4v-system-card"},{"key":"4602_CR24","volume-title":"GPT-4O mini: advancing cost-efficient intelligence","author":"OpenAI.","year":"2024","unstructured":"OpenAI. GPT-4O mini: advancing cost-efficient intelligence. 2024. https:\/\/openai.com\/index\/gpt-4o-mini-advancing-cost-efficient-intelligence\/"},{"key":"4602_CR25","volume-title":"GPT-4O system card","author":"OpenAI.","year":"2024","unstructured":"OpenAI. GPT-4O system card. 2024. https:\/\/openai.com\/index\/gpt-4o-system-card\/"},{"key":"4602_CR26","volume-title":"The Claude 3 model family: opus, sonnet, haiku","author":"Anthropic.","year":"2024","unstructured":"Anthropic. The Claude 3 model family: opus, sonnet, haiku. 2024. https:\/\/assets.anthropic.com\/m\/61e7d27f8c8f5919\/original\/Claude-3-Model-Card.pdf"},{"key":"4602_CR27","first-page":"403","volume-title":"Proceedings of European Conference on Computer Vision","author":"Y Ma","year":"2024","unstructured":"Ma Y, Cao Y, Sun J, et al. Dolphins: multimodal language model for driving. In: Proceedings of European Conference on Computer Vision, 2024. 403\u2013420"},{"key":"4602_CR28","first-page":"535","volume-title":"Proceedings of the Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (System Demonstrations)","author":"Z Zhu","year":"2025","unstructured":"Zhu Z, Tang H, Li Y, et al. MobA: multifaceted memory-enhanced adaptive planning for efficient mobile task automation. In: Proceedings of the Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (System Demonstrations), 2025. 535\u2013549"},{"key":"4602_CR29","doi-asserted-by":"publisher","first-page":"220109","DOI":"10.1007\/s11432-024-4243-0","volume":"67","author":"Z H Zhao","year":"2024","unstructured":"Zhao Z H, Chen B, Li J P, et al. ChemDFM-X: towards large multimodal model for chemistry. Sci China Inf Sci, 2024, 67: 220109","journal-title":"Sci China Inf Sci"},{"key":"4602_CR30","first-page":"2507","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"P Lu","year":"2022","unstructured":"Lu P, Mishra S, Xia T, et al. Learn to explain: multimodal reasoning via thought chains for science question answering. In: Proceedings of Advances in Neural Information Processing Systems, 2022. 2507\u20132521"},{"key":"4602_CR31","first-page":"13299","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"B Li","year":"2024","unstructured":"Li B, Ge Y, Ge Y, et al. SEED-Bench: benchmarking multimodal large language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 13299\u201313308"},{"key":"4602_CR32","first-page":"19053","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"L Sun","year":"2024","unstructured":"Sun L, Han Y, Zhao Z, et al. SciEval: a multi-level large language model evaluation benchmark for scientific research. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2024. 19053\u201319061"},{"key":"4602_CR33","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"Y Huang","year":"2024","unstructured":"Huang Y, Bai Y, Zhu Z, et al. C-Eval: a multi-level multi-discipline Chinese evaluation suite for foundation models. In: Proceedings of Advances in Neural Information Processing Systems, 2024"},{"key":"4602_CR34","first-page":"8748","volume-title":"Proceedings of International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"Radford A, Kim J W, Hallacy C, et al. Learning transferable visual models from natural language supervision. In: Proceedings of International Conference on Machine Learning, 2021. 8748\u20138763"},{"key":"4602_CR35","first-page":"9694","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"J Li","year":"2021","unstructured":"Li J, Selvaraju R, Gotmare A, et al. Align before fuse: vision and language representation learning with momentum distillation. In: Proceedings of Advances in Neural Information Processing Systems, 2021. 9694\u20139705"},{"key":"4602_CR36","first-page":"19730","volume-title":"Proceedings of International Conference on Machine Learning","author":"J Li","year":"2023","unstructured":"Li J, Li D, Savarese S, et al. BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of International Conference on Machine Learning, 2023. 19730\u201319742"},{"key":"4602_CR37","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"H Liu","year":"2024","unstructured":"Liu H, Li C, Wu Q, et al. Visual instruction tuning. In: Proceedings of Advances in Neural Information Processing Systems, 2024"},{"key":"4602_CR38","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"D Zhu","year":"2024","unstructured":"Zhu D, Chen J, Shen X, et al. MiniGPT-4: enhancing vision-language understanding with advanced large language models. In: Proceedings of the 12th International Conference on Learning Representations, 2024"},{"key":"4602_CR39","first-page":"49250","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"W Dai","year":"2023","unstructured":"Dai W, Li J, Li D, et al. InstructBLIP: towards general-purpose vision-language models with instruction tuning. In: Proceedings of Advances in Neural Information Processing Systems, 2023. 49250\u201349267"},{"key":"4602_CR40","volume-title":"InternLM-XComposer: a vision-language large model for advanced text-image comprehension and composition","author":"P Zhang","year":"2023","unstructured":"Zhang P, Wang X D B, Cao Y, et al. InternLM-XComposer: a vision-language large model for advanced text-image comprehension and composition. 2023. ArXiv:2309.15112"},{"key":"4602_CR41","first-page":"2425","volume-title":"Proceedings of the IEEE International Conference on Computer Vision","author":"S Antol","year":"2015","unstructured":"Antol S, Agrawal A, Lu J, et al. VQA: visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision, 2015. 2425\u20132433"},{"key":"4602_CR42","first-page":"6904","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"Y Goyal","year":"2017","unstructured":"Goyal Y, Khot T, Summers-Stay D, et al. Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017. 6904\u20136913"},{"key":"4602_CR43","first-page":"3195","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"K Marino","year":"2019","unstructured":"Marino K, Rastegari M, Farhadi A, et al. OK-VQA: a visual question answering benchmark requiring external knowledge. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019. 3195\u20133204"},{"key":"4602_CR44","first-page":"6700","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"D A Hudson","year":"2019","unstructured":"Hudson D A, Manning C D. GQA: a new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019. 6700\u20136709"},{"key":"4602_CR45","first-page":"740","volume-title":"Proceedings of the 13th European Conference on Computer Vision","author":"T Y Lin","year":"2014","unstructured":"Lin T Y, Maire M, Belongie S, et al. Microsoft COCO: common objects in context. In: Proceedings of the 13th European Conference on Computer Vision, 2014. 740\u2013755"},{"key":"4602_CR46","first-page":"216","volume-title":"Proceedings of European Conference on Computer Vision","author":"Y Liu","year":"2025","unstructured":"Liu Y, Duan H, Zhang Y, et al. MMBench: is your multi-modal model an all-around player? In: Proceedings of European Conference on Computer Vision, 2025. 216\u2013233"},{"key":"4602_CR47","first-page":"57730","volume-title":"Proceedings of International Conference on Machine Learning","author":"W Yu","year":"2024","unstructured":"Yu W, Yang Z, Li L, et al. MM-Vet: evaluating large multimodal models for integrated capabilities. In: Proceedings of International Conference on Machine Learning, 2024. 57730\u201357754"},{"key":"4602_CR48","volume-title":"TouchStone: evaluating vision-language models by language models","author":"S Bai","year":"2023","unstructured":"Bai S, Yang S, Bai J, et al. TouchStone: evaluating vision-language models by language models. 2023. ArXiv:2308.16890"},{"key":"4602_CR49","first-page":"4951","volume-title":"Proceedings of the Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies","author":"W Ge","year":"2025","unstructured":"Ge W, Chen S, Chen H, et al. MLLM-bench: evaluating multimodal LLMs with per-sample criteria. In: Proceedings of the Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies, 2025. 4951\u20134974"},{"key":"4602_CR50","volume-title":"SEED-Bench-2: benchmarking multimodal large language models","author":"B Li","year":"2023","unstructured":"Li B, Ge Y, Ge Y, et al. SEED-Bench-2: benchmarking multimodal large language models. 2023. ArXiv:2311.17092"},{"key":"4602_CR51","first-page":"292","volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing","author":"Y Li","year":"2023","unstructured":"Li Y, Du Y, Zhou K, et al. Evaluating object hallucination in large vision-language models. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, 2023. 292\u2013305"},{"key":"4602_CR52","first-page":"14375","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"T Guan","year":"2024","unstructured":"Guan T, Liu F, Wu X, et al. HallusionBench: an advanced diagnostic suite for entangled language hallucination and visual illusion in large vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2024. 14375\u201314385"},{"key":"4602_CR53","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"W Zhang","year":"2024","unstructured":"Zhang W, Aljunied M, Gao C, et al. M3Exam: a multilingual, multimodal, multilevel benchmark for examining large language models. In: Proceedings of Advances in Neural Information Processing Systems, 2024"},{"key":"4602_CR54","volume-title":"SciGraphQA: a large-scale synthetic multi-turn question-answering dataset for scientific graphs","author":"S Li","year":"2023","unstructured":"Li S, Tajbakhsh N. SciGraphQA: a large-scale synthetic multi-turn question-answering dataset for scientific graphs. 2023. ArXiv:2308.03349"},{"key":"4602_CR55","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"P Lu","year":"2024","unstructured":"Lu P, Bansal H, Xia T, et al. MathVista: evaluating mathematical reasoning of foundation models in visual contexts. In: Proceedings of the 12th International Conference on Learning Representations, 2024"},{"key":"4602_CR56","first-page":"2299","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"W Zhong","year":"2024","unstructured":"Zhong W, Cui R, Guo Y, et al. AGIEval: a human-centric benchmark for evaluating foundation models. In: Proceedings of Findings of the Association for Computational Linguistics, 2024. 2299\u20132314"},{"key":"4602_CR57","first-page":"9556","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X Yue","year":"2024","unstructured":"Yue X, Ni Y, Zhang K, et al. MMMU: a massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 9556\u20139567"},{"key":"4602_CR58","first-page":"50622","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"X Wang","year":"2024","unstructured":"Wang X, Hu Z, Lu P, et al. SciBench: evaluating college-level scientific problem-solving abilities of large language models. In: Proceedings of the 41st International Conference on Machine Learning, 2024. 50622\u201350649"},{"key":"4602_CR59","first-page":"830","volume-title":"Proceedings of the 33rd International Joint Conference on Artificial Intelligence","author":"Z He","year":"2024","unstructured":"He Z, Wu X, Zhou P, et al. CMMU: a benchmark for Chinese multi-modal multi-type question understanding and reasoning. In: Proceedings of the 33rd International Joint Conference on Artificial Intelligence, 2024. 830\u2013838"},{"key":"4602_CR60","first-page":"8817","volume-title":"Proceedings of Findings of the Association for Computational Linguistics ACL","author":"Y Zong","year":"2024","unstructured":"Zong Y, Qiu X. GAOKAO-MM: a Chinese human-level benchmark for multimodal models evaluation. In: Proceedings of Findings of the Association for Computational Linguistics ACL, 2024. 8817\u20138825"},{"key":"4602_CR61","volume-title":"AlignMMBench: ^evaluating Chinese multimodal alignment in large vision-language models","author":"Y Wu","year":"2024","unstructured":"Wu Y, Yu W, Cheng Y, et al. AlignMMBench: evaluating Chinese multimodal alignment in large vision-language models. 2024. ArXiv:2406.09295"},{"key":"4602_CR62","volume-title":"CMMMU: a Chinese massive multi-discipline multimodal understanding benchmark","author":"G Zhang","year":"2024","unstructured":"Zhang G, Du X, Chen B, et al. CMMMU: a Chinese massive multi-discipline multimodal understanding benchmark. 2024. ArXiv:2401.11944"},{"key":"4602_CR63","first-page":"74","volume-title":"Text Summarization Branches Out. Barcelona: Association for Computational Linguistics","author":"C Y Lin","year":"2004","unstructured":"Lin C Y. ROUGE: a package for automatic evaluation of summaries. In: Text Summarization Branches Out. Barcelona: Association for Computational Linguistics, 2004. 74\u201381"},{"key":"4602_CR64","first-page":"22199","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"T Kojima","year":"2022","unstructured":"Kojima T, Gu S S, Reid M, et al. Large language models are zero-shot reasoners. In: Proceedings of Advances in Neural Information Processing Systems, 2022. 22199\u201322213"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4602-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-024-4602-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4602-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,7]],"date-time":"2025-10-07T08:04:17Z","timestamp":1759824257000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-024-4602-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,28]]},"references-count":64,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["4602"],"URL":"https:\/\/doi.org\/10.1007\/s11432-024-4602-x","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,28]]},"assertion":[{"value":"15 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 December 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 July 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 September 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"200107"}}