{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:02:43Z","timestamp":1784268163622,"version":"3.55.0"},"reference-count":57,"publisher":"Tsinghua University Press","issue":"1","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62202014,62332002,62425101"],"award-info":[{"award-number":["62202014,62332002,62425101"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Comp. Visual. Med."],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.26599\/cvm.2025.9450516","type":"journal-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T18:39:45Z","timestamp":1765219185000},"page":"71-84","source":"Crossref","is-referenced-by-count":11,"title":["Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-Based Large Language Models"],"prefix":"10.26599","volume":"12","author":[{"given":"Munan","family":"Ning","sequence":"first","affiliation":[{"name":"Shenzhen Graduate School, Peking University,Shenzhen,China,518055"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bin","family":"Zhu","sequence":"additional","affiliation":[{"name":"Shenzhen Graduate School, Peking University,Shenzhen,China,518055"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yujia","family":"Xie","sequence":"additional","affiliation":[{"name":"Microsoft Cloud AI,Beijing,China,100080"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bin","family":"Lin","sequence":"additional","affiliation":[{"name":"Shenzhen Graduate School, Peking University,Shenzhen,China,518055"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaxi","family":"Cui","sequence":"additional","affiliation":[{"name":"Pandalla.ai,Singapore,Singapore,048624"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lu","family":"Yuan","sequence":"additional","affiliation":[{"name":"Meta AI,Shanghai,China,201203"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongdong","family":"Chen","sequence":"additional","affiliation":[{"name":"Microsoft Cloud AI,Beijing,China,100080"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li","family":"Yuan","sequence":"additional","affiliation":[{"name":"Shenzhen Graduate School, Peking University,Shenzhen,China,518055"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"11138","reference":[{"key":"ref1","volume-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"key":"ref2","author":"Radford","year":"2020","journal-title":"Language models are unsupervised multitask learners"},{"key":"ref3","author":"Brown","year":"2020","journal-title":"Language models are few-shot learners"},{"key":"ref4","author":"Touvron","year":"2023","journal-title":"LLaMA: Open and efficient foundation language models"},{"key":"ref5","author":"Touvron","year":"2023","journal-title":"Llama 2: Open foundation and fine-tuned chat models"},{"key":"ref6","author":"Wang","year":"2022","journal-title":"OmniVL: One foundation model for image-language and video-language tasks"},{"key":"ref7","author":"Chen","year":"2023","journal-title":"X-LLM: Bootstrapping advanced large language models by treating multi-modalities as foreign languages"},{"key":"ref8","author":"Lyu","year":"2023","journal-title":"Macaw-LLM: Multi-modal language modeling with image, audio, video, and text integration"},{"key":"ref9","author":"Wang","year":"2023","journal-title":"ChatVideo: A tracklet-centric multimodal and versatile video understanding system"},{"key":"ref10","author":"Ataallah","year":"2024","journal-title":"MiniGPT4-video: Advancing multimodal LLMs for video understanding with interleaved visual-textual tokens"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01282"},{"key":"ref12","author":"Yu","year":"2024","journal-title":"CREMA: Generalizable and efficient video-language reasoning via multimodal modular fusion"},{"key":"ref13","author":"Ranasinghe","year":"2024","journal-title":"Understanding long videos with multimodal language models"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2024.3507000"},{"key":"ref16","author":"Li","year":"2023","journal-title":"SEED-bench: Benchmarking multimodal LLMs with generative comprehension"},{"key":"ref17","author":"Guo","year":"2025","journal-title":"R-bench: Graduate-level multi-disciplinary benchmarks for LLM & MLLM complex reasoning evaluation"},{"key":"ref18","author":"Hendrycks","year":"2020","journal-title":"Measuring massive multitask language understanding"},{"key":"ref19","author":"Raffel","year":"2019","journal-title":"Exploring the limits of transfer learning with a unified text- to- text transformer"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4321-9"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2025.3571946"},{"key":"ref23","author":"Luo","year":"2023","journal-title":"Valley: Video assistant with large language model enhanced ability"},{"key":"ref24","author":"Su","year":"2023","journal-title":"PandaGPT: One model to instruction-follow them all"},{"key":"ref25","author":"Ye","year":"2023","journal-title":"mPLUG-owl: Modularization empowers large language models with multimodality"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.01300"},{"key":"ref28","author":"Yuan","year":"2021","journal-title":"Florence: A new foundation model for computer vision"},{"key":"ref29","volume-title":"Vicuna: An opensource chatbot impressing gpt-4 with 90%* chatgpt quality","author":"Chiang","year":"2023"},{"key":"ref30","author":"Liu","year":"2023","journal-title":"Visual instruction tuning"},{"key":"ref31","author":"Awadalla","year":"2023","journal-title":"OpenFlamingo: An open-source framework for training large autoregressive vision-language models"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01457"},{"key":"ref33","author":"Bai","year":"2025","journal-title":"Qwen2.5-VL technical report"},{"key":"ref34","author":"Ouyang","year":"2022","journal-title":"Training language models to follow instructions with human feedback"},{"key":"ref35","author":"Soomro","year":"2012","journal-title":"UCF101: A dataset of 101 human actions classes from videos in the wild"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"ref37","author":"Kay","year":"2017","journal-title":"The kinetics human action video dataset"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00633"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3217368"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2010.5539872"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00678"},{"key":"ref42","author":"Leal-Taixe","year":"2015","journal-title":"MOTChallenge 2015: Towards a benchmark for multi-target tracking"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.85"},{"key":"ref44","first-page":"5187","article-title":"Videoinstance segmentation","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Yang","year":"2019"},{"key":"ref45","first-page":"190","article-title":"Collecting highly parallel data for paraphrase evaluation","volume-title":"Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies","author":"Chen","year":"2011"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019127"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.501"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1167"},{"key":"ref52","author":"Clark","year":"2018","journal-title":"Think you have solved question answering? Try ARC, the AI2 reasoning challenge"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1472"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.229"},{"key":"ref55","author":"Li","year":"2022","journal-title":"Elevater: A benchmark and toolkit for evaluating language-augmented visual models"},{"key":"ref56","author":"Liang","year":"2022","journal-title":"Holistic evaluation of language models"},{"key":"ref57","author":"Schulman","year":"2017","journal-title":"Proximalpolicy optimization algorithms"}],"container-title":["Computational Visual Media"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10750449\/11370641\/11284911.pdf?arnumber=11284911","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T20:54:29Z","timestamp":1770152069000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11284911\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2]]},"references-count":57,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.26599\/cvm.2025.9450516","relation":{},"ISSN":["2096-0662","2096-0433"],"issn-type":[{"value":"2096-0662","type":"electronic"},{"value":"2096-0433","type":"print"}],"subject":[],"published":{"date-parts":[[2026,2]]}}}