{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T03:28:43Z","timestamp":1777865323275,"version":"3.51.4"},"reference-count":52,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.01918","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"20626-20636","source":"Crossref","is-referenced-by-count":0,"title":["Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding"],"prefix":"10.1109","author":[{"given":"Yuanhan","family":"Zhang","sequence":"first","affiliation":[{"name":"Nanyang Technological University,S-Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunice","family":"Chew","sequence":"additional","affiliation":[{"name":"Nanyang Technological University,S-Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuhao","family":"Dong","sequence":"additional","affiliation":[{"name":"Nanyang Technological University,S-Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aria","family":"Leo","sequence":"additional","affiliation":[{"name":"Independent Researcher"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Hu","sequence":"additional","affiliation":[{"name":"Independent Researcher"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ziwei","family":"Liu","sequence":"additional","affiliation":[{"name":"Nanyang Technological University,S-Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"ref2","article-title":"Temporalbench: Towards fine-grained temporal understanding for multimodal video models","author":"Cai","year":"2024","journal-title":"arXiv preprint"},{"key":"ref3","first-page":"190","article-title":"Collecting highly parallel data for paraphrase evaluation","volume-title":"Proceedings of the 49th annual meeting of the association for computational linguistics: human language technologies","author":"Chen","year":"2011"},{"key":"ref4","article-title":"Expanding performance boundaries of open-source multimodal models with model, data, and testtime scaling","author":"Chen","year":"2024","journal-title":"arXiv preprint"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.00847"},{"key":"ref6","article-title":"Mmbench-video: A long-form multi-shot benchmark for holistic video understanding","author":"Fang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref7","article-title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal 11 ms in video analysis","author":"Fu","year":"2024","journal-title":"arXiv preprint"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01113"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1037\/0096-1523.25.2.299"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"ref11","article-title":"Video-mmmu: Evaluating knowledge acquisition from multi-discipline professional videos, 2025","author":"Hu","journal-title":"1, 2"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.149"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/s42113-019-00029-y"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-1167"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.02095"},{"key":"ref16","author":"Liu","journal-title":"Tempcompass: Do video llms really understand videos? arXiv preprint arXiv:2403.00476, 2024"},{"key":"ref17","article-title":"Oryx mllm: On-demand spatial-temporal understanding at arbitrary resolution","author":"Liu","year":"2024","journal-title":"arXiv preprint"},{"key":"ref18","article-title":"Chain-of-spot: Interactive reasoning improves large vision-language models","author":"Liu","year":"2024","journal-title":"arXiv preprint"},{"key":"ref19","article-title":"Ola: Pushing the frontiers of omni-modal language model with progressive modality alignment","author":"Liu","year":"2025","journal-title":"arXiv preprint"},{"key":"ref20","article-title":"LMMs-Lab","year":"2024","journal-title":"Video detail caption"},{"key":"ref21","article-title":"Egoschema: A diagnostic benchmark for very longform video language understanding","author":"Mangalam","year":"2024","journal-title":"Advances in Neural Information Processing Systems, 36"},{"key":"ref22","article-title":"Identifying the perceptual dimensions of visual complexity of scenes","volume-title":"Proceedings of the annual meeting of the cognitive science society","author":"Olivia","year":"2004"},{"key":"ref23","article-title":"OpenAI","volume-title":"Hello gpt-4o","year":"2024"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2493"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0275"},{"key":"ref26","first-page":"17","author":"Simons","year":"2014","journal-title":"Complex narratives. In Hollywood puzzle films"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1037\/0278-7393.6.2.174"},{"key":"ref28","article-title":"Visual agents as fast and slow thinkers","author":"Sun","year":"2024","journal-title":"arXiv preprint"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1111\/cogs.12933"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1207\/s15516709cog1202_4"},{"key":"ref31","article-title":"Gemini: a family of highly capable multimodal models","author":"Team","year":"2023","journal-title":"arXiv preprint"},{"key":"ref32","author":"Team","journal-title":"Qwen2.5-vl, 2025"},{"key":"ref33","author":"Wei","year":"2023","journal-title":"Chain-of-thought prompting elicits reasoning in large language models"},{"key":"ref34","article-title":"Star: A benchmark for situated reasoning in real-world videos","volume-title":"Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2)","author":"Wu","year":"2021"},{"key":"ref35","article-title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","author":"Wu","journal-title":"1, 3"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73232-4_3"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00975"},{"key":"ref40","article-title":"Clevrer: Collision events for video representation and reasoning","author":"Yi","year":"2019","journal-title":"arXiv preprint"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019127"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019127"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00901"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/tifs.2024.3520306"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-naacl.51"},{"key":"ref47","article-title":"Long context transfer from language to vision","author":"Zhang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref48","author":"Zhang","year":"2024","journal-title":"Llavanext: A strong zero-shot video understanding model"},{"key":"ref49","author":"Zhang","journal-title":"Video instruction tuning with synthetic data, 2024"},{"key":"ref50","author":"Zhang","year":"2024","journal-title":"Worldqa: Multimodal world knowledge in videos through long-chain reasoning"},{"key":"ref51","article-title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","author":"Zhou","year":"2024","journal-title":"arXiv preprint"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/s00530-003-0076-5"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11445586.pdf?arnumber=11445586","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T06:14:55Z","timestamp":1777529695000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11445586\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":52,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.01918","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}