{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T07:09:05Z","timestamp":1761894545255,"version":"build-2065373602"},"reference-count":35,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/icme59968.2025.11208965","type":"proceedings-article","created":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T17:57:42Z","timestamp":1761847062000},"page":"1-6","source":"Crossref","is-referenced-by-count":0,"title":["HLV-1K: A Large-scale Hour-Long Video Benchmark for Time-Specific Long Video Understanding"],"prefix":"10.1109","author":[{"given":"Heqing","family":"Zou","sequence":"first","affiliation":[{"name":"TikTok"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tianze","family":"Luo","sequence":"additional","affiliation":[{"name":"TikTok"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guiyang","family":"Xie","sequence":"additional","affiliation":[{"name":"TikTok"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Victor Xiao Jie","family":"Zhang","sequence":"additional","affiliation":[{"name":"TikTok"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fengmao","family":"Lv","sequence":"additional","affiliation":[{"name":"Southwest Jiaotong University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangcong","family":"Wang","sequence":"additional","affiliation":[{"name":"Great Bay University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junyang","family":"Chen","sequence":"additional","affiliation":[{"name":"Shenzhen University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhuochen","family":"Wang","sequence":"additional","affiliation":[{"name":"TikTok"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hansheng","family":"Zhang","sequence":"additional","affiliation":[{"name":"TikTok"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huaijian","family":"Zhang","sequence":"additional","affiliation":[{"name":"TikTok"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"12 401","article-title":"MM-LLMs: Recent advances in MultiModal large language models","volume-title":"Findings of the Association for Computational Linguistics: ACL 2024","author":"Zhang"},{"article-title":"Llava-onevision: Easy visual task transfer","year":"2024","author":"Li","key":"ref2"},{"article-title":"Next-gpt: Any-to-any multimodal llm","year":"2023","author":"Wu","key":"ref3"},{"article-title":"Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution","year":"2024","author":"Wang","key":"ref4"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01357"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00796"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00037"},{"article-title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","volume-title":"The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track","author":"Wu","key":"ref8"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01725"},{"article-title":"From seconds to hours: Reviewing multimodal large language models on comprehensive long video understanding","year":"2024","author":"Zou","key":"ref10"},{"article-title":"Hourvideo: 1-hour video-language understanding","year":"2024","author":"Chandrasegaran","key":"ref11"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.149"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"article-title":"Videovista: A versatile benchmark for video understanding and reasoning","year":"2024","author":"Li","key":"ref14"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.02245"},{"article-title":"Lvbench: An extreme long video understanding benchmark","year":"2024","author":"Wang","key":"ref16"},{"article-title":"InstructBLIP: Towards general-purpose vision-language models with instruction tuning","volume-title":"Thirty-seventh Conference on Neural Information Processing Systems","author":"Dai","key":"ref17"},{"key":"ref18","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2024","journal-title":"Advances in neural information processing systems"},{"key":"ref19","first-page":"543","article-title":"Video-LLaMA: An instruction-tuned audio-visual language model for video understanding","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing: System Demonstrations","author":"Zhang"},{"article-title":"Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms","year":"2024","author":"Cheng","key":"ref20"},{"article-title":"Slowfast-llava: A strong training-free baseline for video large language models","year":"2024","author":"Xu","key":"ref21"},{"article-title":"Pllava: Parameter-free llava extension from images to videos for video dense captioning","year":"2024","author":"Xu","key":"ref22"},{"article-title":"Momentor: Advancing video large language model with fine-grained temporal reasoning","volume-title":"Forty-first International Conference on Machine Learning","author":"Qian","key":"ref23"},{"article-title":"Long context transfer from language to vision","year":"2024","author":"Zhang","key":"ref24"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01357"},{"article-title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","year":"2024","author":"Zhou","key":"ref26"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00498"},{"year":"2023","key":"ref28","article-title":"Hello gpt-4o"},{"article-title":"Ultralytics yolov8","year":"2023","author":"Jocher","key":"ref29"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2003.815165"},{"article-title":"Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling","year":"2024","author":"Chen","key":"ref31"},{"article-title":"Kangaroo: A powerful video-language model supporting long-context video input","year":"2024","author":"Liu","key":"ref32"},{"article-title":"Video instruction tuning with synthetic data","year":"2024","author":"Zhang","key":"ref33"},{"year":"2023","key":"ref34","article-title":"Claude 3.5 sonnet"},{"year":"2023","key":"ref35","article-title":"Our next-generation model: Gemini 1.5"}],"event":{"name":"2025 IEEE International Conference on Multimedia and Expo (ICME)","start":{"date-parts":[[2025,6,30]]},"location":"Nantes, France","end":{"date-parts":[[2025,7,4]]}},"container-title":["2025 IEEE International Conference on Multimedia and Expo (ICME)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11208895\/11208897\/11208965.pdf?arnumber=11208965","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T05:30:33Z","timestamp":1761888633000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11208965\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":35,"URL":"https:\/\/doi.org\/10.1109\/icme59968.2025.11208965","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}