{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T20:20:34Z","timestamp":1778617234114,"version":"3.51.4"},"reference-count":65,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62436007"],"award-info":[{"award-number":["62436007"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhejiang NSF","award":["LQK26F020001"],"award-info":[{"award-number":["LQK26F020001"]}]},{"name":"D Program of Zhejiang","award":["2025C01030"],"award-info":[{"award-number":["2025C01030"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["226-2025-00057"],"award-info":[{"award-number":["226-2025-00057"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Ningbo Yongjiang Talent Introduction Programme","award":["2024A-401-G"],"award-info":[{"award-number":["2024A-401-G"]}]},{"name":"Zhejiang University Education Foundation Qizhen Scholar Foundation"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1109\/tpami.2026.3656169","type":"journal-article","created":{"date-parts":[[2026,1,20]],"date-time":"2026-01-20T20:40:09Z","timestamp":1768941609000},"page":"6208-6224","source":"Crossref","is-referenced-by-count":0,"title":["Momentor++: Advancing Video Large Language Models With Fine-Grained Long Video Reasoning"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2258-1291","authenticated-orcid":false,"given":"Juncheng","family":"Li","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9705-5398","authenticated-orcid":false,"given":"Minghe","family":"Gao","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8472-7992","authenticated-orcid":false,"given":"Xiangnan","family":"He","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7356-9711","authenticated-orcid":false,"given":"Siliang","family":"Tang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8327-0003","authenticated-orcid":false,"given":"Wei-Shi","family":"Zheng","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6142-9914","authenticated-orcid":false,"given":"Jun","family":"Xiao","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3094-7735","authenticated-orcid":false,"given":"Meng","family":"Wang","sequence":"additional","affiliation":[{"name":"Hefei University of Technology, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6097-7807","authenticated-orcid":false,"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7361-3084","authenticated-orcid":false,"given":"Yueting","family":"Zhuang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"ChatGPT: Optimizing language models for dialogue","year":"2022"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4321-9"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"ref4","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li","year":"2023"},{"key":"ref5","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref6","article-title":"Vicuna: An open-source chatbot impressing GPT-4 with 90%* chatgpt quality","author":"Chiang","year":"2023"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3796716"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01289"},{"key":"ref9","article-title":"Flash-vstream: Memory-based real-time understanding for long video streams","author":"Zhang","year":"2024"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01282"},{"key":"ref11","first-page":"28828","article-title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","volume":"37","author":"Wu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref12","article-title":"MLVU: A comprehensive benchmark for multi-task long video understanding","author":"Zhou","year":"2024"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02245"},{"key":"ref14","first-page":"41340","article-title":"Momentor: Advancing video large language model with fine-grained temporal reasoning","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","author":"Qian","year":"2024"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.216"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00332"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_4"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.563"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.585"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00304"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3274139"},{"key":"ref23","first-page":"11846","article-title":"Detecting moments and highlights in videos via natural language queries","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"2021","author":"Lei"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00262"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00356"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01032"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-025-02620-2"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02095"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"ref31","article-title":"MiniGPT4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","author":"Ataallah","year":"2024"},{"key":"ref32","first-page":"34892","article-title":"Visual instruction tuning","volume-title":"Proc. Adv. Neural Inform. Process. Syst.","volume":"36","author":"Liu","year":"2024"},{"key":"ref33","article-title":"An image is worth 16 x 16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01725"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.02127"},{"key":"ref36","article-title":"Vidcompress: Memory-enhanced temporal compression for video understanding in large language models","author":"Lan","year":"2024"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/iccv51701.2025.02240"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.02777"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3792"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1024"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i5.32567"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73414-4_26"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.02047"},{"key":"ref44","article-title":"Token merging: Your VIT but faster","author":"Bolya","year":"2022"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"ref46","article-title":"Pyscenedetect: Intelligent scene cut detection and video splitting tool","author":"Castellano","year":"2018"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123380"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01791"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01589"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1088\/1742-5468\/2008\/10\/P10008"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-019-41695-z"},{"key":"ref53","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2021"},{"key":"ref54","article-title":"Qwen2. 5-vl technical report","author":"Bai","year":"2025"},{"key":"ref55","article-title":"Video instruction tuning with synthetic data","author":"Zhang","year":"2024"},{"key":"ref56","doi-asserted-by":"crossref","DOI":"10.32388\/X26ILU","article-title":"Tarsier2: Advancing large vision-language models from detailed video description to comprehensive video understanding","author":"Yuan","year":"2025"},{"key":"ref57","first-page":"32076","article-title":"Et bench: Towards open-ended event-level video-language understanding","volume-title":"Proc. Adv. Neural Inform. Process. Syst.","volume":"37","author":"Liu","year":"2024"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.02131"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019127"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12342"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.01357"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1002\/wics.101"},{"issue":"11","key":"ref65","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/34\/11512030\/11359544-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11512030\/11359544.pdf?arnumber=11359544","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T19:48:51Z","timestamp":1778615331000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11359544\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":65,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2026.3656169","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]}}}