{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T17:22:26Z","timestamp":1783099346195,"version":"3.54.6"},"reference-count":45,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information Sciences"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.ins.2026.123821","type":"journal-article","created":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T23:18:03Z","timestamp":1782170283000},"page":"123821","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["HiLoCo: Efficient long video understanding via hierarchical localization and query-aware token compression"],"prefix":"10.1016","volume":"756","author":[{"given":"Wangqun","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5240-1779","authenticated-orcid":false,"given":"Baoyun","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xingkong","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaojie","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siwen","family":"Jiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huaping","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.ins.2026.123821_bib0005","doi-asserted-by":"crossref","first-page":"1355","DOI":"10.1109\/TCSVT.2025.3566695","article-title":"Video understanding with large language models: a survey","volume":"36","author":"Tang","year":"2026","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.ins.2026.123821_bib0010","series-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023","article-title":"Egoschema: a diagnostic benchmark for very long-form video language understanding","author":"Mangalam","year":"2023"},{"key":"10.1016\/j.ins.2026.123821_bib0015","series-title":"Conference on Computer Vision and Pattern Recognition, CVPR 2025, Nashville, TN, USA, June 11\u201315, 2025, Computer Vision Foundation \/ IEEE","first-page":"24108","article-title":"Video-mme: the first-ever comprehensive evaluation benchmark of multi-modal LLMs in video analysis","author":"Fu","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0020","first-page":"7211","article-title":"ALLVB: all-in-one long video understanding benchmark","author":"Tan","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0025","series-title":"IEEE\/CVF International Conference on Computer Vision, ICCV 2025, Honolulu, HI, USA, October 19\u201325, 2025, IEEE","first-page":"22958","article-title":"Lvbench: an extreme long video understanding benchmark","author":"Wang","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0030","series-title":"Advances in Neural Information Processing Systems 37: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10\u201315, 2024","article-title":"Longvideobench: a benchmark for long-context interleaved video-language understanding","author":"Wu","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0035","series-title":"Computer Vision - ECCV 2024 - 18th European Conference, Milan, Italy, September 29-October 4, 2024, Proceedings, Part LXXX, Volume 15138 of Lecture Notes in Computer Science","first-page":"58","article-title":"Videoagent: long-form video understanding with large language model as agent","author":"Wang","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0040","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, EMNLP 2024, Miami, FL, USA, November 12\u201316, 2024, Association for Computational Linguistics","first-page":"21715","article-title":"A simple LLM framework for long-range video question-answering","author":"Zhang","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0045","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"18936","article-title":"Drvideo: document retrieval based long video understanding","author":"Ma","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0050","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"19792","article-title":"Visionzip: longer is better but not necessary in vision language models","author":"Yang","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0055","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"9392","article-title":"Divprune: diversity-based visual token pruning for large multimodal models","author":"Alvar","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0065","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, EMNLP 2024, Miami, FL, USA, November 12\u201316, 2024, Association for Computational Linguistics","first-page":"5971","article-title":"Video-llava: learning united visual representation by alignment before projection","author":"Lin","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0070","series-title":"Llava-next: improved reasoning, OCR, and world knowledge","author":"Liu","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0075","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024, Seattle, WA, USA, June 16\u201322, 2024, IEEE","first-page":"22195","article-title":"Mvbench: a comprehensive multi-modal video understanding benchmark","author":"Li","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0080","author":"Wang"},{"key":"10.1016\/j.ins.2026.123821_bib0085","series-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, 24\u201328 Apr 2025","article-title":"mplug-owl3: towards long image-sequence understanding in multi-modal large language models","author":"Ye","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0090","author":"Cheng"},{"key":"10.1016\/j.ins.2026.123821_bib0095","series-title":"Computer Vision - ECCV 2024 - 18th European Conference, Milan, Italy, September 29-October 4, 2024, Proceedings, Part XLVI, Lecture Notes in Computer Science","first-page":"323","article-title":"LLaMA-vid: an image is worth 2 tokens in large language models","author":"Li","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0100","series-title":"2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"14313","article-title":"Timechat: a time-sensitive multimodal large language model for long video understanding","author":"Ren","year":"2023"},{"key":"10.1016\/j.ins.2026.123821_bib0105","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2025, Nashville, TN, USA, June 11\u201315, 2025, Computer Vision Foundation \/ IEEE","first-page":"26160","article-title":"Video-XL: extra-long vision language model for hour-scale video understanding","author":"Shu","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0110","series-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, 24\u201328 Apr 2025","article-title":"Longvila: scaling long-context visual language models for long videos","author":"Chen","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0115","article-title":"Long context transfer from language to vision","volume":"2025","author":"Zhang","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.ins.2026.123821_bib0120","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024, Seattle, WA, USA, June 16\u201322, 2024, IEEE","first-page":"18221","article-title":"Moviechat: from dense token to sparse memory for long video understanding","author":"Song","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0125","series-title":"Computer Vision - ECCV 2024 - 18th European Conference, Milan, Italy, September 29-October 4, 2024, Proceedings, Part LXXX, Lecture Notes in Computer Science","first-page":"58","article-title":"Videoagent: long-form video understanding with large language model as agent","author":"Wang","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0130","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024, Seattle, WA, USA, June 16\u201322, 2024, IEEE","first-page":"11996","article-title":"Fairrag: fair human generation via fair retrieval augmentation","author":"Shrestha","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0135","series-title":"Proceedings of the 33rd ACM International Conference on Information and Knowledge Management, CIKM 2024, Boise, ID, USA, October 21\u201325, 2024, ACM","first-page":"4341","article-title":"iRAG: advancing RAG for videos with an incremental approach","author":"Arefeen","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0140","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2025, Nashville, TN, USA, June 11\u201315, 2025, Computer Vision Foundation \/ IEEE","first-page":"4122","article-title":"NVILA: efficient frontier visual language models","author":"Liu","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0145","author":"Chen"},{"key":"10.1016\/j.ins.2026.123821_bib0150","author":"Yang"},{"key":"10.1016\/j.ins.2026.123821_bib0165","author":"Reid"},{"key":"10.1016\/j.ins.2026.123821_bib0170","series-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, 24\u201328 Apr 2025","article-title":"World model on million-length video and language with blockwise ringattention","author":"Liu","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0175","author":"Hong"},{"key":"10.1016\/j.ins.2026.123821_bib0180","series-title":"Advances in Neural Information Processing Systems 37: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024","article-title":"Sharegpt4video: improving video understanding and generation with better captions","author":"Chen","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0185","author":"Chen"},{"key":"10.1016\/j.ins.2026.123821_bib0190","author":"Chen"},{"key":"10.1016\/j.ins.2026.123821_bib0195","series-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, 24\u201328 Apr 2025","article-title":"Oryx MLLM: on-demand spatial-temporal understanding at arbitrary resolution","author":"Liu","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0200","author":"Fei"},{"key":"10.1016\/j.ins.2026.123821_bib0210","doi-asserted-by":"crossref","first-page":"7543","DOI":"10.1109\/TPAMI.2025.3571946","article-title":"Otter: a multi-modal model with in-context instruction tuning","volume":"47","author":"Li","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.ins.2026.123821_bib0215","series-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, 24\u201328 Apr 2025","article-title":"mPLUG-Owl3: towards long image-sequence understanding in multi-modal large language models","author":"Ye","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0220","series-title":"IIEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2025, Nashville, TN, USA, June 11\u201315, 2025, Computer Vision Foundation \/ IEEE","first-page":"8579","article-title":"Re-thinking temporal search for long-form video understanding","author":"Ye","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0225","series-title":"Computer Vision - ECCV 2024 - 18th European Conference, Milan, Italy, September 29-October 4, 2024, Proceedings, Part XXIX, Volume 15087 of Lecture Notes in Computer Science","first-page":"251","article-title":"Goldfish: vision-language understanding of arbitrarily long videos","author":"Ataallah","year":"2024"},{"key":"10.1016\/j.ins.2026.123821_bib0230","series-title":"The Thirty-Ninth Annual Conference on Neural Information Processing Systems","article-title":"Video-RAG: visually-aligned retrieval-augmented long video comprehension","author":"Luo","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0235","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2025, Nashville, TN, USA, June 11\u201315, 2025, Computer Vision Foundation \/ IEEE","first-page":"3352","article-title":"SALOVA: segment-augmented long video assistant for targeted retrieval and routing in long-form video analysis","author":"Kim","year":"2025"},{"key":"10.1016\/j.ins.2026.123821_bib0240","author":"Jiang"},{"key":"10.1016\/j.ins.2026.123821_bib0245","article-title":"Llava-video: video instruction tuning with synthetic data","volume":"2025","author":"Zhang","year":"2025","journal-title":"Trans. Mach. Learn.Res."}],"container-title":["Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0020025526007528?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0020025526007528?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T16:53:22Z","timestamp":1783097602000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0020025526007528"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":45,"alternative-id":["S0020025526007528"],"URL":"https:\/\/doi.org\/10.1016\/j.ins.2026.123821","relation":{},"ISSN":["0020-0255"],"issn-type":[{"value":"0020-0255","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"HiLoCo: Efficient long video understanding via hierarchical localization and query-aware token compression","name":"articletitle","label":"Article Title"},{"value":"Information Sciences","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.ins.2026.123821","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"123821"}}