{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T09:08:54Z","timestamp":1784279334198,"version":"3.55.0"},"reference-count":43,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100012899","name":"Lanzhou University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100012899","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62227807"],"award-info":[{"award-number":["62227807"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U24B20186"],"award-info":[{"award-number":["U24B20186"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113622","type":"journal-article","created":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T06:06:18Z","timestamp":1776233178000},"page":"113622","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["HM-RAG: Long video reasoning and anomaly detection via hierarchical multi-agent retrieval-augmented generation"],"prefix":"10.1016","volume":"179","author":[{"given":"Jisheng","family":"Dang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Quan","family":"Wan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dewei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziyue","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bimei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hong","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3514-5413","authenticated-orcid":false,"given":"Bin","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"4","key":"10.1016\/j.patcog.2026.113622_bib0001","first-page":"1","article-title":"Cross-modal hybrid feature fusion for image-sentence matching","volume":"17","author":"Xu","year":"2021","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl. (TOMM)"},{"key":"10.1016\/j.patcog.2026.113622_bib0002","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2021.108145","article-title":"Generalized pyramid co-attention with learnable aggregation net for video question answering","volume":"120","author":"Gao","year":"2021","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113622_bib0003","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111021","article-title":"Video anomaly detection via self-supervised and spatio-temporal proxy tasks learning","volume":"158","author":"Yang","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113622_bib0004","doi-asserted-by":"crossref","first-page":"8634","DOI":"10.52202\/075280-0377","article-title":"RefLexion: language agents with verbal reinforcement learning","volume":"36","author":"Shinn","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113622_bib0005","series-title":"Advances in Neural Information Processing Systems","first-page":"24824","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume":"35","author":"Wei","year":"2022"},{"key":"10.1016\/j.patcog.2026.113622_bib0006","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"11809","article-title":"Tree of thoughts: deliberate problem solving with large language models","volume":"36","author":"Yao","year":"2023"},{"key":"10.1016\/j.patcog.2026.113622_bib0007","series-title":"The Eleventh International Conference on Learning Representations","article-title":"REACT: synergizing reasoning and acting in language models","author":"Yao","year":"2022"},{"key":"10.1016\/j.patcog.2026.113622_bib0008","series-title":"2024 International Conference on Electrical, Computer and Energy Technologies (ICECET)","first-page":"1","article-title":"RAG beyond text: enhancing image retrieval in RAG systems","author":"Bag","year":"2024"},{"key":"10.1016\/j.patcog.2026.113622_bib0009","unstructured":"M. Bonomo, S. Bianco, Visual RAG: Expanding MLLM visual knowledge without fine-tuning, (2025) arXiv: 2501.10834."},{"key":"10.1016\/j.patcog.2026.113622_bib0010","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"1","article-title":"SiQA: a large multi-modal question answering model for structured images based on RAG","author":"Liu","year":"2025"},{"key":"10.1016\/j.patcog.2026.113622_bib0011","unstructured":"M. Riedler, S. Langer, Beyond text: Optimizing RAG with Multimodal Inputs for Industrial Applications, (2024) arXiv: 2410.21943."},{"key":"10.1016\/j.patcog.2026.113622_bib0012","unstructured":"S. Gupta, R. Ranjan, S.N. Singh, A Comprehensive Survey of Retrieval-Augmented Generation: Evolution, Current Landscape and Future Directions, (2024) arXiv: 2410.12837."},{"key":"10.1016\/j.patcog.2026.113622_bib0013","series-title":"International Conference on Machine Learning (ICML)","first-page":"3929","article-title":"Retrieval augmented language model pre-training","author":"Guu","year":"2020"},{"key":"10.1016\/j.patcog.2026.113622_bib0014","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"9459","article-title":"Retrieval-augmented generation for knowledge-intensive NLP tasks","volume":"33","author":"Lewis","year":"2020"},{"issue":"1","key":"10.1016\/j.patcog.2026.113622_bib0015","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1017\/nlp.2024.53","article-title":"Maximizing RAG efficiency: a comparative analysis of RAG methods","volume":"31","author":"\u015eakar","year":"2025","journal-title":"Nat. Lang. Process."},{"key":"10.1016\/j.patcog.2026.113622_bib0016","series-title":"International Conference on Machine Learning","first-page":"3929","article-title":"Retrieval augmented language model pre-training","author":"Guu","year":"2020"},{"key":"10.1016\/j.patcog.2026.113622_bib0017","doi-asserted-by":"crossref","unstructured":"N. Shinn, F. Cassano, E. Berman, A. Gopinath, K. Narasimhan, S. Yao, Reflexion: Language Agents with Verbal Reinforcement Learning, (2023). arXiv: 2303.11366.","DOI":"10.52202\/075280-0377"},{"key":"10.1016\/j.patcog.2026.113622_bib0018","doi-asserted-by":"crossref","unstructured":"C. Sun, A. Myers, C. Vondrick, K. Murphy, C. Schmid, VideoBERT: A Joint Model for Video and Language Representation Learning, (2019) arXiv: 1904.01766.","DOI":"10.1109\/ICCV.2019.00756"},{"key":"10.1016\/j.patcog.2026.113622_bib0019","series-title":"Icml","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"2","author":"Bertasius","year":"2021"},{"key":"10.1016\/j.patcog.2026.113622_bib0020","unstructured":"C.-Y. Wu, Y. Li, K. Mangalam, et al., MeMViT: Memory-Augmented Multiscale Vision Transformer for Efficient Long-Term Video Recognition, (2022) arXiv: 2201.08383."},{"key":"10.1016\/j.patcog.2026.113622_bib0021","doi-asserted-by":"crossref","unstructured":"H. Zhang, X. Li, L. Bing, Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding, (2023) arXiv: 2306.02858.","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"10.1016\/j.patcog.2026.113622_bib0022","series-title":"Proceedings of the 31st International Conference on Computational Linguistics","first-page":"2814","article-title":"Topology-of-question-decomposition: enhancing large language models with information retrieval for knowledge-intensive tasks","author":"Li","year":"2025"},{"key":"10.1016\/j.patcog.2026.113622_bib0023","doi-asserted-by":"crossref","unstructured":"B. He, H. Li, Y.K. Jang, et al., MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding, (2024) arXiv: 2404.05726.","DOI":"10.1109\/CVPR52733.2024.01282"},{"key":"10.1016\/j.patcog.2026.113622_bib0024","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.patcog.2026.113622_bib0025","doi-asserted-by":"crossref","unstructured":"W. Dai, J. Li, D. Li, et al., InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning, (2023) arXiv: 2305.06500.","DOI":"10.52202\/075280-2142"},{"key":"10.1016\/j.patcog.2026.113622_bib0026","series-title":"Proceedings of the 33rd ACM International Conference on Multimedia","first-page":"2781","article-title":"Hm-RAG: hierarchical multi-agent multimodal retrieval augmented generation","author":"Liu","year":"2025"},{"key":"10.1016\/j.patcog.2026.113622_bib0027","series-title":"Text Summarization Branches Out","first-page":"74","article-title":"ROUGE: a package for automatic evaluation of summaries","author":"Lin","year":"2004"},{"key":"10.1016\/j.patcog.2026.113622_bib0028","series-title":"Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics","first-page":"311","article-title":"BLEU: a method for automatic evaluation of machine translation","author":"Papineni","year":"2002"},{"key":"10.1016\/j.patcog.2026.113622_bib0029","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","article-title":"Next-QA: next phase of question-answering to explaining temporal actions","author":"Xiao","year":"2021"},{"key":"10.1016\/j.patcog.2026.113622_bib0030","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"24108","article-title":"Video-Mme: the first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","author":"Fu","year":"2025"},{"key":"10.1016\/j.patcog.2026.113622_bib0031","series-title":"2024IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"22195","article-title":"MVBench: a comprehensive multi-modal video understanding benchmark","author":"Li","year":"2024"},{"key":"10.1016\/j.patcog.2026.113622_bib0032","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"13691","article-title":"Mlvu: benchmarking multi-task long video understanding","author":"Zhou","year":"2025"},{"key":"10.1016\/j.patcog.2026.113622_bib0033","series-title":"Advances in Neural Information Processing Systems","first-page":"124","article-title":"Zero-shot video question answering via frozen bidirectional language models","volume":"35","author":"Yang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113622_bib0034","series-title":"Proceedings of the Association for Computational Linguistics (ACL)","first-page":"12585","article-title":"Video-ChatGPT: towards detailed video understanding via large vision and language models","author":"Maaz","year":"2024"},{"issue":"9","key":"10.1016\/j.patcog.2026.113622_bib0035","doi-asserted-by":"crossref","first-page":"7543","DOI":"10.1109\/TPAMI.2025.3571946","article-title":"Otter: a multi-modal model with in-context instruction tuning","volume":"47","author":"Li","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113622_bib0036","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"323","article-title":"LLaMA-VID: an image is worth 2 tokens in large language models","author":"Li","year":"2024"},{"key":"10.1016\/j.patcog.2026.113622_bib0037","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"18221","article-title":"MovieChat: from dense token to sparse memory for long video understanding","author":"Song","year":"2024"},{"key":"10.1016\/j.patcog.2026.113622_bib0038","unstructured":"Q. Ye, H. Xu, G. Xu, et al., mPLUG-Owl: Modularization empowers large language models with multimodality, (2024) arXiv: 2304.14178."},{"key":"10.1016\/j.patcog.2026.113622_bib0039","series-title":"European Conference on Computer Vision","first-page":"1","article-title":"St-llm: large language models are effective temporal learners","author":"Liu","year":"2024"},{"key":"10.1016\/j.patcog.2026.113622_bib0040","unstructured":"Y. Liu, K.Q. Lin, C.W. Chen, M.Z. Shou, VideoMind: A Chain-of-LoRA Agent for Long Video Reasoning, (2025) arXiv: 2503.13444."},{"key":"10.1016\/j.patcog.2026.113622_bib0041","unstructured":"J. Jia, Y. Hu, X. Weng, et al., TinyLLaVA Factory: A modularized codebase for small-scale large multimodal models, (2024) arXiv: 2405.11788."},{"key":"10.1016\/j.patcog.2026.113622_bib0042","series-title":"Proceedings of the 1st International Workshop on Efficient Multimedia Computing under Limited","first-page":"18","article-title":"Llava-phi: efficient multi-modal assistant with small language model","author":"Zhu","year":"2024"},{"key":"10.1016\/j.patcog.2026.113622_bib0043","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"26689","article-title":"Vila: on pre-training for visual language models","author":"Lin","year":"2024"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032600587X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032600587X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T08:39:53Z","timestamp":1784277593000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S003132032600587X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":43,"alternative-id":["S003132032600587X"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113622","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"HM-RAG: Long video reasoning and anomaly detection via hierarchical multi-agent retrieval-augmented generation","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113622","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"113622"}}