{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T13:46:25Z","timestamp":1782481585528,"version":"3.54.5"},"reference-count":54,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62472161"],"award-info":[{"award-number":["62472161"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372150"],"award-info":[{"award-number":["62372150"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62202163"],"award-info":[{"award-number":["62202163"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114178","type":"journal-article","created":{"date-parts":[[2026,6,14]],"date-time":"2026-06-14T16:59:19Z","timestamp":1781456359000},"page":"114178","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["HAViG: Hierarchical adaptive visual grounding framework for video question answering"],"prefix":"10.1016","volume":"180","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4569-1429","authenticated-orcid":false,"given":"Lei","family":"Zhu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lingmin","family":"Pan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siqiao","family":"Tan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chengyuan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Deyin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lin Yuanbo","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Farid","family":"Boussaid","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mohammed","family":"Bennamoun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.114178_b1","doi-asserted-by":"crossref","unstructured":"C. Zang, Wang, et al., Discovering the real association: Multimodal causal reasoning in video question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Visionand Pattern Recognition, 2023, pp. 19027\u201319036.","DOI":"10.1109\/CVPR52729.2023.01824"},{"key":"10.1016\/j.patcog.2026.114178_b2","doi-asserted-by":"crossref","unstructured":"D. Gao, L. Zhou, et al., Mist: Multi-modal iterative spatial\u2013temporal transformer for long-form video question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Visionand Pattern Recognition, 2023, pp. 14773\u201314783.","DOI":"10.1109\/CVPR52729.2023.01419"},{"key":"10.1016\/j.patcog.2026.114178_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112362","article-title":"Robust scene text understanding with OCR token and word alignment for text-vqa and text-caption","volume":"172","author":"Jin","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114178_b4","doi-asserted-by":"crossref","first-page":"113740","DOI":"10.1016\/j.patcog.2026.113740","article-title":"Coffee-mate: Assisting keyframe selection via self-distillation for video question answering","author":"Yuan","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114178_b5","doi-asserted-by":"crossref","first-page":"4554","DOI":"10.1109\/TMM.2023.3323878","article-title":"Locate before answering: answer guided question localization for video question answering","volume":"26","author":"Qian","year":"2023","journal-title":"IEEE Trans. Multim."},{"key":"10.1016\/j.patcog.2026.114178_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112288","article-title":"FADMB: Fully attention-based dual memory bank network for weakly supervised video anomaly detection","volume":"172","author":"Luo","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114178_b7","doi-asserted-by":"crossref","first-page":"76749","DOI":"10.52202\/075280-3354","article-title":"Self-chained image-language model for video localization and question answering","volume":"36","author":"Yu","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114178_b8","article-title":"Collaborative aware bidirectional semantic reasoning for video question answering","author":"Wu","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.114178_b9","series-title":"Advances in Neural Information Processing Systems","first-page":"36","article-title":"Self-chained image-language model for video localization and question answering","author":"Yu","year":"2024"},{"key":"10.1016\/j.patcog.2026.114178_b10","doi-asserted-by":"crossref","unstructured":"R. Liao, M. Erler, et al., Videoinsta: Zero-shot long video understanding via informa-tive spatial\u2013temporal reasoning with llms, in: Findings of the Association for Computational Linguistics: EMNLP 2024, 2024, pp. 6577\u20136602.","DOI":"10.18653\/v1\/2024.findings-emnlp.384"},{"key":"10.1016\/j.patcog.2026.114178_b11","doi-asserted-by":"crossref","unstructured":"C. Zhang, T. Lu, et al., A simple llm framework for long-range video question answering, in: Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, 2024, pp. 21715\u201321737.","DOI":"10.18653\/v1\/2024.emnlp-main.1209"},{"key":"10.1016\/j.patcog.2026.114178_b12","doi-asserted-by":"crossref","unstructured":"D. Xu, Z. Zhao, et al., Video question answering via gradually refined attention over appearance and motion, in: Proceedings of the 25th ACM International Conference on Multimedia, 2017, pp. 1645\u20131653.","DOI":"10.1145\/3123266.3123427"},{"key":"10.1016\/j.patcog.2026.114178_b13","doi-asserted-by":"crossref","unstructured":"Z. Zhao, J. Lin, et al., Video question answering via hierarchical dual-level attention network learning, in: Proceedings of the 25th ACM International Con- Ference on Multimedia, 2017, pp. 1050\u20131058.","DOI":"10.1145\/3123266.3123364"},{"key":"10.1016\/j.patcog.2026.114178_b14","doi-asserted-by":"crossref","unstructured":"M. Tapaswi, Y. Zhu, et al., Movieqa: Understanding stories in movies through question-answering, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 4631\u20134640.","DOI":"10.1109\/CVPR.2016.501"},{"key":"10.1016\/j.patcog.2026.114178_b15","doi-asserted-by":"crossref","unstructured":"C. Fan, X. Zhang, et al., Heterogeneous memory enhanced multimodal attention model for video question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 1999\u20132007.","DOI":"10.1109\/CVPR.2019.00210"},{"key":"10.1016\/j.patcog.2026.114178_b16","doi-asserted-by":"crossref","unstructured":"S. Buch, C. Eyzaguirre, et al., Revisiting the \u201cvideo\u201d in video-language understanding, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 2917\u20132927.","DOI":"10.1109\/CVPR52688.2022.00293"},{"key":"10.1016\/j.patcog.2026.114178_b17","doi-asserted-by":"crossref","unstructured":"F. Liu, J. Liu, et al., Hair: Hierarchical visual-semantic relational reasoning for video question answering, in: Proceedings of the IEEE International Conference on Computer Vision, 2021, pp. 1698\u20131707.","DOI":"10.1109\/ICCV48922.2021.00172"},{"key":"10.1016\/j.patcog.2026.114178_b18","unstructured":"T.M. Le, V. Le, et al., Hierarchical conditional relation networks for video question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 9972\u20139981."},{"key":"10.1016\/j.patcog.2026.114178_b19","doi-asserted-by":"crossref","unstructured":"Y. Li, X. Wang, et al., Invariant grounding for video question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 2928\u20132937.","DOI":"10.1109\/CVPR52688.2022.00294"},{"key":"10.1016\/j.patcog.2026.114178_b20","doi-asserted-by":"crossref","unstructured":"N. Kim, S. Ha, et al., Video question answering using language-guided deep compressed-domain video feature, in: Proceedings of the IEEE International Conference on Computer Vision, 2021, pp. 1708\u20131717.","DOI":"10.1109\/ICCV48922.2021.00173"},{"key":"10.1016\/j.patcog.2026.114178_b21","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111080","article-title":"Semantic-aware frame-event fusion based pattern recognition via large vision\u2013language models","volume":"158","author":"Li","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114178_b22","doi-asserted-by":"crossref","unstructured":"Z. Yang, N. Garcia, et al., Bert representations for video question answering, in: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, 2020, pp. 1556\u20131565.","DOI":"10.1109\/WACV45572.2020.9093596"},{"key":"10.1016\/j.patcog.2026.114178_b23","series-title":"TINQ: Temporal inconsistency guided blind video quality assessment","author":"Li","year":"2024"},{"key":"10.1016\/j.patcog.2026.114178_b24","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2021.108145","article-title":"Generalized pyramid co-attention with learnable aggregation net for video question answering","volume":"120","author":"Gao","year":"2021","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114178_b25","series-title":"Internvideo: general video foundation models via generative and discriminative learning","author":"Wang","year":"2022"},{"key":"10.1016\/j.patcog.2026.114178_b26","series-title":"Question-instructed visual descriptions for zero-shot video question answering","author":"Romero","year":"2024"},{"key":"10.1016\/j.patcog.2026.114178_b27","series-title":"Language repository for long video understanding","author":"Kahatapitiya","year":"2024"},{"key":"10.1016\/j.patcog.2026.114178_b28","doi-asserted-by":"crossref","unstructured":"J. Kim, M. Ma, et al., Progressive attention memory network for movie story question answering, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2019, pp. 8337\u20138346.","DOI":"10.1109\/CVPR.2019.00853"},{"issue":"5","key":"10.1016\/j.patcog.2026.114178_b29","doi-asserted-by":"crossref","first-page":"6826","DOI":"10.1109\/TCSVT.2026.3657415","article-title":"DVLTA-VQA: decoupled vision-language modeling with text-guided adaptation for blind video quality assessment","volume":"36","author":"Yu","year":"2026","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.114178_b30","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.108959","article-title":"Dynamic self-attention with vision synchronization networks for video question answering","volume":"132","author":"Liu","year":"2022","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114178_b31","series-title":"2020 IEEE International Conference on Visual Communications and Image Processing","first-page":"338","article-title":"Deep local and global spatiotemporal feature aggregation for blind video quality assessment","author":"Zhou","year":"2020"},{"issue":"5","key":"10.1016\/j.patcog.2026.114178_b32","doi-asserted-by":"crossref","first-page":"2019","DOI":"10.1109\/TIP.2014.2311377","article-title":"Click prediction for web image reranking using multimodal sparse coding","volume":"23","author":"Yu","year":"2014","journal-title":"IEEE Trans. on Image Proc."},{"key":"10.1016\/j.patcog.2026.114178_b33","doi-asserted-by":"crossref","unstructured":"Z. Wang, S. Yu, et al., Videotree: Adaptive tree-based video representation for llm reasoning on long videos, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 3272\u20133283.","DOI":"10.1109\/CVPR52734.2025.00311"},{"key":"10.1016\/j.patcog.2026.114178_b34","doi-asserted-by":"crossref","first-page":"702","DOI":"10.1016\/j.physa.2019.03.012","article-title":"A novel density peaks clustering algorithm based on k nearest neighbors for improving assignment process","volume":"523","author":"Jiang","year":"2019","journal-title":"Phys. A"},{"key":"10.1016\/j.patcog.2026.114178_b35","doi-asserted-by":"crossref","unstructured":"J. Xiao, X. Shang, et al., Next-qa: Next phase of question-answering to explaining temporal actions, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 9777\u20139786.","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"10.1016\/j.patcog.2026.114178_b36","unstructured":"B. Wu, S. Yu, Z. Chen, et al., STAR: A benchmark for situated reasoning in real-world videos, in: Proceedings of the Thirty-Fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2), 2021."},{"key":"10.1016\/j.patcog.2026.114178_b37","doi-asserted-by":"crossref","unstructured":"T. Lin, M. Maire, et al., Microsoft coco: Common objects in context, in: Computer Vision-ECCV 2014: 13th European Conference, 2014, pp. 740\u2013755.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"10.1016\/j.patcog.2026.114178_b38","doi-asserted-by":"crossref","unstructured":"P. Sharma, et al., Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning, in: Proceedings of the 56th Association for Computational Linguistics, 2018, pp. 2556\u20132565.","DOI":"10.18653\/v1\/P18-1238"},{"key":"10.1016\/j.patcog.2026.114178_b39","doi-asserted-by":"crossref","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","article-title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations","volume":"123","author":"Krishna","year":"2017","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.patcog.2026.114178_b40","series-title":"Advances in neural information processing systems","first-page":"24","article-title":"Im2text: describing images using 1 million captioned photographs","author":"Ordonez","year":"2011"},{"key":"10.1016\/j.patcog.2026.114178_b41","doi-asserted-by":"crossref","unstructured":"Q. Ye, G. Xu, et al., Hitea: Hierarchical temporal-aware video-language pre-training, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 15405\u201315416.","DOI":"10.1109\/ICCV51070.2023.01413"},{"key":"10.1016\/j.patcog.2026.114178_b42","doi-asserted-by":"crossref","DOI":"10.1109\/TCSVT.2024.3409453","article-title":"Cfmmc-align: coarse-fine multi-modal contrastive alignment network for traffic event video question answer- ing","author":"Guo","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.114178_b43","doi-asserted-by":"crossref","unstructured":"A. Urooj, H. Kuehne, et al., Learning situation hyper-graphs for video question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 14879\u201314889.","DOI":"10.1109\/CVPR52729.2023.01429"},{"key":"10.1016\/j.patcog.2026.114178_b44","article-title":"Multi-granularity contrastive cross-modal collaborative generation for end-to-end long- term video question answering","author":"Yu","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.patcog.2026.114178_b45","series-title":"Understanding long videos with multimodal language models","author":"Ranasinghe","year":"2024"},{"key":"10.1016\/j.patcog.2026.114178_b46","series-title":"Mixtral of experts","author":"Jiang","year":"2024"},{"key":"10.1016\/j.patcog.2026.114178_b47","doi-asserted-by":"crossref","unstructured":"D. Sur\u00eds, S. Menon, C. Vondrick, Vipergpt: Visual inference via python execution for reasoning, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 11888\u201311898.","DOI":"10.1109\/ICCV51070.2023.01092"},{"key":"10.1016\/j.patcog.2026.114178_b48","doi-asserted-by":"crossref","unstructured":"F. Ma, X. Jin, et al., Vista- llama: Reducing hallucination in video language models via equal distance to visual tokens, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2024, pp. 13151\u201313160.","DOI":"10.1109\/CVPR52733.2024.01249"},{"key":"10.1016\/j.patcog.2026.114178_b49","series-title":"Video-star: self-training enables video instruction tuning with any supervision","author":"Zohar","year":"2024"},{"key":"10.1016\/j.patcog.2026.114178_b50","doi-asserted-by":"crossref","unstructured":"H. Liu, F. Ilievski, C. Snoek, Commonsense video question answering through video-grounded entailment tree reasoning, in: Proceedings of the Computer Vision and Pattern Recognition Conference. Computer Vision Foundation, New York, NY, USA, 2025, pp. 3262\u20133271.","DOI":"10.1109\/CVPR52734.2025.00310"},{"key":"10.1016\/j.patcog.2026.114178_b51","doi-asserted-by":"crossref","unstructured":"X. Wang, Y. Zhang, et al., Videoagent: Long-form video understanding with large language model as agent, in: European Conference on Computer Vision, 2024, Springer, pp. 58\u201376.","DOI":"10.1007\/978-3-031-72989-8_4"},{"key":"10.1016\/j.patcog.2026.114178_b52","doi-asserted-by":"crossref","first-page":"5697","DOI":"10.52202\/079017-0185","article-title":"Topa: Extending large language models for video understanding via text-only pre-alignment","volume":"37","author":"Li","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114178_b53","doi-asserted-by":"crossref","DOI":"10.1109\/ACCESS.2024.3517625","article-title":"An image grid can be worth a video: zero-shot video question answering using a vlm","author":"Kim","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.patcog.2026.114178_b54","doi-asserted-by":"crossref","unstructured":"R. Choudhury, K. Niinuma, et al., Video question answering with procedural programs, in: European Conference on Computer Vision, 2024, pp. 315\u2013332.","DOI":"10.1007\/978-3-031-72920-1_18"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032601143X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032601143X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T12:52:22Z","timestamp":1782478342000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S003132032601143X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":54,"alternative-id":["S003132032601143X"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114178","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"HAViG: Hierarchical adaptive visual grounding framework for video question answering","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114178","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"114178"}}