{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T08:04:24Z","timestamp":1784534664890,"version":"3.55.0"},"reference-count":35,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012165","name":"Key Technologies Research and Development Program","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012165","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Journal of Visual Communication and Image Representation"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.jvcir.2026.104886","type":"journal-article","created":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T07:03:30Z","timestamp":1782803010000},"page":"104886","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MotionViLA: A motion-aware video understanding model for video localization and question answering"],"prefix":"10.1016","volume":"119","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1999-4335","authenticated-orcid":false,"given":"Zehua","family":"Ji","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weifeng","family":"Lv","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0117-3494","authenticated-orcid":false,"given":"Junlin","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zekun","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jian","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.jvcir.2026.104886_b1","doi-asserted-by":"crossref","unstructured":"J. Lei, L. Li, L. Zhou, Z. Gan, T.L. Berg, M. Bansal, J. Liu, Less is more: Clipbert for video-and-language learning via sparse sampling, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 7331\u20137341.","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"10.1016\/j.jvcir.2026.104886_b2","doi-asserted-by":"crossref","first-page":"76749","DOI":"10.52202\/075280-3354","article-title":"Self-chained image-language model for video localization and question answering","volume":"36","author":"Yu","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104886_b3","series-title":"European Conference on Computer Vision","first-page":"186","article-title":"Vila: Efficient video-language alignment for video question answering","author":"Wang","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104886_b4","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104886_b5","series-title":"Large language models are temporal and causal reasoners for video question answering","author":"Ko","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104886_b6","series-title":"European Conference on Computer Vision","first-page":"39","article-title":"Video graph transformer for video question answering","author":"Xiao","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104886_b7","series-title":"Vlap: Efficient video-language alignment via frame prompting and distilling for video question answering","author":"Wang","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104886_b8","series-title":"A simple llm framework for long-range video question-answering","author":"Zhang","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104886_b9","series-title":"European Conference on Computer Vision","first-page":"58","article-title":"Videoagent: Long-form video understanding with large language model as agent","author":"Wang","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104886_b10","doi-asserted-by":"crossref","unstructured":"P. Papalampidi, S. Koppula, S. Pathak, J. Chiu, J. Heyward, V. Patraucean, J. Shen, A. Miech, A. Zisserman, A. Nematzdeh, A simple recipe for contrastively pre-training video-first encoders beyond 16 frames, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 14386\u201314397.","DOI":"10.1109\/CVPR52733.2024.01364"},{"key":"10.1016\/j.jvcir.2026.104886_b11","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2023.111136","article-title":"Vlp2msa: expanding vision-language pre-training to multimodal sentiment analysis","volume":"283","author":"Yi","year":"2024","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.jvcir.2026.104886_b12","article-title":"Hierarchically trusted evidential fusion method with consistency learning for multimodal language understanding","author":"Yang","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.jvcir.2026.104886_b13","series-title":"Valor: Vision-audio-language omni-perception pretraining model and dataset","author":"Chen","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104886_b14","doi-asserted-by":"crossref","first-page":"72842","DOI":"10.52202\/075280-3185","article-title":"Vast: A vision-audio-subtitle-text omni-modality foundation model and dataset","volume":"36","author":"Chen","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104886_b15","doi-asserted-by":"crossref","unstructured":"J. Devlin, M.W. Chang, K. Lee, K. Toutanova, Bert: Pre-training of deep bidirectional transformers for language understanding, in: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), 2019, pp. 4171\u20134186.","DOI":"10.18653\/v1\/N19-1423"},{"key":"10.1016\/j.jvcir.2026.104886_b16","series-title":"Deberta: Decoding-enhanced bert with disentangled attention","author":"He","year":"2020"},{"key":"10.1016\/j.jvcir.2026.104886_b17","series-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"issue":"70","key":"10.1016\/j.jvcir.2026.104886_b18","first-page":"1","article-title":"Scaling instruction-finetuned language models","volume":"25","author":"Chung","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.jvcir.2026.104886_b19","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"9972","article-title":"Hierarchical conditional relation networks for video question answering","author":"Le","year":"2020"},{"key":"10.1016\/j.jvcir.2026.104886_b20","doi-asserted-by":"crossref","unstructured":"J. Xiao, A. Yao, Y. Li, T.-S. Chua, Can i trust your answer? visually grounded video question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13204\u201313214.","DOI":"10.1109\/CVPR52733.2024.01254"},{"issue":"11","key":"10.1016\/j.jvcir.2026.104886_b21","doi-asserted-by":"crossref","first-page":"13265","DOI":"10.1109\/TPAMI.2023.3292266","article-title":"Contrastive video question answering via video graph transformer","volume":"45","author":"Xiao","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.jvcir.2026.104886_b22","doi-asserted-by":"crossref","unstructured":"J. Fang, L.l. Li, J. Zhou, J. Xiao, H. Yu, C. Lv, J. Xue, T.-S. Chua, Abductive Ego-View Accident Video Understanding for Safe Driving Perception, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 22030\u201322040.","DOI":"10.1109\/CVPR52733.2024.02080"},{"key":"10.1016\/j.jvcir.2026.104886_b23","doi-asserted-by":"crossref","unstructured":"J. Xiao, X. Shang, A. Yao, T.-S. Chua, Next-qa: Next phase of question-answering to explaining temporal actions, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 9777\u20139786.","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"10.1016\/j.jvcir.2026.104886_b24","series-title":"Star: A benchmark for situated reasoning in real-world videos","author":"Wu","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104886_b25","doi-asserted-by":"crossref","unstructured":"A. Yang, A. Miech, J. Sivic, I. Laptev, C. Schmid, Just ask: Learning to answer questions from millions of narrated videos, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 1686\u20131697.","DOI":"10.1109\/ICCV48922.2021.00171"},{"key":"10.1016\/j.jvcir.2026.104886_b26","doi-asserted-by":"crossref","unstructured":"J. Wang, Y. Ge, R. Yan, Y. Ge, K.Q. Lin, S. Tsutsui, X. Lin, G. Cai, J. Wu, Y. Shan, et al., All in one: Exploring unified video-language pre-training, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 6598\u20136608.","DOI":"10.1109\/CVPR52729.2023.00638"},{"key":"10.1016\/j.jvcir.2026.104886_b27","doi-asserted-by":"crossref","unstructured":"S. Buch, C. Eyzaguirre, A. Gaidon, J. Wu, L. Fei-Fei, J.C. Niebles, Revisiting the\u201d video\u201d in video-language understanding, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 2917\u20132927.","DOI":"10.1109\/CVPR52688.2022.00293"},{"key":"10.1016\/j.jvcir.2026.104886_b28","series-title":"European Conference on Computer Vision","first-page":"39","article-title":"Video graph transformer for video question answering","author":"Xiao","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104886_b29","doi-asserted-by":"crossref","unstructured":"D. Gao, L. Zhou, L. Ji, L. Zhu, Y. Yang, M.Z. Shou, Mist: Multi-modal iterative spatial-temporal transformer for long-form video question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 14773\u201314783.","DOI":"10.1109\/CVPR52729.2023.01419"},{"key":"10.1016\/j.jvcir.2026.104886_b30","doi-asserted-by":"crossref","unstructured":"L. Momeni, M. Caron, A. Nagrani, A. Zisserman, C. Schmid, Verbs in action: Improving verb understanding in video-language models, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 15579\u201315591.","DOI":"10.1109\/ICCV51070.2023.01428"},{"issue":"11","key":"10.1016\/j.jvcir.2026.104886_b31","doi-asserted-by":"crossref","first-page":"13265","DOI":"10.1109\/TPAMI.2023.3292266","article-title":"Contrastive video question answering via video graph transformer","volume":"45","author":"Xiao","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.jvcir.2026.104886_b32","series-title":"Semi-parametric video-grounded text generation","author":"Kim","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104886_b33","doi-asserted-by":"crossref","unstructured":"Q. Ye, G. Xu, M. Yan, H. Xu, Q. Qian, J. Zhang, F. Huang, Hitea: Hierarchical temporal-aware video-language pre-training, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 15405\u201315416.","DOI":"10.1109\/ICCV51070.2023.01413"},{"key":"10.1016\/j.jvcir.2026.104886_b34","series-title":"Internvideo: General video foundation models via generative and discriminative learning","author":"Wang","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104886_b35","series-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"}],"container-title":["Journal of Visual Communication and Image Representation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326001811?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326001811?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T07:24:34Z","timestamp":1784532274000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1047320326001811"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":35,"alternative-id":["S1047320326001811"],"URL":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104886","relation":{},"ISSN":["1047-3203"],"issn-type":[{"value":"1047-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MotionViLA: A motion-aware video understanding model for video localization and question answering","name":"articletitle","label":"Article Title"},{"value":"Journal of Visual Communication and Image Representation","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104886","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104886"}}