{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T08:04:51Z","timestamp":1784534691753,"version":"3.55.0"},"reference-count":62,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100004739","name":"Youth Innovation Promotion Association of the Chinese Academy of Sciences","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004739","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476068"],"award-info":[{"award-number":["62476068"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62471013"],"award-info":[{"award-number":["62471013"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62406305"],"award-info":[{"award-number":["62406305"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62525212"],"award-info":[{"award-number":["62525212"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23B2051"],"award-info":[{"award-number":["U23B2051"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Journal of Visual Communication and Image Representation"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.jvcir.2026.104879","type":"journal-article","created":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T22:25:23Z","timestamp":1781994323000},"page":"104879","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["CapVA: Towards hierarchical understanding in Text\u2013Video Retrieval With Caption as Video Assistant"],"prefix":"10.1016","volume":"119","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1132-859X","authenticated-orcid":false,"given":"Zhipeng","family":"Yu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yangbangyan","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qianqian","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.jvcir.2026.104879_b1","doi-asserted-by":"crossref","unstructured":"Y. Ma, G. Xu, X. Sun, M. Yan, J. Zhang, R. Ji, X-clip: End-to-end multi-grained contrastive learning for video-text retrieval, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 638\u2013647.","DOI":"10.1145\/3503161.3547910"},{"key":"10.1016\/j.jvcir.2026.104879_b2","doi-asserted-by":"crossref","unstructured":"X. Wang, Y. Li, T. Gan, Z. Zhang, J. Lv, L. Nie, RTQ: Rethinking Video-language Understanding Based on Image-text Model, in: Proceedings of the 31st ACM International Conference on Multimedia, 2023, pp. 557\u2013566.","DOI":"10.1145\/3581783.3612152"},{"key":"10.1016\/j.jvcir.2026.104879_b3","doi-asserted-by":"crossref","unstructured":"H. Fang, Z. Yang, X. Zang, C. Ban, Z. He, H. Sun, L. Zhou, Mask to Reconstruct: Cooperative Semantics Completion for Video-text Retrieval, in: Proceedings of the 31st ACM International Conference on Multimedia, 2023, pp. 3847\u20133856.","DOI":"10.1145\/3581783.3611756"},{"key":"10.1016\/j.jvcir.2026.104879_b4","doi-asserted-by":"crossref","unstructured":"Q. Li, Y. Zhou, C. Ji, F. Lu, J. Gong, S. Wang, J. Li, Multi-Modal Inductive Framework for Text-Video Retrieval, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 2389\u20132398.","DOI":"10.1145\/3664647.3681024"},{"key":"10.1016\/j.jvcir.2026.104879_b5","unstructured":"A. Radford, J.W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, et al., Learning transferable visual models from natural language supervision, in: International Conference on Machine Learning, 2021, pp. 8748\u20138763."},{"key":"10.1016\/j.jvcir.2026.104879_b6","doi-asserted-by":"crossref","first-page":"293","DOI":"10.1016\/j.neucom.2022.07.028","article-title":"Clip4clip: An empirical study of clip for end to end video clip retrieval and captioning","volume":"508","author":"Luo","year":"2022","journal-title":"Neurocomputing"},{"key":"10.1016\/j.jvcir.2026.104879_b7","doi-asserted-by":"crossref","unstructured":"Y. Liu, P. Xiong, L. Xu, S. Cao, Q. Jin, Ts2-net: Token shift and selection transformer for text-video retrieval, in: European Conference on Computer Vision, 2022, pp. 319\u2013335.","DOI":"10.1007\/978-3-031-19781-9_19"},{"key":"10.1016\/j.jvcir.2026.104879_b8","doi-asserted-by":"crossref","unstructured":"Z. Wang, Y.-L. Sung, F. Cheng, G. Bertasius, M. Bansal, Unified coarse-to-fine alignment for video-text retrieval, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 2816\u20132827.","DOI":"10.1109\/ICCV51070.2023.00264"},{"key":"10.1016\/j.jvcir.2026.104879_b9","doi-asserted-by":"crossref","unstructured":"S.K. Gorti, N. Vouitsis, J. Ma, K. Golestan, M. Volkovs, A. Garg, G. Yu, X-pool: Cross-modal language-video attention for text-video retrieval, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 5006\u20135015.","DOI":"10.1109\/CVPR52688.2022.00495"},{"key":"10.1016\/j.jvcir.2026.104879_b10","series-title":"ICLR","article-title":"CLIP-ViP: Adapting pre-trained image-text model to video-language alignment","author":"Xue","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104879_b11","doi-asserted-by":"crossref","unstructured":"W. Wu, H. Luo, B. Fang, J. Wang, W. Ouyang, Cap4video: What can auxiliary captions do for text-video retrieval?, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 10704\u201310713.","DOI":"10.1109\/CVPR52729.2023.01031"},{"key":"10.1016\/j.jvcir.2026.104879_b12","doi-asserted-by":"crossref","unstructured":"V. Gabeur, C. Sun, K. Alahari, C. Schmid, Multi-modal transformer for video retrieval, in: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IV 16, 2020, pp. 214\u2013229.","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"10.1016\/j.jvcir.2026.104879_b13","unstructured":"Y. Liu, S. Albanie, A. Nagrani, A. Zisserman, Use What You Have: Video retrieval using representations from collaborative experts, in: Proceedings of the British Machine Vision Conference, BMVC, 2019."},{"key":"10.1016\/j.jvcir.2026.104879_b14","doi-asserted-by":"crossref","unstructured":"M. Bain, A. Nagrani, G. Varol, A. Zisserman, Frozen in time: A joint video and image encoder for end-to-end retrieval, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 1728\u20131738.","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"10.1016\/j.jvcir.2026.104879_b15","doi-asserted-by":"crossref","unstructured":"A. Miech, J.-B. Alayrac, L. Smaira, I. Laptev, J. Sivic, A. Zisserman, End-to-end learning of visual representations from uncurated instructional videos, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 9879\u20139889.","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"10.1016\/j.jvcir.2026.104879_b16","doi-asserted-by":"crossref","unstructured":"A. Miech, D. Zhukov, J.-B. Alayrac, M. Tapaswi, I. Laptev, J. Sivic, Howto100m: Learning a text-video embedding by watching hundred million narrated video clips, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019, pp. 2630\u20132640.","DOI":"10.1109\/ICCV.2019.00272"},{"key":"10.1016\/j.jvcir.2026.104879_b17","doi-asserted-by":"crossref","unstructured":"W. Wu, X. Wang, H. Luo, J. Wang, Y. Yang, W. Ouyang, Bidirectional cross-modal knowledge exploration for video recognition with pre-trained vision-language models, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 6620\u20136630.","DOI":"10.1109\/CVPR52729.2023.00640"},{"key":"10.1016\/j.jvcir.2026.104879_b18","doi-asserted-by":"crossref","unstructured":"J. Lei, L. Li, L. Zhou, Z. Gan, T.L. Berg, M. Bansal, J. Liu, Less is more: Clipbert for video-and-language learning via sparse sampling, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 7331\u20137341.","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"10.1016\/j.jvcir.2026.104879_b19","series-title":"Clip2video: Mastering video-text retrieval via image clip","author":"Fang","year":"2021"},{"key":"10.1016\/j.jvcir.2026.104879_b20","doi-asserted-by":"crossref","unstructured":"Y. Ma, G. Xu, X. Sun, M. Yan, J. Zhang, R. Ji, X-clip: End-to-end multi-grained contrastive learning for video-text retrieval, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 638\u2013647.","DOI":"10.1145\/3503161.3547910"},{"key":"10.1016\/j.jvcir.2026.104879_b21","doi-asserted-by":"crossref","unstructured":"H. Le, T. Kieu, A. Nguyen, N. Le, Waver: Writing-style agnostic text-video retrieval via distilling vision-language models through open-vocabulary knowledge, in: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, 2024, pp. 3025\u20133029.","DOI":"10.1109\/ICASSP48485.2024.10446193"},{"key":"10.1016\/j.jvcir.2026.104879_b22","article-title":"Tencent text-video retrieval: hierarchical cross-modal interactions with multi-level representations","author":"Jiang","year":"2022","journal-title":"IEEE Access"},{"key":"10.1016\/j.jvcir.2026.104879_b23","doi-asserted-by":"crossref","first-page":"30291","DOI":"10.52202\/068431-2196","article-title":"Expectation-maximization contrastive learning for compact video-and-language representations","volume":"35","author":"Jin","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104879_b24","doi-asserted-by":"crossref","unstructured":"X. Zhuang, H. Li, X. Cheng, Z. Zhu, Y. Xie, Y. Zou, Kdpror: A knowledge-decoupling probabilistic framework for video-text retrieval, in: European Conference on Computer Vision, 2024, pp. 313\u2013331.","DOI":"10.1007\/978-3-031-72754-2_18"},{"key":"10.1016\/j.jvcir.2026.104879_b25","doi-asserted-by":"crossref","unstructured":"P. Jin, J. Huang, P. Xiong, S. Tian, C. Liu, X. Ji, L. Yuan, J. Chen, Video-text as game players: Hierarchical banzhaf interaction for cross-modal representation learning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 2472\u20132482.","DOI":"10.1109\/CVPR52729.2023.00244"},{"key":"10.1016\/j.jvcir.2026.104879_b26","doi-asserted-by":"crossref","unstructured":"P. Jin, H. Li, Z. Cheng, K. Li, X. Ji, C. Liu, L. Yuan, J. Chen, Diffusionret: Generative text-video retrieval with diffusion model, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 2470\u20132481.","DOI":"10.1109\/ICCV51070.2023.00234"},{"key":"10.1016\/j.jvcir.2026.104879_b27","first-page":"1","article-title":"Learning text-to-video retrieval from image captioning","author":"Ventura","year":"2024","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.jvcir.2026.104879_b28","doi-asserted-by":"crossref","unstructured":"C. Deng, Q. Chen, P. Qin, D. Chen, Q. Wu, Prompt switch: Efficient clip adaptation for text-video retrieval, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 15648\u201315658.","DOI":"10.1109\/ICCV51070.2023.01434"},{"key":"10.1016\/j.jvcir.2026.104879_b29","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"8","key":"10.1016\/j.jvcir.2026.104879_b30","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"10.1016\/j.jvcir.2026.104879_b31","doi-asserted-by":"crossref","unstructured":"Y. Tewel, Y. Shalev, I. Schwartz, L. Wolf, Zerocap: Zero-shot image-to-text generation for visual-semantic arithmetic, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 17918\u201317928.","DOI":"10.1109\/CVPR52688.2022.01739"},{"key":"10.1016\/j.jvcir.2026.104879_b32","series-title":"34th British Machine Vision Conference 2023, BMVC 2023, Aberdeen, UK, November 20-24, 2023","article-title":"Zero-shot video captioning by evolving pseudo-tokens","author":"Tewel","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104879_b33","series-title":"FreeVA: Offline MLLM as training-free video assistant","author":"Wu","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104879_b34","doi-asserted-by":"crossref","first-page":"34892","DOI":"10.52202\/075280-1516","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104879_b35","series-title":"ChatGPT","author":"OpenAI","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104879_b36","series-title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","author":"Team","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104879_b37","doi-asserted-by":"crossref","unstructured":"T. Formal, B. Piwowarski, S. Clinchant, SPLADE: Sparse lexical and expansion model for first stage ranking, in: Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval, 2021, pp. 2288\u20132292.","DOI":"10.1145\/3404835.3463098"},{"key":"10.1016\/j.jvcir.2026.104879_b38","doi-asserted-by":"crossref","unstructured":"E.M. Voorhees, Query expansion using lexical-semantic relations, in: SIGIR\u201994: Proceedings of the Seventeenth Annual International ACM-SIGIR Conference on Research and Development in Information Retrieval, Organised By Dublin City University, 1994, pp. 61\u201369.","DOI":"10.1007\/978-1-4471-2099-5_7"},{"key":"10.1016\/j.jvcir.2026.104879_b39","series-title":"ACM SIGIR Forum","first-page":"260","article-title":"Relevance-based language models","volume":"Vol. 51","author":"Lavrenko","year":"2017"},{"key":"10.1016\/j.jvcir.2026.104879_b40","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2020","first-page":"4718","article-title":"BERT-QE: Contextualized query expansion for document re-ranking","author":"Zheng","year":"2020"},{"key":"10.1016\/j.jvcir.2026.104879_b41","unstructured":"Z. Bai, T. Xiao, T. He, P. Wang, Z. Zhang, T. Brox, M.Z. Shou, Bridging Information Asymmetry in Text-video Retrieval: A Data-centric Approach, in: The Thirteenth International Conference on Learning Representations, 2025."},{"key":"10.1016\/j.jvcir.2026.104879_b42","series-title":"Clip2tv: An empirical study on transformer-based methods for video-text retrieval","author":"Gao","year":"2021"},{"key":"10.1016\/j.jvcir.2026.104879_b43","doi-asserted-by":"crossref","unstructured":"B. Ni, H. Peng, M. Chen, S. Zhang, G. Meng, J. Fu, S. Xiang, H. Ling, Expanding language-image pretrained models for general video recognition, in: European Conference on Computer Vision, 2022, pp. 1\u201318.","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"10.1016\/j.jvcir.2026.104879_b44","article-title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","volume":"32","author":"Lu","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104879_b45","unstructured":"M. Patrick, P.-Y. Huang, Y. Asano, F. Metze, A.G. Hauptmann, J.F. Henriques, A. Vedaldi, Support-set bottlenecks for video-text representation learning, in: International Conference on Learning Representations, 2021."},{"key":"10.1016\/j.jvcir.2026.104879_b46","doi-asserted-by":"crossref","first-page":"38655","DOI":"10.52202\/068431-2801","article-title":"Text-adaptive multiple visual prototype matching for video-text retrieval","volume":"35","author":"Lin","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104879_b47","doi-asserted-by":"crossref","unstructured":"H. Fang, X. Zang, C. Ban, Z. Feng, L. Zhou, Z. He, Y. Li, H. Sun, ProTA: Probabilistic Token Aggregation for Text-Video Retrieval, in: 2024 IEEE International Conference on Multimedia and Expo, ICME, 2024, pp. 1\u20136.","DOI":"10.1109\/ICME57554.2024.10687550"},{"key":"10.1016\/j.jvcir.2026.104879_b48","doi-asserted-by":"crossref","unstructured":"K. Tian, Y. Cheng, Y. Liu, X. Hou, Q. Chen, H. Li, Towards efficient and effective text-to-video retrieval with coarse-to-fine visual representation learning, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 38, 2024, pp. 5207\u20135214, (6).","DOI":"10.1609\/aaai.v38i6.28327"},{"key":"10.1016\/j.jvcir.2026.104879_b49","doi-asserted-by":"crossref","unstructured":"J. Xu, T. Mei, T. Yao, Y. Rui, Msr-vtt: A large video description dataset for bridging video and language, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 5288\u20135296.","DOI":"10.1109\/CVPR.2016.571"},{"key":"10.1016\/j.jvcir.2026.104879_b50","doi-asserted-by":"crossref","unstructured":"Y. Yu, J. Kim, G. Kim, A joint sequence fusion model for video question answering and retrieval, in: Proceedings of the European Conference on Computer Vision, ECCV, 2018, pp. 471\u2013487.","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"10.1016\/j.jvcir.2026.104879_b51","series-title":"Frontiers of Multimedia Research","first-page":"3","article-title":"Deep learning for video classification and captioning","author":"Wu","year":"2017"},{"key":"10.1016\/j.jvcir.2026.104879_b52","doi-asserted-by":"crossref","unstructured":"L. Anne Hendricks, O. Wang, E. Shechtman, J. Sivic, T. Darrell, B. Russell, Localizing moments in video with natural language, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 5803\u20135812.","DOI":"10.1109\/ICCV.2017.618"},{"key":"10.1016\/j.jvcir.2026.104879_b53","doi-asserted-by":"crossref","unstructured":"X. Wang, J. Wu, J. Chen, L. Li, Y.-F. Wang, W.Y. Wang, Vatex: A large-scale, high-quality multilingual dataset for video-and-language research, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019, pp. 4581\u20134591.","DOI":"10.1109\/ICCV.2019.00468"},{"key":"10.1016\/j.jvcir.2026.104879_b54","doi-asserted-by":"crossref","unstructured":"R. Krishna, K. Hata, F. Ren, L. Fei-Fei, J. Carlos Niebles, Dense-captioning events in videos, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 706\u2013715.","DOI":"10.1109\/ICCV.2017.83"},{"key":"10.1016\/j.jvcir.2026.104879_b55","doi-asserted-by":"crossref","DOI":"10.1007\/s11263-016-0987-1","article-title":"Movie description","author":"Rohrbach","year":"2017","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.jvcir.2026.104879_b56","doi-asserted-by":"crossref","unstructured":"B. Fang, W. Wu, C. Liu, Y. Zhou, Y. Song, W. Wang, X. Shu, X. Ji, J. Wang, Uatvr: Uncertainty-adaptive text-video retrieval, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 13723\u201313733.","DOI":"10.1109\/ICCV51070.2023.01262"},{"key":"10.1016\/j.jvcir.2026.104879_b57","doi-asserted-by":"crossref","unstructured":"J. Wang, G. Sun, P. Wang, D. Liu, S. Dianat, M. Rabbani, R. Rao, Z. Tao, Text is mass: Modeling as stochastic embedding for text-video retrieval, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 16551\u201316560.","DOI":"10.1109\/CVPR52733.2024.01566"},{"key":"10.1016\/j.jvcir.2026.104879_b58","series-title":"ICLR","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2021"},{"key":"10.1016\/j.jvcir.2026.104879_b59","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TIP.2019.2923608","article-title":"Neural multimodal cooperative learning toward micro-video understanding","volume":"29","author":"Wei","year":"2019","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.jvcir.2026.104879_b60","series-title":"Qwen2. 5-vl technical report","author":"Bai","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104879_b61","series-title":"Qwen2.5-1M: Deploy your own qwen with context length up to 1M tokens","author":"Team","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104879_b62","series-title":"Qwen2.5-1M technical report","author":"Yang","year":"2025"}],"container-title":["Journal of Visual Communication and Image Representation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326001744?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326001744?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T07:27:42Z","timestamp":1784532462000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1047320326001744"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":62,"alternative-id":["S1047320326001744"],"URL":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104879","relation":{},"ISSN":["1047-3203"],"issn-type":[{"value":"1047-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"CapVA: Towards hierarchical understanding in Text\u2013Video Retrieval With Caption as Video Assistant","name":"articletitle","label":"Article Title"},{"value":"Journal of Visual Communication and Image Representation","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104879","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Inc.","name":"copyright","label":"Copyright"}],"article-number":"104879"}}