{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T20:03:37Z","timestamp":1781553817017,"version":"3.54.5"},"reference-count":90,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T00:00:00Z","timestamp":1729555200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T00:00:00Z","timestamp":1729555200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1007\/s11263-024-02202-8","type":"journal-article","created":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T17:03:59Z","timestamp":1729616639000},"page":"1834-1854","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Learning Text-to-Video Retrieval from Image Captioning"],"prefix":"10.1007","volume":"133","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5795-0064","authenticated-orcid":false,"given":"Lucas","family":"Ventura","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cordelia","family":"Schmid","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"G\u00fcl","family":"Varol","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,22]]},"reference":[{"key":"2202_CR1","unstructured":"Alayrac, J. B., Recasens, A., Schneider, R., Arandjelovi\u0107, R., Ramapuram, J., De Fauw, J., Smaira, L., Dieleman, S., & Zisserman, A. (2020). Self-supervised multimodal versatile networks. In: NeurIPS."},{"key":"2202_CR2","unstructured":"Alwassel, H., Mahajan, D., Korbar, B., Torresani, L., Ghanem, B., & Tran, D. (2020). Self-supervised learning by cross-modal audio-video clustering. In: NeurIPS."},{"key":"2202_CR3","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., Teney, D., Johnson, M., Gould, S., & Zhang, L. (2018). Bottom-up and top\u2013down attention for image captioning and visual question answering. In: CVPR.","DOI":"10.1109\/CVPR.2018.00636"},{"key":"2202_CR4","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., & Zisserman, A. (2021). Frozen in time: A joint video and image encoder for end-to-end retrieval. In: ICCV.","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"2202_CR5","unstructured":"Bain, M., Nagrani, A., Varol, G., & Zisserman, A. (2022). A CLIP-hitchhiker\u2019s guide to long video retrieval. arXiv"},{"key":"2202_CR6","unstructured":"Banerjee, S., & Lavie, A. (2005). METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In: Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization."},{"key":"2202_CR7","doi-asserted-by":"crossref","unstructured":"Carreira, J., & Zisserman, A. (2017). Quo vadis, action recognition? A new model and the Kinetics dataset. In CVPR.","DOI":"10.1109\/CVPR.2017.502"},{"key":"2202_CR8","unstructured":"Castro, S., & Heilbron, F.C. (2022). FitCLIP: Refining large-scale pretrained image-text models for zero-shot video understanding tasks. arXiv"},{"key":"2202_CR9","unstructured":"Chen, D., & Dolan, W. (2011). Collecting highly parallel data for paraphrase evaluation. In Annual Meeting of the Association for Computational Linguistics: Human Language Technologies."},{"key":"2202_CR10","unstructured":"Chen, T., Kornblith, S., Norouzi, M., & Hinton, G. E. (2020). A simple framework for contrastive learning of visual representations. In ICML."},{"key":"2202_CR11","doi-asserted-by":"crossref","unstructured":"Chen, X., & Zitnick, C. L. (2014). Learning a recurrent visual representation for image caption generation. arXiv:1411.5654","DOI":"10.1109\/CVPR.2015.7298856"},{"key":"2202_CR12","doi-asserted-by":"crossref","unstructured":"Chen, Y. C., Li, L., Yu, L., Kholy, A. E., Ahmed, F., Gan, Z., Cheng, Y., & Liu, J. (2020). UNITER: Universal image-text representation learning. In: ECCV.","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"2202_CR13","doi-asserted-by":"crossref","unstructured":"Cho, J., Yoon, S., Kale, A., Dernoncourt, F., Bui, T., & Bansal, M. (2022). Fine-grained image captioning with CLIP reward. In NAACL.","DOI":"10.18653\/v1\/2022.findings-naacl.39"},{"key":"2202_CR14","unstructured":"Devlin, J., Chang, M. W., Lee, K., & Toutanova, K. (2019). BERT: Pre-training of deep bidirectional transformers for language understanding. In NAACL."},{"key":"2202_CR15","doi-asserted-by":"crossref","unstructured":"Donahue, J., Hendricks, L. A., Guadarrama, S., Rohrbach, M., Venugopalan, S., Darrell, T., & Saenko, K. (2015). Long-term recurrent convolutional networks for visual recognition and description. In CVPR.","DOI":"10.21236\/ADA623249"},{"key":"2202_CR16","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., & Houlsby, N. (2021). An image is worth 16x16 words: Transformers for image recognition at scale. In ICLR."},{"key":"2202_CR17","unstructured":"Fang, H., Xiong, P., Xu, L., & Chen, Y. (2021). CLIP2Video: Mastering video-text retrieval via image CLIP. arXiv:2106.11097"},{"key":"2202_CR18","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Xiong, B., Girshick, R. B., & He, K. (2021). A large-scale study on unsupervised spatiotemporal representation learning. In CVPR","DOI":"10.1109\/CVPR46437.2021.00331"},{"key":"2202_CR19","doi-asserted-by":"crossref","unstructured":"Gabeur, V., Sun, C., Alahari, K., & Schmid, C. (2020). Multi-modal transformer for video retrieval. In ECCV","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"2202_CR20","unstructured":"Gao, Z., Liu, J., Chen, S., Chang, D., Zhang, H., & Yuan, J. (2021). CLIP2TV: an empirical study on transformer-based methods for video-text retrieval. arXiv:2111.05610"},{"key":"2202_CR21","doi-asserted-by":"crossref","unstructured":"Ge, Y., Ge, Y., Liu, X., Li, D., Shan, Y., Qie, X., & Luo, P. (2022). Bridgeformer: Bridging video-text retrieval with multiple choice questions. In CVPR","DOI":"10.1109\/CVPR52688.2022.01569"},{"key":"2202_CR22","unstructured":"Gordon, D., Ehsani, K., Fox, D., & Farhadi, A. (2020). Watching the world go by: Representation learning from unlabeled videos. arXiv:2003.07990"},{"key":"2202_CR23","unstructured":"Grill, J. B., Strub, F., Altch\u2019e, F., Tallec, C., Richemond, P. H., Buchatskaya, E., Doersch, C., Pires, B. \u00c1., Guo, Z. D., Azar, M. G., Piot, B., Kavukcuoglu, K., Munos, R., & Valko, M. (2020). Bootstrap your own latent: A new approach to self-supervised learning. In NeurIPS"},{"key":"2202_CR24","doi-asserted-by":"crossref","unstructured":"Gu, X., Chen, G., Wang, Y., Zhang, L., Luo, T., & Wen, L. (2023). Text with knowledge graph augmented transformer for video captioning. In CVPR","DOI":"10.1109\/CVPR52729.2023.01816"},{"key":"2202_CR25","doi-asserted-by":"crossref","unstructured":"Hessel, J., Holtzman, A., Forbes, M., Bras, R.L., & Choi, Y. (2021). CLIPScore: A reference-free evaluation metric for image captioning. In EMNLP.","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"2202_CR26","doi-asserted-by":"crossref","unstructured":"Ju, C., Han, T., Zheng, K., Zhang, Y., & Xie, W. (2021). Prompting visual-language models for efficient video understanding. arXiv:2112.04478","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"2202_CR27","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P., Suleyman, M., & Zisserman, A (2017). The Kinetics human action video dataset. arXiv:1705.06950"},{"key":"2202_CR28","unstructured":"Kingma, D. P., & Ba, J. (2015). Adam: A method for stochastic optimization. In ICLR."},{"key":"2202_CR29","doi-asserted-by":"crossref","unstructured":"Krishna, R., Hata, K., Ren, F., Fei-Fei, L., & Niebles, J. C. (2017). Dense-captioning events in videos. In ICCV.","DOI":"10.1109\/ICCV.2017.83"},{"key":"2202_CR30","unstructured":"Lee, D. H. (2013). Pseudo-label: The simple and efficient semi-supervised learning method for deep neural networks. In ICMLW"},{"key":"2202_CR31","unstructured":"Li, J., Li, D., Savarese, S., & Hoi, S. (2023). BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv:2301.12597"},{"key":"2202_CR32","unstructured":"Li, J., Li, D., Xiong, C., & Hoi, S. C. H. (2022). BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In ICML"},{"key":"2202_CR33","doi-asserted-by":"crossref","unstructured":"Li, L., Gan, Z., Lin, K., Lin, C. C., Liu, Z., Liu, C., & Wang, L. (2022). LAVENDER: Unifying video-language understanding as masked language modeling. arXiv","DOI":"10.1109\/CVPR52729.2023.02214"},{"key":"2202_CR34","doi-asserted-by":"crossref","unstructured":"Li, X., Yin, X., Li, C., Zhang, P., Hu, X., Zhang, L., Wang, L., Hu, H., Dong, L., Wei, F., Choi, Y., & Gao, J. (2020). Oscar: Object-semantics aligned pre-training for vision-language tasks. In ECCV.","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"2202_CR35","doi-asserted-by":"crossref","unstructured":"Lin, T. Y., Maire, M., Belongie, S., Bourdev, L., Girshick, R., Hays, J., Perona, P., Ramanan, D., Zitnick, C. L., & Doll\u00e1r, P. (2014). Microsoft coco: Common objects in context. In ECCV.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2202_CR36","unstructured":"Liu, Y., Albanie, S., Nagrani, A., & Zisserman, A. (2019). Use what you have: Video retrieval using representations from collaborative experts. In BMVC"},{"key":"2202_CR37","doi-asserted-by":"crossref","unstructured":"Liu, Y., Xiong, P., Xu, L., Cao, S., & Jin, Q. (2022). TS2-Net: Token shift and selection transformer for text-video retrieval. In ECCV.","DOI":"10.1007\/978-3-031-19781-9_19"},{"key":"2202_CR38","unstructured":"Loshchilov, I., & Hutter, F. (2017). Sgdr: Stochastic gradient descent with warm restarts. In ICLR."},{"key":"2202_CR39","doi-asserted-by":"crossref","unstructured":"Lu, J., Yang, J., Batra, D., & Parikh, D. (2018). Neural baby talk. In CVPR.","DOI":"10.1109\/CVPR.2018.00754"},{"key":"2202_CR40","doi-asserted-by":"crossref","unstructured":"Luo, H., Ji, L., Zhong, M., Chen, Y., Lei, W., Duan, N., & Li, T. (2021). CLIP4Clip: An empirical study of CLIP for end to end video clip retrieval. arXiv:2104.08860","DOI":"10.1016\/j.neucom.2022.07.028"},{"key":"2202_CR41","doi-asserted-by":"crossref","unstructured":"Ma, Y., Xu, G., Sun, X., Yan, M., Zhang, J., & Ji, R. (2022). X-CLIP: End-to-end multi-grained contrastive learning for video-text retrieval. In ACMMM","DOI":"10.1145\/3503161.3547910"},{"key":"2202_CR42","doi-asserted-by":"crossref","unstructured":"Miech, A., Alayrac, J.B., Laptev, I., Sivic, J., & Zisserman, A. (2021). Thinking fast and slow: Efficient text-to-visual retrieval with transformers. In CVPR","DOI":"10.1109\/CVPR46437.2021.00970"},{"key":"2202_CR43","doi-asserted-by":"crossref","unstructured":"Miech, A., Alayrac, J.B., Smaira, L., Laptev, I., Sivic, J., Zisserman, A.: End-to-end learning of visual representations from uncurated instructional videos. In CVPR (2020).","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"2202_CR44","doi-asserted-by":"crossref","unstructured":"Miech, A., Alayrac, J. B., Smaira, L., Laptev, I., Sivic, J., & Zisserman, A. (2020). End-to-end learning of visual representations from uncurated instructional videos. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"2202_CR45","doi-asserted-by":"crossref","unstructured":"Miech, A., Zhukov, D., Alayrac, J.B., Tapaswi, M., Laptev, I., & Sivic, J. (2019). HowTo100M: Learning a text-video embedding by watching hundred million narrated video clips. In ICCV.","DOI":"10.1109\/ICCV.2019.00272"},{"key":"2202_CR46","unstructured":"Mokady, R., Hertz, A., & Bermano, A.H. (2021). ClipCap: CLIP prefix for image captioning. arXiv preprint arXiv:2111.09734"},{"key":"2202_CR47","doi-asserted-by":"crossref","unstructured":"Morgado, P., Vasconcelos, N., & Misra, I. (2020). Audio-visual instance discrimination with cross-modal agreement. arXiv:2004.12943","DOI":"10.1109\/CVPR46437.2021.01274"},{"key":"2202_CR48","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Seo, P. H., Seybold, B. A., Hauth, A., Manen, S., Sun, C., & Schmid, C. (2022). Learning audio-video modalities from image captions. In ECCV.","DOI":"10.1007\/978-3-031-19781-9_24"},{"key":"2202_CR49","unstructured":"Ng, J. Y. H., Hausknecht, M., Vijayanarasimhan, S., Vinyals, O., Monga, R., & Toderici, G. (2015). Beyond short snippets: Deep networks for video classification. In CVPR."},{"key":"2202_CR50","doi-asserted-by":"crossref","unstructured":"Nukrai, D., Mokady, R., Globerson, A.: Text-only training for image captioning using noise-injected CLIP. arXiv:2211.00575 (2022).","DOI":"10.18653\/v1\/2022.findings-emnlp.299"},{"key":"2202_CR51","unstructured":"van\u00a0den Oord, A., Li, Y., & Vinyals, O. (2018). Representation learning with contrastive predictive coding. arXiv:1807.03748"},{"key":"2202_CR52","doi-asserted-by":"crossref","unstructured":"Park, J. S., Rohrbach, M., Darrell, T., & Rohrbach, A. (2019). Adversarial inference for multi-sentence video description. In CVPR.","DOI":"10.1109\/CVPR.2019.00676"},{"key":"2202_CR53","unstructured":"Patrick, M., Huang, P., Asano, Y. M., Metze, F., Hauptmann, A. G., Henriques, J. F., & Vedaldi, A. (2021). Support-set bottlenecks for video-text representation learning. In ICLR."},{"key":"2202_CR54","doi-asserted-by":"crossref","unstructured":"Piergiovanni, A. J., Angelova, A., & Ryoo, M. S. (2020). Evolving losses for unsupervised video representation learning. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00021"},{"key":"2202_CR55","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., Krueger, G., & Sutskever, I. (2021). Learning transferable visual models from natural language supervision. In ICML."},{"key":"2202_CR56","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., & Sutskever, I. (2019). Language models are unsupervised multitask learners. OpenAI blog"},{"key":"2202_CR57","doi-asserted-by":"crossref","unstructured":"Rasheed, H., Khattak, M. U., Maaz, M., Khan, S., & Khan, F. S. (2023). Fine-tuned CLIP models are efficient video learners. In CVPR.","DOI":"10.1109\/CVPR52729.2023.00633"},{"key":"2202_CR58","doi-asserted-by":"crossref","unstructured":"Recasens, A., Luc, P., Alayrac, J. B., Wang, L., Hemsley, R., Strub, F., Tallec, C., Malinowski, M., Patraucean, V., Altch\u00e9, F., Valko, M., Grill, J. B., van\u00a0den Oord, A., Zisserman, A. (2021). Broaden your views for self-supervised video learning. arXiv:2103.16559","DOI":"10.1109\/ICCV48922.2021.00129"},{"key":"2202_CR59","doi-asserted-by":"crossref","unstructured":"Reimers, N., & Gurevych, I. (2019). Sentence-bert: Sentence embeddings using siamese bert-networks. In EMNLP.","DOI":"10.18653\/v1\/D19-1410"},{"key":"2202_CR60","unstructured":"Schuhmann, C., Vencu, R., Beaumont, R., Kaczmarczyk, R., Mullis, C., Katta, A., Coombes, T., Jitsev, J., & Komatsuzaki, A. (2021). LAION-400M: Open dataset of CLIP-filtered 400 million image-text pairs. In Data Centric AI NeurIPS Workshop."},{"key":"2202_CR61","doi-asserted-by":"crossref","unstructured":"Seo, P. H., Nagrani, A., Arnab, A., & Schmid, C. (2022). End-to-end generative pretraining for multimodal video captioning. In CVPR","DOI":"10.1109\/CVPR52688.2022.01743"},{"key":"2202_CR62","doi-asserted-by":"crossref","unstructured":"Sermanet, P., Lynch, C., Chebotar, Y., Hsu, J., Jang, E., Schaal, S., & Levine, S. (2018). Time-contrastive networks: Self-supervised learning from video. In ICRA (2018).","DOI":"10.1109\/ICRA.2018.8462891"},{"key":"2202_CR63","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., & Soricut, R. (2018). Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning. In ACL.","DOI":"10.18653\/v1\/P18-1238"},{"key":"2202_CR64","doi-asserted-by":"crossref","unstructured":"Singh, A., Chakraborty, O., Varshney, A., Panda, R., Feris, R., Saenko, K., & Das, A. (2021). Semi-supervised action recognition with temporal contrastive learning. In CVPR.","DOI":"10.1109\/CVPR46437.2021.01025"},{"key":"2202_CR65","unstructured":"Sohn, K., Berthelot, D., Li, C. L., Zhang, Z., Carlini, N., Cubuk, E.D., Kurakin, A., Zhang, H., & Raffel, C. (2020). FixMatch: Simplifying semi-supervised learning with consistency and confidence. In NeurIPS."},{"key":"2202_CR66","unstructured":"Sun, C., Baradel, F., Murphy, K.P., & Schmid, C. (2019). Contrastive bidirectional transformer for temporal representation learning. arXiv:1906.05743"},{"key":"2202_CR67","doi-asserted-by":"crossref","unstructured":"Tang, M., Wang, Z., LIU, Z., Rao, F., Li, D., & Li, X. (2021). CLIP4Caption: CLIP for video caption. In ACMMM.","DOI":"10.1145\/3474085.3479207"},{"key":"2202_CR68","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., & Paluri, M. (2015). Learning spatiotemporal features with 3D convolutional networks. In ICCV.","DOI":"10.1109\/ICCV.2015.510"},{"key":"2202_CR69","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L.u., & Polosukhin, I. (2017). Attention is all you need. In NeurIPS."},{"key":"2202_CR70","doi-asserted-by":"crossref","unstructured":"Ventura, L., Schmid, C., & Varol, G. (2023). Learning text-to-video retrieval from image captioning. In CVPR Workshop on Learning with Limited Labelled Data for Image and Video Understanding (L3D-IVU).","DOI":"10.1007\/s11263-024-02202-8"},{"key":"2202_CR71","doi-asserted-by":"crossref","unstructured":"Venugopalan, S., Rohrbach, M., Donahue, J., Mooney, R. J., Darrell, T., & Saenko, K. (2015). Sequence to sequence\u2013video to text. In ICCV.","DOI":"10.1109\/ICCV.2015.515"},{"issue":"11","key":"2202_CR72","doi-asserted-by":"crossref","first-page":"2740","DOI":"10.1109\/TPAMI.2018.2868668","volume":"41","author":"L Wang","year":"2019","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., & Van Gool, L. (2019). Temporal segment networks for action recognition in videos. IEEE Transactions on Pattern Analysis and Machine Intelligence, 41(11), 2740\u20132755.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2202_CR73","unstructured":"Wang, M., Xing, J., & Liu, Y. (2021). ActionCLIP: A new paradigm for video action recognition. arXiv:2109.08472"},{"key":"2202_CR74","unstructured":"Wang, P., Yang, A., Men, R., Lin, J., Bai, S., Li, Z., Ma, J., Zhou, C., Zhou, J., & Yang, H. (2022). OFA: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In ICML"},{"key":"2202_CR75","doi-asserted-by":"crossref","first-page":"6079","DOI":"10.1109\/TMM.2022.3204444","volume":"25","author":"X Wang","year":"2022","unstructured":"Wang, X., Zhu, L., Zheng, Z., Xu, M., & Yang, Y. (2022). Align and tell: Boosting text-video retrieval with local alignment and fine-grained supervision. IEEE Transactions on Multimedia, 25, 6079\u20136089.","journal-title":"IEEE Transactions on Multimedia"},{"key":"2202_CR76","unstructured":"Wang, Z., Li, M., Xu, R., Zhou, L., Lei, J., Lin, X., Wang, S., Yang, Z., Zhu, C., Hoiem, D., Chang, S. F., Bansal, M., & Ji, H. (2022). Language models with image descriptors are strong few-shot video-language learners. arXiv"},{"key":"2202_CR77","doi-asserted-by":"crossref","unstructured":"Xu, H., Ghosh, G., Huang, P. Y., Okhonko, D., Aghajanyan, A., Metze, F., Zettlemoyer, L., & Feichtenhofer, C. (2021). VideoCLIP: Contrastive pre-training for zero-shot video-text understanding. In EMNLP.","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"2202_CR78","doi-asserted-by":"crossref","unstructured":"Xu, J., Mei, T., Yao, T., & Rui, Y. (2016). MSR-VTT: A large video description dataset for bridging video and language. In CVPR.","DOI":"10.1109\/CVPR.2016.571"},{"key":"2202_CR79","unstructured":"Xue, H., Sun, Y., Liu, B., Fu, J., Song, R., Li, H., & Luo, J. (2022). CLIP-ViP: Adapting pre-trained image-text model to video-language representation alignment. arXiv."},{"key":"2202_CR80","doi-asserted-by":"crossref","unstructured":"Yang, A., Nagrani, A., Seo, P.H., Miech, A., Pont-Tuset, J., Laptev, I., Sivic, J., & Schmid, C. (2023). Vid2seq: Large-scale pretraining of a visual language model for dense video captioning. In CVPR.","DOI":"10.1109\/CVPR52729.2023.01032"},{"key":"2202_CR81","doi-asserted-by":"crossref","unstructured":"Yang, B., & Zou, Y. (2021). CLIP meets video captioners: Attribute-aware representation learning promotes accurate captioning. arXiv:2111.15162","DOI":"10.1007\/978-3-031-18907-4_29"},{"key":"2202_CR82","unstructured":"Yang, C., Xu, Y., Dai, B., & Zhou, B. (2020). Video representation learning with visual tempo consistency. arXiv:2006.15489"},{"key":"2202_CR83","doi-asserted-by":"crossref","unstructured":"Yang, J., Bisk, Y., & Gao, J. (2021). TACo: Token-aware cascade contrastive learning for video-text alignment. arXiv","DOI":"10.1109\/ICCV48922.2021.01136"},{"key":"2202_CR84","unstructured":"Yao, L., Huang, R., Hou, L., Lu, G., Niu, M., Xu, H., Liang, X., Li, Z., Jiang, X., & Xu, C. (2022). FILIP: Fine-grained interactive language-image pre-training. In ICLR."},{"key":"2202_CR85","unstructured":"Yu, J., Wang, Z., Vasudevan, V., Yeung, L., Seyedhosseini, M., & Wu, Y. (2022). CoCa: Contrastive captioners are image-text foundation models. In Transactions on Machine Learning Research."},{"key":"2202_CR86","doi-asserted-by":"crossref","unstructured":"Yu, Y., Kim, J., & Kim, G. (2018). A joint sequence fusion model for video question answering and retrieval. In ECCV.","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"2202_CR87","doi-asserted-by":"crossref","unstructured":"Zala, A., Cho, J., Kottur, S., Chen, X., O\u011fuz, B., Mehdad, Y., & Bansal, M. (2023). Hierarchical video-moment retrieval and step-captioning. In CVPR.","DOI":"10.1109\/CVPR52729.2023.02208"},{"key":"2202_CR88","doi-asserted-by":"crossref","unstructured":"Zhang, B., Hu, H., & Sha, F. (2018). Cross-modal and hierarchical modeling of video and text. In ECCV (2018).","DOI":"10.1007\/978-3-030-01261-8_23"},{"key":"2202_CR89","doi-asserted-by":"crossref","unstructured":"Zhou, L., Palangi, H., Zhang, L., Hu, H., Corso, J.J., & Gao, J. (2020). Unified vision-language pre-training for image captioning and VQA. In AAAI","DOI":"10.1609\/aaai.v34i07.7005"},{"key":"2202_CR90","doi-asserted-by":"crossref","unstructured":"Zhu, L., & Yang, Y. (2020). ActBERT: Learning global-local video-text representations. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00877"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02202-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02202-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02202-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T22:05:06Z","timestamp":1743372306000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02202-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,22]]},"references-count":90,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["2202"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02202-8","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,22]]},"assertion":[{"value":"3 April 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 July 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}