{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T17:05:08Z","timestamp":1780765508910,"version":"3.54.1"},"reference-count":137,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2024,5,25]],"date-time":"2024-05-25T00:00:00Z","timestamp":1716595200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,5,25]],"date-time":"2024-05-25T00:00:00Z","timestamp":1716595200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100014718","name":"Innovative Research Group Project of the National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61921006"],"award-info":[{"award-number":["61921006"]}],"id":[{"id":"10.13039\/100014718","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010905","name":"Major Research Plan","doi-asserted-by":"publisher","award":["62076119"],"award-info":[{"award-number":["62076119"]}],"id":[{"id":"10.13039\/501100010905","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012165","name":"Key Technologies Research and Development Program","doi-asserted-by":"publisher","award":["2022ZD0160900"],"award-info":[{"award-number":["2022ZD0160900"]}],"id":[{"id":"10.13039\/501100012165","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2024,10]]},"DOI":"10.1007\/s11263-024-02081-z","type":"journal-article","created":{"date-parts":[[2024,5,25]],"date-time":"2024-05-25T11:01:32Z","timestamp":1716634892000},"page":"4792-4817","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["VLG: General Video Recognition with Web Textual Knowledge"],"prefix":"10.1007","volume":"132","author":[{"given":"Jintao","family":"Lin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhaoyang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenhai","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wayne","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3674-7718","authenticated-orcid":false,"given":"Limin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,5,25]]},"reference":[{"key":"2081_CR1","doi-asserted-by":"crossref","unstructured":"Acsintoae, A., Florescu, A., Georgescu, M. I., Mare, T., Sumedrea, P., Ionescu, R. T., Khan, F. S., Shah, M. (2021). Ubnormal: New benchmark for supervised open-set video anomaly detection. arXiv preprint arXiv:2111.08644","DOI":"10.1109\/CVPR52688.2022.01951"},{"key":"2081_CR2","first-page":"24206","volume":"34","author":"H Akbari","year":"2021","unstructured":"Akbari, H., Yuan, L., Qian, R., Chuang, W. H., Chang, S. F., Cui, Y., & Gong, B. (2021). Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text. Advances in Neural Information Processing Systems, 34, 24206\u201324221.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2081_CR3","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., & Schmid, C. (2021). Vivit: A video vision transformer. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 6836\u20136846)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"2081_CR4","unstructured":"Bain, M., Nagrani, A., Varol, G., & Zisserman, A. (2022). A clip-hitchhiker\u2019s guide to long video retrieval. arXiv preprint arXiv:2205.08508"},{"key":"2081_CR5","doi-asserted-by":"crossref","unstructured":"Bao, W., Yu, Q., & Kong, Y. (2021). Evidential deep learning for open set action recognition. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 13349\u201313358)","DOI":"10.1109\/ICCV48922.2021.01310"},{"key":"2081_CR6","doi-asserted-by":"crossref","unstructured":"Bendale, A., & Boult, T. E. (2016). Towards open set deep networks. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 1563\u20131572)","DOI":"10.1109\/CVPR.2016.173"},{"key":"2081_CR7","unstructured":"Bertasius, G., Wang, H., & Torresani, L. (2021). Is space-time attention all you need for video understanding? In ICML, vol\u00a02 (pp. 4)"},{"key":"2081_CR8","unstructured":"Bishay, M., Zoumpourlis, G., & Patras, I. (2019). Tarn: Temporal attentive relation network for few-shot and zero-shot action recognition. arXiv preprint arXiv:1907.09021"},{"key":"2081_CR9","doi-asserted-by":"publisher","first-page":"249","DOI":"10.1016\/j.neunet.2018.07.011","volume":"106","author":"M Buda","year":"2018","unstructured":"Buda, M., Maki, A., & Mazurowski, M. A. (2018). A systematic study of the class imbalance problem in convolutional neural networks. Neural Networks, 106, 249\u2013259.","journal-title":"Neural Networks"},{"key":"2081_CR10","doi-asserted-by":"crossref","unstructured":"Caba\u00a0Heilbron, F., Escorcia, V., Ghanem, B., & Carlos\u00a0Niebles, J. (2015). Activitynet: A large-scale video benchmark for human activity understanding. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 961\u2013970)","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"2081_CR11","doi-asserted-by":"crossref","unstructured":"Cao, D., Xu, L., & Chen, H. (2020a). Action recognition in untrimmed videos with composite self-attention two-stream framework. In Pattern recognition: 5th Asian conference, ACPR 2019, Auckland, New Zealand, November 26\u201329, 2019, Revised Selected Papers, Part II 5 (pp. 27\u201340). Springer","DOI":"10.1007\/978-3-030-41299-9_3"},{"key":"2081_CR12","unstructured":"Cao, K., Wei, C., Gaidon, A., Arechiga, N., & Ma, T. (2019). Learning imbalanced datasets with label-distribution-aware margin loss. In Advances in neural information processing systems 32"},{"key":"2081_CR13","doi-asserted-by":"crossref","unstructured":"Cao, K., Ji, J., Cao, Z., Chang, C. Y., & Niebles, J. C. (2020b). Few-shot video classification via temporal alignment. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 10618\u201310627)","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"2081_CR14","doi-asserted-by":"crossref","unstructured":"Carreira, J., & Zisserman, A. (2017). Quo vadis, action recognition? A new model and the kinetics dataset. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 6299\u20136308)","DOI":"10.1109\/CVPR.2017.502"},{"key":"2081_CR15","unstructured":"Carreira, J., Noland, E., Hillier, C., & Zisserman, A. (2019). A short note on the kinetics-700 human action dataset. arXiv preprint arXiv:1907.06987"},{"key":"2081_CR16","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1613\/jair.953","volume":"16","author":"NV Chawla","year":"2002","unstructured":"Chawla, N. V., Bowyer, K. W., Hall, L. O., & Kegelmeyer, W. P. (2002). Smote: Synthetic minority over-sampling technique. Journal of Artificial Intelligence Research, 16, 321\u2013357.","journal-title":"Journal of Artificial Intelligence Research"},{"key":"2081_CR17","doi-asserted-by":"crossref","unstructured":"Chu, P., Bian, X., Liu, S., & Ling, H. (2020). Feature space augmentation for long-tailed data. In European conference on computer vision (pp. 694\u2013710). Springer","DOI":"10.1007\/978-3-030-58526-6_41"},{"key":"2081_CR18","doi-asserted-by":"crossref","unstructured":"Cui, J., Zhong, Z., Liu, S., Yu, B., & Jia, J. (2021). Parametric contrastive learning. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 715\u2013724)","DOI":"10.1109\/ICCV48922.2021.00075"},{"key":"2081_CR19","doi-asserted-by":"crossref","unstructured":"Cui, Y., Jia, M., Lin, T. Y., Song, Y., & Belongie, S. (2019). Class-balanced loss based on effective number of samples. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 9268\u20139277)","DOI":"10.1109\/CVPR.2019.00949"},{"key":"2081_CR20","doi-asserted-by":"crossref","unstructured":"Diba, A., Fayyaz, M., Sharma, V., Arzani, M. M., Yousefzadeh, R., Gall, J., & Van\u00a0Gool, L. (2018). Spatio-temporal channel correlation networks for action classification. In Proceedings of the European conference on computer vision (ECCV) (pp. 284\u2013299)","DOI":"10.1007\/978-3-030-01225-0_18"},{"key":"2081_CR21","doi-asserted-by":"crossref","unstructured":"Ditria, L., Meyer, BJ., & Drummond, T. (2020). Opengan: Open set generative adversarial networks. In Proceedings of the Asian conference on computer vision","DOI":"10.1007\/978-3-030-69538-5_29"},{"key":"2081_CR22","doi-asserted-by":"crossref","unstructured":"Doll\u00e1r, P., Rabaud, V., Cottrell, G., & Belongie, S. (2005). Behavior recognition via sparse spatio-temporal features. In 2005 IEEE international workshop on visual surveillance and performance evaluation of tracking and surveillance (pp. 65\u201372). IEEE","DOI":"10.1109\/VSPETS.2005.1570899"},{"key":"2081_CR23","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3059295","author":"J Dong","year":"2021","unstructured":"Dong, J., Li, X., Xu, C., Yang, X., Yang, G., Wang, X., & Wang, M. (2021). Dual encoding for video retrieval by text. IEEE Transactions on Pattern Analysis and Machine Intelligence. https:\/\/doi.org\/10.1109\/TPAMI.2021.3059295","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2081_CR24","unstructured":"Drumnond, C., & Holte, R. (2003). Class imbalance and cost sensitivity: Why undersampling beats oversampling. In ICML-KDD 2003 workshop: Learning from imbalanced datasets, vol\u00a03"},{"key":"2081_CR25","doi-asserted-by":"crossref","unstructured":"Fan, H., Xiong, B., Mangalam, K., Li, Y., Yan, Z., Malik, J., & Feichtenhofer, C. (2021). Multiscale vision transformers. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 6824\u20136835)","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"2081_CR26","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C. (2020). X3d: Expanding architectures for efficient video recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 203\u2013213)","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"2081_CR27","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., & Zisserman, A. (2016). Convolutional two-stream network fusion for video action recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 1933\u20131941)","DOI":"10.1109\/CVPR.2016.213"},{"key":"2081_CR28","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., & He, K. (2019). Slowfast networks for video recognition. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 6202\u20136211)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"2081_CR29","unstructured":"Finn, C., Abbeel, P., & Levine, S. (2017). Model-agnostic meta-learning for fast adaptation of deep networks. In International conference on machine learning, PMLR (pp. 1126\u20131135)"},{"key":"2081_CR30","unstructured":"Finn, C., Xu, K., & Levine, S. (2018). Probabilistic model-agnostic meta-learning. In Advances in neural information processing systems 31"},{"key":"2081_CR31","unstructured":"Frome, A., Corrado, G. S., Shlens, J., Bengio, S., Dean, J., Ranzato, M., & Mikolov, T. (2013). Devise: A deep visual-semantic embedding model. In Advances in neural information processing systems 26"},{"key":"2081_CR32","doi-asserted-by":"crossref","unstructured":"Ge, Z., Demyanov, S., Chen, Z., & Garnavi, R. (2017). Generative openmax for multi-class open set classification. arXiv preprint arXiv:1707.07418","DOI":"10.5244\/C.31.42"},{"key":"2081_CR33","doi-asserted-by":"crossref","unstructured":"Goyal, R., Ebrahimi\u00a0Kahou, S., Michalski, V., Materzynska, J., Westphal, S., Kim, H., Haenel, V., Fruend, I., Yianilos, P., Mueller-Freitag, M., & et\u00a0al. (2017). The\" something something\" video database for learning and evaluating visual common sense. In Proceedings of the IEEE international conference on computer vision (pp. 5842\u20135850)","DOI":"10.1109\/ICCV.2017.622"},{"key":"2081_CR34","doi-asserted-by":"crossref","unstructured":"Han, H., Wang, W. Y., & Mao, B. H. (2005). Borderline-smote: A new over-sampling method in imbalanced data sets learning. In International conference on intelligent computing (pp. 878\u2013887). Springer","DOI":"10.1007\/11538059_91"},{"key":"2081_CR35","doi-asserted-by":"crossref","unstructured":"Huang, C., Li, Y., Loy, C. C., & Tang, X. (2016). Learning deep representation for imbalanced classification. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 5375\u20135384)","DOI":"10.1109\/CVPR.2016.580"},{"key":"2081_CR36","doi-asserted-by":"crossref","unstructured":"Jain, LP., Scheirer, W. J., & Boult, T. E. (2014). Multi-class open set recognition using probability of inclusion. In European conference on computer vision (pp. 393\u2013409). Springer","DOI":"10.1007\/978-3-319-10578-9_26"},{"key":"2081_CR37","unstructured":"Jang, E., Gu, S., & Poole, B. (2016). Categorical reparameterization with gumbel-softmax. arXiv preprint arXiv:1611.01144"},{"key":"2081_CR38","unstructured":"Jia, C., Yang, Y., Xia, Y., Chen, Y. T., Parekh, Z., Pham, H., Le, Q., Sung, Y. H., Li, Z., & Duerig, T. (2021). Scaling up visual and vision-language representation learning with noisy text supervision. In International conference on machine learning, PMLR (pp. 4904\u20134916)"},{"key":"2081_CR39","doi-asserted-by":"crossref","unstructured":"Jiang, B., Wang, M., Gan, W., Wu, W., & Yan, J. (2019). Stm: Spatiotemporal and motion encoding for action recognition. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 2000\u20132009)","DOI":"10.1109\/ICCV.2019.00209"},{"key":"2081_CR40","doi-asserted-by":"crossref","unstructured":"Ju, C., Han, T., Zheng, K., Zhang, Y., & Xie, W. (2022). Prompting visual-language models for efficient video understanding. In European conference on computer vision (pp. 105\u2013124). Springer","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"2081_CR41","doi-asserted-by":"crossref","unstructured":"Kahatapitiya, K., Arnab, A., Nagrani, A., & Ryoo, M. S. (2023). Victr: Video-conditioned text representations for activity recognition. arXiv preprint arXiv:2304.02560","DOI":"10.1109\/CVPR52733.2024.01755"},{"key":"2081_CR42","unstructured":"Kang, B., Xie, S., Rohrbach, M., Yan, Z., Gordo, A., Feng, J., & Kalantidis, Y. (2019). Decoupling representation and classifier for long-tailed recognition. arXiv preprint arXiv:1910.09217"},{"key":"2081_CR43","doi-asserted-by":"crossref","unstructured":"Kant, Y., Batra, D., Anderson, P., Schwing, A., Parikh, D., Lu, J., & Agrawal, H. (2020). Spatially aware multimodal transformers for textvqa. In European conference on computer vision (pp. 715\u2013732). Springer","DOI":"10.1007\/978-3-030-58545-7_41"},{"key":"2081_CR44","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P., & et\u00a0al. (2017). The kinetics human action video dataset. arXiv preprint arXiv:1705.06950"},{"issue":"8","key":"2081_CR45","doi-asserted-by":"publisher","first-page":"3573","DOI":"10.1109\/TNNLS.2017.2732482","volume":"29","author":"SH Khan","year":"2017","unstructured":"Khan, S. H., Hayat, M., Bennamoun, M., Sohel, F. A., & Togneri, R. (2017). Cost-sensitive learning of deep feature representations from imbalanced data. IEEE Transactions on Neural Networks and Learning Systems, 29(8), 3573\u20133587.","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"2081_CR46","doi-asserted-by":"crossref","unstructured":"Klaser, A., Marsza\u0142ek, M., & Schmid, C. (2008). A spatio-temporal descriptor based on 3d-gradients. In BMVC 2008-19th British machine vision conference (pp. 275\u20131). British Machine Vision Association","DOI":"10.5244\/C.22.99"},{"key":"2081_CR47","unstructured":"Krishnan, R., Subedar, M., & Tickoo, O. (2018). Bar: Bayesian activity recognition using variational inference. arXiv preprint arXiv:1811.03305"},{"key":"2081_CR48","doi-asserted-by":"crossref","unstructured":"Krishnan, R., Subedar, M., & Tickoo, O. (2020). Specifying weight priors in Bayesian deep neural networks with empirical Bayes. InProceedings of the AAAI conference on artificial intelligence, 34, 4477\u20134484.","DOI":"10.1609\/aaai.v34i04.5875"},{"key":"2081_CR49","unstructured":"Kumar, N., & Narang, S. (2021). Few shot activity recognition using variational inference. arXiv preprint arXiv:2108.08990"},{"key":"2081_CR50","doi-asserted-by":"crossref","unstructured":"Kumar\u00a0Dwivedi, S., Gupta, V., Mitra, R., Ahmed, S., & Jain, A. (2019). Protogan: Towards few shot learning for action recognition. In Proceedings of the IEEE\/CVF International conference on computer vision workshops","DOI":"10.1109\/ICCVW.2019.00166"},{"key":"2081_CR51","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3076522","author":"S Kumawat","year":"2021","unstructured":"Kumawat, S., Verma, M., Nakashima, Y., & Raman, S. (2021). Depthwise spatio-temporal STFT convolutional neural networks for human action recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence. https:\/\/doi.org\/10.1109\/TPAMI.2021.3076522","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"11","key":"2081_CR52","doi-asserted-by":"publisher","first-page":"1686","DOI":"10.1109\/TPAMI.2005.224","volume":"27","author":"F Li","year":"2005","unstructured":"Li, F., & Wechsler, H. (2005). Open set face recognition using transduction. IEEE Transactions on Pattern Analysis and Machine Intelligence, 27(11), 1686\u20131697.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2081_CR53","unstructured":"Li, T., & Wang, L. (2020a). Learning spatiotemporal features via video and text pair discrimination. arXiv preprint arXiv:2001.05691"},{"key":"2081_CR54","unstructured":"Li, T., & Wang, L. (2020b). Learning spatiotemporal features via video and text pair discrimination. CoRR arXiv: 2001.05691"},{"key":"2081_CR55","doi-asserted-by":"crossref","unstructured":"Li, T., Wang, L., & Wu, G. (2021a). Self supervision to distillation for long-tailed visual recognition. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 630\u2013639)","DOI":"10.1109\/ICCV48922.2021.00067"},{"key":"2081_CR56","doi-asserted-by":"crossref","unstructured":"Li, X., Yin, X., Li, C., Zhang, P., Hu, X., Zhang, L., Wang, L., Hu, H., Dong, L., Wei, F., & et\u00a0al. (2020a). Oscar: Object-semantics aligned pre-training for vision-language tasks. In European conference on computer vision (pp. 121\u2013137). Springer","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"2081_CR57","doi-asserted-by":"crossref","unstructured":"Li, Y., Ji, B., Shi, X., Zhang, J., Kang, B., & Wang, L. (2020b). Tea: Temporal excitation and aggregation for action recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 909\u2013918)","DOI":"10.1109\/CVPR42600.2020.00099"},{"key":"2081_CR58","doi-asserted-by":"crossref","unstructured":"Li, Y., Wu, CY., Fan, H., Mangalam, K., Xiong, B., Malik, J., & Feichtenhofer, C. (2021b). Improved multiscale vision transformers for classification and detection. arXiv preprint arXiv:2112.01526","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"2081_CR59","unstructured":"Li, Z., Fan, Z., Tou, H., & Wei, Z. (2022). Mvp: Multi-stage vision-language pre-training via multi-level semantic alignment. arXiv preprint arXiv:2201.12596"},{"key":"2081_CR60","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., & Han, S. (2019). Tsm: Temporal shift module for efficient video understanding. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 7083\u20137093)","DOI":"10.1109\/ICCV.2019.00718"},{"key":"2081_CR61","doi-asserted-by":"crossref","unstructured":"Lin, J., Duan, H., Chen, K., Lin, D., & Wang, L. (2022a) Ocsampler: Compressing videos to one clip with single-step sampling. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 13894\u201313903)","DOI":"10.1109\/CVPR52688.2022.01352"},{"key":"2081_CR62","doi-asserted-by":"crossref","unstructured":"Lin, Z., Geng, S., Zhang, R., Gao, P., de\u00a0Melo, G., Wang, X., Dai, J., Qiao. Y., & Li, H. (2022b) Frozen clip models are efficient video learners. In European conference on computer vision (pp. 388\u2013404). Springer","DOI":"10.1007\/978-3-031-19833-5_23"},{"key":"2081_CR63","doi-asserted-by":"crossref","unstructured":"Liu, Y., Chen, Q., & Albanie, S. (2021a) Adaptive cross-modal prototypes for cross-domain visual-language retrieval. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 14954\u201314964)","DOI":"10.1109\/CVPR46437.2021.01471"},{"key":"2081_CR64","doi-asserted-by":"crossref","unstructured":"Liu, Z., Miao, Z., Zhan, X., Wang, J., Gong, B., & Yu, S. X. (2019) Large-scale long-tailed recognition in an open world. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 2537\u20132546)","DOI":"10.1109\/CVPR.2019.00264"},{"key":"2081_CR65","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wang, L., Wu, W., Qian, C., & Lu, T. (2021b) TAM: Temporal adaptive module for video recognition. In ICCV (pp. 13688\u201313698). IEEE","DOI":"10.1109\/ICCV48922.2021.01345"},{"key":"2081_CR66","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ning, J., Cao, Y., Wei, Y., Zhang, Z., Lin, S., & Hu, H. (2022) Video swin transformer. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 3202\u20133211)","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"2081_CR67","unstructured":"Loshchilov, I., & Hutter, F. (2016) Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983"},{"key":"2081_CR68","unstructured":"Loshchilov, I., & Hutter, F. (2017) Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101"},{"key":"2081_CR69","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1016\/j.neucom.2022.07.028","volume":"508","author":"H Luo","year":"2022","unstructured":"Luo, H., Ji, L., Zhong, M., Chen, Y., Lei, W., Duan, N., & Li, T. (2022). Clip4clip: An empirical study of clip for end to end video clip retrieval and captioning. Neurocomputing, 508, 293\u2013304.","journal-title":"Neurocomputing"},{"key":"2081_CR70","doi-asserted-by":"crossref","unstructured":"Meng, Y., Lin, CC., Panda, R., Sattigeri, P., Karlinsky, L., Oliva, A., Saenko, K., & Feris, R. (2020). Ar-net: Adaptive frame resolution for efficient action recognition. In European conference on computer vision (pp. 86\u2013104). Springer","DOI":"10.1007\/978-3-030-58571-6_6"},{"key":"2081_CR71","doi-asserted-by":"crossref","unstructured":"Miech, A., Alayrac, J. B., Smaira, L., Laptev, I., Sivic, J., & Zisserman, A. (2020a). End-to-end learning of visual representations from uncurated instructional videos. In CVPR","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"2081_CR72","doi-asserted-by":"crossref","unstructured":"Miech, A., Alayrac, J.B., Smaira, L., Laptev, I., Sivic, J., & Zisserman, A. (2020b). End-to-end learning of visual representations from uncurated instructional videos. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 9879\u20139889)","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"2081_CR73","doi-asserted-by":"crossref","unstructured":"Mishra, A., Verma, V. K., Reddy, M. S. K., Arulkumar, S., Rai, P., & Mittal, A. (2018) A generative approach to zero-shot and few-shot action recognition. In 2018 IEEE winter conference on applications of computer vision (WACV) (pp. 372\u2013380). IEEE","DOI":"10.1109\/WACV.2018.00047"},{"issue":"2","key":"2081_CR74","doi-asserted-by":"publisher","first-page":"502","DOI":"10.1109\/TPAMI.2019.2901464","volume":"42","author":"M Monfort","year":"2019","unstructured":"Monfort, M., Andonian, A., Zhou, B., Ramakrishnan, K., Bargal, S. A., Yan, T., Brown, L., Fan, Q., Gutfreund, D., Vondrick, C., et al. (2019). Moments in time dataset: One million videos for event understanding. IEEE Transactions on Pattern Analysis and Machine Intelligence, 42(2), 502\u2013508.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2081_CR75","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3126682","author":"M Monfor","year":"2021","unstructured":"Monfor, M., Pan, B., Ramakrishnan, K., Andonian, A., McNamara, B. A., Lascelles, A., Fan, Q., Gutfreund, D., Feris, R., & Oliva, A. (2021). Multi-moments in time: Learning and interpreting models for multi-action video understanding. IEEE Transactions on Pattern Analysis and Machine Intelligence. https:\/\/doi.org\/10.1109\/TPAMI.2021.3126682","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2081_CR76","unstructured":"Mori, Y., Takahashi, H., & Oka, R. (1999). Image-to-word transformation based on dividing and vector quantizing images with words. In First international workshop on multimedia intelligent storage and retrieval management (pp. 1\u20139). Citeseer"},{"key":"2081_CR77","doi-asserted-by":"crossref","unstructured":"Neal, L., Olson, M., Fern, X., Wong, W. K., & Li, F. (2018). Open set learning with counterfactual images. In Proceedings of the European conference on computer vision (ECCV) (pp. 613\u2013628)","DOI":"10.1007\/978-3-030-01231-1_38"},{"key":"2081_CR78","doi-asserted-by":"crossref","unstructured":"Neimark, D., Bar, O., Zohar, M., & Asselmann, D. (2021). Video transformer network. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 3163\u20133172)","DOI":"10.1109\/ICCVW54120.2021.00355"},{"key":"2081_CR79","doi-asserted-by":"crossref","unstructured":"Ni, B., Peng, H., Chen, M., Zhang, S., Meng, G., Fu, J., Xiang, S., & Ling, H. (2022) Expanding language-image pretrained models for general video recognition. In European conference on computer vision (pp. 1\u201318). Springer","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"2081_CR80","doi-asserted-by":"crossref","unstructured":"Oza, P., & Patel, V. M. (2019). C2ae: Class conditioned auto-encoder for open-set recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 2307\u20132316)","DOI":"10.1109\/CVPR.2019.00241"},{"key":"2081_CR81","first-page":"26462","volume":"35","author":"J Pan","year":"2022","unstructured":"Pan, J., Lin, Z., Zhu, X., Shao, J., & Li, H. (2022). St-adapter: Parameter-efficient image-to-video transfer learning. Advances in Neural Information Processing Systems, 35, 26462\u201326477.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2081_CR82","doi-asserted-by":"crossref","unstructured":"Perrett, T., Masullo, A., Burghardt, T., Mirmehdi, M., & Damen, D. (2021). Temporal-relational crosstransformers for few-shot action recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 475\u2013484)","DOI":"10.1109\/CVPR46437.2021.00054"},{"key":"2081_CR83","unstructured":"Qian, R., Li, Y., Xu, Z., Yang, M. H., Belongie, S., & Cui, Y. (2022). Multimodal open-vocabulary video classification via pre-trained vision and language models. arXiv preprint arXiv:2207.07646"},{"key":"2081_CR84","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., & Clark, J., et\u00a0al. (2021). Learning transferable visual models from natural language supervision. In International conference on machine learning (pp. 8748\u20138763). PMLR"},{"issue":"1","key":"2081_CR85","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1016\/S0165-1765(01)00524-9","volume":"74","author":"WJ Reed","year":"2001","unstructured":"Reed, W. J. (2001). The pareto, zipf and other power laws. Economics letters, 74(1), 15\u201319.","journal-title":"Economics letters"},{"key":"2081_CR86","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.aiopen.2022.01.001","volume":"3","author":"L Ruan","year":"2022","unstructured":"Ruan, L., & Jin, Q. (2022). Survey: Transformer based video-language pre-training. AI Open, 3, 1\u201313.","journal-title":"AI Open"},{"key":"2081_CR87","unstructured":"Ryoo, M. S., Piergiovanni A, Arnab, A., Dehghani, M., & Angelova, A. (2021). Tokenlearner: What can 8 learned tokens do for images and videos? arXiv preprint arXiv:2106.11297"},{"issue":"7","key":"2081_CR88","doi-asserted-by":"publisher","first-page":"1757","DOI":"10.1109\/TPAMI.2012.256","volume":"35","author":"WJ Scheirer","year":"2012","unstructured":"Scheirer, W. J., de Rezende, R. A., Sapkota, A., & Boult, T. E. (2012). Toward open set recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 35(7), 1757\u20131772.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"11","key":"2081_CR89","doi-asserted-by":"publisher","first-page":"2317","DOI":"10.1109\/TPAMI.2014.2321392","volume":"36","author":"WJ Scheirer","year":"2014","unstructured":"Scheirer, W. J., Jain, L. P., & Boult, T. E. (2014). Probability models for open set recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 36(11), 2317\u20132324.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2081_CR90","unstructured":"Sharir, G., Noy, A., Zelnik- & Manor, L. (2021). An image is worth 16x16 words, what is a video worth? arXiv preprint arXiv:2103.13915"},{"key":"2081_CR91","doi-asserted-by":"crossref","unstructured":"Shen, L., Lin, Z., & Huang, Q.(2016). Relay backpropagation for effective learning of deep convolutional neural networks. In European conference on computer vision (pp. 467\u2013482). Springer","DOI":"10.1007\/978-3-319-46478-7_29"},{"key":"2081_CR92","doi-asserted-by":"crossref","unstructured":"Shu, Y., Shi, Y., Wang, Y., Zou, Y., Yuan, Q., & Tian, Y. (2018). Odn: Opening the deep network for open-set action recognition. In 2018 IEEE international conference on multimedia and expo (ICME) (pp. 1\u20136). IEEE","DOI":"10.1109\/ICME.2018.8486601"},{"key":"2081_CR93","unstructured":"Simonyan, K., & Zisserman, A. (2014). Two-stream convolutional networks for action recognition in videos. In Advances in neural information processing systems 27"},{"key":"2081_CR94","doi-asserted-by":"crossref","unstructured":"Singh, A., Natarajan, V., Shah, M., Jiang, Y., Chen, X., Batra, D., Parikh, D., & Rohrbach, M. (2019). Towards VQA models that can read. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 8317\u20138326)","DOI":"10.1109\/CVPR.2019.00851"},{"key":"2081_CR95","unstructured":"Snell, J., Swersky, K., & Zemel, R. (2017). Prototypical networks for few-shot learning. In Advances in neural information processing systems 30"},{"key":"2081_CR96","unstructured":"Soomro, K., Zamir, A. R., & Shah, M (2012). Ucf101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402"},{"key":"2081_CR97","doi-asserted-by":"crossref","unstructured":"Stroud, J., Ross, D., Sun, C., Deng, J., & Sukthankar, R. (2020a). D3d: Distilled 3d networks for video action recognition. In Proceedings of the IEEE\/CVF winter conference on applications of computer vision (pp. 625\u2013634)","DOI":"10.1109\/WACV45572.2020.9093274"},{"key":"2081_CR98","unstructured":"Stroud, J. C, Ross, D. A., Sun, C., Deng, J., Sukthankar, R., & Schmid, C. (2020b). Learning video representations from textual web supervision. CoRR abs\/2007.14937, https:\/\/arxiv.org\/abs\/2007.14937, 2007.14937"},{"key":"2081_CR99","doi-asserted-by":"crossref","unstructured":"Subedar, M., Krishnan, R., Meyer, P. L., Tickoo, O., & Huang, J. (2019). Uncertainty-aware audiovisual activity recognition using deep Bayesian variational inference. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 6301\u20136310)","DOI":"10.1109\/ICCV.2019.00640"},{"key":"2081_CR100","doi-asserted-by":"crossref","unstructured":"Sun, X., Yang, Z., Zhang, C., Ling, K. V., & Peng, G. (2020). Conditional gaussian distribution learning for open set recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 13480\u201313489)","DOI":"10.1109\/CVPR42600.2020.01349"},{"key":"2081_CR101","doi-asserted-by":"crossref","unstructured":"Tan, J., Wang, C., Li, B., Li, Q., Ouyang, W., Yin, C., & Yan, J. (2020). Equalization loss for long-tailed object recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 11662\u201311671)","DOI":"10.1109\/CVPR42600.2020.01168"},{"key":"2081_CR102","doi-asserted-by":"crossref","unstructured":"Thatipelli, A., Narayan, S., Khan, S., Anwer, R. M., Khan, F. S., & Ghanem, B. (2022). Spatio-temporal relation modeling for few-shot action recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 19958\u201319967)","DOI":"10.1109\/CVPR52688.2022.01933"},{"key":"2081_CR103","doi-asserted-by":"crossref","unstructured":"Tian, C., Wang, W., Zhu, X., Dai, J., & Qiao, Y. (2022). VL-LTR: Learning class-wise visual-linguistic representation for long-tailed visual recognition. In European conference on computer vision (pp. 73\u201391). Springer","DOI":"10.1007\/978-3-031-19806-9_5"},{"key":"2081_CR104","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Ray, J., LeCun, Y., & Paluri, M. (2018). A closer look at spatiotemporal convolutions for action recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 6450\u20136459)","DOI":"10.1109\/CVPR.2018.00675"},{"key":"2081_CR105","unstructured":"Vinyals, O., Blundell, C., Lillicrap, T., & Wierstra, D., et al. (2016). Matching networks for one shot learning. In Advances in neural information processing systems 29"},{"issue":"1","key":"2081_CR106","doi-asserted-by":"publisher","first-page":"60","DOI":"10.1007\/s11263-012-0594-8","volume":"103","author":"H Wang","year":"2013","unstructured":"Wang, H., Kl\u00e4ser, A., Schmid, C., & Liu, C. L. (2013). Dense trajectories and motion boundary descriptors for action recognition. International Journal of Computer Vision, 103(1), 60\u201379.","journal-title":"International Journal of Computer Vision"},{"key":"2081_CR107","doi-asserted-by":"crossref","unstructured":"Wang, J., Ge, Y., Yan, R., Ge, Y., Lin, K. Q., Tsutsui, S., Lin, X., Cai, G., Wu, J., & Shan, Y., et\u00a0al. (2023). All in one: Exploring unified video-language pre-training. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 6598\u20136608)","DOI":"10.1109\/CVPR52729.2023.00638"},{"key":"2081_CR108","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., & Gool, L. V. (2016). Temporal segment networks: Towards good practices for deep action recognition. In European conference on computer vision (pp. 20\u201336). Springer","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"2081_CR109","doi-asserted-by":"crossref","unstructured":"Wang, L., Tong, Z., Ji, B., & Wu, G. (2021a). TDN: Temporal difference networks for efficient action recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 1895\u20131904)","DOI":"10.1109\/CVPR46437.2021.00193"},{"key":"2081_CR110","unstructured":"Wang, M., Xing, J., & Liu, Y. (2021b). Actionclip: A new paradigm for video action recognition. arXiv preprint arXiv:2109.08472"},{"key":"2081_CR111","doi-asserted-by":"crossref","unstructured":"Wang, W., Feiszli, M., Wang, H., & Tran, D. (2021c). Unidentified video objects: A benchmark for dense, open-world segmentation. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 10776\u201310785)","DOI":"10.1109\/ICCV48922.2021.01060"},{"key":"2081_CR112","doi-asserted-by":"crossref","unstructured":"Wang, X., Liu, Y., Shen, C., Ng, C. C., Luo, C., Jin, L., & Chan, C. S., Hengel, A., & Wang, L. (2020). On the general value of evidence, and bilingual scene-text visual question answering. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 10126\u201310135)","DOI":"10.1109\/CVPR42600.2020.01014"},{"key":"2081_CR113","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, S., Qing, Z., Tang, M., Zuo, Z., Gao, C., Jin, R., & Sang, N. (2022). Hybrid relation guided set matching for few-shot action recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 19948\u201319957)","DOI":"10.1109\/CVPR52688.2022.01932"},{"key":"2081_CR114","unstructured":"Wang, Y. X., Ramanan, D., & Hebert, M. (2017). Learning to model the tail. In Advances in Neural Information Processing Systems 30"},{"key":"2081_CR115","unstructured":"Weston, J., Bengio, S., & Usunier, N. (2011). Wsabie: Scaling up to large vocabulary image annotation. In Twenty-second international joint conference on artificial intelligence"},{"key":"2081_CR116","unstructured":"wikiHow. (2022). wikhow, the most trusted how-to site on the internet. https:\/\/www.wikihow.com\/Main-Page, Retrieved from May 19, 2022"},{"key":"2081_CR117","unstructured":"Wikipedia. (2022). Wikipedia, the free encyclopedia. https:\/\/www.wikipedia.org\/, Retrieved from May 19, 2022"},{"key":"2081_CR118","doi-asserted-by":"crossref","unstructured":"Wu, W., Sun, Z., & Ouyang, W. (2023). Revisiting classifier: Transferring vision-language models for video recognition. In Proceedings of the AAAI conference on artificial intelligence, 37, 2847\u20132855.","DOI":"10.1609\/aaai.v37i3.25386"},{"key":"2081_CR119","doi-asserted-by":"crossref","unstructured":"Xie, S., Sun, C., Huang, J., Tu, Z., & Murphy, K. (2018). Rethinking spatiotemporal feature learning: Speed-accuracy trade-offs in video classification. In Proceedings of the European conference on computer vision (ECCV) (pp. 305\u2013321)","DOI":"10.1007\/978-3-030-01267-0_19"},{"key":"2081_CR120","doi-asserted-by":"crossref","unstructured":"Xu, H., Ghosh, G., Huang, P. Y., Arora, P., Aminzadeh, M., Feichtenhofer, C., Metze, F., & Zettlemoyer, L. (2021). VLM: Task-agnostic video-language model pre-training for video understanding. arXiv preprint arXiv:2105.09996","DOI":"10.18653\/v1\/2021.findings-acl.370"},{"key":"2081_CR121","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, X., Arnab, A., Lu, Z., Zhang, M., Sun, C., & Schmid, C. (2022). Multiview transformers for video recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 3333\u20133343)","DOI":"10.1109\/CVPR52688.2022.00333"},{"key":"2081_CR122","doi-asserted-by":"crossref","unstructured":"Yang, X., Dong, J., Cao, Y., Wang, X., Wang, M., Chua, T. S. (2020). Tree-augmented cross-modal encoding for complex-query video retrieval. In Proceedings of the 43rd international ACM SIGIR conference on research and development in information retrieval (pp. 1339\u20131348)","DOI":"10.1145\/3397271.3401151"},{"key":"2081_CR123","doi-asserted-by":"crossref","unstructured":"Yin, X., Yu, X., Sohn, K., Liu, X., & Chandraker, M. (2019). Feature transfer learning for face recognition with under-represented data. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 5704\u20135713)","DOI":"10.1109\/CVPR.2019.00585"},{"key":"2081_CR124","unstructured":"Yuan, L., Chen, D., Chen, Y. L., Codella, N., Dai, X., Gao, J., Hu, H., Huang, X., Li, B., & Li, C. et\u00a0al. (2021). Florence: A new foundation model for computer vision. arXiv preprint arXiv:2111.11432"},{"key":"2081_CR125","unstructured":"Zhang, B., Yu, J., Fifty, C., Han, W., Dai, A. M., Pang, R., & Sha, F. (2021a). Co-training transformer with videos and images improves action recognition. arXiv preprint arXiv:2112.07175"},{"key":"2081_CR126","doi-asserted-by":"crossref","unstructured":"Zhang, H., Zhang, L., Qi, X., Li, H., Torr, P. H., & Koniusz, P. (2020). Few-shot action recognition with permutation-invariant attention. In European conference on computer vision (pp. 525\u2013542). Springer","DOI":"10.1007\/978-3-030-58558-7_31"},{"key":"2081_CR127","doi-asserted-by":"crossref","unstructured":"Zhang, P., Li, X., Hu, X., Yang, J., Zhang, L., Wang, L., Choi, Y., & Gao, J. (2021b). Vinvl: Revisiting visual representations in vision-language models. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 5579\u20135588)","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"2081_CR128","unstructured":"Zhang, R., Che, T., Ghahramani, Z., Bengio, Y., & Song, Y. (2018). Metagan: An adversarial approach to few-shot learning. In Advances in neural information processing systems 31"},{"key":"2081_CR129","doi-asserted-by":"crossref","unstructured":"Zhang, X., Wu, Z., Weng, Z., Fu, H., Chen, J., Jiang, Y. G., & Davis, L. S. (2021c). Videolt: Large-scale long-tailed video recognition. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 7960\u20137969)","DOI":"10.1109\/ICCV48922.2021.00786"},{"key":"2081_CR130","doi-asserted-by":"crossref","unstructured":"Zhou, B., Cui, Q., Wei, X. S., & Chen Z. M. (2020). Bbn: Bilateral-branch network with cumulative learning for long-tailed visual recognition. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 9719\u20139728)","DOI":"10.1109\/CVPR42600.2020.00974"},{"key":"2081_CR131","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Sun, X., Zha, Z. J., & Zeng, W. (2018). Mict: Mixed 3d\/2d convolutional tube for human action recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 449\u2013458)","DOI":"10.1109\/CVPR.2018.00054"},{"key":"2081_CR132","doi-asserted-by":"crossref","unstructured":"Zhu, L., & Yang, Y. (2018). Compound memory networks for few-shot video classification. In Proceedings of the European conference on computer vision (ECCV) (pp. 751\u2013766)","DOI":"10.1007\/978-3-030-01234-2_46"},{"key":"2081_CR133","doi-asserted-by":"crossref","unstructured":"Zhu, L., & Yang, Y. (2020a). Actbert: Learning global-local video-text representations. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 8746\u20138755)","DOI":"10.1109\/CVPR42600.2020.00877"},{"key":"2081_CR134","doi-asserted-by":"crossref","unstructured":"Zhu, L., & Yang, Y. (2020b). Inflated episodic memory with region self-attention for long-tailed visual recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 4344\u20134353","DOI":"10.1109\/CVPR42600.2020.00440"},{"key":"2081_CR135","doi-asserted-by":"crossref","unstructured":"Zhu, L., & Yang, Y. (2020). Label independent memory for semi-supervised few-shot video classification. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(1), 273\u2013285.","DOI":"10.1109\/TPAMI.2020.3007511"},{"key":"2081_CR136","unstructured":"Zhu, Y., Li, X., Liu, C., Zolfaghari ,M., Xiong, Y., Wu, C., Zhang, Z., Tighe, J., Manmatha, R., & Li, M. (2020). A comprehensive study of deep video action recognition. arXiv preprint arXiv:2012.06567"},{"key":"2081_CR137","unstructured":"Zhu, Z., Wang, L., Guo, S., & Wu, G., (2021). A closer look at few-shot video classification: A new baseline and benchmark. arXiv preprint arXiv:2110.12358"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02081-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02081-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02081-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,4]],"date-time":"2024-10-04T06:31:01Z","timestamp":1728023461000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02081-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,25]]},"references-count":137,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2024,10]]}},"alternative-id":["2081"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02081-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,5,25]]},"assertion":[{"value":"14 March 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 April 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 May 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}