{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T23:59:54Z","timestamp":1782518394087,"version":"3.54.5"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031783531","type":"print"},{"value":"9783031783548","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,12,4]],"date-time":"2024-12-04T00:00:00Z","timestamp":1733270400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,4]],"date-time":"2024-12-04T00:00:00Z","timestamp":1733270400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-78354-8_21","type":"book-chapter","created":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T10:31:21Z","timestamp":1733221881000},"page":"327-342","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Text-Enhanced Zero-Shot Action Recognition: A Training-Free Approach"],"prefix":"10.1007","author":[{"given":"Massimo","family":"Bosetti","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shibingfeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bendetta","family":"Liberatori","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Giacomo","family":"Zara","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Elisa","family":"Ricci","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Paolo","family":"Rota","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,4]]},"reference":[{"key":"21_CR1","unstructured":"An, B., Zhu, S., Panaitescu-Liess, M.A., Mummadi, C.K., Huang, F.: More context, less distraction: Visual classification by inferring and conditioning on contextual attributes. arXiv (2023)"},{"key":"21_CR2","doi-asserted-by":"crossref","unstructured":"Brattoli, B., Tighe, J., Zhdanov, F., Perona, P., Chalupka, K.: Rethinking zero-shot video classification: End-to-end training for realistic applications. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00467"},{"key":"21_CR3","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et\u00a0al.: Language models are few-shot learners. NeurIPS (2020)"},{"key":"21_CR4","unstructured":"Carreira, J., Noland, E., Banki-Horvath, A., Hillier, C., Zisserman, A.: A short note about kinetics-600. arXiv (2018)"},{"key":"21_CR5","doi-asserted-by":"crossref","unstructured":"Chen, S., Huang, D.: Elaborative rehearsal for zero-shot action recognition. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01338"},{"key":"21_CR6","doi-asserted-by":"crossref","unstructured":"Deng, A., Yang, T., Chen, C.: A large-scale study of spatiotemporal representation learning with a new benchmark on action recognition. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01876"},{"key":"21_CR7","doi-asserted-by":"crossref","unstructured":"Doshi, K., Yilmaz, Y.: Zero-shot action recognition with transformer-based video semantic embedding. In: CVPRW (2023)","DOI":"10.1109\/CVPRW59228.2023.00514"},{"key":"21_CR8","doi-asserted-by":"crossref","unstructured":"Estevam, Laroca, P.e.a.: Tell me what you see: A zero-shot action recognition method based on natural language descriptions. In: Multimed Tools Appl (2024)","DOI":"10.1007\/s11042-023-16566-5"},{"key":"21_CR9","doi-asserted-by":"crossref","unstructured":"Gao, J., Hou, Y., Guo, Z., Zheng, H.: Learning spatio-temporal semantics and cluster relation for zero-shot action recognition. IEEE Transactions on Circuits and Systems for Video Technology (2023)","DOI":"10.1109\/TCSVT.2023.3272627"},{"key":"21_CR10","unstructured":"Gu, X., Lin, T.Y., Kuo, W., Cui, Y.: Open-vocabulary object detection via vision and language knowledge distillation. In: ICLR (2022)"},{"key":"21_CR11","unstructured":"Huang, X., Zhou, H., Yao, K., Han, K.: FROSTER: Frozen CLIP is a strong teacher for open-vocabulary action recognition. In: ICLR (2024)"},{"key":"21_CR12","unstructured":"Jia, C., Yang, Y., Xia, Y., Chen, Y.T., Parekh, Z., Pham, H., Le, Q., Sung, Y.H., Li, Z., Duerig, T.: Scaling up visual and vision-language representation learning with noisy text supervision. In: ICML (2021)"},{"key":"21_CR13","unstructured":"Jiang, A., Sablayrolles, A., Mensch, A., Bamford, C., Chaplot, D., de\u00a0las Casas, D., Bressand, F., Lengyel, G., Lample, G., Saulnier, L., et\u00a0al.: Mistral 7b (2023). arXiv (2023)"},{"key":"21_CR14","doi-asserted-by":"crossref","unstructured":"Ju, C., Han, T., Zheng, K., Zhang, Y., Xie, W.: Prompting visual-language models for efficient video understanding. In: ECCV (2022)","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"21_CR15","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P., et\u00a0al.: The kinetics human action video dataset. arXiv (2017)"},{"key":"21_CR16","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H., Garrote, E., Poggio, T., Serre, T.: Hmdb: A large video database for human motion recognition. In: ICCV (2011)","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"21_CR17","doi-asserted-by":"crossref","unstructured":"Liberatori, B., Conti, A., Rota, P., Wang, Y., Ricci, E.: Test-time zero-shot temporal action localization. arXiv (2024)","DOI":"10.1109\/CVPR52733.2024.01771"},{"key":"21_CR18","doi-asserted-by":"crossref","unstructured":"Lin, W., Karlinsky, L., Shvetsova, N., Possegger, H., Kozinski, M., Panda, R., Feris, R., Kuehne, H., Bischof, H.: Match, expand and improve: Unsupervised finetuning for zero-shot action recognition with language knowledge. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00267"},{"key":"21_CR19","doi-asserted-by":"crossref","unstructured":"Liu, J., Kuipers, B., Savarese, S.: Recognizing human actions by attributes. In: CVPR (2011)","DOI":"10.1109\/CVPR.2011.5995353"},{"key":"21_CR20","doi-asserted-by":"crossref","unstructured":"Mandal, D., Narayan, S., Dwivedi, S.K., Gupta, V., Ahmed, S., Khan, F.S., Shao, L.: Out-of-distribution detection for generalized zero-shot action recognition. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01022"},{"key":"21_CR21","unstructured":"Menon, S., Vondrick, C.: Visual classification via description from large language models. In: ICLR (2023)"},{"key":"21_CR22","unstructured":"Meta, A.: Introducing meta llama 3: The most capable openly available llm to date. Meta AI (2024)"},{"key":"21_CR23","doi-asserted-by":"crossref","unstructured":"Mettes, P.: Universal prototype transport for zero-shot action recognition and localization. IJCV (2023)","DOI":"10.1007\/s11263-023-01846-2"},{"key":"21_CR24","doi-asserted-by":"crossref","unstructured":"Momeni, L., Caron, M., Nagrani, A., Zisserman, A., Schmid, C.: Verbs in action: Improving verb understanding in video-language models. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01428"},{"key":"21_CR25","doi-asserted-by":"crossref","unstructured":"Nag, S., Zhu, X., Song, Y.Z., Xiang, T.: Zero-shot temporal action detection via vision-language prompting. In: ECCV (2022)","DOI":"10.1007\/978-3-031-20062-5_39"},{"key":"21_CR26","doi-asserted-by":"crossref","unstructured":"Ni, B., Peng, H., Chen, M., Zhang, S., Meng, G., Fu, J., Xiang, S., Ling, H.: Expanding language-image pretrained models for general video recognition. In: ECCV (2022)","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"21_CR27","unstructured":"OpenAI: Chatgpt: Gpt-4 (2024), https:\/\/www.openai.com\/, accessed: 2024-07-05"},{"key":"21_CR28","doi-asserted-by":"crossref","unstructured":"Park, J.S., Shen, S., Farhadi, A., Darrell, T., Choi, Y., Rohrbach, A.: Exposing the limits of video-text models through contrast sets. In: ACL (2022)","DOI":"10.18653\/v1\/2022.naacl-main.261"},{"key":"21_CR29","doi-asserted-by":"crossref","unstructured":"Pratt, S., Covert, I., Liu, R., Farhadi, A.: What does a platypus look like? generating customized prompts for zero-shot image classification. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01438"},{"key":"21_CR30","doi-asserted-by":"crossref","unstructured":"Qi, C., Feng, Z., Xing, M., Su, Y., Zheng, J., Zhang, Y.: Energy-based temporal summarized attentive network for zero-shot action recognition. IEEE Transactions on Multimedia (2023)","DOI":"10.1109\/TMM.2023.3264847"},{"key":"21_CR31","doi-asserted-by":"crossref","unstructured":"Qian, Y., Yu, L., Liu, W., Hauptmann, A.G.: Rethinking zero-shot action recognition: Learning from latent atomic actions. In: ECCV (2022)","DOI":"10.1007\/978-3-031-19772-7_7"},{"key":"21_CR32","doi-asserted-by":"crossref","unstructured":"Qin, J., Liu, L., Shao, L., Shen, F., Ni, B., Chen, J., Wang, Y.: Zero-shot action recognition with error-correcting output codes. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.117"},{"key":"21_CR33","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"21_CR34","unstructured":"Ranasinghe, K., Ryoo, M.S.: Language-based action concept spaces improve video self-supervised learning. In: NeurIPS (2024)"},{"key":"21_CR35","doi-asserted-by":"crossref","unstructured":"Rao, Y., Zhao, W., Chen, G., Tang, Y., Zhu, Z., Huang, G., Zhou, J., Lu, J.: Denseclip: Language-guided dense prediction with context-aware prompting. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01755"},{"key":"21_CR36","doi-asserted-by":"crossref","unstructured":"Rasheed, H., Khattak, M.U., Maaz, M., Khan, S., Khan, F.S.: Fine-tuned clip models are efficient video learners. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00633"},{"key":"21_CR37","unstructured":"Rasheed, H., Maaz, M., Khattak, M.U., Khan, S., Khan, F.S.: Bridging the gap between object and image-level representations for open-vocabulary detection. In: NeurIPS (2022)"},{"key":"21_CR38","doi-asserted-by":"crossref","unstructured":"Roth, K., Kim, J.M., Koepke, A.S., Vinyals, O., Schmid, C., Akata, Z.: Waffling around for performance: Visual classification with random words and broad concepts. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01443"},{"key":"21_CR39","doi-asserted-by":"crossref","unstructured":"Shao, H., Qian, S., Liu, Y.: Temporal interlacing network. In: AAAI (2020)","DOI":"10.1609\/aaai.v34i07.6872"},{"key":"21_CR40","unstructured":"Soomro, K., Zamir, A.R., Shah, M.: Ucf101: A dataset of 101 human actions classes from videos in the wild. arXiv (2012)"},{"key":"21_CR41","unstructured":"Wang, M., Xing, J., Liu, Y.: Actionclip: A new paradigm for video action recognition. arXiv (2021)"},{"key":"21_CR42","unstructured":"Wang, Z., Blume, A., Li, S., Liu, G., Cho, J., Tang, Z., Bansal, M., Ji, H.: Paxion: Patching action knowledge in video-language foundation models. NeurIPS (2024)"},{"key":"21_CR43","doi-asserted-by":"crossref","unstructured":"Wu, W., Sun, Z., Ouyang, W.: Revisiting classifier: Transferring vision-language models for video recognition. In: AAAI (2023)","DOI":"10.1609\/aaai.v37i3.25386"},{"key":"21_CR44","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, X., Nagrani, A., Arnab, A., Wang, Z., Ge, W., Ross, D., Schmid, C.: Unloc: A unified framework for video localization tasks. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01253"},{"key":"21_CR45","doi-asserted-by":"crossref","unstructured":"Yang, Z., An, G., Zheng, Z., Cao, S., Wang, F.: Epk-clip: External and priori knowledge clip for action recognition. Expert Systems with Applications (2024)","DOI":"10.1016\/j.eswa.2024.124183"},{"key":"21_CR46","doi-asserted-by":"crossref","unstructured":"Zellers, R., Choi, Y.: Zero-shot activity recognition with verb attribute induction. In: EMNLP (2017)","DOI":"10.18653\/v1\/D17-1099"},{"key":"21_CR47","unstructured":"Zhang, R., Fang, R., Zhang, W., Gao, P., Li, K., Dai, J., Qiao, Y., Li, H.: Tip-adapter: Training-free clip-adapter for better vision-language modeling. arXiv (2021)"},{"key":"21_CR48","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Zhuo, J., Ma, B., Geng, J., Wei, X., Wei, X., Wang, S.: Orthogonal temporal interpolation for zero-shot video recognition. In: ACMM-MM (2023)","DOI":"10.1145\/3581783.3611903"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-78354-8_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T11:30:22Z","timestamp":1733225422000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-78354-8_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,4]]},"ISBN":["9783031783531","9783031783548"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-78354-8_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,4]]},"assertion":[{"value":"4 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kolkata","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2024.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}