{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T19:21:48Z","timestamp":1785698508300,"version":"3.56.0"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031781247","type":"print"},{"value":"9783031781254","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,12,5]],"date-time":"2024-12-05T00:00:00Z","timestamp":1733356800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,5]],"date-time":"2024-12-05T00:00:00Z","timestamp":1733356800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-78125-4_25","type":"book-chapter","created":{"date-parts":[[2024,12,4]],"date-time":"2024-12-04T06:09:48Z","timestamp":1733292588000},"page":"362-379","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["MCANet: Multimodal Caption Aware Training-Free Video Anomaly Detection via\u00a0Large Language Model"],"prefix":"10.1007","author":[{"given":"Prabhu Prasad","family":"Dev","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Raju","family":"Hazari","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pranesh","family":"Das","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,5]]},"reference":[{"key":"25_CR1","doi-asserted-by":"crossref","unstructured":"Zhao, M., Liu, Y., Liu, J., Zeng, X.: Exploiting spatial-temporal correlations for video anomaly detection. In: 2022 26th International Conference on Pattern Recognition (ICPR), pp. 1727\u20131733. IEEE (2022)","DOI":"10.1109\/ICPR56361.2022.9956287"},{"key":"25_CR2","doi-asserted-by":"crossref","unstructured":"Lee, J., Nam, W.-J., Lee, S.-W.: Multi-contextual predictions with vision transformer for video anomaly detection. In: 2022 26th International Conference on Pattern Recognition (ICPR), pp. 1012\u20131018. IEEE (2022)","DOI":"10.1109\/ICPR56361.2022.9956507"},{"key":"25_CR3","doi-asserted-by":"crossref","unstructured":"Deng, H., Zhang, Z., Zou, S., Li, X.: Bi-directional frame interpolation for unsupervised video anomaly detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2634\u20132643 (2023)","DOI":"10.1109\/WACV56688.2023.00266"},{"key":"25_CR4","doi-asserted-by":"crossref","unstructured":"Zaheer, M.Z., Mahmood, A., Khan, M.H., Segu, M., Yu, F., Lee, S.-I.: Generative cooperative learning for unsupervised video anomaly detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14744\u201314754 (2022)","DOI":"10.1109\/CVPR52688.2022.01433"},{"key":"25_CR5","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110500","volume":"153","author":"Z Sun","year":"2024","unstructured":"Sun, Z., Wang, P., Zheng, W., Zhang, M.: Dual GroupGAN: an unsupervised four-competitor (2V2) approach for video anomaly detection. Pattern Recogn. 153, 110500 (2024)","journal-title":"Pattern Recogn."},{"key":"25_CR6","doi-asserted-by":"crossref","unstructured":"Al-lahham, A., Tastan, N., Zaheer, M.Z., Nandakumar, K.: A coarse-to-fine pseudo-labeling (C2FPL) framework for unsupervised video anomaly detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6793\u20136802 (2024)","DOI":"10.1109\/WACV57701.2024.00665"},{"key":"25_CR7","doi-asserted-by":"crossref","unstructured":"Sultani, W., Chen, C., Shah, M.: Real-world anomaly detection in surveillance videos. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6479\u20136488 (2018)","DOI":"10.1109\/CVPR.2018.00678"},{"key":"25_CR8","doi-asserted-by":"crossref","unstructured":"Huang, C., et al.: Weakly supervised video anomaly detection via self-guided temporal discriminative transformer. IEEE Trans. Cybern. 54(5), 3197\u20133210 (2022)","DOI":"10.1109\/TCYB.2022.3227044"},{"key":"25_CR9","doi-asserted-by":"crossref","unstructured":"Ullah, W., Ullah, F.U.M., Khan, Z.A., Baik, S.W.: Sequential attention mechanism for weakly supervised video anomaly detection. Expert Syst. Appl. 230, 120599 (2023)","DOI":"10.1016\/j.eswa.2023.120599"},{"key":"25_CR10","doi-asserted-by":"crossref","unstructured":"Karim, H., Doshi, K., Yilmaz, Y.: Real-time weakly supervised video anomaly detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 6848\u20136856 (2024)","DOI":"10.1109\/WACV57701.2024.00670"},{"key":"25_CR11","doi-asserted-by":"crossref","unstructured":"Yan, L., Han, C., Xu, Z., Liu, D., Wang, Q.: Prompt learns prompt: exploring knowledge-aware generative prompt collaboration for video captioning. In: Proceedings of International Joint Conference on Artificial Intelligence (IJCAI), pp. 1622\u20131630 (2023)","DOI":"10.24963\/ijcai.2023\/180"},{"key":"25_CR12","unstructured":"Lin, K., et\u00a0al.: MM-VID: Advancing video understanding with GPT-4v (ision). arXiv preprint arXiv:2310.19773 (2023)"},{"key":"25_CR13","unstructured":"Chen, G., et\u00a0al.: VideoLLM: Modeling video sequence with large language models. arXiv preprint arXiv:2305.13292 (2023)"},{"key":"25_CR14","doi-asserted-by":"crossref","unstructured":"Jiang, C., et al.: BUS: efficient and effective vision-language pre-training with bottom-up patch summarization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2900\u20132910 (2023)","DOI":"10.1109\/ICCV51070.2023.00271"},{"key":"25_CR15","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1007\/978-3-030-58577-8_8","volume-title":"Computer Vision \u2013 ECCV 2020","author":"X Li","year":"2020","unstructured":"Li, X., et al.: Oscar: object-semantics aligned pre-training for vision-language tasks. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 121\u2013137. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_8"},{"key":"25_CR16","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., Zisserman, A.: Frozen in time: a joint video and image encoder for end-to-end retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1728\u20131738 (2021)","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"25_CR17","unstructured":"Li, K., et al.: VideoChat: Chat-centric video understanding. arXiv preprint arXiv:2305.06355 (2023)"},{"key":"25_CR18","doi-asserted-by":"crossref","unstructured":"Zhang, H., Li, X., Bing, L.: Video-LLaMa: An instruction-tuned audio-visual language model for video understanding. arXiv preprint arXiv:2306.02858 (2023)","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"25_CR19","doi-asserted-by":"crossref","unstructured":"Lin, B., Zhu, B., Ye, Y., Ning, M., Jin, P., Yuan, L.: Video-LLaVA: Learning united visual representation by alignment before projection. arXiv preprint arXiv:2311.10122 (2023)","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"25_CR20","doi-asserted-by":"crossref","unstructured":"He, B., et al.: MA-LMM: Memory-augmented large multimodal model for long-term video understanding. arXiv preprint arXiv:2404.05726 (2024)","DOI":"10.1109\/CVPR52733.2024.01282"},{"key":"25_CR21","doi-asserted-by":"crossref","unstructured":"Rotstein, N., Bensa\u00efd, D., Brody, S., Ganz, R., Kimmel, R.: FuseCap: leveraging large language models for enriched fused image captions. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 5689\u20135700 (2024)","DOI":"10.1109\/WACV57701.2024.00559"},{"key":"25_CR22","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"322","DOI":"10.1007\/978-3-030-58577-8_20","volume-title":"Computer Vision \u2013 ECCV 2020","author":"P Wu","year":"2020","unstructured":"Wu, P., et al.: Not only look, but also listen: learning multimodal violence detection under weak supervision. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 322\u2013339. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_20"},{"key":"25_CR23","doi-asserted-by":"crossref","unstructured":"Lv, H., Yue, Z., Sun, Q., Luo, B., Cui, Z., Zhang, H.: Unbiased multiple instance learning for weakly supervised video anomaly detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8022\u20138031 (2023)","DOI":"10.1109\/CVPR52729.2023.00775"},{"key":"25_CR24","doi-asserted-by":"crossref","unstructured":"Girdhar, R., et al.: ImageBind: one embedding space to bind them all. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15180\u201315190 (2023)","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"25_CR25","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"25_CR26","doi-asserted-by":"crossref","unstructured":"Hershey, S., et\u00a0al.: CNN architectures for large-scale audio classification. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 131\u2013135. IEEE (2017)","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"25_CR27","first-page":"18090","volume":"36","author":"S Deshmukh","year":"2023","unstructured":"Deshmukh, S., Elizalde, B., Singh, R., Wang, H.: Pengi: an audio language model for audio tasks. Adv. Neural. Inf. Process. Syst. 36, 18090\u201318108 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"25_CR28","doi-asserted-by":"crossref","unstructured":"Wu, P., Liu, X., Liu, J.: Weakly supervised audio-visual violence detection. IEEE Transactions on Multimedia (2022)","DOI":"10.1109\/TMM.2022.3147369"},{"key":"25_CR29","doi-asserted-by":"crossref","unstructured":"Zhen, Y., Guo, Y., Wei, J., Bao, X., Huang, D.: Multi-scale background suppression anomaly detection in surveillance videos. In: 2021 IEEE International Conference on Image Processing (ICIP), pp. 1114\u20131118. IEEE (2021)","DOI":"10.1109\/ICIP42928.2021.9506580"},{"key":"25_CR30","doi-asserted-by":"crossref","unstructured":"Dev, P.P., Das, P., Hazari, R.: MSDeepNet: a novel multi-stream deep neural network for real-world anomaly detection in surveillance videos. In: International Conference on Deep Learning Theory and Applications, pp. 157\u2013172. Springer (2023)","DOI":"10.1007\/978-3-031-39059-3_11"},{"key":"25_CR31","doi-asserted-by":"crossref","unstructured":"Park, S., Kim, H., Kim, M., Kim, D., Sohn, K.: Normality guided multiple instance learning for weakly supervised video anomaly detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2665\u20132674 (2023)","DOI":"10.1109\/WACV56688.2023.00269"},{"key":"25_CR32","doi-asserted-by":"crossref","unstructured":"Tian, Y., Pang, G., Chen, Y., Singh, R., Verjans, J.W., Carneiro, G.: Weakly-supervised video anomaly detection with robust temporal feature magnitude learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4975\u20134986 (2021)","DOI":"10.1109\/ICCV48922.2021.00493"},{"key":"25_CR33","doi-asserted-by":"crossref","unstructured":"Zanella, L., Liberatori, B., Menapace, W., Poiesi, F., Wang, Y., Ricci, E.: Delving into clip latent space for video anomaly recognition. arXiv preprint arXiv:2310.02835 (2023)","DOI":"10.2139\/ssrn.4768666"},{"key":"25_CR34","doi-asserted-by":"crossref","unstructured":"Chen, Y., Liu, Z., Zhang, B., Fok, W., Qi, X., Yik-Chung, W.: MGFN: magnitude-contrastive glance-and-focus network for weakly-supervised video anomaly detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 387\u2013395 (2023)","DOI":"10.1609\/aaai.v37i1.25112"},{"key":"25_CR35","doi-asserted-by":"crossref","unstructured":"Zhou, H., Junqing, Yu., Yang, W.: Dual memory units with uncertainty regulation for weakly supervised video anomaly detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 3769\u20133777 (2023)","DOI":"10.1609\/aaai.v37i3.25489"},{"key":"25_CR36","doi-asserted-by":"crossref","unstructured":"Joo, H.K., Vo, K., Yamazaki, K., Le, N.: CLIP-TSA: clip-assisted temporal self-attention for weakly-supervised video anomaly detection. In: 2023 IEEE International Conference on Image Processing (ICIP), pp. 3230\u20133234. IEEE (2023)","DOI":"10.1109\/ICIP49359.2023.10222289"},{"key":"25_CR37","doi-asserted-by":"crossref","unstructured":"Hasan, M., Choi, J., Neumann, J., Roy-Chowdhury, A.K., Davis, L.S.: Learning temporal regularity in video sequences. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 733\u2013742 (2016)","DOI":"10.1109\/CVPR.2016.86"},{"key":"25_CR38","doi-asserted-by":"crossref","unstructured":"Sohrab, F., Raitoharju, J., Gabbouj, M., Iosifidis, A.: Subspace support vector data description. In: 2018 24th International Conference on Pattern Recognition (ICPR), pp. 722\u2013727. IEEE (2018)","DOI":"10.1109\/ICPR.2018.8545819"},{"key":"25_CR39","doi-asserted-by":"crossref","unstructured":"Wang, J., Cherian, A.: GODS: generalized one-class discriminative subspaces for anomaly detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8201\u20138211 (2019)","DOI":"10.1109\/ICCV.2019.00829"},{"key":"25_CR40","doi-asserted-by":"crossref","unstructured":"Sun, C., Jia, Y., Hu, Y., Wu, Y.: Scene-aware context reasoning for unsupervised abnormal event detection in videos. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 184\u2013192 (2020)","DOI":"10.1145\/3394171.3413887"},{"key":"25_CR41","doi-asserted-by":"crossref","unstructured":"Tur, A.O., Dall\u2019Asen, N., Beyan, C., Ricci, E.: Exploring diffusion models for unsupervised video anomaly detection. In: 2023 IEEE International Conference on Image Processing (ICIP), pp. 2540\u20132544. IEEE (2023)","DOI":"10.1109\/ICIP49359.2023.10222594"},{"key":"25_CR42","doi-asserted-by":"crossref","unstructured":"Tur, A.O., Dall\u2019Asen, N., Beyan, C., Ricci, E.: Unsupervised video anomaly detection with diffusion models conditioned on compact motion representations. In: International Conference on Image Analysis and Processing, pp. 49\u201362. Springer (2023)","DOI":"10.1007\/978-3-031-43153-1_5"},{"key":"25_CR43","doi-asserted-by":"crossref","unstructured":"Thakare, K.V., Raghuwanshi, Y., Dogra, D.P., Choi, H., Kim, I.-J.: DyAnNet: a scene dynamicity guided self-trained video anomaly detection network. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 5541\u20135550 (2023)","DOI":"10.1109\/WACV56688.2023.00550"},{"key":"25_CR44","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"25_CR45","doi-asserted-by":"crossref","unstructured":"Zhaopeng, G., Zhu, B., Zhu, G., Chen, Y., Tang, M., Wang, J.: AnomalyGPT: detecting industrial anomalies using large vision-language models. In Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 1932\u20131940 (2024)","DOI":"10.1609\/aaai.v38i3.27963"},{"key":"25_CR46","doi-asserted-by":"crossref","unstructured":"Zanella, L., Menapace, W., Mancini, M., Wang, Y., Ricci, E.: Harnessing large language models for training-free video anomaly detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18527\u201318536 (2024)","DOI":"10.1109\/CVPR52733.2024.01753"},{"key":"25_CR47","doi-asserted-by":"crossref","unstructured":"Lu, C., Shi, J., Jia, J.: Abnormal event detection at 150 FPS in MATLAB. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2720\u20132727 (2013)","DOI":"10.1109\/ICCV.2013.338"},{"key":"25_CR48","doi-asserted-by":"crossref","unstructured":"Thakare, K.V., Dogra, D.P., Choi, H., Kim, H., Kim, I.-J.: RareAnom: a benchmark video dataset for rare type anomalies. Pattern Recog. 140, 109567 (2023)","DOI":"10.1016\/j.patcog.2023.109567"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-78125-4_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,4]],"date-time":"2024-12-04T07:08:06Z","timestamp":1733296086000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-78125-4_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,5]]},"ISBN":["9783031781247","9783031781254"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-78125-4_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,5]]},"assertion":[{"value":"5 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kolkata","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2024.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}