{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T07:04:02Z","timestamp":1784531042418,"version":"3.55.0"},"reference-count":87,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Image and Vision Computing"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.imavis.2026.106072","type":"journal-article","created":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T23:44:37Z","timestamp":1781048677000},"page":"106072","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["HCATRE-AVAD: Hierarchical cross-alignment and temporal relational encoding for weakly supervised audio-visual anomaly detection"],"prefix":"10.1016","volume":"173","author":[{"given":"Nuku Atta Kordzo","family":"Abiew","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lijian","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Godbless","family":"Mensah","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenlong","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qirong","family":"Mao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.imavis.2026.106072_b1","series-title":"2017 IEEE International Conference on Image Processing","first-page":"1577","article-title":"Abnormal event detection in videos using generative adversarial nets","author":"Ravanbakhsh","year":"2017"},{"key":"10.1016\/j.imavis.2026.106072_b2","doi-asserted-by":"crossref","unstructured":"R. Hinami, T. Mei, S. Satoh, Joint detection and recounting of abnormal events by learning deep generic knowledge, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 3619\u20133627.","DOI":"10.1109\/ICCV.2017.391"},{"key":"10.1016\/j.imavis.2026.106072_b3","doi-asserted-by":"crossref","unstructured":"W. Luo, W. Liu, S. Gao, A revisit of sparse coding based anomaly detection in stacked rnn framework, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 341\u2013349.","DOI":"10.1109\/ICCV.2017.45"},{"key":"10.1016\/j.imavis.2026.106072_b4","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2021.116306","article-title":"Enhanced the moving object detection and object tracking for traffic surveillance using rbf-fdlnn and cbf algorithm","volume":"191","author":"Chandrakar","year":"2022","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.imavis.2026.106072_b5","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6479","article-title":"Real-world anomaly detection in surveillance videos","author":"Sultani","year":"2018"},{"issue":"21","key":"10.1016\/j.imavis.2026.106072_b6","doi-asserted-by":"crossref","first-page":"10709","DOI":"10.1007\/s10489-024-05775-6","article-title":"Crowd behavior detection: leveraging video swin transformer for crowd size and violence level analysis","volume":"54","author":"Qaraqe","year":"2024","journal-title":"Appl. Intell."},{"key":"10.1016\/j.imavis.2026.106072_b7","doi-asserted-by":"crossref","unstructured":"H. Karim, K. Doshi, Y. Yilmaz, Real-time weakly supervised video anomaly detection, in: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, 2024, pp. 6848\u20136856.","DOI":"10.1109\/WACV57701.2024.00670"},{"issue":"2","key":"10.1016\/j.imavis.2026.106072_b8","doi-asserted-by":"crossref","first-page":"423","DOI":"10.1109\/TPAMI.2018.2798607","article-title":"Multimodal machine learning: A survey and taxonomy","volume":"41","author":"Baltru\u0161aitis","year":"2019","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"10","key":"10.1016\/j.imavis.2026.106072_b9","doi-asserted-by":"crossref","first-page":"12113","DOI":"10.1109\/TPAMI.2023.3275156","article-title":"Multimodal learning with transformers: A survey","volume":"45","author":"Xu","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.imavis.2026.106072_b10","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110898","article-title":"Semantic-driven dual consistency learning for weakly supervised video anomaly detection","volume":"157","author":"Su","year":"2025","journal-title":"Pattern Recognit."},{"issue":"8","key":"10.1016\/j.imavis.2026.106072_b11","doi-asserted-by":"crossref","first-page":"2939","DOI":"10.1007\/s00371-021-02166-7","article-title":"A survey on deep multimodal learning for computer vision: advances, trends, applications, and datasets","volume":"38","author":"Bayoudh","year":"2022","journal-title":"Vis. Comput."},{"key":"10.1016\/j.imavis.2026.106072_b12","series-title":"2024 IEEE International Conference on Image Processing","first-page":"1106","article-title":"Mavad: Audio-visual dataset and method for anomaly detection in traffic videos","author":"Leporowski","year":"2024"},{"issue":"5","key":"10.1016\/j.imavis.2026.106072_b13","doi-asserted-by":"crossref","first-page":"829","DOI":"10.1162\/neco_a_01273","article-title":"A survey on deep learning for multimodal data fusion","volume":"32","author":"Gao","year":"2020","journal-title":"Neural Comput."},{"key":"10.1016\/j.imavis.2026.106072_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2020.103915","article-title":"Anomaly detection in surveillance video based on bidirectional prediction","volume":"98","author":"Chen","year":"2020","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.106072_b15","series-title":"CVPR 2011","first-page":"3449","article-title":"Sparse reconstruction cost for abnormal event detection","author":"Cong","year":"2011"},{"key":"10.1016\/j.imavis.2026.106072_b16","series-title":"2010 IEEE Computer Society Conference on Computer Vision and Pattern Recognition","first-page":"1975","article-title":"Anomaly detection in crowded scenes","author":"Mahadevan","year":"2010"},{"key":"10.1016\/j.imavis.2026.106072_b17","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2023.103798","article-title":"End-to-end learning for weakly supervised video anomaly detection using absorbing markov chain","volume":"236","author":"Park","year":"2023","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.imavis.2026.106072_b18","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2021.104229","article-title":"Intelligent video anomaly detection and classification using faster rcnn with deep reinforcement learning model","volume":"112","author":"Mansour","year":"2021","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.106072_b19","doi-asserted-by":"crossref","unstructured":"G.A. Noghre, A.D. Pazho, H. Tabkhi, An exploratory study on human-centric video anomaly detection through variational autoencoders and trajectory prediction, in: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, 2024, pp. 995\u20131004.","DOI":"10.1109\/WACVW60836.2024.00109"},{"key":"10.1016\/j.imavis.2026.106072_b20","doi-asserted-by":"crossref","unstructured":"A. Al-Lahham, M.Z. Zaheer, N. Tastan, K. Nandakumar, Collaborative learning of anomalies with privacy (clap) for unsupervised video anomaly detection: A new baseline, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 12416\u201312425.","DOI":"10.1109\/CVPR52733.2024.01180"},{"key":"10.1016\/j.imavis.2026.106072_b21","article-title":"Multi-level feature splicing 3d network based on multi-task joint learning for video anomaly detection","author":"Li","year":"2025","journal-title":"Neurocomputing"},{"issue":"23","key":"10.1016\/j.imavis.2026.106072_b22","doi-asserted-by":"crossref","first-page":"28133","DOI":"10.1007\/s10489-023-04940-7","article-title":"Stemgan: spatio-temporal generative adversarial network for video anomaly detection","volume":"53","author":"Singh","year":"2023","journal-title":"Appl. Intell."},{"issue":"3","key":"10.1016\/j.imavis.2026.106072_b23","doi-asserted-by":"crossref","first-page":"3240","DOI":"10.1007\/s10489-022-03613-1","article-title":"Attention-based residual autoencoder for video anomaly detection","volume":"53","author":"Le","year":"2023","journal-title":"Appl. Intell."},{"key":"10.1016\/j.imavis.2026.106072_b24","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2025.105644","article-title":"Mg-kg: Unsupervised video anomaly detection based on motion guidance and knowledge graph","volume":"162","author":"Sun","year":"2025","journal-title":"Image Vis. Comput."},{"issue":"3","key":"10.1016\/j.imavis.2026.106072_b25","doi-asserted-by":"crossref","first-page":"61","DOI":"10.1007\/s00138-025-01676-x","article-title":"Vicap-ad: video caption-based weakly supervised video anomaly detection","volume":"36","author":"Lim","year":"2025","journal-title":"Mach. Vis. Appl."},{"key":"10.1016\/j.imavis.2026.106072_b26","doi-asserted-by":"crossref","unstructured":"Y. Tian, G. Pang, Y. Chen, R. Singh, J.W. Verjans, G. Carneiro, Weakly-supervised video anomaly detection with robust temporal feature magnitude learning, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 4975\u20134986.","DOI":"10.1109\/ICCV48922.2021.00493"},{"key":"10.1016\/j.imavis.2026.106072_b27","doi-asserted-by":"crossref","unstructured":"J.-X. Zhong, N. Li, W. Kong, S. Liu, T.H. Li, G. Li, Graph convolutional label noise cleaner: Train a plug-and-play action classifier for anomaly detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 1237\u20131246.","DOI":"10.1109\/CVPR.2019.00133"},{"key":"10.1016\/j.imavis.2026.106072_b28","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.111942","article-title":"Normality learning reinforcement for anomaly detection in surveillance videos","volume":"297","author":"Cheng","year":"2024","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.imavis.2026.106072_b29","doi-asserted-by":"crossref","unstructured":"J.-C. Feng, F.-T. Hong, W.-S. Zheng, Mist: Multiple instance self-training framework for video anomaly detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 14009\u201314018.","DOI":"10.1109\/CVPR46437.2021.01379"},{"key":"10.1016\/j.imavis.2026.106072_b30","doi-asserted-by":"crossref","unstructured":"J. Gao, M. Chen, C. Xu, Fine-grained temporal contrastive learning for weakly-supervised temporal action localization, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 19999\u201320009.","DOI":"10.1109\/CVPR52688.2022.01937"},{"issue":"24","key":"10.1016\/j.imavis.2026.106072_b31","doi-asserted-by":"crossref","first-page":"30607","DOI":"10.1007\/s10489-023-05072-8","article-title":"Weakly-supervised video anomaly detection via temporal resolution feature learning","volume":"53","author":"Peng","year":"2023","journal-title":"Appl. Intell."},{"key":"10.1016\/j.imavis.2026.106072_b32","doi-asserted-by":"crossref","first-page":"4505","DOI":"10.1109\/TIP.2021.3072863","article-title":"Localizing anomalies from weakly-labeled videos","volume":"30","author":"Lv","year":"2021","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.imavis.2026.106072_b33","doi-asserted-by":"crossref","unstructured":"P. Wu, X. Zhou, G. Pang, Z. Yang, Q. Yan, P. Wang, Y. Zhang, Weakly supervised video anomaly detection and localization with spatio-temporal prompts, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 9301\u20139310.","DOI":"10.1145\/3664647.3681442"},{"key":"10.1016\/j.imavis.2026.106072_b34","article-title":"Anomaly-aware self-supervised feature learning for weakly supervised video anomaly detection","author":"Yang","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.imavis.2026.106072_b35","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2023.103656","article-title":"Ssmtl++: Revisiting self-supervised multi-task learning for video anomaly detection","volume":"229","author":"Barbalau","year":"2023","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.imavis.2026.106072_b36","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2024.103955","article-title":"Human-scene network: A novel baseline with self-rectifying loss for weakly supervised video anomaly detection","volume":"241","author":"Majhi","year":"2024","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.imavis.2026.106072_b37","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105169","article-title":"Event-driven weakly supervised video anomaly detection","volume":"149","author":"Sun","year":"2024","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.106072_b38","article-title":"Degree-aware weakly supervised video anomaly detection with soft consistency learning and rule attention fuzzy neural network assistance","author":"Zhang","year":"2026","journal-title":"Comput. Vis. Image Underst."},{"issue":"5","key":"10.1016\/j.imavis.2026.106072_b39","doi-asserted-by":"crossref","first-page":"3003","DOI":"10.1007\/s00371-024-03584-z","article-title":"Video anomaly detection with both normal and anomaly memory modules","volume":"41","author":"Zhang","year":"2025","journal-title":"Vis. Comput."},{"key":"10.1016\/j.imavis.2026.106072_b40","article-title":"Multimodal learning for anomaly detection in videos","author":"Tian","year":"2021","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.imavis.2026.106072_b41","doi-asserted-by":"crossref","DOI":"10.1109\/TMM.2025.3535377","article-title":"Audio-visual collaborative learning for weakly supervised video anomaly detection","author":"Meng","year":"2025","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.imavis.2026.106072_b42","series-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10491","article-title":"Self-supervised video forensics by audio-visual anomaly detection","author":"Feng","year":"2023"},{"key":"10.1016\/j.imavis.2026.106072_b43","series-title":"2021 IEEE International Conference on Image Processing","first-page":"1569","article-title":"Learning of linear video prediction models in a multi-modal framework for anomaly detection","author":"Slavic","year":"2021"},{"key":"10.1016\/j.imavis.2026.106072_b44","series-title":"European Conference on Computer Vision","first-page":"322","article-title":"Not only look, but also listen: Learning multimodal violence detection under weak supervision","author":"Wu","year":"2020"},{"key":"10.1016\/j.imavis.2026.106072_b45","doi-asserted-by":"crossref","unstructured":"H. Xuan, Z. Zhang, S. Chen, J. Yang, Y. Yan, Cross-modal attention network for temporal inconsistent audio-visual event localization, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 34, 2020, pp. 279\u2013286.","DOI":"10.1609\/aaai.v34i01.5361"},{"key":"10.1016\/j.imavis.2026.106072_b46","doi-asserted-by":"crossref","first-page":"7878","DOI":"10.1109\/TIP.2021.3106814","article-title":"Discriminative cross-modality attention network for temporal inconsistent audio-visual event localization","volume":"30","author":"Xuan","year":"2021","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.imavis.2026.106072_b47","doi-asserted-by":"crossref","unstructured":"L. Zhou, P. Wu, M. Zhang, Q. Wang, G. Pang, P. Wang, Targetvau: Multimodal anomaly-aware reasoning for target behavior understanding in videos, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 40, 2026, pp. 13710\u201313718.","DOI":"10.1609\/aaai.v40i16.38378"},{"issue":"20","key":"10.1016\/j.imavis.2026.106072_b48","first-page":"21017","article-title":"Federated weakly supervised video anomaly detection with multimodal prompt","volume":"39","author":"Wang","year":"2025","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.imavis.2026.106072_b49","series-title":"2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18297","article-title":"Open-vocabulary video anomaly detection","author":"Wu","year":"2024"},{"issue":"6","key":"10.1016\/j.imavis.2026.106072_b50","first-page":"6074","article-title":"Vadclip: Adapting vision-language models for weakly supervised video anomaly detection","volume":"38","author":"Wu","year":"2024","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.imavis.2026.106072_b51","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113709","article-title":"Aligning first, then fusing: A novel weakly supervised multimodal violence detection method","author":"Jin","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.imavis.2026.106072_b52","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.127726","article-title":"Video anomaly detection: A systematic review of issues and prospects","volume":"591","author":"Samaila","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.imavis.2026.106072_b53","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105286","article-title":"Learning weakly supervised audio-visual violence detection in hyperbolic space","volume":"151","author":"Zhou","year":"2024","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.106072_b54","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2026.105959","article-title":"Mrtp: Multiscale video anomaly detection with representative text prompt","author":"Yi","year":"2026","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.106072_b55","doi-asserted-by":"crossref","first-page":"4923","DOI":"10.1109\/TIP.2024.3451935","article-title":"Learning prompt-enhanced context features for weakly-supervised video anomaly detection","volume":"33","author":"Pu","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.imavis.2026.106072_b56","series-title":"European Conference on Computer Vision","first-page":"776","author":"Tian","year":"2020"},{"key":"10.1016\/j.imavis.2026.106072_b57","doi-asserted-by":"crossref","first-page":"1","DOI":"10.2352\/EI.2023.35.14.COIMG-173","article-title":"Multimodal contrastive learning for unsupervised video representation learning","volume":"35","author":"Hiremath","year":"2023","journal-title":"Electron. Imaging"},{"key":"10.1016\/j.imavis.2026.106072_b58","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7937","article-title":"Improving cross-modal understanding in visual dialog via contrastive learning","author":"Chen","year":"2022"},{"key":"10.1016\/j.imavis.2026.106072_b59","doi-asserted-by":"crossref","unstructured":"M. Zolfaghari, Y. Zhu, P. Gehler, T. Brox, Crossclr: Cross-modal contrastive learning for multi-modal video representations, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, ICCV, 2021, pp. 1450\u20131459.","DOI":"10.1109\/ICCV48922.2021.00148"},{"key":"10.1016\/j.imavis.2026.106072_b60","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113600","article-title":"Learning opposite prompts for weakly supervised video anomaly detection","author":"Qiu","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.imavis.2026.106072_b61","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105205","article-title":"Triplet-set feature proximity learning for video anomaly detection","volume":"150","author":"Biradar","year":"2024","journal-title":"Image Vis. Comput."},{"issue":"5","key":"10.1016\/j.imavis.2026.106072_b62","doi-asserted-by":"crossref","first-page":"3197","DOI":"10.1109\/TCYB.2022.3227044","article-title":"Weakly supervised video anomaly detection via self-guided temporal discriminative transformer","volume":"54","author":"Huang","year":"2024","journal-title":"IEEE Trans. Cybern."},{"key":"10.1016\/j.imavis.2026.106072_b63","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.128698","article-title":"A lightweight video anomaly detection model with weak supervision and adaptive instance selection","volume":"613","author":"Wang","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.imavis.2026.106072_b64","article-title":"Hierarchical temporal sequence segmentation for weakly supervised video anomaly detection","author":"Abiew","year":"2025","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.imavis.2026.106072_b65","article-title":"A video anomaly detection and classification method based on cross-modal feature alignment","author":"Fu","year":"2025","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.106072_b66","doi-asserted-by":"crossref","unstructured":"A. Ghadiya, P. Kar, V. Chudasama, P. Wasnik, Cross-modal fusion and attention mechanism for weakly supervised video anomaly detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 1965\u20131974.","DOI":"10.1109\/CVPRW63382.2024.00202"},{"key":"10.1016\/j.imavis.2026.106072_b67","doi-asserted-by":"crossref","unstructured":"J. Carreira, A. Zisserman, Quo vadis, action recognition? a new model and the kinetics dataset, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 6299\u20136308.","DOI":"10.1109\/CVPR.2017.502"},{"key":"10.1016\/j.imavis.2026.106072_b68","series-title":"2017 Ieee International Conference on Acoustics, Speech and Signal Processing","first-page":"131","article-title":"Cnn architectures for large-scale audio classification","author":"Hershey","year":"2017"},{"key":"10.1016\/j.imavis.2026.106072_b69","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2022.109348","article-title":"Attention-based anomaly detection in multi-view surveillance videos","volume":"252","author":"Li","year":"2022","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.imavis.2026.106072_b70","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113530","article-title":"Anomaly detection method of surveillance video based on global-local information","volume":"317","author":"Wu","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.imavis.2026.106072_b71","doi-asserted-by":"crossref","unstructured":"W. Liu, W. Luo, D. Lian, S. Gao, Future frame prediction for anomaly detection\u2013a new baseline, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 6536\u20136545.","DOI":"10.1109\/CVPR.2018.00684"},{"key":"10.1016\/j.imavis.2026.106072_b72","series-title":"2024 IEEE\/CVF Winter Conference on Applications of Computer Vision Workshops","first-page":"132","article-title":"A multi-head approach with shuffled segments for weakly-supervised video anomaly detection","author":"AlMarri","year":"2024"},{"key":"10.1016\/j.imavis.2026.106072_b73","series-title":"ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"2260","article-title":"Violence detection in videos based on fusing visual and audio information","author":"Pang","year":"2021"},{"key":"10.1016\/j.imavis.2026.106072_b74","doi-asserted-by":"crossref","unstructured":"C. Zhang, G. Li, Y. Qi, S. Wang, L. Qing, Q. Huang, M.-H. Yang, Exploiting completeness and uncertainty of pseudo labels for weakly supervised video anomaly detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 16271\u201316280.","DOI":"10.1109\/CVPR52729.2023.01561"},{"key":"10.1016\/j.imavis.2026.106072_b75","doi-asserted-by":"crossref","unstructured":"H. Karim, K. Doshi, Y. Yilmaz, Real-time weakly supervised video anomaly detection, in: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, WACV, 2024, pp. 6848\u20136856.","DOI":"10.1109\/WACV57701.2024.00670"},{"key":"10.1016\/j.imavis.2026.106072_b76","series-title":"Proceedings of the 30th ACM International Conference on Multimedia","first-page":"6278","article-title":"Modality-aware contrastive instance learning with self-distillation for weakly-supervised audio-visual violence detection","author":"Yu","year":"2022"},{"key":"10.1016\/j.imavis.2026.106072_b77","doi-asserted-by":"crossref","unstructured":"H. Zhou, J. Yu, W. Yang, Dual memory units with uncertainty regulation for weakly supervised video anomaly detection, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 37, 2023, pp. 3769\u20133777.","DOI":"10.1609\/aaai.v37i3.25489"},{"issue":"2","key":"10.1016\/j.imavis.2026.106072_b78","first-page":"1395","article-title":"Self-training multi-sequence learning with transformer for weakly supervised video anomaly detection","volume":"36","author":"Li","year":"2022","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.imavis.2026.106072_b79","series-title":"European Conference on Computer Vision","first-page":"729","article-title":"Self-supervised sparse representation for video anomaly detection","author":"Wu","year":"2022"},{"key":"10.1016\/j.imavis.2026.106072_b80","doi-asserted-by":"crossref","unstructured":"M. Ye, X. Peng, W. Gan, W. Wu, Y. Qiao, Anopcn: Video anomaly detection via deep predictive coding network, in: Proceedings of the 27th ACM International Conference on Multimedia, 2019, pp. 1805\u20131813.","DOI":"10.1145\/3343031.3350899"},{"key":"10.1016\/j.imavis.2026.106072_b81","doi-asserted-by":"crossref","unstructured":"R. Cai, H. Zhang, W. Liu, S. Gao, Z. Hao, Appearance-motion memory consistency network for video anomaly detection, Vol. 35, 2021, pp. 938\u2013946.","DOI":"10.1609\/aaai.v35i2.16177"},{"key":"10.1016\/j.imavis.2026.106072_b82","series-title":"ICASSP 2023\u20132023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"A video anomaly detection framework based on appearance-motion semantics representation consistency","author":"Huang","year":"2023"},{"key":"10.1016\/j.imavis.2026.106072_b83","doi-asserted-by":"crossref","unstructured":"G. Yu, S. Wang, Z. Cai, E. Zhu, C. Xu, J. Yin, M. Kloft, Cloze test helps: Effective video anomaly detection via learning to complete video events, in: Proceedings of the 28th ACM International Conference on Multimedia, 2020, pp. 583\u2013591.","DOI":"10.1145\/3394171.3413973"},{"key":"10.1016\/j.imavis.2026.106072_b84","doi-asserted-by":"crossref","unstructured":"H. Park, J. Noh, B. Ham, Learning memory-guided normality for anomaly detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 14372\u201314381.","DOI":"10.1109\/CVPR42600.2020.01438"},{"key":"10.1016\/j.imavis.2026.106072_b85","doi-asserted-by":"crossref","unstructured":"C. Chen, Y. Xie, S. Lin, A. Yao, G. Jiang, W. Zhang, Y. Qu, R. Qiao, B. Ren, L. Ma, Comprehensive regularization in a bi-directional predictive network for video anomaly detection, in: Proceedings of the AAAI conference on artificial intelligence, Vol. 36, 2022, pp. 230\u2013238.","DOI":"10.1609\/aaai.v36i1.19898"},{"key":"10.1016\/j.imavis.2026.106072_b86","doi-asserted-by":"crossref","unstructured":"A. Singh, M.J. Jones, E.G. Learned-Miller, Eval: Explainable video anomaly localization, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 18717\u201318726.","DOI":"10.1109\/CVPR52729.2023.01795"},{"key":"10.1016\/j.imavis.2026.106072_b87","doi-asserted-by":"crossref","unstructured":"W. Liu, H. Chang, B. Ma, S. Shan, X. Chen, Diversity-measurable anomaly detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 12147\u201312156.","DOI":"10.1109\/CVPR52729.2023.01169"}],"container-title":["Image and Vision Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626001794?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626001794?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T06:25:02Z","timestamp":1784528702000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0262885626001794"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":87,"alternative-id":["S0262885626001794"],"URL":"https:\/\/doi.org\/10.1016\/j.imavis.2026.106072","relation":{},"ISSN":["0262-8856"],"issn-type":[{"value":"0262-8856","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"HCATRE-AVAD: Hierarchical cross-alignment and temporal relational encoding for weakly supervised audio-visual anomaly detection","name":"articletitle","label":"Article Title"},{"value":"Image and Vision Computing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.imavis.2026.106072","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"106072"}}