{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T05:08:16Z","timestamp":1777871296380,"version":"3.51.4"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100019033","name":"Key Research and Development Program of Liaoning Province","doi-asserted-by":"publisher","award":["2025080041-JH2\/1028"],"award-info":[{"award-number":["2025080041-JH2\/1028"]}],"id":[{"id":"10.13039\/501100019033","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["DUT25YG235"],"award-info":[{"award-number":["DUT25YG235"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013804","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013804","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100009965","name":"Dalian Science and Technology Bureau","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100009965","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100017683","name":"Dalian Science and Technology Innovation Fund","doi-asserted-by":"publisher","award":["2025JJ12CG028"],"award-info":[{"award-number":["2025JJ12CG028"]}],"id":[{"id":"10.13039\/501100017683","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.knosys.2026.115903","type":"journal-article","created":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T03:56:15Z","timestamp":1774929375000},"page":"115903","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Semantic consistency learning across temporal scales for weakly supervised video anomaly detection"],"prefix":"10.1016","volume":"342","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-9212-6373","authenticated-orcid":false,"given":"Guanghui","family":"Xu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7060-1133","authenticated-orcid":false,"given":"Sifan","family":"Long","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8937-1515","authenticated-orcid":false,"given":"Hongwei","family":"Ge","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-9384-6482","authenticated-orcid":false,"given":"Enxuan","family":"Gu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5143-8619","authenticated-orcid":false,"given":"Zidi","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8657-637X","authenticated-orcid":false,"given":"Zhaoqin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.115903_bib0001","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2022.109348","article-title":"Attention-based anomaly detection in multi-view surveillance videos","volume":"252","author":"Li","year":"2022","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.115903_bib0002","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2023.110986","article-title":"Stochastic video normality network for abnormal event detection in surveillance videos","volume":"280","author":"Liu","year":"2023","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.115903_bib0003","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113530","article-title":"Anomaly detection method of surveillance video based on global-local information","volume":"317","author":"Wu","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.115903_bib0004","doi-asserted-by":"crossref","first-page":"4527","DOI":"10.1109\/TIP.2022.3184250","article-title":"A self-supervised residual feature learning model for multifocus image fusion","volume":"31","author":"Wang","year":"2022","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.knosys.2026.115903_bib0005","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.111978","article-title":"VPE-WSVAD: visual prompt exemplars for weakly-supervised video anomaly detection","volume":"299","author":"Su","year":"2024","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.115903_bib0006","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"6479","article-title":"Real-world anomaly detection in surveillance videos","author":"Sultani","year":"2018"},{"key":"10.1016\/j.knosys.2026.115903_bib0007","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"12637","article-title":"Highlight what you want: weakly-supervised instance-level controllable infrared-visible image fusion","author":"Wang","year":"2025"},{"key":"10.1016\/j.knosys.2026.115903_bib0008","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.knosys.2026.115903_bib0009","first-page":"1","article-title":"Multi-text guidance is important: multi-modality image fusion via large generative vision-language model","author":"Wang","year":"2025","journal-title":"Int. J. Comput. Vis."},{"issue":"6","key":"10.1016\/j.knosys.2026.115903_bib0010","first-page":"6074","article-title":"VadCLIP: adapting vision-language models for weakly supervised video anomaly detection","volume":"38","author":"Wu","year":"2024","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.knosys.2026.115903_bib0011","series-title":"2023\u202fIEEE International Conference on Image Processing (ICIP)","first-page":"3230","article-title":"Clip-tsa: clip-assisted temporal self-attention for weakly-supervised video anomaly detection","author":"Joo","year":"2023"},{"key":"10.1016\/j.knosys.2026.115903_bib0012","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"18899","article-title":"Text prompt with normality guidance for weakly supervised video anomaly detection","author":"Yang","year":"2024"},{"issue":"10","key":"10.1016\/j.knosys.2026.115903_bib0013","doi-asserted-by":"crossref","first-page":"2529","DOI":"10.1007\/s11263-023-01806-w","article-title":"When multi-focus image fusion networks meet traditional edge-preservation technology","volume":"131","author":"Wang","year":"2023","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.115903_bib0014","doi-asserted-by":"crossref","DOI":"10.1109\/TMI.2025.3579213","article-title":"Rethinking brain tumor segmentation from the frequency domain perspective","author":"Shao","year":"2025","journal-title":"IEEE Trans. Med. Imaging"},{"key":"10.1016\/j.knosys.2026.115903_bib0015","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1237","article-title":"Graph convolutional label noise cleaner: train a plug-and-play action classifier for anomaly detection","author":"Zhong","year":"2019"},{"key":"10.1016\/j.knosys.2026.115903_bib0016","unstructured":"T.N. Kipf, M. Welling, Semi-supervised classification with graph convolutional networks, 2017. arXiv: 1609.02907."},{"key":"10.1016\/j.knosys.2026.115903_bib0017","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"14009","article-title":"Mist: multiple instance self-training framework for video anomaly detection","author":"Feng","year":"2021"},{"key":"10.1016\/j.knosys.2026.115903_bib0018","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"4975","article-title":"Weakly-supervised video anomaly detection with robust temporal feature magnitude learning","author":"Tian","year":"2021"},{"key":"10.1016\/j.knosys.2026.115903_bib0019","unstructured":"F. Yu, V. Koltun, Multi-scale context aggregation by dilated convolutions, (2015). arXiv preprint arXiv: 1511.07122."},{"key":"10.1016\/j.knosys.2026.115903_bib0020","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"1395","article-title":"Self-training multi-sequence learning with transformer for weakly supervised video anomaly detection","volume":"36","author":"Li","year":"2022"},{"key":"10.1016\/j.knosys.2026.115903_bib0021","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A.N. Gomez, L. Kaiser, I. Polosukhin, Attention is all you need, 2017. arXiv: 1706.03762."},{"key":"10.1016\/j.knosys.2026.115903_bib0022","doi-asserted-by":"crossref","unstructured":"H. Lv, Z. Yue, Q. Sun, B. Luo, Z. Cui, H. Zhang, Unbiased multiple instance learning for weakly supervised video anomaly detection, (2023). arXiv preprint arXiv: 2303.12369.","DOI":"10.1109\/CVPR52729.2023.00775"},{"key":"10.1016\/j.knosys.2026.115903_bib0023","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.124013","article-title":"Diffusion-based normality pre-training for weakly supervised video anomaly detection","volume":"251","author":"Basak","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.knosys.2026.115903_bib0024","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.124846","article-title":"TDS-Net: transformer enhanced dual-stream network for video anomaly detection","volume":"256","author":"Hussain","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.knosys.2026.115903_bib0025","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","first-page":"8574","article-title":"OE-CTST: outlier-embedded cross temporal scale transformer for weakly-supervised video anomaly detection","author":"Majhi","year":"2024"},{"key":"10.1016\/j.knosys.2026.115903_bib0026","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.128753","article-title":"Hierarchical Temporal Sequence Segmentation for weakly supervised video anomaly detection","volume":"295","author":"Abiew","year":"2026","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.knosys.2026.115903_bib0027","doi-asserted-by":"crossref","unstructured":"Y. Du, Z. Liu, J. Li, W.X. Zhao, A survey of vision-language pre-trained models, (2022). arXiv preprint arXiv: 2202.10936.","DOI":"10.24963\/ijcai.2022\/762"},{"key":"10.1016\/j.knosys.2026.115903_bib0028","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10334","article-title":"Beyond attentive tokens: incorporating token importance and diversity for efficient vision transformers","author":"Long","year":"2023"},{"issue":"3","key":"10.1016\/j.knosys.2026.115903_bib0029","doi-asserted-by":"crossref","first-page":"1258","DOI":"10.1007\/s11263-024-02243-z","article-title":"Mutual prompt leaning for vision language models","volume":"133","author":"Long","year":"2025","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.115903_bib0030","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"21959","article-title":"Task-Oriented Multi-Modal Mutual leaning for vision-language models","author":"Long","year":"2023"},{"key":"10.1016\/j.knosys.2026.115903_bib0031","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"18319","article-title":"Prompt-enhanced multiple instance learning for weakly supervised video anomaly detection","author":"Chen","year":"2024"},{"key":"10.1016\/j.knosys.2026.115903_bib0032","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"18527","article-title":"Harnessing large language models for training-free video anomaly detection","author":"Zanella","year":"2024"},{"key":"10.1016\/j.knosys.2026.115903_bib0033","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.115903_bib0034","unstructured":"H. Zhang, X. Xu, X. Wang, J. Zuo, C. Han, X. Huang, C. Gao, Y. Wang, N. Sang, Holmes-VAD: towards unbiased and explainable video anomaly detection via multi-modal LLM, 2024. arXiv: 2406.12235."},{"key":"10.1016\/j.knosys.2026.115903_bib0035","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.127857","article-title":"MMVAD: a vision\u2013language model for cross-domain video anomaly detection with contrastive learning and scale-adaptive frame segmentation","volume":"285","author":"Biswas","year":"2025","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.knosys.2026.115903_bib0036","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"16816","article-title":"Conditional prompt learning for vision-language models","author":"Zhou","year":"2022"},{"key":"10.1016\/j.knosys.2026.115903_bib0037","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXX 16","first-page":"322","article-title":"Not only look, but also listen: learning multimodal violence detection under weak supervision","author":"Wu","year":"2020"},{"key":"10.1016\/j.knosys.2026.115903_bib0038","article-title":"Support vector method for novelty detection","volume":"12","author":"Sch\u00f6lkopf","year":"1999","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.115903_bib0039","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"733","article-title":"Learning temporal regularity in video sequences","author":"Hasan","year":"2016"},{"key":"10.1016\/j.knosys.2026.115903_bib0040","series-title":"Computer Vision\u2013ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXXV","first-page":"105","article-title":"Prompting visual-language models for efficient video understanding","author":"Ju","year":"2022"},{"key":"10.1016\/j.knosys.2026.115903_bib0041","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"12137","article-title":"Look around for anomalies: weakly-supervised anomaly detection via context-motion relational learning","author":"Cho","year":"2023"},{"key":"10.1016\/j.knosys.2026.115903_bib0042","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","article-title":"Dual Memory Units with Uncertainty Regulation for Weakly Supervised Video Anomaly Detection","author":"Zhou","year":"2023"},{"key":"10.1016\/j.knosys.2026.115903_bib0043","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"387","article-title":"MGFN: magnitude-contrastive glance-and-focus network for weakly-supervised video anomaly detection","volume":"37","author":"Chen","year":"2023"},{"key":"10.1016\/j.knosys.2026.115903_bib0044","unstructured":"C. Zhang, G. Li, Y. Qi, H. Ye, L. Qing, M.-H. Yang, Q. Huang, Dynamic erasing network based on multi-scale temporal features for weakly supervised video anomaly detection, 2023. arXiv: 2312.01764."},{"key":"10.1016\/j.knosys.2026.115903_bib0045","series-title":"2024\u202fIEEE International Conference on Multimedia and Expo (ICME)","first-page":"1","article-title":"FE-VAD: high-low frequency enhanced weakly supervised video anomaly detection","author":"Pi","year":"2024"},{"key":"10.1016\/j.knosys.2026.115903_bib0046","series-title":"ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"4020","article-title":"Learning Spatio-Temporal Relations with Multi-Scale Integrated Perception for Video Anomaly Detection","author":"Ye","year":"2024"},{"key":"10.1016\/j.knosys.2026.115903_bib0047","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2024.104163","article-title":"Delving into CLIP latent space for video anomaly recognition","volume":"249","author":"Zanella","year":"2024","journal-title":"Comput. Vis. Image Understand."},{"key":"10.1016\/j.knosys.2026.115903_bib0048","doi-asserted-by":"crossref","first-page":"4923","DOI":"10.1109\/TIP.2024.3451935","article-title":"Learning prompt-enhanced context features for weakly-supervised video anomaly detection","volume":"33","author":"Pu","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.knosys.2026.115903_bib0049","series-title":"Proceedings of the 32nd ACM International Conference on Multimedia","first-page":"9301","article-title":"Weakly supervised video anomaly detection and localization with spatio-temporal prompts","author":"Wu","year":"2024"},{"key":"10.1016\/j.knosys.2026.115903_bib0050","first-page":"1674","article-title":"Weakly supervised audio-visual violence detection","author":"Wu","year":"2022","journal-title":"IEEE Trans. Multimedia"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126006295?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126006295?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:10:31Z","timestamp":1777569031000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126006295"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":50,"alternative-id":["S0950705126006295"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.115903","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Semantic consistency learning across temporal scales for weakly supervised video anomaly detection","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.115903","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"115903"}}