{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T13:55:53Z","timestamp":1777125353365,"version":"3.51.4"},"reference-count":56,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100014103","name":"Key Technology Research and Development Program of Shandong Province","doi-asserted-by":"publisher","award":["2021CXGC010506"],"award-info":[{"award-number":["2021CXGC010506"]}],"id":[{"id":"10.13039\/100014103","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100014103","name":"Key Technology Research and Development Program of Shandong Province","doi-asserted-by":"publisher","award":["2021SFGC0104"],"award-info":[{"award-number":["2021SFGC0104"]}],"id":[{"id":"10.13039\/100014103","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100012905","name":"Department of Science and Technology of Shandong Province","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100012905","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Shandong Province Natural Science Foundation","doi-asserted-by":"publisher","award":["ZR2020LZH008"],"award-info":[{"award-number":["ZR2020LZH008"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Shandong Province Natural Science Foundation","doi-asserted-by":"publisher","award":["ZR2022LZH003"],"award-info":[{"award-number":["ZR2022LZH003"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Shandong Province Natural Science Foundation","doi-asserted-by":"publisher","award":["ZR2021MF118"],"award-info":[{"award-number":["ZR2021MF118"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Signal Processing"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.sigpro.2026.110638","type":"journal-article","created":{"date-parts":[[2026,4,11]],"date-time":"2026-04-11T08:20:59Z","timestamp":1775895659000},"page":"110638","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Self-supervised multi-label spatiotemporal proxy task-based video anomaly detection"],"prefix":"10.1016","volume":"246","author":[{"given":"Sinan","family":"Jia","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiangwei","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lifeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenshuo","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"issue":"10","key":"10.1016\/j.sigpro.2026.110638_bib0001","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3729222","article-title":"Networking systems for video anomaly detection: a tutorial and survey","volume":"57","author":"Liu","year":"2025","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.sigpro.2026.110638_bib0002","doi-asserted-by":"crossref","first-page":"471","DOI":"10.1016\/j.procs.2022.01.057","article-title":"Visual anomaly detection for images: a systematic survey","volume":"199","author":"Yang","year":"2022","journal-title":"Procedia Comput. Sci."},{"issue":"2","key":"10.1016\/j.sigpro.2026.110638_bib0003","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3439950","article-title":"Deep learning for anomaly detection: a review","volume":"54","author":"Pang","year":"2021","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"10.1016\/j.sigpro.2026.110638_bib0004","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"13588","article-title":"A hybrid video anomaly detection framework via memory-augmented flow reconstruction and flow-guided frame prediction","author":"Liu","year":"2021"},{"key":"10.1016\/j.sigpro.2026.110638_bib0005","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"14372","article-title":"Learning memory-guided normality for anomaly detection","author":"Park","year":"2020"},{"key":"10.1016\/j.sigpro.2026.110638_bib0006","article-title":"A semi-supervised deep learning based video anomaly detection framework using RGB-D for surveillance of real-world critical environments","volume":"40","author":"Khaire","year":"2022","journal-title":"Forensic Sci. Int. Digit. Invest."},{"key":"10.1016\/j.sigpro.2026.110638_bib0007","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"20392","article-title":"A new comprehensive benchmark for semi-supervised video anomaly detection and anticipation","author":"Cao","year":"2023"},{"key":"10.1016\/j.sigpro.2026.110638_bib0008","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110898","article-title":"Semantic-driven dual consistency learning for weakly supervised video anomaly detection","volume":"157","author":"Su","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.sigpro.2026.110638_bib0009","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103256","article-title":"Federated weakly-supervised video anomaly detection with mixture of local-to-global experts","author":"Su","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.sigpro.2026.110638_bib0010","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.111978","article-title":"VPE-WSVAD: visual prompt exemplars for weakly-supervised video anomaly detection","volume":"299","author":"Su","year":"2024","journal-title":"Knowl. Based Syst."},{"key":"10.1016\/j.sigpro.2026.110638_bib0011","doi-asserted-by":"crossref","unstructured":"H. Shen, L. Shi, W. Xu, Y. Cen, L. Zhang, G. An, Patch spatio-temporal relation prediction for video anomaly detection,arXiv preprint arXiv: 2403.19111, 2024.","DOI":"10.2139\/ssrn.4892938"},{"key":"10.1016\/j.sigpro.2026.110638_bib0012","series-title":"2024 IEEE International Conference on Multimedia and Expo (ICME)","first-page":"1","article-title":"Video anomaly detection via self-supervised learning with frame interval and rotation prediction","author":"Jia","year":"2024"},{"key":"10.1016\/j.sigpro.2026.110638_bib0013","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112198","article-title":"Self-supervised learning video anomaly detection based on time interval prediction and noise classification","volume":"171","author":"Liu","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.sigpro.2026.110638_bib0014","unstructured":"J. Liu, Z. Ma, Z. Wang, C. Zou, J. Ren, Z. Wang, L. Song, B. Hu, Y. Liu, V. Leung, A survey on diffusion models for anomaly detection, 2025. arXiv preprint arXiv: 2501.11430."},{"key":"10.1016\/j.sigpro.2026.110638_bib0015","unstructured":"Y. Liu, J. Liu, C. Li, R. Xi, W. Li, L. Cao, J. Wang, L.T. Yang, J. Yuan, W. Zhou, Anomaly detection and generation with diffusion models: A survey, 2025, arXiv preprint arXiv: 2506.09368."},{"key":"10.1016\/j.sigpro.2026.110638_bib0016","first-page":"1","article-title":"Deep learning for video anomaly detection: a review","volume":"37","author":"Wu","year":"2026","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.sigpro.2026.110638_bib0017","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7799","article-title":"Vision-language pseudo-labels for single-positive multi-label learning","author":"Xing","year":"2024"},{"key":"10.1016\/j.sigpro.2026.110638_bib0018","series-title":"2025 International Conference on Innovations in Intelligent Systems: Advancements in Computing, Communication, and Cybersecurity (ISAC3)","first-page":"1","article-title":"MSPL-VAD: multi-stage pseudo-labeling framework for video anomaly detection","author":"Senapati","year":"2025"},{"key":"10.1016\/j.sigpro.2026.110638_bib0019","article-title":"Three-dimensional view relationship-based context-aware emotion recognition","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.sigpro.2026.110638_bib0020","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"1373","article-title":"CVNet: contour vibration network for building extraction","author":"Xu","year":"2022"},{"key":"10.1016\/j.sigpro.2026.110638_bib0021","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2025.104379","article-title":"Anomaly-aware self-supervised feature learning for weakly supervised video anomaly detection","volume":"257","author":"Yang","year":"2025","journal-title":"Comput. Vision Image Understanding"},{"key":"10.1016\/j.sigpro.2026.110638_bib0022","series-title":"CVPR 2011","first-page":"3313","article-title":"Online detection of unusual events in videos via dynamic sparse coding","author":"Zhao","year":"2011"},{"issue":"7","key":"10.1016\/j.sigpro.2026.110638_bib0023","doi-asserted-by":"crossref","first-page":"1885","DOI":"10.1007\/s11760-022-02148-9","article-title":"Video anomaly detection based on 3D convolutional auto-encoder","volume":"16","author":"Hu","year":"2022","journal-title":"Signal Image Video Process."},{"issue":"1","key":"10.1016\/j.sigpro.2026.110638_bib0024","doi-asserted-by":"crossref","first-page":"215","DOI":"10.1007\/s11760-020-01740-1","article-title":"Residual spatiotemporal autoencoder for unsupervised video anomaly detection","volume":"15","author":"Deepak","year":"2021","journal-title":"Signal Image Video Process."},{"key":"10.1016\/j.sigpro.2026.110638_bib0025","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2024.106597","article-title":"An attribution graph-based interpretable method for CNNs","volume":"179","author":"Zheng","year":"2024","journal-title":"Neural Netw."},{"key":"10.1016\/j.sigpro.2026.110638_bib0026","doi-asserted-by":"crossref","first-page":"172425","DOI":"10.1109\/ACCESS.2019.2954540","article-title":"Spatio-temporal unity networking for video anomaly detection","volume":"7","author":"Li","year":"2019","journal-title":"IEEE Access"},{"key":"10.1016\/j.sigpro.2026.110638_bib0027","series-title":"International Symposium on Visual Computing","first-page":"472","article-title":"Future video prediction from a single frame for video anomaly detection","author":"Baradaran","year":"2023"},{"key":"10.1016\/j.sigpro.2026.110638_bib0028","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"14592","article-title":"Video event restoration based on keyframes for video anomaly detection","author":"Yang","year":"2023"},{"issue":"6","key":"10.1016\/j.sigpro.2026.110638_bib0029","doi-asserted-by":"crossref","first-page":"2301","DOI":"10.1109\/TNNLS.2021.3083152","article-title":"Robust unsupervised video anomaly detection by multipath frame prediction","volume":"33","author":"Wang","year":"2021","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.sigpro.2026.110638_bib0030","doi-asserted-by":"crossref","first-page":"4505","DOI":"10.1109\/TIP.2021.3072863","article-title":"Localizing anomalies from weakly-labeled videos","volume":"30","author":"Lv","year":"2021","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.sigpro.2026.110638_bib0031","series-title":"European Conference on Computer Vision","first-page":"69","article-title":"Unsupervised learning of visual representations by solving jigsaw puzzles","author":"Noroozi","year":"2016"},{"key":"10.1016\/j.sigpro.2026.110638_bib0032","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"1422","article-title":"Unsupervised visual representation learning by context prediction","author":"Doersch","year":"2015"},{"key":"10.1016\/j.sigpro.2026.110638_bib0033","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"803","article-title":"Temporal relational reasoning in videos","author":"Zhou","year":"2018"},{"issue":"11","key":"10.1016\/j.sigpro.2026.110638_bib0034","doi-asserted-by":"crossref","first-page":"2740","DOI":"10.1109\/TPAMI.2018.2868668","article-title":"Temporal segment networks for action recognition in videos","volume":"41","author":"Wang","year":"2018","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.sigpro.2026.110638_bib0035","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"12742","article-title":"Anomaly detection in video via self-supervised and multi-task learning","author":"Georgescu","year":"2021"},{"key":"10.1016\/j.sigpro.2026.110638_bib0036","series-title":"European Conference on Computer Vision","first-page":"494","article-title":"Video anomaly detection by solving decoupled spatio-temporal jigsaw puzzles","author":"Wang","year":"2022"},{"key":"10.1016\/j.sigpro.2026.110638_bib0037","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2023.103656","article-title":"SSMTL++: revisiting self-supervised multi-task learning for video anomaly detection","volume":"229","author":"Barbalau","year":"2023","journal-title":"Comput. Vision Image Understanding"},{"key":"10.1016\/j.sigpro.2026.110638_bib0038","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111021","article-title":"Video anomaly detection via self-supervised and spatio-temporal proxy tasks learning","volume":"158","author":"Yang","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.sigpro.2026.110638_bib0039","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.129576","article-title":"Spatio-temporal graph-based self-labeling for video anomaly detection","volume":"627","author":"Xing","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.sigpro.2026.110638_bib0040","unstructured":"J. Redmon, A. Farhadi, YOLOv3: an incremental improvement, arXiv preprint arXiv: 1804.02767(2018)."},{"key":"10.1016\/j.sigpro.2026.110638_bib0041","unstructured":"S. Gidaris, P. Singh, N. Komodakis, Unsupervised representation learning by predicting image rotations, arXiv preprint arXiv: 1803.07728(2018)."},{"key":"10.1016\/j.sigpro.2026.110638_bib0042","series-title":"2021 5th Asian Conference on Artificial Intelligence Technology (ACAIT)","first-page":"526","article-title":"Temporal-aware self-supervised learning for unsupervised video anomaly detection","author":"Shang","year":"2021"},{"issue":"1","key":"10.1016\/j.sigpro.2026.110638_bib0043","first-page":"18","article-title":"Anomaly detection and localization in crowded scenes","volume":"36","author":"Li","year":"2013","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.sigpro.2026.110638_bib0044","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"2720","article-title":"Abnormal event detection at 150 fps in matlab","author":"Lu","year":"2013"},{"key":"10.1016\/j.sigpro.2026.110638_bib0045","series-title":"Proceedings of the 29th ACM International Conference on Multimedia","first-page":"5546","article-title":"Convolutional transformer based dual discriminator generative adversarial networks for video anomaly detection","author":"Feng","year":"2021"},{"key":"10.1016\/j.sigpro.2026.110638_bib0046","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"14744","article-title":"Generative cooperative learning for unsupervised video anomaly detection","author":"Zaheer","year":"2022"},{"key":"10.1016\/j.sigpro.2026.110638_bib0047","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7842","article-title":"Object-centric auto-encoders and dummy anomalies for abnormal event detection in video","author":"Ionescu","year":"2019"},{"key":"10.1016\/j.sigpro.2026.110638_bib0048","series-title":"Proceedings of the 28th ACM International Conference on Multimedia","first-page":"583","article-title":"Cloze test helps: effective video anomaly detection via learning to complete video events","author":"Yu","year":"2020"},{"key":"10.1016\/j.sigpro.2026.110638_bib0049","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15425","article-title":"Learning normal dynamics in videos with meta prototype network","author":"Lv","year":"2021"},{"key":"10.1016\/j.sigpro.2026.110638_bib0050","series-title":"European Conference on Computer Vision","first-page":"404","article-title":"Dynamic local aggregation network with adaptive clusterer for anomaly detection","author":"Yang","year":"2022"},{"issue":"12","key":"10.1016\/j.sigpro.2026.110638_bib0051","doi-asserted-by":"crossref","first-page":"8285","DOI":"10.1109\/TCSVT.2022.3190539","article-title":"Bidirectional spatio-temporal feature learning with multiscale evaluation for video anomaly detection","volume":"32","author":"Zhong","year":"2022","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.sigpro.2026.110638_bib0052","doi-asserted-by":"crossref","first-page":"159","DOI":"10.1016\/j.comcom.2024.01.004","article-title":"Memory-enhanced appearance-motion consistency framework for video anomaly detection","volume":"216","author":"Ning","year":"2024","journal-title":"Comput. Commun."},{"key":"10.1016\/j.sigpro.2026.110638_bib0053","series-title":"2024 14th International Conference on System Engineering and Technology (ICSET)","first-page":"148","article-title":"Self-supervised multi-task learning using StridedConv3D-VideoSwin for video anomaly detection","author":"Kesuma","year":"2024"},{"key":"10.1016\/j.sigpro.2026.110638_bib0054","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2024.103946","article-title":"Enhancing video anomaly detection with learnable memory network: a new approach to memory-based auto-encoders","volume":"241","author":"Wang","year":"2024","journal-title":"Comput. Vision Image Understanding"},{"issue":"5","key":"10.1016\/j.sigpro.2026.110638_bib0055","doi-asserted-by":"crossref","first-page":"3003","DOI":"10.1007\/s00371-024-03584-z","article-title":"Video anomaly detection with both normal and anomaly memory modules","volume":"41","author":"Zhang","year":"2025","journal-title":"Vis. Comput."},{"key":"10.1016\/j.sigpro.2026.110638_bib0056","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.125581","article-title":"Rethinking prediction-based video anomaly detection from local\u2013global normality perspective","volume":"262","author":"Zhao","year":"2025","journal-title":"Expert Syst. Appl."}],"container-title":["Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0165168426001520?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0165168426001520?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T13:31:34Z","timestamp":1777123894000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0165168426001520"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":56,"alternative-id":["S0165168426001520"],"URL":"https:\/\/doi.org\/10.1016\/j.sigpro.2026.110638","relation":{},"ISSN":["0165-1684"],"issn-type":[{"value":"0165-1684","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Self-supervised multi-label spatiotemporal proxy task-based video anomaly detection","name":"articletitle","label":"Article Title"},{"value":"Signal Processing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.sigpro.2026.110638","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"110638"}}