{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T14:13:18Z","timestamp":1767795198272,"version":"3.49.0"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T00:00:00Z","timestamp":1767744000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T00:00:00Z","timestamp":1767744000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100002322","name":"Coordena\u00e7\u00e3o de Aperfei\u00e7oamento de Pessoal de N\u00edvel Superior","doi-asserted-by":"publisher","award":["This study was financed in part by the Coordena\u00e7\u00e3o de Aperfei\u00e7oamento de Pessoal de N\u00edvel Superior-Brasil (CAPES)-Finance Code 001."],"award-info":[{"award-number":["This study was financed in part by the Coordena\u00e7\u00e3o de Aperfei\u00e7oamento de Pessoal de N\u00edvel Superior-Brasil (CAPES)-Finance Code 001."]}],"id":[{"id":"10.13039\/501100002322","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SN COMPUT. SCI."],"DOI":"10.1007\/s42979-025-04600-2","type":"journal-article","created":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T11:17:50Z","timestamp":1767784670000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Self-Supervised Learning for Extracting Interpersonal Dynamics in Video Robbery Detection"],"prefix":"10.1007","volume":"7","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0230-2865","authenticated-orcid":false,"given":"Davi Duarte","family":"de Paula","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Denis Henrique Pinheiro","family":"Salvadeo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jean Pierre Brik L\u00f3pez","family":"Vargas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,7]]},"reference":[{"key":"4600_CR1","doi-asserted-by":"crossref","unstructured":"Arnab A, Dehghani M, Heigold G, et\u00a0al. Vivit: a video vision transformer. 2021. arXiv:2103.15691.","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"4600_CR2","unstructured":"Ba JL, Kiros JR, Hinton GE. Layer normalization. 2016. arXiv preprint arXiv:1607.06450."},{"key":"4600_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2023.103656","volume":"229","author":"A Barbalau","year":"2023","unstructured":"Barbalau A, Ionescu RT, Georgescu MI, et al. Ssmtl++: revisiting self-supervised multi-task learning for video anomaly detection. Comput Vis Image Underst. 2023;229:103656.","journal-title":"Comput Vis Image Underst"},{"key":"4600_CR4","unstructured":"Bardes A, Garrido Q, Ponce J, et\u00a0al. Revisiting feature prediction for learning visual representations from video. 2024. arXiv preprint arXiv:2404.08471."},{"key":"4600_CR5","doi-asserted-by":"crossref","unstructured":"Basharat A, Gritai A, Shah M. Learning object motion patterns for anomaly detection and improved object detection. In: 2008 IEEE conference on computer vision and pattern recognition. IEEE; 2008. pp. 1\u20138.","DOI":"10.1109\/CVPR.2008.4587510"},{"key":"4600_CR6","unstructured":"Bertasius G, Wang H, Torresani L. Is space-time attention all you need for video understanding? In: ICML; 2021. p\u00a04."},{"issue":"3","key":"4600_CR7","first-page":"1","volume":"2","author":"J Carreira","year":"2017","unstructured":"Carreira J, Zisserman A. Quo vadis, action recognition. A new model and the kinetics dataset. CoRR. 2017;2(3):1 arXiv:1705.07750.","journal-title":"CoRR"},{"key":"4600_CR8","doi-asserted-by":"publisher","unstructured":"Chen Y, Liu Z, Zhang B, Fok W, Qi X, Wu Y. Mgfn: magnitude-contrastive glance-and-focus network for weakly-supervised video anomaly detection. In: Proceedings of the Thirty-Seventh AAAI Conference on Artificial Intelligence. 2023. pp. 387\u201395. https:\/\/doi.org\/10.1609\/aaai.v37i1.25112","DOI":"10.1609\/aaai.v37i1.25112"},{"key":"4600_CR9","doi-asserted-by":"publisher","unstructured":"Dalal N, Triggs B. Histograms of oriented gradients for human detection. In: 2005 IEEE computer society conference on computer vision and pattern recognition (CVPR\u201905), vol. 1. 2005. pp. 886\u201393. https:\/\/doi.org\/10.1109\/CVPR.2005.177.","DOI":"10.1109\/CVPR.2005.177"},{"key":"4600_CR10","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, et\u00a0al. An image is worth 16x16 words: transformers for image recognition at scale. 2021. arXiv:2010.11929."},{"key":"4600_CR11","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2020.102920","volume":"195","author":"Y Fan","year":"2020","unstructured":"Fan Y, Wen G, Li D, et al. Video anomaly detection and localization via gaussian mixture fully convolutional variational autoencoder. Comput Vis Image Underst. 2020;195:102920.","journal-title":"Comput Vis Image Underst"},{"key":"4600_CR12","doi-asserted-by":"publisher","unstructured":"Feichtenhofer C. X3d: expanding architectures for efficient video recognition. In: 2020 IEEE\/CVF Conference on computer vision and pattern recognition (CVPR), Seattle, WA, USA, 2020, pp. 200-210 https:\/\/doi.org\/10.1609\/aaai.v37i1.25112","DOI":"10.1609\/aaai.v37i1.25112"},{"key":"4600_CR13","doi-asserted-by":"crossref","unstructured":"Goyal R, Ebrahimi Kahou S, Michalski V, Materzynska J, Westphal S, Kim H, Haenel V, Fruend I, Yianilos P, Mueller-Freitag M, Hoppe F, Thurau C, Bax I, Memisevic R. The \u201csomething something\u201d video database for learning and evaluating visual common sense. In: Proceedings of the IEEE international conference on computer vision. 2017. pp. 5842\u201350.","DOI":"10.1109\/ICCV.2017.622"},{"key":"4600_CR14","doi-asserted-by":"crossref","unstructured":"Graham B, El-Nouby A, Touvron H, Stock P, Joulin A, Jegou H, Douze M. Levit: a vision transformer in convnet\u2019s clothing for faster inference. In: Proceedings of the IEEE\/CVF international conference on computer vision. 2021. pp. 12259\u201369.","DOI":"10.1109\/ICCV48922.2021.01204"},{"key":"4600_CR15","doi-asserted-by":"publisher","unstructured":"Gu C, Sun C, Ross D,Vondrick C, Pantofaru C, Li Y, Vijayanarasimhan S, Toderici G, Ricco S, Sukthankar R, Schmid C, Malik J. Ava: a video dataset of spatio-temporally localized atomic visual actions. In: Proceedings of the IEEE conference on computer vision and pattern recognition. 2018. pp. 6047\u201356. https:\/\/doi.org\/10.1109\/CVPR.2018.00633","DOI":"10.1109\/CVPR.2018.00633"},{"issue":"1","key":"4600_CR16","first-page":"307","volume":"13","author":"MU Gutmann","year":"2012","unstructured":"Gutmann MU, Hyv\u00e4rinen A. Noise-contrastive estimation of unnormalized statistical models, with applications to natural image statistics. J Mach Learn Res. 2012;13(1):307\u201361.","journal-title":"J Mach Learn Res"},{"issue":"11","key":"4600_CR17","doi-asserted-by":"publisher","first-page":"9389","DOI":"10.1109\/TNNLS.2022.3159538","volume":"34","author":"C Huang","year":"2023","unstructured":"Huang C, Wen J, Xu Y, et al. Self-supervised attentive generative adversarial networks for video anomaly detection. IEEE Trans Neural Netw Learn Syst. 2023;34(11):9389\u2013403. https:\/\/doi.org\/10.1109\/TNNLS.2022.3159538.","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"11","key":"4600_CR18","doi-asserted-by":"publisher","first-page":"4037","DOI":"10.1109\/TPAMI.2020.2992393","volume":"43","author":"L Jing","year":"2020","unstructured":"Jing L, Tian Y. Self-supervised visual feature learning with deep neural networks: a survey. IEEE Trans Pattern Anal Mach Intell. 2020;43(11):4037\u201358.","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"4600_CR19","unstructured":"Kay W, Carreira J, Simonyan K, et\u00a0al. The kinetics human action video dataset. 2017. arXiv preprint arXiv:1705.06950"},{"key":"4600_CR20","doi-asserted-by":"crossref","unstructured":"Li D, Qiu Z, Pan Y, Yao T, Li H, Mei T. Representing videos as discriminative sub-graphs for action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2021. pp. 3310\u201319. https:\/\/doi.ieeecomputersociety.org\/10.1109\/CVPR46437.2021.00332","DOI":"10.1109\/CVPR46437.2021.00332"},{"key":"4600_CR21","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1016\/j.neucom.2022.01.026","volume":"481","author":"N Li","year":"2022","unstructured":"Li N, Zhong JX, Shu X, et al. Weakly-supervised anomaly detection in video surveillance via graph convolutional label noise cleaning. Neurocomputing. 2022;481:154\u201367.","journal-title":"Neurocomputing"},{"key":"4600_CR22","unstructured":"Li R, Wu XJ, Xu T. Video is graph: Structured graph module for video action recognition. 2021. arXiv preprint arXiv:2110.05904."},{"key":"4600_CR23","first-page":"851","volume":"2","author":"M Loper","year":"2023","unstructured":"Loper M, Mahmood N, Romero J, et al. Smpl: a skinned multi-person linear model. Semin Graph Pap Push Bound. 2023;2:851\u201366.","journal-title":"Semin Graph Pap Push Bound"},{"key":"4600_CR24","doi-asserted-by":"publisher","unstructured":"Lowe D. Object recognition from local scale-invariant features. In: Proceedings of the seventh IEEE international conference on computer vision, vol. 2. 1999. pp. 1150\u201357. https:\/\/doi.org\/10.1109\/ICCV.1999.790410","DOI":"10.1109\/ICCV.1999.790410"},{"key":"4600_CR25","doi-asserted-by":"crossref","unstructured":"Lv H, Zhou C, Xu C, et\u00a0al. Localizing anomalies from weakly-labeled videos. CoRR. 2020. arXiv:2008.08944.","DOI":"10.1109\/TIP.2021.3072863"},{"key":"4600_CR26","doi-asserted-by":"publisher","first-page":"4505","DOI":"10.1109\/TIP.2021.3072863","volume":"30","author":"H Lv","year":"2021","unstructured":"Lv H, Zhou C, Cui Z, et al. Localizing anomalies from weakly-labeled videos. IEEE Trans Image Process. 2021;30:4505\u201315.","journal-title":"IEEE Trans Image Process"},{"issue":"12","key":"4600_CR27","doi-asserted-by":"publisher","first-page":"18693","DOI":"10.1007\/s11042-021-10570-3","volume":"80","author":"R Maqsood","year":"2021","unstructured":"Maqsood R, Bajwa UI, Saleem G, et al. Anomaly recognition from surveillance videos using 3d convolution neural network. Multimed Tools Appl. 2021;80(12):18693\u2013716.","journal-title":"Multimed Tools Appl"},{"key":"4600_CR28","unstructured":"Mathieu M, Couprie C, LeCun Y. Deep multi-scale video prediction beyond mean square error. 2015. arXiv preprint arXiv:1511.05440."},{"issue":"8","key":"4600_CR29","doi-asserted-by":"publisher","first-page":"873","DOI":"10.1109\/34.946990","volume":"23","author":"G Medioni","year":"2001","unstructured":"Medioni G, Cohen I, Br\u00e9mond F, et al. Event detection and analysis from video streams. IEEE Trans Pattern Anal Mach Intell. 2001;23(8):873\u201389.","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"24","key":"4600_CR30","doi-asserted-by":"publisher","first-page":"10016","DOI":"10.3390\/s222410016","volume":"22","author":"DD de Paula","year":"2022","unstructured":"de Paula DD, Salvadeo DH, de Araujo DM. Camnuvem: a robbery dataset for video anomaly detection. Sensors. 2022;22(24):10016.","journal-title":"Sensors"},{"key":"4600_CR31","doi-asserted-by":"publisher","unstructured":"de\u00a0Paula D, Salvadeo D, Silva L, Junior U. Self-supervised feature extraction for video surveillance anomaly detection. In: 2023 36th SIBGRAPI conference on graphics, patterns and images (SIBGRAPI). 2023. pp. 115\u2013120 https:\/\/doi.org\/10.1109\/SIBGRAPI59091.2023.10347173","DOI":"10.1109\/SIBGRAPI59091.2023.10347173"},{"key":"4600_CR32","unstructured":"Radford A, Narasimhan K, Salimans T, et\u00a0al. Improving language understanding by generative pre-training. 2018. https:\/\/cdn.openai.com\/research-covers\/language-unsupervised\/language_understanding_paper.pdf. Accessed 07 Jan 2025."},{"key":"4600_CR33","unstructured":"Rajasegaran J, Pavlakos G, Kanazawa A, et\u00a0al. Tracking people with 3d representations. CoRR. 2021. arXiv:2111.07868."},{"key":"4600_CR34","doi-asserted-by":"publisher","unstructured":"Rajasegaran J, Pavlakos G, Kanazawa A, Malik J. Tracking people by predicting 3d appearance, location and pose. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2022. pp. 2740\u201349. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00276","DOI":"10.1109\/CVPR52688.2022.00276"},{"key":"4600_CR35","doi-asserted-by":"crossref","unstructured":"Rajasegaran J, Pavlakos G, Kanazawa A, Christoph F, Jitendra M . On the benefits of 3d pose and tracking for human action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2023. pp. 640\u201349. https:\/\/doi.ieeecomputersociety.org\/10.1109\/CVPR52729.2023.00069","DOI":"10.1109\/CVPR52729.2023.00069"},{"key":"4600_CR36","unstructured":"Ryoo MS, Piergiovanni A, Arnab A, et\u00a0al. Tokenlearner: what can 8 learned tokens do for images and videos? 2021. arXiv preprint arXiv:2106.11297."},{"key":"4600_CR37","unstructured":"Srivastava N, Mansimov E, Salakhudinov R. Unsupervised learning of video representations using lstms. In: Proceedings of the 32nd International conference on machine learning - Volume 37 . PMLR; 2015. pp. 843\u201352. https:\/\/dl.acm.org\/doi\/10.5555\/3045118.3045209"},{"key":"4600_CR38","doi-asserted-by":"publisher","unstructured":"Sultani W, Chen C, Shah M. Real-world anomaly detection in surveillance videos. In: IEEE conference on computer vision and pattern recognition. 2018. pp. 6479\u201388. https:\/\/doi.org\/10.1109\/CVPR.2018.00678","DOI":"10.1109\/CVPR.2018.00678"},{"key":"4600_CR39","doi-asserted-by":"publisher","unstructured":"Tian Y, Pang G, Chen Y,Singh R, Verjans J, Carneiro G . Weakly-supervised video anomaly detection with robust temporal feature magnitude learning. In: IEEE\/CVF international conference on computer vision. 2021. pp. 4975\u201386. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00493","DOI":"10.1109\/ICCV48922.2021.00493"},{"key":"4600_CR40","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2021.103187","volume":"206","author":"M Tomei","year":"2021","unstructured":"Tomei M, Baraldi L, Calderara S, et al. Video action detection by learning graph-based spatio-temporal interactions. Comput Vis Image Underst. 2021;206:103187.","journal-title":"Comput Vis Image Underst"},{"key":"4600_CR41","doi-asserted-by":"publisher","unstructured":"Tran D, Bourdev L, Fergus R,Torresani L, Paluri M . Learning spatiotemporal features with 3d convolutional networks. In: IEEE international conference on computer vision. 2015. pp. 4489\u201397. https:\/\/doi.org\/10.1109\/ICCV.2015.510","DOI":"10.1109\/ICCV.2015.510"},{"key":"4600_CR42","unstructured":"Ultralytics. YOLOv5: a state-of-the-art real-time object detection system. 2021. https:\/\/docs.ultralytics.com. Accessed 14 Oct 2024."},{"key":"4600_CR43","unstructured":"Vaswani A, Shazeer N, Parmar N , Uszkoreit J, Jones L, Gomez A, Kaiser L, Polosukhin I . Attention is all you need. In: Advances in neural information processing systems. 2017. pp. 5998\u20136008. https:\/\/dl.acm.org\/doi\/10.5555\/3295222.3295349"},{"key":"4600_CR44","unstructured":"Villegas R, Yang J, Hong S, et\u00a0al. Decomposing motion and content for natural video sequence prediction. 2017. arXiv preprint arXiv:1706.08033."},{"key":"4600_CR45","doi-asserted-by":"crossref","unstructured":"Wang J, Song Y, Leung T, et\u00a0al. Learning fine-grained image similarity with deep ranking. In: Proceedings of the IEEE conference on computer vision and pattern recognition. 2014. pp. 1386\u201393.","DOI":"10.1109\/CVPR.2014.180"},{"key":"4600_CR46","doi-asserted-by":"crossref","unstructured":"Wang X, Gupta A. Videos as space-time region graphs. In: Proceedings of the European conference on computer vision (ECCV). 2018. pp. 399\u2013417. https:\/\/dl.acm.org\/doi\/10.1007\/978-3-030-01228-1_25","DOI":"10.1007\/978-3-030-01228-1_25"},{"key":"4600_CR47","doi-asserted-by":"publisher","unstructured":"Wang X, Girshick R, Gupta A, He K. Non-local neural networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition. 2018. pp. 7794\u2013803. https:\/\/doi.org\/10.1109\/CVPR.2018.00813","DOI":"10.1109\/CVPR.2018.00813"},{"key":"4600_CR48","doi-asserted-by":"publisher","unstructured":"Wei C, Fan H, Xie S, Wu C, Yuille A, Feichtenhofer C. Masked feature prediction for self-supervised visual pre-training. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2022. pp. 14668\u2013678. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01426","DOI":"10.1109\/CVPR52688.2022.01426"},{"issue":"10","key":"4600_CR49","doi-asserted-by":"publisher","first-page":"6753","DOI":"10.1109\/TCSVT.2022.3169894","volume":"32","author":"B Wu","year":"2022","unstructured":"Wu B, Niu G, Yu J, et al. Towards knowledge-aware video captioning via transitive visual relationship detection. IEEE Trans Circuits Syst Video Technol. 2022;32(10):6753\u201365.","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"4600_CR50","doi-asserted-by":"publisher","unstructured":"Wu CY, Krahenbuhl P. Towards long-form video understanding. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2021. pp. 1884\u201394. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00192","DOI":"10.1109\/CVPR46437.2021.00192"},{"key":"4600_CR51","doi-asserted-by":"crossref","unstructured":"Wu JC, Hsieh HY, Chen DJ, et\u00a0al. Self-supervised sparse representation for video anomaly detection. In: European conference on computer vision. Springer; 2022. pp. 729\u201345.","DOI":"10.1007\/978-3-031-19778-9_42"},{"key":"4600_CR52","doi-asserted-by":"crossref","unstructured":"Zhang T, Lu H, Li SZ. Learning semantic scene models by object classification and trajectory clustering. In: 2009 IEEE conference on computer vision and pattern recognition. IEEE; 2009. pp. 1940\u201347.","DOI":"10.1109\/CVPR.2009.5206809"},{"key":"4600_CR53","unstructured":"Zhou Y, Qu Y, Xu X, et\u00a0al. Batchnorm-based weakly supervised video anomaly detection. 2023. arXiv preprint arXiv:2311.15367."},{"key":"4600_CR54","unstructured":"Zong B, Song Q, Min MR, Bouguila, N . Deep autoencoding gaussian mixture model for unsupervised anomaly detection. In: International conference on learning representations. 2018 https:\/\/dl.acm.org\/doi\/10.1145\/3591569.3591605"}],"container-title":["SN Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-025-04600-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42979-025-04600-2","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-025-04600-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T11:17:55Z","timestamp":1767784675000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42979-025-04600-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,7]]},"references-count":54,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,1]]}},"alternative-id":["4600"],"URL":"https:\/\/doi.org\/10.1007\/s42979-025-04600-2","relation":{},"ISSN":["2661-8907"],"issn-type":[{"value":"2661-8907","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,7]]},"assertion":[{"value":"19 April 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 January 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The conducted research is not related to either human or animal use. The videos from the CamNuvem dataset used in the experiments were collected from public sources like social media.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}],"article-number":"62"}}