{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T09:08:30Z","timestamp":1765357710657,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680592","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"5865-5873","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["A Multilevel Guidance-Exploration Network and Behavior-Scene Matching Method for Human Behavior Anomaly Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-6475-0188","authenticated-orcid":false,"given":"Guoqing","family":"Yang","sequence":"first","affiliation":[{"name":"Department of Artificial Intelligence, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3411-9582","authenticated-orcid":false,"given":"Zhiming","family":"Luo","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0215-5984","authenticated-orcid":false,"given":"Jianzhe","family":"Gao","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3769-7350","authenticated-orcid":false,"given":"Yingxin","family":"Lai","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5064-1488","authenticated-orcid":false,"given":"Kun","family":"Yang","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8664-037X","authenticated-orcid":false,"given":"Yifan","family":"He","sequence":"additional","affiliation":[{"name":"Reconova Technologies Co., Ltd, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5403-9945","authenticated-orcid":false,"given":"Shaozi","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Fahad Shahbaz Khan, and Mubarak Shah.","author":"Acsintoae Andra","year":"2022","unstructured":"Andra Acsintoae, Andrei Florescu, Mariana-Iuliana Georgescu, Tudor Mare, Paul Sumedrea, Radu Tudor Ionescu, Fahad Shahbaz Khan, and Mubarak Shah. 2022. Ubnormal: New benchmark for supervised open-set video anomaly detection. In CVPR. 20143--20153."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"e_1_3_2_1_3_1","volume-title":"Beit: Bert pre-training of image transformers. arXiv preprint arXiv:2106.08254","author":"Bao Hangbo","year":"2021","unstructured":"Hangbo Bao, Li Dong, Songhao Piao, and Furu Wei. 2021. Beit: Bert pre-training of image transformers. arXiv preprint arXiv:2106.08254 (2021)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2023.103656"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Paul Bergmann Michael Fauser David Sattlegger and Carsten Steger. 2020. Uninformed students: Student-teacher anomaly detection with discriminative latent embeddings. In CVPR. 4183--4192.","DOI":"10.1109\/CVPR42600.2020.00424"},{"key":"e_1_3_2_1_6_1","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"2","year":"2021","unstructured":"Bertasius. 2021. Is space-time attention all you need for video understanding?. In ICML, Vol. 2. 4.","journal-title":"ICML"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i2.16177"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108213"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19898"},{"key":"e_1_3_2_1_10_1","volume-title":"International Journal of Computer Vision","author":"Chen Xiaokang","year":"2023","unstructured":"Xiaokang Chen, Mingyu Ding, Xiaodi Wang, Ying Xin, Shentong Mo, Yunhao Wang, Shumin Han, Ping Luo, Gang Zeng, and Jingdong Wang. 2023. Context autoencoder for self-supervised representation learning. International Journal of Computer Vision (2023), 1--16."},{"key":"e_1_3_2_1_11_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_12_1","volume-title":"Alphapose: Whole-body regional multi-person pose estimation and tracking in real-time","author":"Fang Hao-Shu","year":"2022","unstructured":"Hao-Shu Fang, Jiefeng Li, Hongyang Tang, Chao Xu, Haoyi Zhu, Yuliang Xiu, Yong-Lu Li, and Cewu Lu. 2022. Alphapose: Whole-body regional multi-person pose estimation and tracking in real-time. IEEE Transactions on Pattern Analysis and Machine Intelligence (2022)."},{"volume-title":"Mist: Multiple instance self-training framework for video anomaly detection. In CVPR. 14009--14018.","year":"2021","key":"e_1_3_2_1_13_1","unstructured":"Feng. 2021. Mist: Multiple instance self-training framework for video anomaly detection. In CVPR. 14009--14018."},{"key":"e_1_3_2_1_14_1","volume-title":"Stefano D'arrigo, Marco Aurelio Sterpa, Alessio Sampieri, and Fabio Galasso.","author":"Flaborea Alessandro","year":"2023","unstructured":"Alessandro Flaborea, Guido Maria D'Amely di Melendugno, Stefano D'arrigo, Marco Aurelio Sterpa, Alessio Sampieri, and Fabio Galasso. 2023. Contracting Skeletal Kinematic Embeddings for Anomaly Detection. arXiv preprint arXiv:2301.09489 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"Fahad Shahbaz Khan, Marius Popescu, and Mubarak Shah.","author":"Georgescu Mariana-Iuliana","year":"2021","unstructured":"Mariana-Iuliana Georgescu, Antonio Barbalau, Radu Tudor Ionescu, Fahad Shahbaz Khan, Marius Popescu, and Mubarak Shah. 2021. Anomaly detection in video via self-supervised and multi-task learning. In CVPR. 12742--12752."},{"key":"e_1_3_2_1_16_1","volume-title":"Fahad Shahbaz Khan, Marius Popescu, and Mubarak Shah.","author":"Georgescu Mariana Iuliana","year":"2022","unstructured":"Mariana Iuliana Georgescu, Radu Tudor Ionescu, Fahad Shahbaz Khan, Marius Popescu, and Mubarak Shah. 2022. A background-agnostic framework with adversarial training for abnormal event detection in video. IEEE transactions on pattern analysis and machine intelligence, Vol. 44, 9 (2022), 4505--4523."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW56347.2022.00477"},{"volume-title":"Normalizing Flows for Human Pose Anomaly Detection. arXiv preprint arXiv:2211.10946","year":"2022","key":"e_1_3_2_1_18_1","unstructured":"Hirschorn. 2022. Normalizing Flows for Human Pose Anomaly Detection. arXiv preprint arXiv:2211.10946 (2022)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10097199"},{"key":"e_1_3_2_1_20_1","volume-title":"Selective Domain-Invariant Feature for Generalizable Deepfake Detection. In ICASSP","author":"Lai Yingxin","year":"2024","unstructured":"Yingxin Lai, Guoqing Yang, Yifan He, Zhiming Luo, and Shaozi Li. 2024. Selective Domain-Invariant Feature for Generalizable Deepfake Detection. In ICASSP 2024. IEEE, 2335--2339."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00958"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i2.20028"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548232"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Wen Liu Weixin Luo Dongze Lian and Shenghua Gao. 2018. Future frame prediction for anomaly detection--a new baseline. In CVPR. 6536--6545.","DOI":"10.1109\/CVPR.2018.00684"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Zhian Liu Yongwei Nie Chengjiang Long Qing Zhang and Guiqing Li. 2021. A hybrid video anomaly detection framework via memory-augmented flow reconstruction and flow-guided frame prediction. In CVPR. 13588--13597.","DOI":"10.1109\/ICCV48922.2021.01333"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.45"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Amir Markovitz Gilad Sharir Itamar Friedman Lihi Zelnik-Manor and Shai Avidan. 2020. Graph embedded pose clustering for anomaly detection. In CVPR. 10539--10547.","DOI":"10.1109\/CVPR42600.2020.01055"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Romero Morais Vuong Le Truyen Tran Budhaditya Saha Moussa Mansour and Svetha Venkatesh. 2019. Learning regularity in skeleton trajectories for anomaly detection in videos. In CVPR. 11996--12004.","DOI":"10.1109\/CVPR.2019.01227"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Hyunjong Park Jongyoun Noh and Bumsub Ham. 2020. Learning memory-guided normality for anomaly detection. In CVPR. 14372--14381.","DOI":"10.1109\/CVPR42600.2020.01438"},{"key":"e_1_3_2_1_30_1","volume-title":"Kamal Nasrollahi, Fahad Shahbaz Khan, Thomas B Moeslund, and Mubarak Shah.","author":"Ristea Nicolae-Cuatualin","year":"2022","unstructured":"Nicolae-Cuatualin Ristea, Neelu Madan, Radu Tudor Ionescu, Kamal Nasrollahi, Fahad Shahbaz Khan, Thomas B Moeslund, and Mubarak Shah. 2022. sspcab. In CVPR. 13576--13586."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00262"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Sultani. 2018. Real-world anomaly detection in surveillance videos. In CVPR. 6479--6488.","DOI":"10.1109\/CVPR.2018.00678"},{"key":"e_1_3_2_1_33_1","volume-title":"Long-Short Temporal Co-Teaching for Weakly Supervised Video Anomaly Detection. arXiv preprint arXiv:2303.18044","author":"Sun Shengyang","year":"2023","unstructured":"Shengyang Sun and Gong. 2023. Long-Short Temporal Co-Teaching for Weakly Supervised Video Anomaly Detection. arXiv preprint arXiv:2303.18044 (2023)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Shengyang Sun and Xiaojin Gong. 2023. Hierarchical Semantic Contrast for Scene-aware Video Anomaly Detection. In CVPR. 22846--22856.","DOI":"10.1109\/CVPR52729.2023.02188"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Yu Tian Guansong Pang Yuanhong Chen Rajvinder Singh Johan W Verjans and Gustavo Carneiro. 2021. Weakly-supervised video anomaly detection with robust temporal feature magnitude learning. In ICCV. 4975--4986.","DOI":"10.1109\/ICCV48922.2021.00493"},{"key":"e_1_3_2_1_36_1","volume-title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems","author":"Tong Zhan","year":"2022","unstructured":"Zhan Tong, Yibing Song, Jue Wang, and Limin Wang. 2022. Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems, Vol. 35 (2022), 10078--10093."},{"volume-title":"Video anomaly detection by solving decoupled spatio-temporal jigsaw puzzles","author":"Wang Guodong","key":"e_1_3_2_1_37_1","unstructured":"Guodong Wang and Wang. 2022. Video anomaly detection by solving decoupled spatio-temporal jigsaw puzzles. In ECCV. Springer, 494--511."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00611"},{"key":"e_1_3_2_1_39_1","volume-title":"Making Reconstruction-based Method Great Again for Video Anomaly Detection. In 2022 IEEE International Conference on Data Mining (ICDM). IEEE, 1215--1220","author":"Wang Yizhou","year":"2022","unstructured":"Yizhou Wang, Can Qin, Yue Bai, Yi Xu, Xu Ma, and Yun Fu. 2022. Making Reconstruction-based Method Great Again for Video Anomaly Detection. In 2022 IEEE International Conference on Data Mining (ICDM). IEEE, 1215--1220."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20056-4_20"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3062192"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448084"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"e_1_3_2_1_44_1","volume-title":"Pose Flow: Efficient Online Pose Tracking. In BMVC.","author":"Xiu Yuliang","year":"2018","unstructured":"Yuliang Xiu, Jiefeng Li, Haoyu Wang, Yinghong Fang, and Cewu Lu. 2018. Pose Flow: Efficient Online Pose Tracking. In BMVC."},{"volume-title":"Reconstructed Student-Teacher and Discriminative Networks for Anomaly Detection. In 2022 IROS","author":"Yamada Shinji","key":"e_1_3_2_1_45_1","unstructured":"Shinji Yamada, Satoshi Kamiya, and Kazuhiro Hotta. 2022. Reconstructed Student-Teacher and Discriminative Networks for Anomaly Detection. In 2022 IROS. IEEE, 2725--2732."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"Zhiwei Yang Jing Liu Zhaoyang Wu Peng Wu and Xiaotao Liu. 2023. Video Event Restoration Based on Keyframes for Video Anomaly Detection. In CVPR. 14592--14601.","DOI":"10.1109\/CVPR52729.2023.01402"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"M Zaigham Zaheer Arif Mahmood M Haris Khan Mattia Segu Fisher Yu and Seung-Ik Lee. 2022. Generative cooperative learning for unsupervised video anomaly detection. In CVPR. 14744--14754.","DOI":"10.1109\/CVPR52688.2022.01433"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611774"},{"key":"e_1_3_2_1_50_1","unstructured":"Xinyu Zhang Jiahui Chen Junkun Yuan Qiang Chen Jian Wang Xiaodi Wang Shumin Han Xiaokang Chen Jimin Pi Kun Yao et al. 2022. Cae v2: Context autoencoder with clip target. arXiv preprint arXiv:2211.09799 (2022)."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Melbourne VIC Australia","acronym":"MM '24"},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680592","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680592","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:56Z","timestamp":1750295876000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680592"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":50,"alternative-id":["10.1145\/3664647.3680592","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680592","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}