{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:57:43Z","timestamp":1781539063679,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T00:00:00Z","timestamp":1781481600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Beijing Natural Science Foundation","award":["L245025"],"award-info":[{"award-number":["L245025"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,16]]},"DOI":"10.1145\/3805622.3810731","type":"proceedings-article","created":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:42:57Z","timestamp":1781534577000},"page":"2768-2772","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["VAV-R1: Difficulty-Aware Multimodal Reasoning for Video Anomaly Validation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-9961-9507","authenticated-orcid":false,"given":"Jiang","family":"Liu","sequence":"first","affiliation":[{"name":"Xi\u2019an Jiaotong University, Xi'an, China and China Telecom Artificial Intelligence Technology (Beijing) Co., Ltd, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0299-6887","authenticated-orcid":false,"given":"Yuhang","family":"Liu","sequence":"additional","affiliation":[{"name":"China Telecom Artificial Intelligence Technology (Beijing) Co., Ltd, Beijing, China and Zhongke Jingyu (Beijing) Sensing Technology Co., Ltd, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4867-2494","authenticated-orcid":false,"given":"Juan","family":"Yang","sequence":"additional","affiliation":[{"name":"China Telecom Artificial Intelligence Technology (Beijing) Co., Ltd, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6161-3373","authenticated-orcid":false,"given":"Qianhao","family":"Ren","sequence":"additional","affiliation":[{"name":"China Telecom Artificial Intelligence Technology (Beijing) Co., Ltd, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7429-031X","authenticated-orcid":false,"given":"Yutong","family":"Wang","sequence":"additional","affiliation":[{"name":"Zhongke Jingyu (Beijing) Sensing Technology Co., Ltd, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1835-9271","authenticated-orcid":false,"given":"Zhongjiang","family":"He","sequence":"additional","affiliation":[{"name":"China Telecom Artificial Intelligence Technology (Beijing) Co., Ltd, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1906-2327","authenticated-orcid":false,"given":"Jingmin","family":"Xin","sequence":"additional","affiliation":[{"name":"Xi\u2019an Jiaotong University, Xi\u2019an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7917-1628","authenticated-orcid":false,"given":"Hao","family":"Sun","sequence":"additional","affiliation":[{"name":"China Telecom Artificial Intelligence Technology (Beijing) Co., Ltd, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,15]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Sanghwan Bae Jiwoo Hong Min\u00a0Young Lee Hanbyul Kim JeongYeon Nam and Donghyun Kwak. 2025. Online difficulty filtering for reasoning oriented reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.03380 (2025)."},{"key":"e_1_3_3_1_3_2","unstructured":"Rohit Bharadwaj Hanan Gani Muzammal Naseer Fahad\u00a0Shahbaz Khan and Salman Khan. 2024. Vane-bench: Video anomaly evaluation benchmark for conversational lmms. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.10326 (2024)."},{"key":"e_1_3_3_1_4_2","unstructured":"Yunkang Cao Xiaohao Xu Jiangning Zhang Yuqi Cheng Xiaonan Huang Guansong Pang and Weiming Shen. 2024. A survey on visual anomaly detection: Challenge approach and prospect. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.16402 (2024)."},{"key":"e_1_3_3_1_5_2","unstructured":"Kaituo Feng Kaixiong Gong Bohao Li Zonghao Guo Yibing Wang Tianshuo Peng Junfei Wu Xiaoying Zhang Benyou Wang and Xiangyu Yue. 2025. Video-r1: Reinforcing video reasoning in mllms. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.21776 (2025)."},{"key":"e_1_3_3_1_6_2","unstructured":"Aaron Foss Chloe Evans Sasha Mitts Koustuv Sinha Ammar Rizvi and Justine\u00a0T Kao. 2025. CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.09943 (2025)."},{"key":"e_1_3_3_1_7_2","unstructured":"Chao Huang Benfeng Wang Jie Wen Chengliang Liu Wei Wang Li Shen and Xiaochun Cao. 2025. Vad-R1: Towards Video Anomaly Reasoning via Perception-to-Cognition Chain-of-Thought. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.19877 (2025)."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Chao Huang Jie Wen Yong Xu Qiuping Jiang Jian Yang Yaowei Wang and David Zhang. 2022. Self-supervised attentive generative adversarial networks for video anomaly detection. IEEE transactions on neural networks and learning systems 34 11 (2022) 9389\u20139403.","DOI":"10.1109\/TNNLS.2022.3159538"},{"key":"e_1_3_3_1_9_2","unstructured":"Xinhao Li Ziang Yan Desen Meng Lu Dong Xiangyu Zeng Yinan He Yali Wang Yu Qiao Yi Wang and Limin Wang. 2025. Videochat-r1: Enhancing spatio-temporal perception via reinforcement fine-tuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.06958 (2025)."},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"crossref","unstructured":"Yuhang Liu Yutong Wang Yuhang Li Chaoyue Dai and Fei-Yue Wang. 2024. Sensingagent: Advancing vehicular sensing systems for spatiotemporal cognitive intelligence. IEEE Transactions on Intelligent Vehicles (2024).","DOI":"10.1109\/TIV.2024.3492543"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01333"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00775"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"crossref","unstructured":"Hui Lv Chuanwei Zhou Zhen Cui Chunyan Xu Yong Li and Jian Yang. 2021. Localizing anomalies from weakly-labeled videos. IEEE transactions on image processing 30 (2021) 4505\u20134515.","DOI":"10.1109\/TIP.2021.3072863"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754500"},{"key":"e_1_3_3_1_16_2","unstructured":"Zhihong Shao Peiyi Wang Qihao Zhu Runxin Xu Junxiao Song Xiao Bi Haowei Zhang Mingchuan Zhang YK Li Yang Wu and Daya Guo. 2024. Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.03300 (2024)."},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00678"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Jiaqi Tang Hao Lu Ruizheng Wu Xiaogang Xu Ke Ma Cheng Fang Bin Guo Jiangbo Lu Qifeng Chen and Yingcong Chen. 2024. Hawk: Learning to understand open-world video anomalies. Advances in Neural Information Processing Systems 37 (2024) 139751\u2013139785.","DOI":"10.52202\/079017-4435"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00493"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"Boyang Wan Wenhui Jiang Yuming Fang Zhiyuan Luo and Guanqun Ding. 2021. Anomaly detection in video sequences: A benchmark and computational model. IET Image Processing 15 14 (2021) 3454\u20133465.","DOI":"10.1049\/ipr2.12258"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Shaohua Wan Xiaolong Xu Tian Wang and Zonghua Gu. 2020. An intelligent video analysis method for abnormal event detection in intelligent transportation systems. IEEE Transactions on Intelligent Transportation Systems 22 7 (2020) 4487\u20134495.","DOI":"10.1109\/TITS.2020.3017505"},{"key":"e_1_3_3_1_22_2","unstructured":"Yi Wang Xinhao Li Ziang Yan Yinan He Jiashuo Yu Xiangyu Zeng Chenting Wang Changlian Ma Haian Huang Jianfei Gao Min Dou Kai Chen Wenhai Wang Yu Qiao Yali Wang and Limin Wang. 2025. Internvideo2. 5: Empowering video mllms with long and rich context modeling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.12386 (2025)."},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_20"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"crossref","unstructured":"Yu Yao Xizi Wang Mingze Xu Zelin Pu Yuchen Wang Ella Atkins and David\u00a0J Crandall. 2022. DoTA: Unsupervised detection of traffic anomaly in driving videos. IEEE transactions on pattern analysis and machine intelligence 45 1 (2022) 444\u2013459.","DOI":"10.1109\/TPAMI.2022.3150763"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00811"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01753"},{"key":"e_1_3_3_1_27_2","unstructured":"Huaxin Zhang Xiaohao Xu Xiang Wang Jialong Zuo Chuchu Han Xiaonan Huang Changxin Gao Yuehuan Wang and Nong Sang. 2024. Holmes-vad: Towards unbiased and explainable video anomaly detection via multi-modal llm. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.12235 (2024)."},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01292"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"crossref","unstructured":"Jiajie Zhang Nianyi Lin Lei Hou Ling Feng and Juanzi Li. 2025. Adaptthink: Reasoning models can learn when to think. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.13417 (2025).","DOI":"10.18653\/v1\/2025.emnlp-main.184"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Jixiao Zhang and Chunsheng Zuo. 2025. Grpo-lead: A difficulty-aware reinforcement learning approach for concise mathematical reasoning in language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.09696 (2025).","DOI":"10.18653\/v1\/2025.emnlp-main.287"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","unstructured":"Liyun Zhu Lei Wang Arjun Raj Tom Gedeon and Chen Chen. 2024. Advancing video anomaly detection: A concise review and a new dataset. Advances in Neural Information Processing Systems 37 (2024) 89943\u201389977.","DOI":"10.52202\/079017-2856"}],"event":{"name":"ICMR '26: International Conference on Multimedia Retrieval","location":"Amsterdam The Netherlands","acronym":"ICMR '26","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2026 International Conference on Multimedia Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:45:30Z","timestamp":1781538330000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805622.3810731"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,15]]},"references-count":30,"alternative-id":["10.1145\/3805622.3810731","10.1145\/3805622"],"URL":"https:\/\/doi.org\/10.1145\/3805622.3810731","relation":{},"subject":[],"published":{"date-parts":[[2026,6,15]]},"assertion":[{"value":"2026-06-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}