{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:20:56Z","timestamp":1783153256902,"version":"3.54.6"},"reference-count":65,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["RS-2024-00352566"],"award-info":[{"award-number":["RS-2024-00352566"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neucom.2026.134178","type":"journal-article","created":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T16:28:53Z","timestamp":1780417733000},"page":"134178","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["DTVAD: Dual-branch aligned temporal-aware framework for open-vocabulary video anomaly detection"],"prefix":"10.1016","volume":"697","author":[{"given":"Jeongyeon","family":"Kim","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3262-1798","authenticated-orcid":false,"given":"Chaeyoung","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Taeyeon","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongseon","family":"Lee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7450-6600","authenticated-orcid":false,"given":"Hanul","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.134178_bib0005","series-title":"Proc. NeurIPS","first-page":"582","article-title":"Support vector method for novelty detection","author":"Sch\u00f6lkopf","year":"1999"},{"key":"10.1016\/j.neucom.2026.134178_bib0010","series-title":"Proc. ICML","first-page":"4390","article-title":"Deep one-class classification","author":"Ruff","year":"2018"},{"key":"10.1016\/j.neucom.2026.134178_bib0015","series-title":"Proc. IEEE CVPR","first-page":"2898","article-title":"Ocgan: one-class novelty detection using GANs with constrained latent representations","author":"Perera","year":"2019"},{"issue":"4","key":"10.1016\/j.neucom.2026.134178_bib0020","first-page":"4167","article-title":"Adversarially robust one-class novelty detection","volume":"45","author":"Lo","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.134178_bib0025","series-title":"Proc. IEEE CVPR","first-page":"3449","article-title":"Sparse reconstruction cost for abnormal event detection","author":"Cong","year":"2011"},{"key":"10.1016\/j.neucom.2026.134178_bib0030","series-title":"Proc. IEEE CVPR","first-page":"733","article-title":"Learning temporal regularity in video sequences","author":"Hasan","year":"2016"},{"key":"10.1016\/j.neucom.2026.134178_bib0035","series-title":"Proc. IEEE ICCV","first-page":"1705","article-title":"Memorizing normality to detect anomaly: memory-augmented deep autoencoder for unsupervised anomaly detection","author":"Gong","year":"2019"},{"key":"10.1016\/j.neucom.2026.134178_bib0040","series-title":"Proc. IEEE CVPR","first-page":"14372","article-title":"Learning memory-guided normality for anomaly detection","author":"Park","year":"2020"},{"key":"10.1016\/j.neucom.2026.134178_bib0045","author":"Ravanbakhsh"},{"key":"10.1016\/j.neucom.2026.134178_bib0050","series-title":"Proc. IEEE CVPR","first-page":"6536","article-title":"Future frame prediction for anomaly detection \u2013 a new baseline","author":"Liu","year":"2018"},{"key":"10.1016\/j.neucom.2026.134178_bib0055","series-title":"Proc. IEEE ICASSP","first-page":"1323","article-title":"Stan: spatio-temporal adversarial networks for abnormal event detection","author":"Lee","year":"2018"},{"key":"10.1016\/j.neucom.2026.134178_bib0060","series-title":"Proc. IEEE CVPR","first-page":"6479","article-title":"Real-world anomaly detection in surveillance videos","author":"Sultani","year":"2018"},{"key":"10.1016\/j.neucom.2026.134178_bib0065","series-title":"Proc. ECCV","first-page":"322","article-title":"Not only look, but also listen: learning multimodal violence detection under weak supervision","author":"Wu","year":"2020"},{"key":"10.1016\/j.neucom.2026.134178_bib0070","series-title":"Proc. IEEE ICCV","first-page":"4975","article-title":"Weakly-supervised video anomaly detection with robust temporal feature magnitude learning","author":"Tian","year":"2021"},{"key":"10.1016\/j.neucom.2026.134178_bib0075","series-title":"Proc. IEEE CVPR","first-page":"3212","article-title":"Bayesian nonparametric submodular video partition for robust anomaly detection","author":"Sapkota","year":"2022"},{"key":"10.1016\/j.neucom.2026.134178_bib0080","series-title":"Proc. AAAI","first-page":"387","article-title":"Mgfn: magnitude-contrastive glance-and-focus network for weakly-supervised video anomaly detection","author":"Chen","year":"2023"},{"key":"10.1016\/j.neucom.2026.134178_bib0085","series-title":"Proc. IEEE CVPR","first-page":"12137","article-title":"Look around for anomalies: weakly-supervised anomaly detection via context-motion relational learning","author":"Cho","year":"2023"},{"key":"10.1016\/j.neucom.2026.134178_bib0090","series-title":"Proc. AAAI","first-page":"3769","article-title":"Dual memory units with uncertainty regulation for weakly supervised video anomaly detection","author":"Zhou","year":"2023"},{"key":"10.1016\/j.neucom.2026.134178_bib0095","series-title":"Proc. IEEE CVPR","first-page":"5549","article-title":"Tevad: improved video anomaly detection with captions","author":"Chen","year":"2023"},{"key":"10.1016\/j.neucom.2026.134178_bib0100","series-title":"Proc. IEEE WACV","first-page":"202","article-title":"Overlooked video classification in weakly supervised video anomaly detection","author":"Tan","year":"2024"},{"key":"10.1016\/j.neucom.2026.134178_bib0105","series-title":"Proc. AAAI","first-page":"6074","article-title":"Vadclip: adapting vision-language models for weakly supervised video anomaly detection","author":"Wu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134178_bib0110","doi-asserted-by":"crossref","first-page":"4923","DOI":"10.1109\/TIP.2024.3451935","article-title":"Learning prompt-enhanced context features for weakly-supervised video anomaly detection","volume":"33","author":"Pu","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.neucom.2026.134178_bib0115","series-title":"Proceedings of the 32nd ACM International Conference on Multimedia","first-page":"9301","article-title":"Weakly supervised video anomaly detection and localization with spatio-temporal prompts","author":"Wu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134178_bib0120","series-title":"Proc. IEEE CVPR","first-page":"18899","article-title":"Text prompt with normality guidance for weakly supervised video anomaly detection","author":"Yang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134178_bib0125","doi-asserted-by":"crossref","first-page":"5907","DOI":"10.1109\/TIP.2024.3477351","article-title":"Injecting text clues for improving anomalous event detection from weakly labeled videos","volume":"33","author":"Liu","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.neucom.2026.134178_bib0130","doi-asserted-by":"crossref","first-page":"3132","DOI":"10.1109\/TMM.2025.3557682","article-title":"Multimodal evidential learning for open-world weakly-supervised video anomaly detection","volume":"27","author":"Huang","year":"2025","journal-title":"IEEE Trans. Multimedia"},{"key":"10.1016\/j.neucom.2026.134178_bib0135","series-title":"Proc. AAAI","first-page":"21017","article-title":"Federated weakly supervised video anomaly detection with multimodal prompt","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134178_bib0140","series-title":"Proc. IEEE CVPR","first-page":"18527","article-title":"Harnessing large language models for training-free video anomaly detection","author":"Zanella","year":"2024"},{"key":"10.1016\/j.neucom.2026.134178_bib0145","series-title":"Proc. IEEE ICIP","first-page":"3230","article-title":"Clip-tsa: clip-assisted temporal self-attention for weakly-supervised video anomaly detection","author":"Joo","year":"2023"},{"key":"10.1016\/j.neucom.2026.134178_bib0150","series-title":"Proc. NeurIPS","article-title":"A framework for multiple-instance learning","volume":"vol. 10","author":"Maron","year":"1997"},{"issue":"1","key":"10.1016\/j.neucom.2026.134178_bib0155","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1016\/S0004-3702(96)00034-3","article-title":"Solving the multiple instance problem with axis-parallel rectangles","volume":"89","author":"Dietterich","year":"1997","journal-title":"Artif. Intell."},{"key":"10.1016\/j.neucom.2026.134178_bib0160","series-title":"Proc. IEEE CVPR","first-page":"18297","article-title":"Open-vocabulary video anomaly detection","author":"Wu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134178_bib0165","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"29203","article-title":"Anomize: better open vocabulary video anomaly detection","author":"Li","year":"2025"},{"issue":"6","key":"10.1016\/j.neucom.2026.134178_bib0170","doi-asserted-by":"crossref","first-page":"5925","DOI":"10.1109\/TCSVT.2025.3528108","article-title":"PLOVAD: prompting vision-language models for open vocabulary video anomaly detection","volume":"35","author":"Xu","year":"2025","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.neucom.2026.134178_bib0175","series-title":"Proc. ICML","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.neucom.2026.134178_bib0180","series-title":"Proc. ICML","first-page":"4904","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","author":"Jia","year":"2021"},{"key":"10.1016\/j.neucom.2026.134178_bib0185","series-title":"Proc. ICLR","first-page":"12888","article-title":"BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.neucom.2026.134178_bib0190","series-title":"Proc. IEEE ICCV","first-page":"11941","article-title":"Sigmoid loss for language image pre-training","author":"Zhai","year":"2023"},{"key":"10.1016\/j.neucom.2026.134178_bib0195","series-title":"Proc. ICLR","article-title":"Graph attention networks","author":"Veli\u010dkovi\u0107","year":"2018"},{"issue":"10","key":"10.1016\/j.neucom.2026.134178_bib0200","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3729222","article-title":"Networking systems for video anomaly detection: a tutorial and survey","volume":"57","author":"Liu","year":"2025","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.neucom.2026.134178_bib0205","series-title":"Proc. IEEE CVPR","first-page":"1587","article-title":"A survey of state of the art large vision language models: benchmark evaluations and challenges","author":"Li","year":"2025"},{"key":"10.1016\/j.neucom.2026.134178_bib0210","series-title":"Proc. ECCV","first-page":"395","article-title":"Towards open set video anomaly detection","author":"Zhu","year":"2022"},{"issue":"9","key":"10.1016\/j.neucom.2026.134178_bib0215","doi-asserted-by":"crossref","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","article-title":"Learning to prompt for vision-language models","volume":"130","author":"Zhou","year":"2022","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.134178_bib0220","series-title":"Proc. IEEE CVPR","first-page":"16816","article-title":"Conditional prompt learning for vision-language models","author":"Zhou","year":"2022"},{"key":"10.1016\/j.neucom.2026.134178_bib0225","series-title":"Proc. ICLR","article-title":"Open-vocabulary object detection via vision and language knowledge distillation","author":"Gu","year":"2022"},{"key":"10.1016\/j.neucom.2026.134178_bib0230","series-title":"Proc. IEEE CVPR","first-page":"14393","article-title":"Open-vocabulary object detection using captions","author":"Zareian","year":"2021"},{"key":"10.1016\/j.neucom.2026.134178_bib0235","first-page":"9125","article-title":"Detclip: dictionary-enriched visual-concept paralleled pre-training for open-world detection","volume":"35","author":"Yao","year":"2022","journal-title":"Proc. NeurIPS"},{"key":"10.1016\/j.neucom.2026.134178_bib0240","series-title":"Proc. IEEE CVPR","first-page":"18082","article-title":"Denseclip: language-guided dense prediction with context-aware prompting","author":"Rao","year":"2022"},{"key":"10.1016\/j.neucom.2026.134178_bib0245","series-title":"Proc. IEEE CVPR","first-page":"4113","article-title":"Cat-seg: cost aggregation for open-vocabulary semantic segmentation","author":"Cho","year":"2024"},{"key":"10.1016\/j.neucom.2026.134178_bib0250","series-title":"Proc. IEEE CVPR","first-page":"3426","article-title":"Sed: a simple encoder-decoder for open-vocabulary semantic segmentation","author":"Xie","year":"2024"},{"key":"10.1016\/j.neucom.2026.134178_bib0255","series-title":"Proc. IEEE CVPR","first-page":"25346","article-title":"Dpseg: dual-prompt cost volume learning for open-vocabulary semantic segmentation","author":"Zhao","year":"2025"},{"key":"10.1016\/j.neucom.2026.134178_bib0260","author":"Gandhamal"},{"key":"10.1016\/j.neucom.2026.134178_bib0265","series-title":"Proc. ECCV","first-page":"696","article-title":"Extract free dense labels from clip","author":"Zhou","year":"2022"},{"key":"10.1016\/j.neucom.2026.134178_bib0270","first-page":"32215","article-title":"Convolutions die hard: open-vocabulary segmentation with single frozen convolutional clip","volume":"36","author":"Yu","year":"2023","journal-title":"Proc. NeurIPS"},{"key":"10.1016\/j.neucom.2026.134178_bib0275","series-title":"Proc. IEEE CVPR","first-page":"19113","article-title":"Maple: multi-modal prompt learning","author":"Khattak","year":"2023"},{"key":"10.1016\/j.neucom.2026.134178_bib0280","series-title":"Proc. IEEE CVPR","first-page":"6757","article-title":"Visual-language prompt tuning with knowledge-guided context optimization","author":"Yao","year":"2023"},{"key":"10.1016\/j.neucom.2026.134178_bib0285","series-title":"Proc. IEEE CVPR","first-page":"23232","article-title":"Lasp: text-to-text optimization for language-aware soft prompting of vision & language models","author":"Bulat","year":"2023"},{"issue":"2","key":"10.1016\/j.neucom.2026.134178_bib0290","doi-asserted-by":"crossref","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","article-title":"Clip-adapter: better vision-language models with feature adapters","volume":"132","author":"Gao","year":"2024","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.134178_bib0295","series-title":"Proc. ECCV","first-page":"493","article-title":"Tip-adapter: training-free adaption of clip for few-shot classification","author":"Zhang","year":"2022"},{"key":"10.1016\/j.neucom.2026.134178_bib0300","author":"Seputis"},{"key":"10.1016\/j.neucom.2026.134178_bib0305","series-title":"Proc. ICLR","article-title":"Fast and accurate deep network learning by exponential linear units (elus)","author":"Clevert","year":"2016"},{"key":"10.1016\/j.neucom.2026.134178_bib0310","author":"Hendrycks"},{"key":"10.1016\/j.neucom.2026.134178_bib0315","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1016\/j.neunet.2017.12.012","article-title":"Sigmoid-weighted linear units for neural network function approximation in reinforcement learning","volume":"107","author":"Elfwing","year":"2018","journal-title":"Neural Netw."},{"key":"10.1016\/j.neucom.2026.134178_bib0320","series-title":"Proc. ICLR","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2018"},{"issue":"8","key":"10.1016\/j.neucom.2026.134178_bib0325","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"Hochreiter","year":"1997","journal-title":"Neural Comput."}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226015766?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226015766?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:13:36Z","timestamp":1783152816000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226015766"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":65,"alternative-id":["S0925231226015766"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134178","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"DTVAD: Dual-branch aligned temporal-aware framework for open-vocabulary video anomaly detection","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134178","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"134178"}}