{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T12:49:42Z","timestamp":1785934182974,"version":"3.56.0"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["NO.NJ2024031"],"award-info":[{"award-number":["NO.NJ2024031"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013804","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013804","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100011246","name":"State Key Laboratory of Novel Software Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100011246","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.eswa.2026.133253","type":"journal-article","created":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T20:07:15Z","timestamp":1781208435000},"page":"133253","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PB","title":["CAP-HiCLIP: A class-aware prompting model with hierarchical consistency for zero-shot anomaly detection"],"prefix":"10.1016","volume":"331","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-3949-1883","authenticated-orcid":false,"given":"Aimin","family":"Feng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7404-6973","authenticated-orcid":false,"given":"Keyang","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4963-3777","authenticated-orcid":false,"given":"Junjie","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6435-6028","authenticated-orcid":false,"given":"Yihao","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7003-0838","authenticated-orcid":false,"given":"Yifeng","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0428-0331","authenticated-orcid":false,"given":"Wei","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133253_bib0001","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"9592","article-title":"MVTec AD\u2013A comprehensive real-world dataset for unsupervised anomaly detection","author":"Bergmann","year":"2019"},{"key":"10.1016\/j.eswa.2026.133253_bib0002","unstructured":"Cao, Y., Xu, X., Sun, C., Cheng, Y., Du, Z., Gao, L., & Shen, W. (2023). Segment any anomaly without training via hybrid prompt regularization. arXiv: 2305.10724."},{"key":"10.1016\/j.eswa.2026.133253_bib0003","unstructured":"Chen, X., Han, Y., & Zhang, J. (2023). APRIL-GAN: A zero-\/few-shot anomaly classification and segmentation method for CVPR 2023 VAND workshop challenge tracks 1&2: 1st place on zero-shot AD and 4th place on few-shot AD. arXiv: 2305.17382."},{"key":"10.1016\/j.eswa.2026.133253_bib0004","series-title":"International joint conference on artificial intelligence","doi-asserted-by":"crossref","first-page":"17","DOI":"10.5772\/intechopen.107726","article-title":"CLIP-AD: A language-guided staged dual-path model for zero-shot anomaly detection","author":"Chen","year":"2024"},{"key":"10.1016\/j.eswa.2026.133253_bib0005","unstructured":"Deng, H., Zhang, Z., Bao, J., & Li, X. (2023). Bootstrap fine-grained vision-language alignment for unified zero-shot anomaly localization. arXiv: 2308.15939."},{"key":"10.1016\/j.eswa.2026.133253_bib0006","series-title":"2009\u202fIEEE conference on computer vision and pattern recognition","first-page":"248","article-title":"ImageNet: A large-scale hierarchical image database","author":"Deng","year":"2009"},{"key":"10.1016\/j.eswa.2026.133253_bib0007","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv: 2010.11929."},{"key":"10.1016\/j.eswa.2026.133253_bib0008","unstructured":"Gu, X., Lin, T.-Y., Kuo, W., & Cui, Y. (2021). Open-vocabulary object detection via vision and language knowledge distillation. arXiv: 2104.13921."},{"key":"10.1016\/j.eswa.2026.133253_bib0009","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"1932","article-title":"AnomalyGPT: Detecting industrial anomalies using large vision-language models","volume":"vol. 38","author":"Gu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133253_bib0010","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.eswa.2026.133253_bib0011","first-page":"4003","article-title":"Cross attention network for few-shot classification","volume":"32","author":"Hou","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133253_bib0012","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11375","article-title":"Adapting visual-language models for generalizable anomaly detection in medical images","author":"Huang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133253_bib0013","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19606","article-title":"WinCLIP: Zero-\/few-shot anomaly classification and segmentation","author":"Jeong","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0014","series-title":"2021 13th international congress on ultra modern telecommunications and control systems and workshops (ICUMT)","first-page":"66","article-title":"Deep learning-based defect detection of metal parts: Evaluating current methods in complex conditions","author":"Jezek","year":"2021"},{"key":"10.1016\/j.eswa.2026.133253_bib0015","series-title":"International conference on machine learning","first-page":"4904","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","author":"Jia","year":"2021"},{"key":"10.1016\/j.eswa.2026.133253_bib0016","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"4535","article-title":"Clip-count: Towards text-guided zero-shot object counting","author":"Jiang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0017","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15190","article-title":"Self-regulating prompts: Foundational model adaptation without forgetting","author":"Khattak","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0018","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19903","article-title":"HIER: Metric learning beyond class labels via hierarchical regularization","author":"Kim","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0019","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"4015","article-title":"Segment anything","author":"Kirillov","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0020","article-title":"Prompting across perception and recognition: A unified CLIP-based visual-text prompt framework for zero-shot anomaly detection","volume":"229","author":"Lai","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.133253_bib0021","unstructured":"Li, B., Weinberger, K. Q., Belongie, S., Koltun, V., & Ranftl, R. (2022a). Language-driven semantic segmentation. arXiv: 2201.03546."},{"key":"10.1016\/j.eswa.2026.133253_bib0022","doi-asserted-by":"crossref","unstructured":"Li, C., Zhou, S., Kong, J., Qi, L., & Xue, H. (2025a). KAnoCLIP: Zero-shot anomaly detection through knowledge-driven prompt learning and enhanced cross-modal integration. arXiv: 2501.03786.","DOI":"10.1109\/ICASSP49660.2025.10888834"},{"key":"10.1016\/j.eswa.2026.133253_bib0023","series-title":"International conference on machine learning","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0024","series-title":"International conference on machine learning","first-page":"12888","article-title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.eswa.2026.133253_bib0025","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133253_bib0026","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.129122","article-title":"ClipSAM: CLIP and SAM collaboration for zero-shot anomaly segmentation","volume":"618","author":"Li","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.133253_bib0027","doi-asserted-by":"crossref","first-page":"13448","DOI":"10.52202\/075280-0591","article-title":"GraphAdapter: Tuning vision-language models with dual knowledge graph","volume":"36","author":"Li","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133253_bib0028","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"16838","article-title":"PromptAD: Learning prompts with only normal samples for few-shot anomaly detection","author":"Li","year":"2024"},{"key":"10.1016\/j.eswa.2026.133253_bib0029","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"2980","article-title":"Focal loss for dense object detection","author":"Lin","year":"2017"},{"key":"10.1016\/j.eswa.2026.133253_bib0030","series-title":"European conference on computer vision","first-page":"38","article-title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133253_bib0031","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11976","article-title":"A convnet for the 2020s","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133253_bib0032","doi-asserted-by":"crossref","unstructured":"Ma, W., Zhang, X., Yao, Q., Tang, F., Wu, C., Li, Y., Yan, R., Jiang, Z., & Zhou, S. K. (2025). AA-CLIP: Enhancing zero-shot anomaly detection via anomaly-aware clip. arXiv: 2503.06661.","DOI":"10.1109\/CVPR52734.2025.00447"},{"issue":"11","key":"10.1016\/j.eswa.2026.133253_bib0033","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"Van der Maaten","year":"2008","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.eswa.2026.133253_bib0034","series-title":"2016 Fourth international conference on 3D vision (3DV)","first-page":"565","article-title":"V-Net: Fully convolutional neural networks for volumetric medical image segmentation","author":"Milletari","year":"2016"},{"key":"10.1016\/j.eswa.2026.133253_bib0035","series-title":"2021\u202fIEEE 30th international symposium on industrial electronics (ISIE)","first-page":"01","article-title":"VT-ADL: A vision transformer network for image anomaly detection and localization","author":"Mishra","year":"2021"},{"key":"10.1016\/j.eswa.2026.133253_bib0036","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15579","article-title":"Verbs in action: Improving verb understanding in video-language models","author":"Momeni","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0037","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"3170","article-title":"Teaching clip to count to ten","author":"Paiss","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0038","series-title":"European conference on computer vision","first-page":"301","article-title":"VCP-CLIP: A visual context prompting model for zero-shot anomaly segmentation","author":"Qu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133253_bib0039","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.133253_bib0040","unstructured":"Roy, S., & Etemad, A. (2023). Consistency-guided prompt learning for vision-language models. arXiv: 2306.01195."},{"issue":"3","key":"10.1016\/j.eswa.2026.133253_bib0041","doi-asserted-by":"crossref","first-page":"759","DOI":"10.1007\/s10845-019-01476-x","article-title":"Segmentation-based deep-learning approach for surface-defect detection","volume":"31","author":"Tabernik","year":"2020","journal-title":"Journal of Intelligent Manufacturing"},{"key":"10.1016\/j.eswa.2026.133253_bib0042","series-title":"Proceedings of the 2023 conference on empirical methods in natural language processing","first-page":"14333","article-title":"When are lemons purple? The concept association bias of vision-language models","author":"Tang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0043","first-page":"1","article-title":"Deep learning for unsupervised anomaly localization in industrial images: A survey","volume":"71","author":"Tao","year":"2022","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"10.1016\/j.eswa.2026.133253_bib0044","series-title":"Dagm symposium in","first-page":"11","article-title":"Weakly supervised learning for industrial optical inspection","volume":"vol. 6","author":"Wieler","year":"2007"},{"key":"10.1016\/j.eswa.2026.133253_bib0045","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"18938","article-title":"Diffusion model as representation learner","author":"Yang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0046","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"6868","article-title":"Can language understand depth?","author":"Zhang","year":"2022"},{"key":"10.1016\/j.eswa.2026.133253_bib0047","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"16816","article-title":"Conditional prompt learning for vision-language models","author":"Zhou","year":"2022"},{"key":"10.1016\/j.eswa.2026.133253_bib0048","unstructured":"Zhou, Q., Pang, G., Tian, Y., He, S., & Chen, J. (2023). AnomalyCLIP: Object-agnostic prompt learning for zero-shot anomaly detection. arXiv: 2310.18961."},{"key":"10.1016\/j.eswa.2026.133253_bib0049","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15659","article-title":"Prompt-aligned gradient for prompt tuning","author":"Zhu","year":"2023"},{"key":"10.1016\/j.eswa.2026.133253_bib0050","series-title":"European conference on computer vision","first-page":"392","article-title":"Spot-the-difference self-supervised pre-training for anomaly detection and segmentation","author":"Zou","year":"2022"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426021627?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426021627?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T12:36:31Z","timestamp":1785933391000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426021627"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":50,"alternative-id":["S0957417426021627"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133253","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"CAP-HiCLIP: A class-aware prompting model with hierarchical consistency for zero-shot anomaly detection","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133253","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133253"}}