{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,20]],"date-time":"2026-08-20T09:28:31Z","timestamp":1787218111121,"version":"build-2736575974"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["52335001"],"award-info":[{"award-number":["52335001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.engappai.2026.115711","type":"journal-article","created":{"date-parts":[[2026,7,30]],"date-time":"2026-07-30T09:23:01Z","timestamp":1785403381000},"page":"115711","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"P7","title":["From detection data to structured knowledge: Augmenting multimodal large language models for substation safety inspection"],"prefix":"10.1016","volume":"181","author":[{"given":"Yutong","family":"Lai","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuoshuo","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lei","family":"Fei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9611-9687","authenticated-orcid":false,"given":"Dejun","family":"Ning","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.engappai.2026.115711_b1","series-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"10.1016\/j.engappai.2026.115711_b2","series-title":"Qwen2. 5-vl technical report","author":"Bai","year":"2025"},{"key":"10.1016\/j.engappai.2026.115711_b3","doi-asserted-by":"crossref","unstructured":"Bergmann, P., Fauser, M., Sattlegger, D., Steger, C., 2019. MVTec AD\u2013A comprehensive real-world dataset for unsupervised anomaly detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 9592\u20139600.","DOI":"10.1109\/CVPR.2019.00982"},{"key":"10.1016\/j.engappai.2026.115711_b4","series-title":"The revolution of multimodal large language models: a survey","author":"Caffagni","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b5","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2023.106677","article-title":"Itran: A novel transformer-based approach for industrial anomaly detection and localization","volume":"125","author":"Cai","year":"2023","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.engappai.2026.115711_b6","series-title":"A survey on visual anomaly detection: Challenge, approach, and prospect","author":"Cao","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b7","series-title":"Deep learning for anomaly detection: A survey","author":"Chalapathy","year":"2019"},{"key":"10.1016\/j.engappai.2026.115711_b8","unstructured":"Chaoyou, F., Peixian, C., Yunhang, S., Yulei, Q., Mengdan, Z., Xu, L., Jinrui, Y., Xiawu, Z., Ke, L., Xing, S., et al., 2023, Mme: A comprehensive evaluation benchmark for multimodal large language models, 3arXiv preprint arXiv:2306.13394."},{"key":"10.1016\/j.engappai.2026.115711_b9","series-title":"Can multimodal large language models be guided to improve industrial anomaly detection?","author":"Chen","year":"2025"},{"key":"10.1016\/j.engappai.2026.115711_b10","doi-asserted-by":"crossref","first-page":"27056","DOI":"10.52202\/079017-0850","article-title":"Are we on the right way for evaluating large vision-language models?","volume":"37","author":"Chen","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.115711_b11","article-title":"Vmad: Visual-enhanced multimodal large language model for zero-shot anomaly detection","author":"Deng","year":"2025","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"key":"10.1016\/j.engappai.2026.115711_b12","series-title":"The llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b13","series-title":"Retrieval-augmented generation for large language models: A survey","author":"Gao","year":"2023"},{"key":"10.1016\/j.engappai.2026.115711_b14","series-title":"Synergizing rag and reasoning: A systematic review","author":"Gao","year":"2025"},{"key":"10.1016\/j.engappai.2026.115711_b15","doi-asserted-by":"crossref","unstructured":"Gu, Z., Zhu, B., Zhu, G., Chen, Y., Tang, M., Wang, J., 2024. Anomalygpt: Detecting industrial anomalies using large vision-language models. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 38, pp. 1932\u20131940.","DOI":"10.1609\/aaai.v38i3.27963"},{"key":"10.1016\/j.engappai.2026.115711_b16","series-title":"DeepRAG: Thinking to retrieve step by step for large language mod- els","author":"Guan","year":"2025"},{"key":"10.1016\/j.engappai.2026.115711_b17","series-title":"RAG-anything: All-in-one RAG framework","author":"Guo","year":"2025"},{"key":"10.1016\/j.engappai.2026.115711_b18","series-title":"Critical review for one-class classification: recent advances and the reality behind them","author":"Hayashi","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b19","series-title":"Enhancing llm reasoning with multi-path collaborative reactive and reflection agents","author":"He","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b20","series-title":"Mrag-bench: Vision-centric evaluation for retrieval-augmented multimodal models","author":"Hu","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b21","series-title":"AD-copilot: A vision-language assistant for industrial anomaly detection via visual in-context comparison","author":"Jiang","year":"2026"},{"key":"10.1016\/j.engappai.2026.115711_b22","series-title":"Mmad: The first-ever comprehensive benchmark for multimodal large language models in industrial anomaly detection","author":"Jiang","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b23","doi-asserted-by":"crossref","unstructured":"Jiang, Y., Lu, X., Jin, Q., Sun, Q., Wu, H., Zhuo, C., 2024. Fabgpt: An efficient large multimodal model for complex wafer defect knowledge queries. In: Proceedings of the 43rd IEEE\/ACM International Conference on Computer-Aided Design. pp. 1\u20138.","DOI":"10.1145\/3676536.3676750"},{"key":"10.1016\/j.engappai.2026.115711_b24","series-title":"A survey of visual sensory anomaly detection","author":"Jiang","year":"2022"},{"key":"10.1016\/j.engappai.2026.115711_b25","first-page":"9459","article-title":"Retrieval-augmented generation for knowledge-intensive nlp tasks","volume":"33","author":"Lewis","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.115711_b26","doi-asserted-by":"crossref","unstructured":"Li, Y., Goodge, A., Liu, F., Foo, C.-S., 2024a. Promptad: Zero-shot anomaly detection using text prompts. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 1093\u20131102.","DOI":"10.1109\/WACV57701.2024.00113"},{"key":"10.1016\/j.engappai.2026.115711_b27","series-title":"Benchmarking multimodal retrieval augmented generation with dynamic vqa dataset and self-adaptive planning agent","author":"Li","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b28","series-title":"Myriad: Large multimodal model by applying vision experts for industrial anomaly detection","author":"Li","year":"2023"},{"key":"10.1016\/j.engappai.2026.115711_b29","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J., 2024. Improved baselines with visual instruction tuning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 26296\u201326306.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"10.1016\/j.engappai.2026.115711_b30","doi-asserted-by":"crossref","unstructured":"Liu, J., Wang, W., Yihang, S., Huang, J., Zhang, Y., Li, C.-Y., Chen, W., Xing, X., Chang, K.-J., Shen, L., et al., 2025. Asclepius: A spectrum evaluation benchmark for medical multi-modal large language models. In: Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). pp. 24181\u201324201.","DOI":"10.18653\/v1\/2025.acl-long.1178"},{"issue":"1","key":"10.1016\/j.engappai.2026.115711_b31","doi-asserted-by":"crossref","first-page":"104","DOI":"10.1007\/s11633-023-1459-z","article-title":"Deep industrial image anomaly detection: A survey","volume":"21","author":"Liu","year":"2024","journal-title":"Mach. Intell. Res."},{"key":"10.1016\/j.engappai.2026.115711_b32","series-title":"KILT: a benchmark for knowledge intensive language tasks","author":"Petroni","year":"2020"},{"key":"10.1016\/j.engappai.2026.115711_b33","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2024","first-page":"247","article-title":"SnapNTell: Enhancing entity-centric visual question answering with retrieval augmented multimodal LLM","author":"Qiu","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b34","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"issue":"2","key":"10.1016\/j.engappai.2026.115711_b35","doi-asserted-by":"crossref","first-page":"661","DOI":"10.1007\/s40684-021-00343-6","article-title":"State of the art in defect detection based on machine vision","volume":"9","author":"Ren","year":"2022","journal-title":"Int. J. Precis. Eng. Manufacturing-Green Technol."},{"key":"10.1016\/j.engappai.2026.115711_b36","series-title":"Findings of the Association for Computational Linguistics: NAACL 2025","first-page":"2026","article-title":"Unirag: Universal retrieval augmentation for large vision language models","author":"Sharifymoghaddam","year":"2025"},{"key":"10.1016\/j.engappai.2026.115711_b37","series-title":"Agentic retrieval-augmented generation: A survey on agentic rag","author":"Singh","year":"2025"},{"key":"10.1016\/j.engappai.2026.115711_b38","doi-asserted-by":"crossref","unstructured":"Wyatt, J., Leach, A., Schmon, S.M., Willcocks, C.G., 2022. Anoddpm: Anomaly detection with denoising diffusion probabilistic models using simplex noise. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 650\u2013656.","DOI":"10.1109\/CVPRW56347.2022.00080"},{"key":"10.1016\/j.engappai.2026.115711_b39","series-title":"Minicpm-v: A gpt-4v level mllm on your phone","author":"Yao","year":"2024"},{"key":"10.1016\/j.engappai.2026.115711_b40","first-page":"94327","article-title":"Gmai-mmbench: A comprehensive multimodal evaluation benchmark towards general medical ai","volume":"37","author":"Ye","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.115711_b41","doi-asserted-by":"crossref","unstructured":"Yi, J., Yoon, S., 2020. Patch svdd: Patch-level svdd for anomaly detection and segmentation. In: Proceedings of the Asian Conference on Computer Vision. pp. 375\u2013390.","DOI":"10.1007\/978-3-030-69544-6_23"},{"key":"10.1016\/j.engappai.2026.115711_b42","series-title":"MME-industry: A cross-industry multimodal evaluation benchmark","author":"Yi","year":"2025"},{"key":"10.1016\/j.engappai.2026.115711_b43","series-title":"Mm-vet: Evaluating large multimodal models for integrated capabilities","author":"Yu","year":"2023"},{"key":"10.1016\/j.engappai.2026.115711_b44","doi-asserted-by":"crossref","unstructured":"Yue, X., Ni, Y., Zhang, K., Zheng, T., Liu, R., Zhang, G., Stevens, S., Jiang, D., Ren, W., Sun, Y., et al., 2024. Mmmu: A massive multi-discipline multimodal understanding and reasoning benchmark for expert agi. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 9556\u20139567.","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"10.1016\/j.engappai.2026.115711_b45","article-title":"Logicode: an llm-driven framework for logical anomaly detection","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"issue":"3","key":"10.1016\/j.engappai.2026.115711_b46","doi-asserted-by":"crossref","first-page":"333","DOI":"10.1108\/IR-10-2016-0260","article-title":"Development and implementation of a robotic inspection system for power substations","volume":"44","author":"Zhang","year":"2017","journal-title":"Ind. Robot.: An Int. J."},{"key":"10.1016\/j.engappai.2026.115711_b47","series-title":"Pmc-vqa: Visual instruction tuning for medical visual question answering","author":"Zhang","year":"2023"},{"key":"10.1016\/j.engappai.2026.115711_b48","series-title":"Anomalyclip: Object-agnostic prompt learning for zero-shot anomaly detection","author":"Zhou","year":"2023"}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626019950?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626019950?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,20]],"date-time":"2026-08-20T09:18:41Z","timestamp":1787217521000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197626019950"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":48,"alternative-id":["S0952197626019950"],"URL":"https:\/\/doi.org\/10.1016\/j.engappai.2026.115711","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"From detection data to structured knowledge: Augmenting multimodal large language models for substation safety inspection","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.engappai.2026.115711","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"115711"}}