{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T14:46:38Z","timestamp":1782485198750,"version":"3.54.5"},"reference-count":52,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100016804","name":"Natural Science Foundation of Shenzhen Municipality","doi-asserted-by":"publisher","award":["JCYJ20250604145532041"],"award-info":[{"award-number":["JCYJ20250604145532041"]}],"id":[{"id":"10.13039\/100016804","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002920","name":"Research Grants Council, University Grants Committee","doi-asserted-by":"publisher","award":["15216225"],"award-info":[{"award-number":["15216225"]}],"id":[{"id":"10.13039\/501100002920","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62502117"],"award-info":[{"award-number":["62502117"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.knosys.2026.116321","type":"journal-article","created":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T15:32:53Z","timestamp":1780759973000},"page":"116321","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Stage-aware industrial defect understanding via multi-agent collaboration"],"prefix":"10.1016","volume":"349","author":[{"given":"Jiayuan","family":"Xie","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinting","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuxi","family":"Tu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yangyang","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3370-471X","authenticated-orcid":false,"given":"Qing","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116321_b1","series-title":"IEEE Conference on Computer Vision and Pattern Recognition","first-page":"4401","article-title":"A style-based generator architecture for generative adversarial networks","author":"Karras","year":"2019"},{"key":"10.1016\/j.knosys.2026.116321_b2","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"19606","article-title":"WinCLIP: Zero-\/few-shot anomaly classification and segmentation","author":"Jeong","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b3","series-title":"AnomalyCLIP: Object-agnostic prompt learning for zero-shot anomaly detection","author":"Zhou","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b4","series-title":"27th Mediterranean Conference on Control and Automation","first-page":"75","article-title":"Surface defect detection for automated inspection systems using convolutional neural networks","author":"Konrad","year":"2019"},{"issue":"3","key":"10.1016\/j.knosys.2026.116321_b5","doi-asserted-by":"crossref","first-page":"127","DOI":"10.1080\/15980316.2021.1876174","article-title":"A more reliable defect detection and performance improvement method for panel inspection based on artificial intelligence","volume":"22","author":"Jeong","year":"2021","journal-title":"J. Inf. Disp."},{"key":"10.1016\/j.knosys.2026.116321_b6","series-title":"IEEE Conference on Computer Vision and Pattern Recognition","first-page":"9664","article-title":"CutPaste: Self-supervised learning for anomaly detection and localization","author":"Li","year":"2021"},{"key":"10.1016\/j.knosys.2026.116321_b7","series-title":"IEEE Winter Conference on Applications of Computer Vision","first-page":"2523","article-title":"Defect-GAN: High-fidelity defect synthesis for automated defect inspection","author":"Zhang","year":"2021"},{"key":"10.1016\/j.knosys.2026.116321_b8","series-title":"Learning feature inversion for multi-class anomaly detection under general-purpose COCO-AD benchmark","author":"Zhang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116321_b9","series-title":"IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"1819","article-title":"CFLOW-AD: real-time unsupervised anomaly detection with localization via conditional normalizing flows","author":"Gudovskiy","year":"2022"},{"key":"10.1016\/j.knosys.2026.116321_b10","series-title":"Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI 2024, Thirty-Sixth Conference on Innovative Applications of Artificial Intelligence, IAAI 2024, Fourteenth Symposium on Educational Advances in Artificial Intelligence","first-page":"19306","article-title":"Automated defect report generation for enhanced industrial quality control","author":"Xie","year":"2024"},{"key":"10.1016\/j.knosys.2026.116321_b11","series-title":"Improving factuality and reasoning in language models through multiagent debate","author":"Du","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b12","series-title":"The Eleventh International Conference on Learning Representations","article-title":"Self-consistency improves chain of thought reasoning in language models","author":"Wang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b13","series-title":"More agents is all you need","author":"Li","year":"2024"},{"key":"10.1016\/j.knosys.2026.116321_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112520","article-title":"LiFSO-Net: A lightweight feature screening optimization network for complex-scale flat metal defect detection","volume":"304","author":"Zhong","year":"2024","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116321_b15","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2023.111066","article-title":"GRA-Net: Global receptive attention network for surface defect detection","volume":"280","author":"Xiao","year":"2023","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116321_b16","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112320","article-title":"A drift detection method for industrial images based on a defect segmentation model","volume":"301","author":"Li","year":"2024","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116321_b17","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2022.110176","article-title":"Progressive refined redistribution pyramid network for defect detection in complex scenarios","volume":"260","author":"Yu","year":"2023","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116321_b18","series-title":"CLIP-AD: A language-guided staged dual-path model for zero-shot anomaly detection","author":"Chen","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b19","series-title":"Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI 2024, Thirty-Sixth Conference on Innovative Applications of Artificial Intelligence, IAAI 2024, Fourteenth Symposium on Educational Advances in Artificial Intelligence","first-page":"1932","article-title":"Anomalygpt: Detecting industrial anomalies using large vision-language models","author":"Gu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116321_b20","series-title":"PandaGPT: One model to instruction-follow them all","author":"Su","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b21","series-title":"Exploring grounding potential of VQA-oriented GPT-4V for zero-shot anomaly detection","author":"Zhang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b22","series-title":"IEEE Conference on Computer Vision and Pattern Recognition","first-page":"9592","article-title":"MVTec AD - A comprehensive real-world dataset for unsupervised anomaly detection","author":"Bergmann","year":"2019"},{"key":"10.1016\/j.knosys.2026.116321_b23","series-title":"Computer Vision - ECCV 2022 - 17th European Conference","first-page":"392","article-title":"Spot-the-difference self-supervised pre-training for anomaly detection and segmentation","volume":"vol. 13690","author":"Zou","year":"2022"},{"key":"10.1016\/j.knosys.2026.116321_b24","series-title":"13th International Congress on Ultra Modern Telecommunications and Control Systems and Workshops","first-page":"66","article-title":"Deep learning-based defect detection of metal parts: evaluating current methods in complex conditions","author":"Jezek","year":"2021"},{"key":"10.1016\/j.knosys.2026.116321_b25","series-title":"Real-IAD: A real-world multi-view dataset for benchmarking versatile industrial anomaly detection","author":"Wang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116321_b26","series-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023","article-title":"PAD: A dataset and benchmark for pose-agnostic anomaly detection","author":"Zhou","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b27","series-title":"Proceedings of the 17th International Joint Conference on Computer Vision, Imaging and Computer Graphics Theory and Applications","first-page":"202","article-title":"The MVTec 3D-AD dataset for unsupervised 3D anomaly detection and localization","author":"Bergmann","year":"2022"},{"issue":"4","key":"10.1016\/j.knosys.2026.116321_b28","doi-asserted-by":"crossref","first-page":"947","DOI":"10.1007\/s11263-022-01578-9","article-title":"Beyond dents and scratches: Logical constraints in unsupervised anomaly detection and localization","volume":"130","author":"Bergmann","year":"2022","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116321_b29","series-title":"2021 IEEE\/CVF International Conference on Computer Vision","first-page":"2054","article-title":"TRAR: routing the attention spans in transformer for visual question answering","author":"Zhou","year":"2021"},{"key":"10.1016\/j.knosys.2026.116321_b30","series-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"21","article-title":"Stacked attention networks for image question answering","author":"Yang","year":"2016"},{"key":"10.1016\/j.knosys.2026.116321_b31","series-title":"IEEE Conference on Computer Vision and Pattern Recognition","first-page":"6281","article-title":"Deep modular co-attention networks for visual question answering","author":"Yu","year":"2019"},{"key":"10.1016\/j.knosys.2026.116321_b32","doi-asserted-by":"crossref","first-page":"2986","DOI":"10.1109\/TMM.2021.3091882","article-title":"Explicit cross-modal representation learning for visual commonsense reasoning","volume":"24","author":"Zhang","year":"2022","journal-title":"IEEE Trans. Multim."},{"key":"10.1016\/j.knosys.2026.116321_b33","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15555","article-title":"Query and attention augmentation for knowledge-based explainable reasoning","author":"Zhang","year":"2022"},{"key":"10.1016\/j.knosys.2026.116321_b34","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"19102","article-title":"VQACL: A novel visual question answering continual learning setting","author":"Zhang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b35","series-title":"Computer Vision - ECCV 2022 - 17th European Conference","first-page":"497","article-title":"Learning efficient multi-agent cooperative visual exploration","volume":"vol. 13699","author":"Yu","year":"2022"},{"key":"10.1016\/j.knosys.2026.116321_b36","series-title":"Multi-agent VQA: exploring multi-agent foundation models in zero-shot visual question answering","author":"Jiang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116321_b37","series-title":"Towards top-down reasoning: An explainable multi-agent approach for visual question answering","author":"Wang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b38","unstructured":"H. Chaudhary, S.K. Saha, Devices and Machines in Handmade Carpet Manufacturing, in: Proc. of the 14th ISME Int. Conf. on Mechanical Engineering in Knowledge Age, 2005, pp. 66\u201372."},{"issue":"1","key":"10.1016\/j.knosys.2026.116321_b39","first-page":"74","article-title":"Integration of value stream map and strategic layout planning into DMAIC approach to improve carpeting process","volume":"10","author":"Nagi","year":"2017","journal-title":"J. Ind. Eng. Manag."},{"key":"10.1016\/j.knosys.2026.116321_b40","series-title":"Gpt-4o system card","author":"Hurst","year":"2024"},{"key":"10.1016\/j.knosys.2026.116321_b41","series-title":"VisualBERT: A simple and performant baseline for vision and language","author":"Li","year":"2019"},{"key":"10.1016\/j.knosys.2026.116321_b42","series-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","first-page":"4171","article-title":"BERT: pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.knosys.2026.116321_b43","series-title":"Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015","first-page":"91","article-title":"Faster R-CNN: towards real-time object detection with region proposal networks","author":"Ren","year":"2015"},{"key":"10.1016\/j.knosys.2026.116321_b44","series-title":"Computer Vision - ECCV 2020 - 16th European Conference","first-page":"104","article-title":"UNITER: universal image-text representation learning","volume":"vol. 12375","author":"Chen","year":"2020"},{"key":"10.1016\/j.knosys.2026.116321_b45","series-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing","first-page":"5099","article-title":"LXMERT: Learning cross-modality encoder representations from transformers","author":"Tan","year":"2019"},{"key":"10.1016\/j.knosys.2026.116321_b46","series-title":"Proceedings of the 38th International Conference on Machine Learning","first-page":"5583","article-title":"ViLT: Vision-and-language transformer without convolution or region supervision","volume":"vol. 139","author":"Kim","year":"2021"},{"key":"10.1016\/j.knosys.2026.116321_b47","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"10757","article-title":"A multi-modal context reasoning approach for conditional inference on joint textual and visual clues","author":"Li","year":"2023"},{"key":"10.1016\/j.knosys.2026.116321_b48","series-title":"Computer Vision - ECCV 2020 - 16th European Conference","first-page":"121","article-title":"Oscar: Object-semantics aligned pre-training for vision-language tasks","volume":"vol. 12375","author":"Li","year":"2020"},{"key":"10.1016\/j.knosys.2026.116321_b49","series-title":"RoBERTa: A robustly optimized BERT pretraining approach","author":"Liu","year":"2019"},{"key":"10.1016\/j.knosys.2026.116321_b50","series-title":"Llavanext: Improved reasoning, ocr, and world knowledge","author":"Liu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116321_b51","series-title":"Detectron2","author":"Wu","year":"2019"},{"key":"10.1016\/j.knosys.2026.116321_b52","series-title":"3rd International Conference on Learning Representations","article-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2015"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126010476?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126010476?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T13:54:32Z","timestamp":1782482072000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126010476"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":52,"alternative-id":["S0950705126010476"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116321","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Stage-aware industrial defect understanding via multi-agent collaboration","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116321","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"116321"}}