{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T16:03:53Z","timestamp":1781193833890,"version":"3.54.1"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100013804","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013804","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003397","name":"Huazhong University of Science and Technology","doi-asserted-by":"publisher","award":["2021GCRC058"],"award-info":[{"award-number":["2021GCRC058"]}],"id":[{"id":"10.13039\/501100003397","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132129","type":"journal-article","created":{"date-parts":[[2026,3,20]],"date-time":"2026-03-20T15:41:39Z","timestamp":1774021299000},"page":"132129","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Bidirectional adaptive transformers for multimodal anomaly detection"],"prefix":"10.1016","volume":"320","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-2419-0149","authenticated-orcid":false,"given":"Yuxin","family":"Jiang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7619-6618","authenticated-orcid":false,"given":"Yunkang","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5204-7992","authenticated-orcid":false,"given":"Weiming","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132129_bib0001","unstructured":"Ba, J. L., Kiros, J. R., & Hinton, G. E. (2016). Layer normalization. arXiv: 1607.06450."},{"key":"10.1016\/j.eswa.2026.132129_bib0002","doi-asserted-by":"crossref","unstructured":"Bergmann, P., Jin, X., Sattlegger, D., & Steger, C. (2021). The mvtec 3d-ad dataset for unsupervised 3d anomaly detection and localization. arXiv: 2112.09045.","DOI":"10.5220\/0010865000003124"},{"key":"10.1016\/j.eswa.2026.132129_bib0003","series-title":"Proceedings of the asian conference on computer vision","first-page":"3586","article-title":"The eyecandies dataset for unsupervised multimodal anomaly detection and localization","author":"Bonfiglioli","year":"2022"},{"key":"10.1016\/j.eswa.2026.132129_bib0004","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2022.108846","article-title":"Informative knowledge distillation for image anomaly segmentation","volume":"248","author":"Cao","year":"2022","journal-title":"Knowledge-Based Systems"},{"issue":"11","key":"10.1016\/j.eswa.2026.132129_bib0005","doi-asserted-by":"crossref","first-page":"10674","DOI":"10.1109\/TII.2023.3241579","article-title":"Collaborative discrepancy optimization for reliable image anomaly localization","volume":"19","author":"Cao","year":"2023","journal-title":"IEEE Transactions on Industrial Informatics"},{"key":"10.1016\/j.eswa.2026.132129_bib0006","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110761","article-title":"Complementary pseudo multimodal feature for point cloud anomaly detection","volume":"156","author":"Cao","year":"2024","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132129_bib0007","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"9650","article-title":"Emerging properties in self-supervised vision transformers","author":"Caron","year":"2021"},{"key":"10.1016\/j.eswa.2026.132129_bib0008","unstructured":"Chang, A. X., Funkhouser, T., Guibas, L., Hanrahan, P., Huang, Q., Li, Z., Savarese, S., Savva, M., Song, S., Su, H. et al. (2015). ShapeNet: An information-rich 3d model repository. arXiv: 1512.03012."},{"key":"10.1016\/j.eswa.2026.132129_bib0009","doi-asserted-by":"crossref","first-page":"16664","DOI":"10.52202\/068431-1212","article-title":"Adaptformer: Adapting vision transformers for scalable visual recognition","volume":"35","author":"Chen","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"6","key":"10.1016\/j.eswa.2026.132129_bib0010","doi-asserted-by":"crossref","first-page":"5000","DOI":"10.1109\/TII.2025.3552723","article-title":"Multimodal industrial anomaly detection via uni-modal and cross-modal fusion","volume":"21","author":"Cheng","year":"2025","journal-title":"IEEE Transactions on Industrial Informatics"},{"key":"10.1016\/j.eswa.2026.132129_bib0011","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"17234","article-title":"Multimodal industrial anomaly detection by crossmodal feature mapping","author":"Costanzino","year":"2024"},{"key":"10.1016\/j.eswa.2026.132129_bib0012","series-title":"International conference on pattern recognition","first-page":"475","article-title":"Padim: A patch distribution modeling framework for anomaly detection and localization","author":"Defard","year":"2021"},{"key":"10.1016\/j.eswa.2026.132129_bib0013","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"9737","article-title":"Anomaly detection via reverse distillation from one-class embedding","author":"Deng","year":"2022"},{"key":"10.1016\/j.eswa.2026.132129_bib0014","series-title":"2009 IEEE conference on computer vision and pattern recognition","first-page":"248","article-title":"ImageNet: A large-scale hierarchical image database","author":"Deng","year":"2009"},{"key":"10.1016\/j.eswa.2026.132129_bib0015","doi-asserted-by":"crossref","DOI":"10.1016\/j.compind.2025.104301","article-title":"A simple and reliable semi-supervised anomaly detection network for detecting crack in stamped parts","volume":"169","author":"Dong","year":"2025","journal-title":"Computers in Industry"},{"key":"10.1016\/j.eswa.2026.132129_bib0016","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv: 2010.11929."},{"issue":"6","key":"10.1016\/j.eswa.2026.132129_bib0017","doi-asserted-by":"crossref","first-page":"381","DOI":"10.1145\/358669.358692","article-title":"Random sample consensus: A paradigm for model fitting with applications to image analysis and automated cartography","volume":"24","author":"Fischler","year":"1981","journal-title":"Communications of the ACM"},{"key":"10.1016\/j.eswa.2026.132129_bib0018","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"1705","article-title":"Memorizing normality to detect anomaly: Memory-augmented deep autoencoder for unsupervised anomaly detection","author":"Gong","year":"2019"},{"key":"10.1016\/j.eswa.2026.132129_bib0019","series-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision","first-page":"98","article-title":"Cflow-ad: Real-time unsupervised anomaly detection with localization via conditional normalizing flows","author":"Gudovskiy","year":"2022"},{"issue":"3","key":"10.1016\/j.eswa.2026.132129_bib0020","doi-asserted-by":"crossref","first-page":"1102","DOI":"10.1109\/TMI.2023.3327720","article-title":"Encoder-decoder contrast for unsupervised anomaly detection in medical images","volume":"43","author":"Guo","year":"2023","journal-title":"IEEE Transactions on Medical Imaging"},{"key":"10.1016\/j.eswa.2026.132129_bib0021","doi-asserted-by":"crossref","first-page":"10721","DOI":"10.52202\/075280-0471","article-title":"Recontrast: Domain-specific anomaly detection via contrastive reconstruction","volume":"36","author":"Guo","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132129_bib0022","unstructured":"Hendrycks, D., & Gimpel, K. (2016). Gaussian error linear units (gelus). arXiv: 1606.08415."},{"key":"10.1016\/j.eswa.2026.132129_bib0023","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"2967","article-title":"Back to the feature: Classical 3d features are (almost) all you need for 3d anomaly detection","author":"Horwitz","year":"2023"},{"issue":"1","key":"10.1016\/j.eswa.2026.132129_bib0024","doi-asserted-by":"crossref","first-page":"762","DOI":"10.1109\/TII.2024.3459612","article-title":"Unsupervised wind turbine blade damage detection with memory-aided denoising reconstruction","volume":"21","author":"Jia","year":"2025","journal-title":"IEEE Transactions on Industrial Informatics"},{"key":"10.1016\/j.eswa.2026.132129_bib0025","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2023.110982","article-title":"A masked reverse knowledge distillation method incorporating global and local information for image anomaly detection","volume":"280","author":"Jiang","year":"2023","journal-title":"Knowledge-Based Systems"},{"issue":"6","key":"10.1016\/j.eswa.2026.132129_bib0026","doi-asserted-by":"crossref","first-page":"84","DOI":"10.1145\/3065386","article-title":"Imagenet classification with deep convolutional neural networks","volume":"60","author":"Krizhevsky","year":"2017","journal-title":"Communications of the ACM"},{"key":"10.1016\/j.eswa.2026.132129_bib0027","first-page":"109","article-title":"Scaling & shifting your features: A new baseline for efficient model tuning","volume":"35","author":"Lian","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132129_bib0028","doi-asserted-by":"crossref","first-page":"4327","DOI":"10.1109\/TIP.2023.3293772","article-title":"Omni-frequency channel-selection representations for unsupervised anomaly detection","volume":"32","author":"Liang","year":"2023","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.132129_bib0029","doi-asserted-by":"crossref","DOI":"10.1016\/j.aei.2022.101822","article-title":"A two-stage anomaly detection framework: Towards low omission rate in industrial vision applications","volume":"55","author":"Liu","year":"2023","journal-title":"Advanced Engineering Informatics"},{"key":"10.1016\/j.eswa.2026.132129_bib0030","doi-asserted-by":"crossref","DOI":"10.1016\/j.aei.2024.102759","article-title":"Surface defect detection of stay cable sheath based on autoencoder and auxiliary anomaly location","volume":"62","author":"Liu","year":"2024","journal-title":"Advanced Engineering Informatics"},{"key":"10.1016\/j.eswa.2026.132129_bib0031","series-title":"Advances in neural information processing systems","article-title":"An intriguing failing of convolutional neural networks and the coordconv solution","volume":"31","author":"Liu","year":"2018"},{"key":"10.1016\/j.eswa.2026.132129_bib0032","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.111563","article-title":"Reb: Reducing biases in representation for industrial anomaly detection","volume":"290","author":"Lyu","year":"2024","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.132129_bib0033","series-title":"European conference on computer vision","first-page":"604","article-title":"Masked autoencoders for point cloud self-supervised learning","author":"Pang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132129_bib0034","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"14318","article-title":"Towards total recall in industrial anomaly detection","author":"Roth","year":"2022"},{"key":"10.1016\/j.eswa.2026.132129_bib0035","series-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision","first-page":"2592","article-title":"Asymmetric student-teacher networks for industrial anomaly detection","author":"Rudolph","year":"2023"},{"issue":"7","key":"10.1016\/j.eswa.2026.132129_bib0036","doi-asserted-by":"crossref","first-page":"1443","DOI":"10.1162\/089976601750264965","article-title":"Estimating the support of a high-dimensional distribution","volume":"13","author":"Sch\u00f6lkopf","year":"2001","journal-title":"Neural Computation"},{"key":"10.1016\/j.eswa.2026.132129_bib0037","doi-asserted-by":"crossref","DOI":"10.1016\/j.compind.2023.103994","article-title":"A two-stage unsupervised approach for surface anomaly detection in wire and arc additive manufacturing","volume":"151","author":"Song","year":"2023","journal-title":"Computers in Industry"},{"key":"10.1016\/j.eswa.2026.132129_bib0038","first-page":"1","article-title":"Deep learning for unsupervised anomaly localization in industrial images: A survey","volume":"71","author":"Tao","year":"2022","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"10.1016\/j.eswa.2026.132129_bib0039","doi-asserted-by":"crossref","DOI":"10.1016\/j.compind.2025.104315","article-title":"MinimaxAD: A lightweight autoencoder for feature-rich anomaly detection","volume":"171","author":"Wang","year":"2025","journal-title":"Computers in Industry"},{"key":"10.1016\/j.eswa.2026.132129_bib0040","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8032","article-title":"Multimodal industrial anomaly detection via hybrid fusion","author":"Wang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132129_bib0041","doi-asserted-by":"crossref","DOI":"10.1016\/j.compind.2024.104192","article-title":"Tdad: Self-supervised industrial anomaly detection with a two-stage diffusion model","volume":"164","author":"Wei","year":"2025","journal-title":"Computers in Industry"},{"issue":"7","key":"10.1016\/j.eswa.2026.132129_bib0042","doi-asserted-by":"crossref","first-page":"5666","DOI":"10.1109\/TII.2025.3556083","article-title":"Multitask hybrid knowledge distillation for unsupervised anomaly detection","volume":"21","author":"Xu","year":"2025","journal-title":"IEEE Transactions on Industrial Informatics"},{"key":"10.1016\/j.eswa.2026.132129_bib0043","doi-asserted-by":"crossref","first-page":"116","DOI":"10.1109\/TMM.2020.3046884","article-title":"Attribute restoration framework for anomaly detection","volume":"24","author":"Ye","year":"2020","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132129_bib0044","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19313","article-title":"Point-bert: Pre-training 3d point cloud transformers with masked point modeling","author":"Yu","year":"2022"},{"issue":"1","key":"10.1016\/j.eswa.2026.132129_bib0045","doi-asserted-by":"crossref","first-page":"300","DOI":"10.1109\/TCSVT.2024.3462433","article-title":"Surveillance video-and-language understanding: from small to large multimodal models","volume":"35","author":"Yuan","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"issue":"9","key":"10.1016\/j.eswa.2026.132129_bib0046","doi-asserted-by":"crossref","first-page":"6222","DOI":"10.1109\/TVCG.2023.3329578","article-title":"EGST: Enhanced geometric structure transformer for point cloud registration","volume":"30","author":"Yuan","year":"2024","journal-title":"IEEE Transactions on Visualization and Computer Graphics"},{"issue":"9","key":"10.1016\/j.eswa.2026.132129_bib0047","doi-asserted-by":"crossref","first-page":"8343","DOI":"10.1109\/TCSVT.2024.3379220","article-title":"Learning discriminative features via multi-hierarchical mutual information for unsupervised point cloud registration","volume":"34","author":"Yuan","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132129_bib0048","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3914","article-title":"DestSeg: Segmentation guided denoising student-teacher for anomaly detection","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132129_bib0049","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"16259","article-title":"Point transformer","author":"Zhao","year":"2021"},{"key":"10.1016\/j.eswa.2026.132129_bib0050","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"14707","article-title":"Dynamic adapter meets prompt tuning: Parameter-efficient transfer learning for point cloud analysis","author":"Zhou","year":"2024"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426010420?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426010420?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T15:51:58Z","timestamp":1781193118000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426010420"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":50,"alternative-id":["S0957417426010420"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132129","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Bidirectional adaptive transformers for multimodal anomaly detection","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132129","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132129"}}