{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T17:54:19Z","timestamp":1782150859669,"version":"3.54.5"},"reference-count":51,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2025ZD0123602"],"award-info":[{"award-number":["2025ZD0123602"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114261","type":"journal-article","created":{"date-parts":[[2026,6,14]],"date-time":"2026-06-14T16:38:25Z","timestamp":1781455105000},"page":"114261","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["Attacking hard-label large vision\u2013language models with model-sensitive adversarial patch designs"],"prefix":"10.1016","volume":"180","author":[{"given":"Nian","family":"Ai","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guangke","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaowen","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongliang","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8179-4508","authenticated-orcid":false,"given":"Daizong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ognjen","family":"Arandjelovi\u0107","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.114261_b1","article-title":"Instructblip: Towards general-purpose vision-language models with instruction tuning","volume":"36","author":"Dai","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114261_b2","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114261_b3","unstructured":"D. Zhu, J. Chen, X. Shen, X. Li, M. Elhoseiny, MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models, in: The Twelfth International Conference on Learning Representations, 2023."},{"key":"10.1016\/j.patcog.2026.114261_b4","doi-asserted-by":"crossref","DOI":"10.1109\/TGRS.2026.3687072","article-title":"HyperR3SNet: Leveraging hyperbolic space and vision foundation models for remote sensing semantic segmentation","author":"Fu","year":"2026","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"issue":"1","key":"10.1016\/j.patcog.2026.114261_b5","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3776560","article-title":"Autonomous domain adaptation self-optimization approach for cross-domain industrial agents","volume":"17","author":"Zuo","year":"2026","journal-title":"ACM Trans. Intell. Syst. Technol."},{"key":"10.1016\/j.patcog.2026.114261_b6","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114261_b7","series-title":"Multi-timescale distributed control for multi-energy virtual power plant clusters via cloud-edge collaboration","author":"Gao","year":"2025"},{"key":"10.1016\/j.patcog.2026.114261_b8","article-title":"Distributed model predictive control strategy for multi-energy virtual power plant based on digital twin","author":"Gao","year":"2025","journal-title":"IEEE Trans. Smart Grid"},{"key":"10.1016\/j.patcog.2026.114261_b9","doi-asserted-by":"crossref","DOI":"10.1109\/TMM.2025.3618564","article-title":"Learning efficient and adaptive cross-channel dependencies for weakly-supervised object detection","author":"Chen","year":"2025","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.114261_b10","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"7114","article-title":"Spatial-spectral homogeneous attacks on physical-world large vision-language models","volume":"vol. 40","author":"Liu","year":"2026"},{"key":"10.1016\/j.patcog.2026.114261_b11","series-title":"Hierarchical text-conditional image generation with clip latents","first-page":"3","author":"Ramesh","year":"2022"},{"key":"10.1016\/j.patcog.2026.114261_b12","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.patcog.2026.114261_b13","article-title":"ReID: Re-ranking through image description for object re-identification","author":"Yang","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114261_b14","article-title":"Underwater image enhancement by diffusion model with customized clip-classifier","author":"Liu","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114261_b15","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110847","article-title":"Sentiment analysis based on text information enhancement and multimodal feature fusion","volume":"156","author":"Liu","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114261_b16","unstructured":"X. Liu, Y. Zhu, Y. Lan, C. Yang, Y. Qiao, Safety of multimodal large language models on images and text, in: Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, 2024, pp. 8151\u20138159."},{"key":"10.1016\/j.patcog.2026.114261_b17","series-title":"2024 IEEE International Conference on Systems, Man, and Cybernetics","first-page":"3428","article-title":"Unbridled icarus: A survey of the potential perils of image inputs in multimodal large language model security","author":"Fan","year":"2024"},{"key":"10.1016\/j.patcog.2026.114261_b18","article-title":"Generating transferable attacks across large vision-language models using adversarial deformation learning","author":"Liu","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114261_b19","article-title":"On evaluating adversarial robustness of large vision-language models","volume":"36","author":"Zhao","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114261_b20","unstructured":"L. Bailey, E. Ong, S. Russell, S. Emmons, Image Hijacks: adversarial images can control generative models at runtime, in: Proceedings of the 41st International Conference on Machine Learning, 2024, pp. 2443\u20132455."},{"key":"10.1016\/j.patcog.2026.114261_b21","series-title":"How robust is google\u2019s bard to adversarial image attacks?","author":"Dong","year":"2023"},{"key":"10.1016\/j.patcog.2026.114261_b22","article-title":"Are large vision-language models robust to adversarial visual transformations?","author":"Liu","year":"2026","journal-title":"IEEE Trans. Inf. Forensics Secur."},{"key":"10.1016\/j.patcog.2026.114261_b23","doi-asserted-by":"crossref","unstructured":"X. Cui, A. Aparcedo, Y.K. Jang, S.-N. Lim, On the robustness of large multimodal models against image adversarial attacks, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 24625\u201324634.","DOI":"10.1109\/CVPR52733.2024.02325"},{"key":"10.1016\/j.patcog.2026.114261_b24","unstructured":"K. Gao, Y. Bai, J. Bai, Y. Yang, S.-T. Xia, Adversarial Robustness for Visual Grounding of Multimodal Large Language Models, in: ICLR 2024 Workshop on Reliable and Responsible Foundation Models, 2024."},{"key":"10.1016\/j.patcog.2026.114261_b25","unstructured":"Z. Wang, Z. Han, S. Chen, F. Xue, Z. Ding, X. Xiao, V. Tresp, P. Torr, J. Gu, Stop Reasoning! When Multimodal LLM with Chain-of-Thought Reasoning Meets Adversarial Image, in: First Conference on Language Modeling, 2024."},{"key":"10.1016\/j.patcog.2026.114261_b26","series-title":"Test-time backdoor attacks on multimodal large language models","author":"Lu","year":"2024"},{"key":"10.1016\/j.patcog.2026.114261_b27","unstructured":"H. Luo, J. Gu, F. Liu, P. Torr, An Image Is Worth 1000 Lies: Adversarial Transferability across Prompts on Vision-Language Models, in: The International Conference on Learning Representations, 2024."},{"key":"10.1016\/j.patcog.2026.114261_b28","unstructured":"K. Gao, Y. Bai, J. Gu, S.-T. Xia, P. Torr, Z. Li, W. Liu, Inducing High Energy-Latency of Large Vision-Language Models with Verbose Images, in: The Twelfth International Conference on Learning Representations, 2024."},{"key":"10.1016\/j.patcog.2026.114261_b29","unstructured":"E. Shayegani, Y. Dong, N. Abu-Ghazaleh, Jailbreak in pieces: Compositional adversarial attacks on multi-modal language models, in: The Twelfth International Conference on Learning Representations, 2023."},{"key":"10.1016\/j.patcog.2026.114261_b30","unstructured":"H. Yan, H. Ma, X. Cai, D. Liu, Z. Yuan, X. Qu, J. Dong, R. Guan, X. Fang, H. He, et al., Fit the Distribution: Cross-Image\/Prompt Adversarial Attacks on Multimodal Large Language Models, in: The Thirty-Ninth Annual Conference on Neural Information Processing Systems, 2025."},{"key":"10.1016\/j.patcog.2026.114261_b31","doi-asserted-by":"crossref","first-page":"52936","DOI":"10.52202\/075280-2303","article-title":"Vlattack: Multimodal adversarial attacks on vision-language tasks via pre-trained models","volume":"36","author":"Yin","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114261_b32","unstructured":"M. Cheng, T. Le, P.-Y. Chen, H. Zhang, J. Yi, C.-J. Hsieh, Query-Efficient Hard-label Black-box Attack: An Optimization-based Approach, in: International Conference on Learning Representations, 2018."},{"key":"10.1016\/j.patcog.2026.114261_b33","series-title":"Adversarial patch","author":"Brown","year":"2017"},{"key":"10.1016\/j.patcog.2026.114261_b34","doi-asserted-by":"crossref","unstructured":"R. Duan, X. Ma, Y. Wang, J. Bailey, A.K. Qin, Y. Yang, Adversarial camouflage: Hiding physical-world attacks with natural styles, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 1000\u20131008.","DOI":"10.1109\/CVPR42600.2020.00108"},{"key":"10.1016\/j.patcog.2026.114261_b35","series-title":"Intriguing properties of neural networks","author":"Szegedy","year":"2013"},{"key":"10.1016\/j.patcog.2026.114261_b36","series-title":"International Conference on Machine Learning","first-page":"2507","article-title":"Lavan: Localized and visible adversarial noise","author":"Karmon","year":"2018"},{"key":"10.1016\/j.patcog.2026.114261_b37","doi-asserted-by":"crossref","unstructured":"K. Eykholt, I. Evtimov, E. Fernandes, B. Li, A. Rahmati, C. Xiao, A. Prakash, T. Kohno, D. Song, Robust physical-world attacks on deep learning visual classification, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 1625\u20131634.","DOI":"10.1109\/CVPR.2018.00175"},{"key":"10.1016\/j.patcog.2026.114261_b38","series-title":"International Conference on Machine Learning","first-page":"284","article-title":"Synthesizing robust adversarial examples","author":"Athalye","year":"2018"},{"key":"10.1016\/j.patcog.2026.114261_b39","unstructured":"Y. Liu, X. Chen, C. Liu, D. Song, Delving into Transferable Adversarial Examples and Black-box Attacks, in: International Conference on Learning Representations, 2017."},{"key":"10.1016\/j.patcog.2026.114261_b40","article-title":"B-avibench: Towards evaluating the robustness of large vision-language model on black-box adversarial visual-instructions","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Inf. Forensics Secur."},{"key":"10.1016\/j.patcog.2026.114261_b41","doi-asserted-by":"crossref","unstructured":"K. He, X. Chen, S. Xie, Y. Li, P. Doll\u00e1r, R. Girshick, Masked autoencoders are scalable vision learners, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 16000\u201316009.","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"10.1016\/j.patcog.2026.114261_b42","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"issue":"9","key":"10.1016\/j.patcog.2026.114261_b43","doi-asserted-by":"crossref","first-page":"1145","DOI":"10.1088\/0034-4885\/43\/9\/002","article-title":"Monte Carlo theory and practice","volume":"43","author":"James","year":"1980","journal-title":"Rep. Progr. Phys."},{"key":"10.1016\/j.patcog.2026.114261_b44","series-title":"Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.patcog.2026.114261_b45","series-title":"2009 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"248","article-title":"Imagenet: A large-scale hierarchical image database","author":"Deng","year":"2009"},{"key":"10.1016\/j.patcog.2026.114261_b46","series-title":"International Conference on Machine Learning","first-page":"8821","article-title":"Zero-shot text-to-image generation","author":"Ramesh","year":"2021"},{"key":"10.1016\/j.patcog.2026.114261_b47","series-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"10.1016\/j.patcog.2026.114261_b48","series-title":"2020 Ieee Symposium on Security and Privacy (Sp)","first-page":"1277","article-title":"Hopskipjumpattack: A query-efficient decision-based attack","author":"Chen","year":"2020"},{"key":"10.1016\/j.patcog.2026.114261_b49","doi-asserted-by":"crossref","unstructured":"H. Li, X. Xu, X. Zhang, S. Yang, B. Li, Qeba: Query-efficient boundary-based blackbox attack, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 1221\u20131230.","DOI":"10.1109\/CVPR42600.2020.00130"},{"key":"10.1016\/j.patcog.2026.114261_b50","doi-asserted-by":"crossref","unstructured":"X. Cui, A. Aparcedo, Y.K. Jang, S.-N. Lim, On the robustness of large multimodal models against image adversarial attacks, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 24625\u201324634.","DOI":"10.1109\/CVPR52733.2024.02325"},{"key":"10.1016\/j.patcog.2026.114261_b51","doi-asserted-by":"crossref","unstructured":"S.-M. Moosavi-Dezfooli, A. Fawzi, O. Fawzi, P. Frossard, Universal adversarial perturbations, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 1765\u20131773.","DOI":"10.1109\/CVPR.2017.17"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326012264?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326012264?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T17:45:07Z","timestamp":1782150307000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326012264"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":51,"alternative-id":["S0031320326012264"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114261","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Attacking hard-label large vision\u2013language models with model-sensitive adversarial patch designs","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114261","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114261"}}