{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T09:06:30Z","timestamp":1784279190217,"version":"3.55.0"},"reference-count":39,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113729","type":"journal-article","created":{"date-parts":[[2026,4,16]],"date-time":"2026-04-16T06:58:48Z","timestamp":1776322728000},"page":"113729","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["Boosting adversarial transferability of vision-language pre-trained models via optimal transport"],"prefix":"10.1016","volume":"179","author":[{"given":"Simeng","family":"Qin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sensen","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongchen","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaojun","family":"Jia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Bai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jindong","family":"Gu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaochun","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113729_b1","article-title":"Commonsense-guided semantic and relational consistencies for image-text retrieval","author":"Li","year":"2023","journal-title":"IEEE Trans. Multimed."},{"issue":"3","key":"10.1016\/j.patcog.2026.113729_b2","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3617592","article-title":"Deep learning approaches on image captioning: A review","volume":"56","author":"Ghandi","year":"2023","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.patcog.2026.113729_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111609","article-title":"Cycle-VQA: A cycle-consistent framework for robust medical visual question answering","volume":"165","author":"Fan","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113729_b4","doi-asserted-by":"crossref","unstructured":"Z. Yang, K. Kafle, F. Dernoncourt, V. Ordonez, Improving Visual Grounding by Encouraging Consistent Gradient-based Explanations, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 19165\u201319174.","DOI":"10.1109\/CVPR52729.2023.01837"},{"key":"10.1016\/j.patcog.2026.113729_b5","series-title":"Explaining and harnessing adversarial examples","author":"Goodfellow","year":"2014"},{"key":"10.1016\/j.patcog.2026.113729_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109828","article-title":"Revisiting the transferability of adversarial examples via source-agnostic adversarial feature inducing method","volume":"144","author":"Xiao","year":"2023","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113729_b7","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112076","article-title":"Improving adversarial transferability and imperceptibility with loss landscape and diffusion model","volume":"170","author":"Liu","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113729_b8","doi-asserted-by":"crossref","unstructured":"J. Zhang, Q. Yi, J. Sang, Towards adversarial attack on vision-language pre-training models, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 5005\u20135013.","DOI":"10.1145\/3503161.3547801"},{"key":"10.1016\/j.patcog.2026.113729_b9","doi-asserted-by":"crossref","unstructured":"D. Lu, Z. Wang, T. Wang, W. Guan, H. Gao, F. Zheng, Set-level guidance attack: Boosting adversarial transferability of vision-language pre-training models, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 102\u2013111.","DOI":"10.1109\/ICCV51070.2023.00016"},{"key":"10.1016\/j.patcog.2026.113729_b10","series-title":"European Conference on Computer Vision","first-page":"442","article-title":"Boosting transferability in vision-language attacks via diversification along the intersection region of adversarial trajectory","author":"Gao","year":"2024"},{"key":"10.1016\/j.patcog.2026.113729_b11","volume":"vol. 338","author":"Villani","year":"2009"},{"key":"10.1016\/j.patcog.2026.113729_b12","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113729_b13","doi-asserted-by":"crossref","unstructured":"J. Yang, J. Duan, S. Tran, Y. Xu, S. Chanda, L. Chen, B. Zeng, T. Chilimbi, J. Huang, Vision-language pre-training with triple contrastive learning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 15671\u201315680.","DOI":"10.1109\/CVPR52688.2022.01522"},{"key":"10.1016\/j.patcog.2026.113729_b14","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113729_b15","doi-asserted-by":"crossref","unstructured":"B.A. Plummer, L. Wang, C.M. Cervantes, J.C. Caicedo, J. Hockenmaier, S. Lazebnik, Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models, in: Proceedings of the IEEE International Conference on Computer Vision, 2015, pp. 2641\u20132649.","DOI":"10.1109\/ICCV.2015.303"},{"key":"10.1016\/j.patcog.2026.113729_b16","series-title":"Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.patcog.2026.113729_b17","series-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"issue":"1","key":"10.1016\/j.patcog.2026.113729_b18","doi-asserted-by":"crossref","first-page":"87","DOI":"10.1109\/TPAMI.2022.3152247","article-title":"A survey on vision transformer","volume":"45","author":"Han","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113729_b19","series-title":"Image-text retrieval: A survey on recent research and development","author":"Cao","year":"2022"},{"issue":"6","key":"10.1016\/j.patcog.2026.113729_b20","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3295748","article-title":"A comprehensive survey of deep learning for image captioning","volume":"51","author":"Hossain","year":"2019","journal-title":"ACM Comput. Surv. (CsUR)"},{"key":"10.1016\/j.patcog.2026.113729_b21","doi-asserted-by":"crossref","unstructured":"C. Deng, Q. Wu, Q. Wu, F. Hu, F. Lyu, M. Tan, Visual grounding via accumulated attention, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 7746\u20137755.","DOI":"10.1109\/CVPR.2018.00808"},{"key":"10.1016\/j.patcog.2026.113729_b22","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"16158","article-title":"Admix: Enhancing the transferability of adversarial attacks","author":"Wang","year":"2021"},{"key":"10.1016\/j.patcog.2026.113729_b23","doi-asserted-by":"crossref","unstructured":"J. Zhang, J.-t. Huang, W. Wang, Y. Li, W. Wu, X. Wang, Y. Su, M.R. Lyu, Improving the Transferability of Adversarial Samples by Path-Augmented Method, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 8173\u20138182.","DOI":"10.1109\/CVPR52729.2023.00790"},{"key":"10.1016\/j.patcog.2026.113729_b24","series-title":"Towards deep learning models resistant to adversarial attacks","author":"Madry","year":"2017"},{"key":"10.1016\/j.patcog.2026.113729_b25","series-title":"Bert-attack: Adversarial attack against bert using bert","author":"Li","year":"2020"},{"key":"10.1016\/j.patcog.2026.113729_b26","doi-asserted-by":"crossref","DOI":"10.1109\/TPAMI.2025.3581476","article-title":"Semantic-aligned adversarial evolution triangle for high-transferability vision-language attack","author":"Jia","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113729_b27","doi-asserted-by":"crossref","unstructured":"Q. Yu, Z. Zeng, Y. Yan, L. Ying, R. Srikant, H. Tong, Joint optimal transport and embedding for network alignment, in: Proceedings of the ACM on Web Conference 2025, 2025, pp. 2064\u20132075.","DOI":"10.1145\/3696410.3714937"},{"key":"10.1016\/j.patcog.2026.113729_b28","series-title":"International Conference on Intelligent Computing","first-page":"340","article-title":"Optimal transport-based prompt alignment for unsupervised domain adaptation","author":"Zhou","year":"2025"},{"issue":"2","key":"10.1016\/j.patcog.2026.113729_b29","doi-asserted-by":"crossref","first-page":"1330","DOI":"10.1137\/23M1581443","article-title":"Fast computation of optimal transport via entropy-regularized extragradient methods","volume":"35","author":"Li","year":"2025","journal-title":"SIAM J. Optim."},{"key":"10.1016\/j.patcog.2026.113729_b30","article-title":"Sinkhorn distances: Lightspeed computation of optimal transport","volume":"26","author":"Cuturi","year":"2013","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113729_b31","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep residual learning for image recognition, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.patcog.2026.113729_b32","series-title":"Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, the Netherlands, October 11-14, 2016, Proceedings, Part II 14","first-page":"69","article-title":"Modeling context in referring expressions","author":"Yu","year":"2016"},{"key":"10.1016\/j.patcog.2026.113729_b33","series-title":"International Conference on Machine Learning","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.patcog.2026.113729_b34","doi-asserted-by":"crossref","unstructured":"K. Papineni, S. Roukos, T. Ward, W.-J. Zhu, Bleu: a method for automatic evaluation of machine translation, in: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, 2002, pp. 311\u2013318.","DOI":"10.3115\/1073083.1073135"},{"key":"10.1016\/j.patcog.2026.113729_b35","unstructured":"S. Banerjee, A. Lavie, et al., An automatic metric for mt evaluation with improved correlation with human judgments, in: Proceedings of the ACL-2005 Workshop on Intrinsic and Extrinsic Evaluation Measures for MT and\/Or Summarization, 2005, pp. 65\u201372."},{"key":"10.1016\/j.patcog.2026.113729_b36","series-title":"Text Summarization Branches Out","first-page":"74","article-title":"Rouge: A package for automatic evaluation of summaries","author":"Lin","year":"2004"},{"key":"10.1016\/j.patcog.2026.113729_b37","doi-asserted-by":"crossref","unstructured":"R. Vedantam, C. Lawrence Zitnick, D. Parikh, Cider: Consensus-based image description evaluation, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2015, pp. 4566\u20134575.","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"10.1016\/j.patcog.2026.113729_b38","series-title":"Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, the Netherlands, October 11-14, 2016, Proceedings, Part V 14","first-page":"382","article-title":"Spice: Semantic propositional image caption evaluation","author":"Anderson","year":"2016"},{"key":"10.1016\/j.patcog.2026.113729_b39","series-title":"PAC-BAYES information bottleneck","author":"Wang","year":"2022"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326006941?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326006941?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T08:33:12Z","timestamp":1784277192000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326006941"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":39,"alternative-id":["S0031320326006941"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113729","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Boosting adversarial transferability of vision-language pre-trained models via optimal transport","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113729","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"113729"}}