{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T09:00:56Z","timestamp":1784797256183,"version":"3.55.0"},"reference-count":61,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T00:00:00Z","timestamp":1783123200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T00:00:00Z","timestamp":1783123200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach. Intell. Res."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1007\/s11633-026-1667-4","type":"journal-article","created":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T05:13:28Z","timestamp":1783142008000},"page":"823-840","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Visual Adversarial Attack on Vision-language Models for Autonomous Driving"],"prefix":"10.1007","volume":"23","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9874-6828","authenticated-orcid":false,"given":"Tianyuan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinwei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yitong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Boyi","family":"Jia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siyuan","family":"Liang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shengshan","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiang","family":"Fu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4224-1318","authenticated-orcid":false,"given":"Aishan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xianglong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,4]]},"reference":[{"issue":"1","key":"1667_CR1","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1007\/s11633-022-1369-5","volume":"20","author":"F L Chen","year":"2023","unstructured":"F. L. Chen, D. Z. Zhang, M. L. Han, X. Y. Chen, J. Shi, S. Xu, B. Xu. VLP: A survey on vision-language pretraining. Machine Intelligence Research, vol. 20, no. 1, pp. 38\u201356, 2023. DOI: https:\/\/doi.org\/10.1007\/s11633-022-1369-5.","journal-title":"Machine Intelligence Research"},{"issue":"4","key":"1667_CR2","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1007\/s11633-022-1410-8","volume":"20","author":"X Wang","year":"2023","unstructured":"X. Wang, G. Chen, G. Qian, P. Gao, X. Y. Wei, Y. Wang, Y. Tian, W. Gao. Large-scale multi-modal pretrained models: A comprehensive survey. Machine Intelligence Research, vol. 20, no. 4, pp. 447\u2013482, 2023. DOI: https:\/\/doi.org\/10.1007\/s11633-022-1410-8.","journal-title":"Machine Intelligence Research"},{"key":"1667_CR3","doi-asserted-by":"publisher","first-page":"14093","DOI":"10.1109\/ICRA57147.2024.10611018","volume-title":"Proceedings of IEEE International Conference on Robotics and Automation","author":"L Chen","year":"2024","unstructured":"L. Chen, O. Sinavski, J. H\u00fcnermann, A. Karnsund, A. J. Willmott, D. Birch, D. Maund, J. Shotton. Driving with LLMs: Fusing object-level vector modality for explainable autonomous driving. In Proceedings of IEEE International Conference on Robotics and Automation, Yokohama, Japan, pp. 14093\u201314100, 2024. DOI: https:\/\/doi.org\/10.1109\/ICRA57147.2024.10611018."},{"key":"1667_CR4","unstructured":"J. Mao, Y. Qian, J. Ye, H. Zhao, Y. Wang. GPT-driver: Learning to drive with GPT, [Online], Available: https:\/\/arxiv.org\/abs\/2310.01415, 2023."},{"key":"1667_CR5","doi-asserted-by":"publisher","first-page":"15120","DOI":"10.1109\/CVPR52733.2024.01432","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"H Shao","year":"2024","unstructured":"H. Shao, Y. Hu, L. Wang, G. Song, S. L. Waslander, Y. Liu, H. Li. LMDrive: Closed-loop end-to-end driving with large language models. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Seattle, USA, pp. 15120\u201315130, 2024. DOI: https:\/\/doi.org\/10.1109\/CVPR52733.2024.01432."},{"issue":"10","key":"1667_CR6","doi-asserted-by":"publisher","first-page":"8186","DOI":"10.1109\/LRA.2024.3440097","volume":"9","author":"Z Xu","year":"2024","unstructured":"Z. Xu, Y. Zhang, E. Xie, Z. Zhao, Y. Guo, K. Y. K. Wong, Z. Li, H. Zhao. DriveGPT4: Interpretable end-to-end autonomous driving via large language model. IEEE Robotics and Automation Letters, vol. 9, no. 10, pp. 8186\u20138193, 2024. DOI: https:\/\/doi.org\/10.1109\/LRA.2024.3440097.","journal-title":"IEEE Robotics and Automation Letters"},{"key":"1667_CR7","doi-asserted-by":"publisher","unstructured":"Z. Ying, A. Liu, S. Liang, L. Huang, J. Guo, W. Zhou, X. Liu, D. Tao. SafeBench: A safety evaluation framework for multimodal large language models. International Journal of Computer Vision, vol. 134, no. 1, Article number 18, 2026. DOI: https:\/\/doi.org\/10.1007\/S11263-025-02613-1.","DOI":"10.1007\/S11263-025-02613-1"},{"key":"1667_CR8","doi-asserted-by":"publisher","first-page":"7153","DOI":"10.1109\/TIFS.2025.3583249","volume":"20","author":"Z Ying","year":"2025","unstructured":"Z. Ying, A. Liu, T. Zhang, Z. Yu, S. Liang, X. Liu, D. Tao. Jailbreak vision language models via bi-modal adversarial prompt. IEEE Transactions on Information Forensics Security, vol. 20, pp. 7153\u20137165, 2025. DOI: https:\/\/doi.org\/10.1109\/TIFS.2025.3583249.","journal-title":"IEEE Transactions on Information Forensics Security"},{"key":"1667_CR9","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Z Yin","year":"2023","unstructured":"Z. Yin, M. Ye, T. Zhang, T. Du, J. Zhu, H. Liu, J. Chen, T. Wang, F. Ma. VLATTACK: Multimodal adversarial attacks on vision-language tasks via pre-trained models. In Proceedings of the 37th International Conference on Neural Information Processing Systems, New Orleans, USA, Article number 2303, 2023."},{"key":"1667_CR10","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"H Luo","year":"2024","unstructured":"H. Luo, J. Gu, F. Liu, P. Torr. An image is worth 1000 lies: Transferability of adversarial images across prompts on vision-language models. In Proceedings of the 12th International Conference on Learning Representations, Vienna, Austria, 2024."},{"key":"1667_CR11","unstructured":"J. Li, K. Gao, Y. Bai, J. Zhang, S. T. Xia, Y. Wang. FMM-attack: A flow-based multi-modal adversarial attack on video-based LLMs, [Online], Available: https:\/\/arxiv.org\/abs\/2403.13507, 2024."},{"key":"1667_CR12","unstructured":"L. Wang, T. Zhang, Y. Qu, S. Liang, Y. Chen, A. Liu, X. Liu, D. Tao. Black-box adversarial attack on vision language models for autonomous driving, [Online], Available: https:\/\/arxiv.org\/abs\/2501.13563, 2025."},{"key":"1667_CR13","doi-asserted-by":"publisher","first-page":"1022","DOI":"10.1109\/CVPR52729.2023.00105","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Dong","year":"2023","unstructured":"Y. Dong, C. Kang, J. Zhang, Z. Zhu, Y. Wang, X. Yang, H. Su, X. Wei, J. Zhu. Benchmarking robustness of 3D object detection to common corruptions in autonomous driving. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Vancouver, Canada, pp. 1022\u20131032, 2023. DOI: https:\/\/doi.org\/10.1109\/CVPR52729.2023.00105."},{"key":"1667_CR14","unstructured":"X. Wei, B. Pu, J. Lu, B. Wu. Physically adversarial attacks and defenses in computer vision: A survey, [Online], Available: https:\/\/arxiv.org\/abs\/2211.01671v1, 2022."},{"key":"1667_CR15","doi-asserted-by":"publisher","first-page":"21527","DOI":"10.1609\/aaai.v38i19.30150","volume-title":"Proceedings of the 38th AAAI Conference on Artificial Intelligence","author":"X Qi","year":"2024","unstructured":"X. Qi, K. Huang, A. Panda, P. Henderson, M. Wang, P. Mittal. Visual adversarial examples jailbreak aligned large language models. In Proceedings of the 38th AAAI Conference on Artificial Intelligence, Vancouver, Canada, pp. 21527\u201321536, 2024. DOI: https:\/\/doi.org\/10.1609\/aaai.v38i19.30150."},{"key":"1667_CR16","unstructured":"Y. Dong, H. Chen, J. Chen, Z. Fang, X. Yang, Y. Zhang, Y. Tian, H. Su, J. Zhu. How robust is Google\u2019s bard to adversarial image attacks? [Online], Available: https:\/\/arxiv.org\/abs\/2309.11751, 2023."},{"key":"1667_CR17","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"N Carlini","year":"2023","unstructured":"N. Carlini, M. Nasr, C. A. Choquette-Choo, M. Jagielski, I. Gao, A. Awadalla, P. W. Koh, D. Ippolito, K. Lee, F. Tramer, L. Schmidt. Are aligned neural networks adversarially aligned? In Proceedings of the 37th International Conference on Neural Information Processing Systems, New Orleans, USA, Article number 2687, 2023."},{"key":"1667_CR18","unstructured":"X. Jia, S. Gao, S. Qin, T. Pang, C. Du, Y. Huang, X. Li, Y. Li, B. Li, Y. Liu. Adversarial attacks against closed-source MLLMs via feature optimal alignment, [Online], Available: https:\/\/arxiv.org\/abs\/2505.21494, 2025."},{"issue":"6","key":"1667_CR19","doi-asserted-by":"publisher","first-page":"999","DOI":"10.1007\/s11633-025-1558-0","volume":"22","author":"Y Hu","year":"2025","unstructured":"Y. Hu, L. Hu, Q. Kong, B. Fan. A survey on end-to-end perception and prediction for autonomous driving. Machine Intelligence Research, vol. 22, no. 6, pp. 999\u20131030, 2025. DOI: https:\/\/doi.org\/10.1007\/s11633-025-1558-0.","journal-title":"Machine Intelligence Research"},{"issue":"6","key":"1667_CR20","doi-asserted-by":"publisher","first-page":"550","DOI":"10.1007\/s11633-022-1339-y","volume":"19","author":"D Wu","year":"2022","unstructured":"D. Wu, M. W. Liao, W. T. Zhang, X. G. Wang, X. Bai, W. Q. Cheng, W. Y. Liu. YOLOP: You only look once for panoptic driving perception. Machine Intelligence Research, vol. 19, no. 6, pp. 550\u2013562, 2022. DOI: https:\/\/doi.org\/10.1007\/s11633-022-1339-y.","journal-title":"Machine Intelligence Research"},{"key":"1667_CR21","doi-asserted-by":"publisher","first-page":"8565","DOI":"10.1109\/CVPR46437.2021.00846","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J Wang","year":"2021","unstructured":"J. Wang, A. Liu, Z. Yin, S. Liu, S. Tang, X. Liu. Dual attention suppression attack: Generate adversarial camouflage in physical world. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Nashville, USA, pp. 8565\u20138574, 2021. DOI: https:\/\/doi.org\/10.1109\/CVPR46437.2021.00846."},{"key":"1667_CR22","volume-title":"Proceedings of the 32nd USENIX Conference on Security Symposium","author":"A Liu","year":"2023","unstructured":"A. Liu, J. Guo, J. Wang, S. Liang, R. Tao, W. Zhou, C. Liu, X. Liu, D. Tao. X-adv: Physical adversarial object attacks against X-ray prohibited item detection. In Proceedings of the 32nd USENIX Conference on Security Symposium, Anaheim, USA, Article number 212, 2023."},{"key":"1667_CR23","doi-asserted-by":"publisher","first-page":"12324","DOI":"10.1109\/CVPR52729.2023.01186","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"S Li","year":"2023","unstructured":"S. Li, S. Zhang, G. Chen, D. Wang, P. Feng, J. Wang, A. Liu, X. Yi, X. Liu. Towards benchmarking and assessing visual naturalness of physical world adversarial attacks. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Vancouver, Canada, pp. 12324\u201312333, 2023. DOI: https:\/\/doi.org\/10.1109\/CVPR52729.2023.01186."},{"key":"1667_CR24","doi-asserted-by":"publisher","first-page":"1028","DOI":"10.1609\/aaai.v33i01.33011028","volume-title":"Proceedings of the 33rd AAAI Conference on Artificial Intelligence","author":"A Liu","year":"2019","unstructured":"A. Liu, X. Liu, J. Fan, Y. Ma, A. Zhang, H. Xie, D. Tao. Perceptual-sensitive GAN for generating adversarial patches. In Proceedings of the 33rd AAAI Conference on Artificial Intelligence, Honolulu, USA, pp. 1028\u20131035, 2019. DOI: https:\/\/doi.org\/10.1609\/aaai.v33i01.33011028."},{"key":"1667_CR25","doi-asserted-by":"publisher","first-page":"395","DOI":"10.1007\/978-3-030-58601-0_24","volume-title":"Proceedings of the 16th European Conference on Computer Vision","author":"A Liu","year":"2020","unstructured":"A. Liu, J. Wang, X. Liu, B. Cao, C. Zhang, H. Yu. Bias-based universal adversarial patch attack for automatic check-out. In Proceedings of the 16th European Conference on Computer Vision, Springer, Glasgow, UK, pp. 395\u2013410, 2020. DOI: https:\/\/doi.org\/10.1007\/978-3-030-58601-0_24."},{"key":"1667_CR26","doi-asserted-by":"publisher","first-page":"5005","DOI":"10.1145\/3503161.3547801","volume-title":"Proceedings of the 30th ACM International Conference on Multimedia","author":"J Zhang","year":"2022","unstructured":"J. Zhang, Q. Yi, J. Sang. Towards adversarial attack on vision-language pre-training models. In Proceedings of the 30th ACM International Conference on Multimedia, Lisboa, Portugal, pp. 5005\u20135013, 2022. DOI: https:\/\/doi.org\/10.1145\/3503161.3547801."},{"key":"1667_CR27","doi-asserted-by":"publisher","first-page":"1722","DOI":"10.1109\/SP54263.2024.00102","volume-title":"Proceedings of IEEE Symposium on Security and Privacy","author":"H Wang","year":"2024","unstructured":"H. Wang, K. Dong, Z. Zhu, H. Qin, A. Liu, X. Fang, J. Wang, X. Liu. Transferable multimodal attack on vision-language pre-training models. In Proceedings of IEEE Symposium on Security and Privacy, San Francisco, USA, pp. 1722\u20131740, 2024. DOI: https:\/\/doi.org\/10.1109\/SP54263.2024.00102."},{"key":"1667_CR28","doi-asserted-by":"publisher","first-page":"748","DOI":"10.1145\/3664647.3681538","volume-title":"Proceedings of the 32nd ACM International Conference on Multimedia","author":"W Xu","year":"2024","unstructured":"W. Xu, K. Chen, Z. Gao, Z. Wei, J. Chen, Y. G. Jiang. Highly transferable diffusion-based unrestricted adversarial attack on pre-trained vision-language models. In Proceedings of the 32nd ACM International Conference on Multimedia, Melbourne, Australia, pp. 748\u2013757, 2024. DOI: https:\/\/doi.org\/10.1145\/3664647.3681538."},{"issue":"5","key":"1667_CR29","doi-asserted-by":"publisher","first-page":"831","DOI":"10.1007\/s11633-022-1411-7","volume":"21","author":"A Mumuni","year":"2024","unstructured":"A. Mumuni, F. Mumuni, N. K. Gerrar. A survey of synthetic data augmentation methods in machine vision. Machine Intelligence Research, vol. 21, no. 5, pp. 831\u2013869, 2024. DOI: https:\/\/doi.org\/10.1007\/s11633-022-1411-7.","journal-title":"Machine Intelligence Research"},{"key":"1667_CR30","doi-asserted-by":"publisher","first-page":"442","DOI":"10.1007\/978-3-031-72998-025","volume-title":"Proceedings of the 18th European Conference on Computer Vision","author":"S Gao","year":"2025","unstructured":"S. Gao, X. Jia, X. Ren, I. Tsang, Q. Guo. Boosting transferability in vision-language attacks via diversification along the intersection region of adversarial trajectory. In Proceedings of the 18th European Conference on Computer Vision, Springer, Milan, Italy, pp. 442\u2013460, 2025. DOI: https:\/\/doi.org\/10.1007\/978-3-031-72998-025."},{"key":"1667_CR31","doi-asserted-by":"publisher","first-page":"6311","DOI":"10.1145\/3581783.3612454","volume-title":"Proceedings of the 31st ACM International Conference on Multimedia","author":"Z Zhou","year":"2023","unstructured":"Z. Zhou, S. Hu, M. Li, H. Zhang, Y. Zhang, H. Jin. AdvCLIP: Downstream-agnostic adversarial examples in multimodal contrastive learning. In Proceedings of the 31st ACM International Conference on Multimedia, Ottawa, Canada, pp. 6311\u20136320, 2023. DOI: https:\/\/doi.org\/10.1145\/3581783.3612454."},{"key":"1667_CR32","first-page":"43685","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"C Schlarmann","year":"2024","unstructured":"C. Schlarmann, N. D. Singh, F. Croce, M. Hein. Robust CLIP: Unsupervised adversarial fine-tuning of vision embeddings for robust large vision-language models. In Proceedings of the 41st International Conference on Machine Learning, Vienna, Austria, pp. 43685\u201343704, 2024."},{"key":"1667_CR33","first-page":"2443","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"L Bailey","year":"2024","unstructured":"L. Bailey, E. Ong, S. Russell, S. Emmons. Image hijacks: Adversarial images can control generative models at runtime. In Proceedings of the 41st International Conference on Machine Learning, Vienna, Austria, pp. 2443\u20132455, 2024."},{"key":"1667_CR34","unstructured":"J. Zhang, J. Ye, X. Ma, Y. Li, Y. Yang, J. Sang, D. Y. Yeung. AnyAttack: Towards large-scale self-supervised generation of targeted adversarial examples for vision-language models, [Online], Available: https:\/\/arxiv.org\/abs\/2410.05346v1, 2024."},{"key":"1667_CR35","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Y Zhao","year":"2023","unstructured":"Y. Zhao, T. Pang, C. Du, X. Yang, C. Li, N. M. Cheung, M. Lin. On evaluating adversarial robustness of large vision-language models. In Proceedings of the 37th International Conference on Neural Information Processing Systems, New Orleans, USA, Article number 2355, 2023."},{"issue":"3","key":"1667_CR36","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1007\/s11633-022-1330-7","volume":"19","author":"M Ren","year":"2022","unstructured":"M. Ren, Y. L. Wang, Z. F. He. Towards interpretable defense against adversarial attacks via causal inference. Machine Intelligence Research, vol. 19, no. 3, pp. 209\u2013226, 2022. DOI: https:\/\/doi.org\/10.1007\/s11633-022-1330-7.","journal-title":"Machine Intelligence Research"},{"key":"1667_CR37","unstructured":"Z. Ni, R. Ye, Y. Wei, Z. Xiang, Y. Wang, S. Chen. Physical backdoor attack can jeopardize driving with vision-large-language models, [Online], Available: https:\/\/arxiv.org\/abs\/2404.12916, 2024."},{"key":"1667_CR38","unstructured":"N. Chung, S. Gao, T. A. Vu, J. Zhang, A. Liu, Y. Lin, J. S. Dong, Q. Guo. Towards transferable attacks against vision-LLMs in autonomous driving with typography, [Online], Available: https:\/\/arxiv.org\/abs\/2405.14169, 2024."},{"key":"1667_CR39","unstructured":"J. Fu, Z. Chen, K. Jiang, H. Guo, S. Gao, W. Zhang. PG-attack: A precision-guided adversarial attack framework against vision foundation models for autonomous driving, [Online], Available: https:\/\/arxiv.org\/abs\/2407.13111, 2024."},{"key":"1667_CR40","doi-asserted-by":"publisher","first-page":"292","DOI":"10.1007\/978-3-031-73347-517","volume-title":"Proceedings of the 18th European Conference on Computer Vision","author":"M Nie","year":"2025","unstructured":"M. Nie, R. Peng, C. Wang, X. Cai, J. Han, H. Xu, L. Zhang. Reason2Drive: Towards interpretable and chain-based reasoning for autonomous driving. In Proceedings of the 18th European Conference on Computer Vision, Springer, Milan, Italy, pp. 292\u2013308, 2025. DOI: https:\/\/doi.org\/10.1007\/978-3-031-73347-517."},{"key":"1667_CR41","doi-asserted-by":"publisher","first-page":"252","DOI":"10.1007\/978-3-031-72980-5_15","volume-title":"Proceedings of the 18th European Conference on Computer Vision","author":"A M Marcu","year":"2024","unstructured":"A. M. Marcu, L. Chen, J. H\u00fcnermann, A. Karnsund, B. Hanotte, P. Chidananda, S. Nair, V. Badrinarayanan, A. Kendall, J. Shotton, E. Arani, O. Sinavski. LingoQA: Visual question answering for autonomous driving. In Proceedings of the 18th European Conference on Computer Vision, Springer, Milan, Italy, pp. 252\u2013269, 2024. https:\/\/doi.org\/10.1007\/978-3-031-72980-5_15."},{"key":"1667_CR42","doi-asserted-by":"publisher","first-page":"403","DOI":"10.1007\/978-3-031-72995-9_23","volume-title":"Proceedings of the 18th European Conference on Computer Vision","author":"Y Ma","year":"2025","unstructured":"Y. Ma, Y. Cao, J. Sun, M. Pavone, C. Xiao. Dolphins: Multimodal language model for driving. In Proceedings of the 18th European Conference on Computer Vision, Springer, Milan, Italy, pp. 403\u2013420, 2025. DOI: https:\/\/doi.org\/10.1007\/978-3-031-72995-9_23."},{"key":"1667_CR43","doi-asserted-by":"publisher","first-page":"5154","DOI":"10.1109\/ITSC57777.2023.10421993","volume-title":"Proceedings of the 26th International Conference on Intelligent Transportation Systems","author":"J Liu","year":"2023","unstructured":"J. Liu, P. Hang, X. Qi, J. Wang, J. Sun. MTD-GPT: A multi-task decision-making GPT model for autonomous driving at unsignalized intersections. In Proceedings of the 26th International Conference on Intelligent Transportation Systems, IEEE, Bilbao, Spain, pp. 5154\u20135161, 2023. DOI: https:\/\/doi.org\/10.1109\/ITSC57777.2023.10421993."},{"key":"1667_CR44","doi-asserted-by":"publisher","first-page":"256","DOI":"10.1007\/978-3-031-72943-0_15","volume-title":"Proceedings of the 18th European Conference on Computer Vision","author":"C Sima","year":"2025","unstructured":"C. Sima, K. Renz, K. Chitta, L. Chen, H. Zhang, C. Xie, J. Bei\u00dfwenger, P. Luo, A. Geiger, H. Li. DriveLM: Driving with graph visual question answering. In Proceedings of the 18th European Conference on Computer Vision, Springer, Milan, Italy, pp. 256\u2013274, 2025. DOI: https:\/\/doi.org\/10.1007\/978-3-031-72943-0_15."},{"key":"1667_CR45","doi-asserted-by":"publisher","unstructured":"E. Cui, W. Wang, Z. Li, J. Xie, H. Zou, H. Deng, G. Luo, L. Lu, X. Zhu, J. Dai. DriveMLM: Aligning multimodal large language models with behavioral planning states for autonomous driving. Visual Intelligence, vol. 3, no. 1, Article number 22, 2025. DOI: https:\/\/doi.org\/10.1007\/S44267-025-00095-W.","DOI":"10.1007\/S44267-025-00095-W"},{"issue":"8017","key":"1667_CR46","doi-asserted-by":"publisher","first-page":"625","DOI":"10.1038\/s41586-024-07421-0","volume":"630","author":"S Farquhar","year":"2024","unstructured":"S. Farquhar, J. Kossen, L. Kuhn, Y. Gal. Detecting hallucinations in large language models using semantic entropy. Nature, vol. 630, no. 8017, pp. 625\u2013630, 2024. DOI: https:\/\/doi.org\/10.1038\/S41586-024-07421-0.","journal-title":"Nature"},{"key":"1667_CR47","unstructured":"OpenAI. GPT-4 technical report, [Online], Available: https:\/\/arxiv.org\/abs\/2303.08774, 2023."},{"issue":"4","key":"1667_CR48","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Z. Wang, A. C. Bovik, H. R. Sheikh, E. P. Simoncelli. Image quality assessment: From error visibility to structural similarity. IEEE Transactions on Image Processing, vol. 13, no. 4, pp. 600\u2013612, 2004. DOI: https:\/\/doi.org\/10.1109\/TIP.2003.819861.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1667_CR49","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"D Zhu","year":"2024","unstructured":"D. Zhu, J. Chen, X. Shen, X. Li, M. Elhoseiny. MiniG-PT-4: Enhancing vision-language understanding with advanced large language models. In Proceedings of the 12th International Conference on Learning Representations, Vienna, Austria, 2024."},{"key":"1667_CR50","unstructured":"T. Gong, C. Lyu, S. Zhang, Y. Wang, M. Zheng, Q. Zhao, K. Liu, W. Zhang, P. Luo, K. Chen. MultiModal-GPT: A vision and language model for dialogue with humans, [Online], Available: https:\/\/arxiv.org\/abs\/2305.04790, 2023."},{"key":"1667_CR51","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"H Liu","year":"2023","unstructured":"H. Liu, C. Li, Q. Wu, Y. J. Lee. Visual instruction tuning. In Proceedings of the 37th International Conference on Neural Information Processing Systems, New Orleans, USA, Article number 1516, 2023."},{"key":"1667_CR52","first-page":"1","volume-title":"Proceedings of the 1st Annual Conference on Robot Learning","author":"A Dosovitskiy","year":"2017","unstructured":"A. Dosovitskiy, G. Ros, F. Codevilla, A. M. L\u00f3pez, V. Koltun. CARLA: An open urban driving simulator. In Proceedings of the 1st Annual Conference on Robot Learning, Mountain View, USA, pp. 1\u201316, 2017."},{"key":"1667_CR53","volume-title":"Proceedings of the 3rd International Conference on Learning Representations","author":"I J Goodfellow","year":"2015","unstructured":"I. J. Goodfellow, J. Shlens, C. Szegedy. Explaining and harnessing adversarial examples. In Proceedings of the 3rd International Conference on Learning Representations, San Diego, USA, 2015."},{"key":"1667_CR54","volume-title":"Proceedings of the 6th International Conference on Learning Representations","author":"A Madry","year":"2018","unstructured":"A. Madry, A. Makelov, L. Schmidt, D. Tsipras, A. Vladu. Towards deep learning models resistant to adversarial attacks. In Proceedings of the 6th International Conference on Learning Representations, Vancouver, Canada, 2018."},{"key":"1667_CR55","doi-asserted-by":"publisher","first-page":"14679","DOI":"10.1109\/CVPR52734.2025.01368","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"P Xie","year":"2025","unstructured":"P. Xie, Y. Bie, J. Mao, Y. Song, Y. Wang, H. Chen, K. Chen. Chain of attack: On the robustness of vision-language models against transfer-based adversarial attacks. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Nashville, USA, pp. 14679\u201314689, 2025. DOI: https:\/\/doi.org\/10.1109\/CVPR52734.2025.01368."},{"key":"1667_CR56","first-page":"3839","volume-title":"Proceedings of the 32nd International Conference on Neural Information Processing Systems","author":"K Scaman","year":"2018","unstructured":"K. Scaman, A. Virmaux. Lipschitz regularity of deep neural networks: Analysis and efficient estimation. In Proceedings of the 32nd International Conference on Neural Information Processing Systems, Montr\u00e9al, Canada, pp. 3839\u20133848, 2018."},{"key":"1667_CR57","volume-title":"Proceedings of the 8th International Conference on Learning Representations","author":"F Latorre","year":"2020","unstructured":"F. Latorre, P. Rolland, V. Cevher. Lipschitz constant estimation of neural networks via sparse polynomial optimization. In Proceedings of the 8th International Conference on Learning Representations, Addis Ababa, Ethiopia, 2020."},{"key":"1667_CR58","first-page":"371","volume":"9","author":"G Shafer","year":"2008","unstructured":"G. Shafer, V. Vovk. A tutorial on conformal prediction. The Journal of Machine Learning Research, vol. 9, pp. 371\u2013421, 2008","journal-title":"The Journal of Machine Learning Research"},{"key":"1667_CR59","unstructured":"R. I. Oliveira, P. Orenstein, T. Ramos, J. V. Romano. Split conformal prediction and non-exchangeable data. The Journal of Machine Learning Research, vol. 25, no. 1, Article number 225, 2024"},{"key":"1667_CR60","unstructured":"PIXLOOP, [Online], Available: https:\/\/www.pixmoving-com\/pixloop, 2024."},{"key":"1667_CR61","first-page":"3309","volume-title":"Proceedings of the 300th USENIX Security Symposium","author":"T Sato","year":"2021","unstructured":"T. Sato, J. Shen, N. Wang, Y. Jia, X. Lin, Q. A. Chen. Dirty road can attack: Security of deep learning based automated lane centering under physical-world attack. In Proceedings of the 300th USENIX Security Symposium, pp. 3309\u20133326, 2021."}],"container-title":["Machine Intelligence Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-026-1667-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11633-026-1667-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-026-1667-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T08:02:56Z","timestamp":1784793776000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11633-026-1667-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,4]]},"references-count":61,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,8]]}},"alternative-id":["1667"],"URL":"https:\/\/doi.org\/10.1007\/s11633-026-1667-4","relation":{},"ISSN":["2731-538X","2731-5398"],"issn-type":[{"value":"2731-538X","type":"print"},{"value":"2731-5398","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,4]]},"assertion":[{"value":"28 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declared that they have no conflicts of interest to this work.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations of conflict of interest"}}]}}