{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T21:45:55Z","timestamp":1782510355335,"version":"3.54.5"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.eswa.2026.133310","type":"journal-article","created":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T15:48:14Z","timestamp":1781884094000},"page":"133310","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["VLM-driven causal auditing: A counterfactual framework for revealing causal confusion in end-to-end driving"],"prefix":"10.1016","volume":"331","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7889-4695","authenticated-orcid":false,"given":"Guofa","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4872-9839","authenticated-orcid":false,"given":"Yuhao","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9143-5449","authenticated-orcid":false,"given":"Chen","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4822-4057","authenticated-orcid":false,"given":"Qi","family":"Lan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3718-5593","authenticated-orcid":false,"given":"Jie","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5926-7856","authenticated-orcid":false,"given":"Xiangyun","family":"Ren","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133310_bib0001","unstructured":"Bai, S., Cai, Y., Chen, R. et al. (2025). Qwen3-VL technical report. arXiv: 2511.21631."},{"key":"10.1016\/j.eswa.2026.133310_bib0002","series-title":"Proceedings of the 2025\u202fIEEE\/RSJ international conference on intelligent robots and systems (IROS)","first-page":"13505","article-title":"GABRIL: Gaze-based regularization for mitigating causal confusion in imitation learning","author":"Banayeeanzade","year":"2025"},{"key":"10.1016\/j.eswa.2026.133310_bib0003","series-title":"2018 IEEE International conference on robotics and automation (ICRA)","first-page":"4701","article-title":"VisualBackProp: Efficient visualization of CNNs for autonomous driving","author":"Bojarski","year":"2018"},{"key":"10.1016\/j.eswa.2026.133310_bib0004","unstructured":"Bojarski, M., Yeres, P., Choroma\u0144ska, A. et al. (2017). Explaining how a deep neural network trained with end-to-end learning steers a car. arXiv: 1704.07911."},{"key":"10.1016\/j.eswa.2026.133310_bib0005","series-title":"Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence","first-page":"6276","article-title":"Counterfactuals in explainable artificial intelligence (XAI): Evidence from human reasoning","author":"Byrne","year":"2019"},{"issue":"12","key":"10.1016\/j.eswa.2026.133310_bib0006","doi-asserted-by":"crossref","first-page":"10164","DOI":"10.1109\/TPAMI.2024.3435937","article-title":"End-to-end autonomous driving: Challenges and frontiers","volume":"46","author":"Chen","year":"2024","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"1","key":"10.1016\/j.eswa.2026.133310_bib0007","doi-asserted-by":"crossref","first-page":"103","DOI":"10.1109\/TIV.2023.3318070","article-title":"Recent advancements in end-to-end autonomous driving using deep learning: A survey","volume":"9","author":"Chib","year":"2024","journal-title":"IEEE Transactions on Intelligent Vehicles"},{"key":"10.1016\/j.eswa.2026.133310_bib0008","series-title":"2021 IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"15773","article-title":"NEAT: Neural attention fields for end-to-end autonomous driving","author":"Chitta","year":"2021"},{"key":"10.1016\/j.eswa.2026.133310_bib0009","doi-asserted-by":"crossref","first-page":"591","DOI":"10.1007\/s42154-025-00361-z","article-title":"A survey of human intelligence augmented artificial intelligence: An autonomous driving perspective","volume":"8","author":"Li","year":"2025","journal-title":"Automotive Innovation"},{"key":"10.1016\/j.eswa.2026.133310_bib0010","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops (CVPRW)","first-page":"1389","article-title":"Explaining autonomous driving by learning end-to-end visual attention","author":"Cultrera","year":"2020"},{"key":"10.1016\/j.eswa.2026.133310_bib0011","series-title":"Proceedings of the 1st Annual Conference on Robot Learning","first-page":"1","article-title":"CARLA: An open urban driving simulator","volume":"78","author":"Dosovitskiy","year":"2017"},{"key":"10.1016\/j.eswa.2026.133310_bib0012","unstructured":"GLM-V Team, Hong, W., Gu, X., Pan, Z. et al. (2026). GLM-5V-Turbo: Toward a native foundation model for multimodal agents. arXiv: 2604.26752."},{"key":"10.1016\/j.eswa.2026.133310_bib0013","series-title":"Proceedings of the 36th International Conference on Machine Learning","first-page":"2376","article-title":"Counterfactual visual explanations","volume":"97","author":"Goyal","year":"2019"},{"key":"10.1016\/j.eswa.2026.133310_bib0014","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"17853","article-title":"Planning-oriented autonomous driving","author":"Hu","year":"2023"},{"key":"10.1016\/j.eswa.2026.133310_bib0015","series-title":"Advances in Neural Information Processing Systems","first-page":"49956","article-title":"Prioritizing perception-guided self-supervision: A new paradigm for causal modeling in end-to-end autonomous driving","volume":"38","author":"Huang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133310_bib0016","series-title":"Proceedings of the 2025\u202fIEEE\/RSJ international conference on intelligent robots and systems (IROS)","first-page":"20501","article-title":"DriveLMM-o1: A step-by-step reasoning dataset and large multimodal model for driving scenario understanding","author":"Ishaq","year":"2025"},{"key":"10.1016\/j.eswa.2026.133310_bib0017","series-title":"Computer vision \u2013 ECCV 2022","first-page":"387","article-title":"STEEX: Steering counterfactual explanations with semantics","author":"Jacob","year":"2022"},{"key":"10.1016\/j.eswa.2026.133310_bib0018","series-title":"2023 IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"8206","article-title":"Hidden biases of end-to-end driving models","author":"Jaeger","year":"2023"},{"key":"10.1016\/j.eswa.2026.133310_bib0019","series-title":"Advances in Neural Information Processing Systems","first-page":"819","article-title":"Bench2Drive: Towards multi-ability benchmarking of closed-loop end-to-end autonomous driving","volume":"37","author":"Jia","year":"2024"},{"key":"10.1016\/j.eswa.2026.133310_bib0020","series-title":"Proceedings of the 2023\u202fIEEE\/CVF international conference on computer vision (ICCV)","first-page":"8306","article-title":"VAD: Vectorized scene representation for efficient autonomous driving","author":"Jiang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133310_bib0021","series-title":"2023 IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"3992","article-title":"Segment anything","author":"Kirillov","year":"2023"},{"issue":"12","key":"10.1016\/j.eswa.2026.133310_bib0022","doi-asserted-by":"crossref","first-page":"19342","DOI":"10.1109\/TITS.2024.3474469","article-title":"Explainable AI for safe and trustworthy autonomous driving: A systematic review","volume":"25","author":"Kuznietsov","year":"2024","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"issue":"10","key":"10.1016\/j.eswa.2026.133310_bib0023","doi-asserted-by":"crossref","first-page":"14905","DOI":"10.1109\/TITS.2024.3393634","article-title":"SVCE: Shapley value guided counterfactual explanation for machine learning-based autonomous driving","volume":"25","author":"Li","year":"2024","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"issue":"11","key":"10.1016\/j.eswa.2026.133310_bib0024","doi-asserted-by":"crossref","first-page":"16704","DOI":"10.1109\/TVT.2025.3575760","article-title":"SVAP: Shapley value guided attribution prior for neural network-based autonomous driving","volume":"74","author":"Li","year":"2025","journal-title":"IEEE Transactions on Vehicular Technology"},{"issue":"11","key":"10.1016\/j.eswa.2026.133310_bib0025","doi-asserted-by":"crossref","first-page":"6881","DOI":"10.1109\/TIV.2024.3390426","article-title":"Relevance inference based on direct contribution: Counterfactual explanation to deep networks for intelligent decision-making","volume":"9","author":"Li","year":"2024","journal-title":"IEEE Transactions on Intelligent Vehicles"},{"key":"10.1016\/j.eswa.2026.133310_bib0026","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision workshops (ICCVW)","first-page":"794","article-title":"Interpretable decision-making for end-to-end autonomous driving","author":"Mirzaie","year":"2025"},{"key":"10.1016\/j.eswa.2026.133310_bib0027","series-title":"Computer vision \u2013 ECCV 2024","first-page":"292","article-title":"Reason2Drive: Towards interpretable and chain-based reasoning for autonomous driving","author":"Nie","year":"2024"},{"issue":"8","key":"10.1016\/j.eswa.2026.133310_bib0028","doi-asserted-by":"crossref","first-page":"10142","DOI":"10.1109\/TITS.2021.3122865","article-title":"Explanations in autonomous driving: A survey","volume":"23","author":"Omeiza","year":"2022","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.eswa.2026.133310_bib0029","series-title":"Causality: Models, Reasoning, and Inference","author":"Pearl","year":"2009"},{"key":"10.1016\/j.eswa.2026.133310_bib0030","series-title":"Proceedings of the 2024\u202fIEEE intelligent vehicles symposium (IV)","first-page":"2353","article-title":"Guiding attention in end-to-end driving models","author":"Porres","year":"2024"},{"key":"10.1016\/j.eswa.2026.133310_bib0031","series-title":"Proceedings of The 6th Conference on Robot Learning","first-page":"459","article-title":"PlanT: Explainable planning transformers via object-level representations","volume":"205","author":"Renz","year":"2023"},{"key":"10.1016\/j.eswa.2026.133310_bib0032","series-title":"2023 IEEE Intelligent Transportation Systems Conference (ITSC)","first-page":"5655","article-title":"SAFE: Saliency-aware counterfactual explanations for DNN-based automated driving systems","author":"Samadi","year":"2023"},{"issue":"2","key":"10.1016\/j.eswa.2026.133310_bib0033","doi-asserted-by":"crossref","first-page":"336","DOI":"10.1007\/s11263-019-01228-7","article-title":"Grad-CAM: Visual explanations from deep networks via gradient-based localization","volume":"128","author":"Selvaraju","year":"2020","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.133310_bib0034","series-title":"Proceedings of The 6th Conference on Robot Learning","first-page":"726","article-title":"Safety-enhanced autonomous driving using interpretable sensor fusion transformer","volume":"205","author":"Shao","year":"2023"},{"key":"10.1016\/j.eswa.2026.133310_bib0035","series-title":"Computer vision \u2013 ECCV 2024","first-page":"256","article-title":"DriveLM: Driving with graph visual question answering","author":"Sima","year":"2024"},{"key":"10.1016\/j.eswa.2026.133310_bib0036","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.111638","article-title":"Semantic shapley-based counterfactual explanations for end-to-end autonomous driving","volume":"159","author":"Sun","year":"2025","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.133310_bib0037","doi-asserted-by":"crossref","unstructured":"Sun, W., Lin, X., Shi, Y., Zhang, C., Wu, H., & Zheng, S. (2024). SparseDrive: End-to-end autonomous driving via sparse scene representation. arXiv: 2405.19620.","DOI":"10.1109\/ICRA55743.2025.11128800"},{"key":"10.1016\/j.eswa.2026.133310_bib0038","series-title":"Proceedings of the 2022\u202fIEEE\/CVF winter conference on applications of computer vision (WACV)","first-page":"3172","article-title":"Resolution-robust large mask inpainting with Fourier convolutions","author":"Suvorov","year":"2022"},{"key":"10.1016\/j.eswa.2026.133310_bib0039","unstructured":"Verma, S., Dickerson, J., & Hines, K. E. (2021). Counterfactual explanations for machine learning: Challenges revisited. arXiv: 2106.07756."},{"key":"10.1016\/j.eswa.2026.133310_bib0040","series-title":"Advances in Neural Information Processing Systems","first-page":"2564","article-title":"Fighting copycat agents in behavioral cloning from observation histories","volume":"33","author":"Wen","year":"2020"},{"key":"10.1016\/j.eswa.2026.133310_bib0041","series-title":"Advances in Neural Information Processing Systems","first-page":"6119","article-title":"Trajectory-guided control prediction for end-to-end autonomous driving: A simple yet strong baseline","volume":"35","author":"Wu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133310_bib0042","unstructured":"Xiaomi MiMo Team (2026). Xiaomi MiMo-V2-Omni: See, hear, act in the agentic era. https:\/\/mimo.xiaomi.com\/mimo-v2-omni."},{"key":"10.1016\/j.eswa.2026.133310_bib0043","series-title":"Proceedings of the 2025\u202fIEEE\/CVF winter conference on applications of computer vision workshops","first-page":"911","article-title":"OpenEMMA: Open-source multimodal model for end-to-end autonomous driving","author":"Xing","year":"2025"},{"key":"10.1016\/j.eswa.2026.133310_bib0044","unstructured":"Yu, T., Feng, R., Feng, R. et al. (2023). Inpaint anything: Segment anything meets image inpainting. arXiv: 2304.06790."},{"key":"10.1016\/j.eswa.2026.133310_bib0045","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"15062","article-title":"OCTET: Object-aware counterfactual explanations","author":"Zemni","year":"2023"},{"key":"10.1016\/j.eswa.2026.133310_bib0046","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"586","article-title":"The unreasonable effectiveness of deep features as a perceptual metric","author":"Zhang","year":"2018"},{"issue":"5","key":"10.1016\/j.eswa.2026.133310_bib0047","doi-asserted-by":"crossref","first-page":"6322","DOI":"10.1109\/TNNLS.2022.3213246","article-title":"Imitation learning: Progress, taxonomies and challenges","volume":"35","author":"Zheng","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.133310_bib0048","first-page":"1","article-title":"Vision language models in autonomous driving: A survey and outlook","author":"Zhou","year":"2024","journal-title":"IEEE Transactions on Intelligent Vehicles"},{"issue":"9","key":"10.1016\/j.eswa.2026.133310_bib0049","doi-asserted-by":"crossref","first-page":"14043","DOI":"10.1109\/TITS.2021.3134702","article-title":"A survey of deep RL and IL for autonomous driving policy learning","volume":"23","author":"Zhu","year":"2022","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.eswa.2026.133310_bib0050","unstructured":"Zimmerlin, J., Bei\u00dfwenger, J., Jaeger, B., Geiger, A., & Chitta, K. (2024). Hidden biases of end-to-end driving datasets. arxiv: 2412.09602."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426022190?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426022190?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T20:54:06Z","timestamp":1782507246000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426022190"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":50,"alternative-id":["S0957417426022190"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133310","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"VLM-driven causal auditing: A counterfactual framework for revealing causal confusion in end-to-end driving","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133310","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133310"}}