{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T18:08:08Z","timestamp":1778782088786,"version":"3.51.4"},"reference-count":196,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,4,10]],"date-time":"2026-04-10T00:00:00Z","timestamp":1775779200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100004489","name":"Mitacs Inc","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004489","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004054","name":"King Abdulaziz University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004054","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.engappai.2026.114805","type":"journal-article","created":{"date-parts":[[2026,4,11]],"date-time":"2026-04-11T09:40:13Z","timestamp":1775900413000},"page":"114805","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"P2","title":["Foundation models for autonomous driving: A comprehensive survey"],"prefix":"10.1016","volume":"176","author":[{"given":"Sonda","family":"Fourati","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4378-9999","authenticated-orcid":false,"given":"Wael","family":"Jaafar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Noura","family":"Baccar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Safwan","family":"Alfattani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rami","family":"Langar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"issue":"3","key":"10.1016\/j.engappai.2026.114805_b1","doi-asserted-by":"crossref","first-page":"706","DOI":"10.3390\/s21030706","article-title":"A survey of autonomous vehicles: Enabling communication technologies and challenges","volume":"21","author":"Ahangar","year":"2021","journal-title":"Sensors"},{"key":"10.1016\/j.engappai.2026.114805_b2","series-title":"Lingo-2: Driving with language","author":"AI","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b3","doi-asserted-by":"crossref","unstructured":"Aldeen, M., MohajerAnsari, P., Ma, J., Chowdhury, M., Cheng, L., Pes\u00e9, M.D., 2024. WIP: A First Look At Employing Large Multimodal Models Against Autonomous Vehicle Attacks. In: Proc. Symp. Veh. Secu. Priv. (VehicleSec).","DOI":"10.14722\/vehiclesec.2024.23044"},{"key":"10.1016\/j.engappai.2026.114805_b4","doi-asserted-by":"crossref","unstructured":"Alibeigi, M., Ljungbergh, W., Tonderski, A., Hess, G., Lilja, A., Lindstr\u00f6m, C., Motorniuk, D., Fu, J., Widahl, J., Petersson, C., 2023. Zenseact Open Dataset: A large-scale and diverse multimodal dataset for autonomous driving. In: Proc. IEEE\/CVF Int. Conf. Comput. Vis.. pp. 20178\u201320188.","DOI":"10.1109\/ICCV51070.2023.01846"},{"key":"10.1016\/j.engappai.2026.114805_b5","series-title":"Proc. Int. Conf. Mob. Netw. Wireless Commun.","first-page":"1","article-title":"LLM\u2019s for autonomous driving: A new way to teach machines to drive","author":"Ananthajothi","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b6","series-title":"Frontier AI regulation: Managing emerging risks to public safety","author":"Anderljung","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b7","series-title":"ApolloScape dataset","author":"ApolloScape","year":"2018"},{"key":"10.1016\/j.engappai.2026.114805_b8","series-title":"Hybrid reasoning based on large language models for autonomous car driving","author":"Azarafza","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b9","series-title":"ManeuverGPT agentic control for safe autonomous stunt maneuvers","author":"Azdam","year":"2025"},{"key":"10.1016\/j.engappai.2026.114805_b10","series-title":"Beyond efficiency: A systematic survey of resource-efficient large language models","author":"Bai","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b11","doi-asserted-by":"crossref","unstructured":"Bai, Y., Geng, X., Mangalam, K., Bar, A., Yuille, A., Darrell, T., Malik, J., Efros, A.A., 2024. Sequential modeling enables scalable learning for large vision models. In: Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recogn.. CVPR, pp. 22861\u201322872.","DOI":"10.1109\/CVPR52733.2024.02157"},{"key":"10.1016\/j.engappai.2026.114805_b12","series-title":"Hallucination of multimodal large language models: A survey","author":"Bai","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b13","doi-asserted-by":"crossref","first-page":"73","DOI":"10.58496\/BJAI\/2024\/010","article-title":"Improving bus passenger flow prediction using Bi-LSTM fusion model and SMO algorithm","volume":"2024","author":"Balasubramani","year":"2024","journal-title":"Babylon. J. Artif. Intell."},{"key":"10.1016\/j.engappai.2026.114805_b14","series-title":"Beit: BERT pre-training of image transformers","author":"Bao","year":"2021"},{"key":"10.1016\/j.engappai.2026.114805_b15","series-title":"Proc. Europ. Conf. Comput. Vis.","first-page":"44","article-title":"Segmentation and recognition using structure from motion point clouds","author":"Brostow","year":"2008"},{"issue":"1\u20132","key":"10.1016\/j.engappai.2026.114805_b16","doi-asserted-by":"crossref","first-page":"33","DOI":"10.1177\/02783649231160195","article-title":"Boreas: A multi-season autonomous driving dataset","volume":"42","author":"Burnett","year":"2023","journal-title":"Int. J. Robot. Res."},{"key":"10.1016\/j.engappai.2026.114805_b17","doi-asserted-by":"crossref","unstructured":"Caesar, H., Bankiti, V., Lang, A.H., Vora, S., Liong, V.E., Xu, Q., Krishnan, A., Pan, Y., Baldan, G., Beijbom, O., 2020. nuScenes: A multimodal dataset for autonomous driving. In: Proc. IEEE\/CVF Conf. Comput. Vis. Patt. Recogn.. pp. 11621\u201311631.","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"10.1016\/j.engappai.2026.114805_b18","series-title":"A comprehensive survey of AI-generated content (AIGC): A history of generative AI from GAN to ChatGPT","author":"Cao","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b19","series-title":"Dense reward for free in reinforcement learning from human feedback","author":"Chan","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b20","series-title":"2023 8th International Conference on Information Technology Research","first-page":"1","article-title":"Cross-vit: Cross-attention vision transformer for image duplicate detection","author":"Chandrasiri","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b21","doi-asserted-by":"crossref","unstructured":"Chang, M.-F., Lambert, J., Sangkloy, P., Singh, J., Bak, S., Hartnett, A., Wang, D., Carr, P., Lucey, S., Ramanan, D., et al., 2019. Argoverse: 3D tracking and forecasting with rich maps. In: Proc. IEEE\/CVF Conf. Comput. Vis. Patt. Recogn.. pp. 8748\u20138757.","DOI":"10.1109\/CVPR.2019.00895"},{"issue":"3","key":"10.1016\/j.engappai.2026.114805_b22","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3641289","article-title":"A survey on evaluation of large language models","volume":"15","author":"Chang","year":"2024","journal-title":"ACM Trans. Intell. Syst. Technol."},{"issue":"5","key":"10.1016\/j.engappai.2026.114805_b23","first-page":"10","article-title":"Supplementary material uniter: universal image-text representation learning","volume":"1","author":"Chen","year":"2023","journal-title":"ReCALL"},{"key":"10.1016\/j.engappai.2026.114805_b24","series-title":"Proc. IEEE Int. Conf. Robot. Automa.","first-page":"14093","article-title":"Driving with LLMs: Fusing Object-Level vector modality for explainable autonomous driving","author":"Chen","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b25","series-title":"Enhancing emergency Decision-making with knowledge graphs and large language models","author":"Chen","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b26","series-title":"PCA-Bench: Evaluating multimodal large language models in Perception-Cognition-Action chain","author":"Chen","year":"2024"},{"issue":"11","key":"10.1016\/j.engappai.2026.114805_b27","doi-asserted-by":"crossref","first-page":"12878","DOI":"10.1109\/TPAMI.2022.3200245","article-title":"Transfuser: Imitation with transformer-based sensor fusion for autonomous driving","volume":"45","author":"Chitta","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2026.114805_b28","series-title":"Language-image models with 3D understanding","author":"Cho","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b29","series-title":"Semantics-guided Transformer-based sensor fusion for improved waypoint prediction","author":"Choi","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b30","series-title":"OpenPilot \u2013 An open-source ADAS driving software system","author":"comma.ai","year":"2025"},{"key":"10.1016\/j.engappai.2026.114805_b31","series-title":"Chain-of-Thought for autonomous driving: A comprehensive survey and future prospects","author":"Cui","year":"2025"},{"key":"10.1016\/j.engappai.2026.114805_b32","series-title":"Proc. IEEE\/ACM Symp. Edge Comput.","first-page":"319","article-title":"Human-autonomy teaming on autonomous vehicles with large language Model-Enabled human digital twins","author":"Cui","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b33","doi-asserted-by":"crossref","unstructured":"Cui, C., Ma, Y., Cao, X., Ye, W., Wang, Z., 2024a. Drive as you speak: Enabling human-like interaction with large language models in autonomous vehicles. In: Proc. IEEE\/CVF Winter Conf. Appl. Comput. Vis.. pp. 902\u2013909.","DOI":"10.1109\/WACVW60836.2024.00101"},{"key":"10.1016\/j.engappai.2026.114805_b34","article-title":"Receive, reason, and react: Drive as you say, with large language models in autonomous vehicles","author":"Cui","year":"2024","journal-title":"IEEE Intell. Transp. Syst. Mag."},{"key":"10.1016\/j.engappai.2026.114805_b35","doi-asserted-by":"crossref","unstructured":"Cui, C., Ma, Y., Cao, X., Ye, W., Zhou, Y., Liang, K., Chen, J., Lu, J., Yang, Z., Liao, K.-D., et al., 2024c. A survey on multimodal large language models for autonomous driving. In: Proc. IEEE\/CVF Winter Conf. Appl. Comput. Vis.. pp. 958\u2013979.","DOI":"10.1109\/WACVW60836.2024.00106"},{"key":"10.1016\/j.engappai.2026.114805_b36","series-title":"Large language models for autonomous driving (llm4ad): Concept, benchmark, experiments, and challenges","author":"Cui","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b37","series-title":"Large language models for autonomous driving: Real-world experiments","author":"Cui","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b38","doi-asserted-by":"crossref","DOI":"10.1109\/TIV.2024.3396450","article-title":"VistaRAG: Toward safe and trustworthy autonomous driving through retrieval-augmented generation","author":"Dai","year":"2024","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.engappai.2026.114805_b39","series-title":"QLoRA: Efficient finetuning of quantized LLMs","author":"Dettmers","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b40","doi-asserted-by":"crossref","unstructured":"Ding, X., Han, J., Xu, H., Liang, X., Zhang, W., Li, X., 2024. Holistic Autonomous Driving Understanding by Bird\u2019s-Eye-View Injected Multi-Modal Large Models. In: Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recogn.. CVPR, pp. 13668\u201313677.","DOI":"10.1109\/CVPR52733.2024.01297"},{"key":"10.1016\/j.engappai.2026.114805_b41","series-title":"HiLM-D: Towards High-Resolution understanding in multimodal large language models for autonomous driving","author":"Ding","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b42","series-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"10.1016\/j.engappai.2026.114805_b43","series-title":"PaLM-E: An embodied multimodal language model","author":"Driess","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b44","unstructured":"Driess, D., Xia, F., Sajjadi, M.S., Lynch, C., Chowdhery, A., Ichter, B., Wahid, A., Tompson, J., Vuong, Q., Yu, T., et al., 2023b. Palm-E: An embodied multimodal language model. In: Proc. Int. Conf. ML. ICML, pp. 8469\u20138488."},{"key":"10.1016\/j.engappai.2026.114805_b45","series-title":"Prompting Multi-Modal tokens to enhance End-to-End autonomous driving imitation learning with LLMs","author":"Duan","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b46","doi-asserted-by":"crossref","first-page":"1","DOI":"10.69709\/CAIC.2025.177363","article-title":"A novel MLLM-based approach for autonomous driving in different weather conditions","volume":"2","author":"Fourati","year":"2025","journal-title":"Comput. AI Connect."},{"key":"10.1016\/j.engappai.2026.114805_b47","article-title":"GPTQ: Accurate post-training quantization for generative pre-trained transformers","author":"Frantar","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b48","series-title":"LimSim++: A Closed-Loop platform for deploying multimodal LLMs in autonomous driving","author":"Fu","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b49","series-title":"A survey for foundation models in autonomous driving","author":"Gao","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b50","doi-asserted-by":"crossref","DOI":"10.1109\/TIV.2024.3399813","article-title":"LLM-based operating systems for automated vehicles: A new perspective","author":"Ge","year":"2024","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.engappai.2026.114805_b51","article-title":"Development of a liver Disease-Specific large language model chat interface using retrieval augmented generation","author":"Ge","year":"2023","journal-title":"MedRxiv"},{"key":"10.1016\/j.engappai.2026.114805_b52","series-title":"KITTI vision benchmark suite","author":"Geiger","year":"2012"},{"key":"10.1016\/j.engappai.2026.114805_b53","series-title":"Multi-Frame, lightweight & efficient Vision-Language models for question answering in autonomous driving","author":"Gopalkrishnan","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b54","article-title":"World models for autonomous driving: An initial survey","author":"Guan","year":"2024","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.engappai.2026.114805_b55","series-title":"Co-driver: VLM-based autonomous driving assistant with human-like behavior and understanding for complex road scenes","author":"Guo","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b56","series-title":"A survey on large language models: Applications, challenges, limitations, and practical usage","author":"Hadi","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b57","series-title":"Parameter-efficient fine-tuning for large models: A comprehensive survey","author":"Han","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b58","series-title":"DME-Driver: Integrating human decision logic and 3D scene perception in autonomous driving","author":"Han","year":"2024"},{"issue":"6","key":"10.1016\/j.engappai.2026.114805_b59","doi-asserted-by":"crossref","first-page":"3335","DOI":"10.3390\/s23063335","article-title":"Sensor fusion in autonomous vehicle with traffic surveillance camera system: detection, localization, and AI networking","volume":"23","author":"Hasanujjaman","year":"2023","journal-title":"Sensors"},{"key":"10.1016\/j.engappai.2026.114805_b60","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R., 2022. Masked autoencoders are scalable vision learners. In: Proc. IEEE\/CVF Conf. Comput. Vis. Patt. Recogn.. pp. 16000\u201316009.","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"10.1016\/j.engappai.2026.114805_b61","series-title":"GAIA-1: a generative world model for autonomous driving","first-page":"1","author":"Hu","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b62","series-title":"Leveraging large language models for enhanced NLP task performance through knowledge distillation and optimized training strategies","author":"Huang","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b63","series-title":"Applications of large scale foundation models for autonomous driving","author":"Huang","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b64","article-title":"Language is not all you need: Aligning perception with language models","volume":"36","author":"Huang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b65","unstructured":"Huang, Y., Sansom, J., Ma, Z., Gervits, F., Chai, J., 2024. DriVLMe: Exploring Foundation Models as Autonomous Driving Agents That Perceive, Communicate, and Navigate. In: Proc. Vis. & Lang. for Autonom. Driv. Robot. Wrkshp.. pp. 1\u201312."},{"key":"10.1016\/j.engappai.2026.114805_b66","doi-asserted-by":"crossref","DOI":"10.1016\/j.trc.2025.105321","article-title":"Vlm-rl: A unified vision language models and reinforcement learning framework for safe autonomous driving","volume":"180","author":"Huang","year":"2025","journal-title":"Transp. Res. C"},{"key":"10.1016\/j.engappai.2026.114805_b67","doi-asserted-by":"crossref","unstructured":"Inoue, Y., Yada, Y., Tanahashi, K., Yamaguchi, Y., 2024. nuScenes-MQA: Integrated evaluation of captions and QA for autonomous driving datasets using markup annotations. In: Proc. IEEE\/CVF Winter Conf. Appl. Comput. Vis.. pp. 930\u2013938.","DOI":"10.1109\/WACVW60836.2024.00104"},{"issue":"1\u20133","key":"10.1016\/j.engappai.2026.114805_b68","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1561\/0600000079","article-title":"Computer vision for autonomous vehicles: Problems, datasets and state of the art","volume":"12","author":"Janai","year":"2020","journal-title":"Found. Trends Comput. Graph. Vis."},{"key":"10.1016\/j.engappai.2026.114805_b69","series-title":"Adriver-I: A general world model for autonomous driving","author":"Jia","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b70","series-title":"Proc. Int. Conf. Mach. Learn.","first-page":"4904","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","author":"Jia","year":"2021"},{"key":"10.1016\/j.engappai.2026.114805_b71","series-title":"A survey on Vision-Language-Action models for autonomous driving","author":"Jiang","year":"2025"},{"key":"10.1016\/j.engappai.2026.114805_b72","series-title":"SurrealDriver: Designing generative driver agent simulation framework in urban contexts based on large language model","author":"Jin","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b73","series-title":"Multi-agent LLM Framework for Conversational Data Retrieval from Structured Data","author":"Khan","year":"2024"},{"issue":"2","key":"10.1016\/j.engappai.2026.114805_b74","doi-asserted-by":"crossref","first-page":"258","DOI":"10.1016\/j.acra.2022.03.028","article-title":"Completeness of reporting of systematic reviews and meta-analysis of diagnostic test accuracy (DTA) of radiological articles based on the PRISMA-DTA reporting guideline","volume":"30","author":"Kim","year":"2023","journal-title":"Acad. Radiol."},{"issue":"1","key":"10.1016\/j.engappai.2026.114805_b75","first-page":"568","article-title":"The role of Self-Driving vehicles in sustainability and road safety from a generation-specific perspective","volume":"7","author":"Kiss","year":"2024","journal-title":"Decis. Mak.: Appl. Manag. Eng."},{"key":"10.1016\/j.engappai.2026.114805_b76","series-title":"A superalignment framework in autonomous driving with large language models","author":"Kong","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b77","series-title":"pFedLVM: A large vision model (LVM)-Driven and latent feature-based personalized federated learning framework in autonomous driving","author":"Kou","year":"2024"},{"issue":"9","key":"10.1016\/j.engappai.2026.114805_b78","doi-asserted-by":"crossref","first-page":"12772","DOI":"10.1109\/TNNLS.2023.3264730","article-title":"Bvit: Broad attention-based vision transformer","volume":"35","author":"Li","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b79","series-title":"Europ. Conf. Comput. Vis.","first-page":"406","article-title":"CODA: A real-world road corner case dataset for object detection in autonomous driving","author":"Li","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b80","series-title":"BLIP-2: Bootstrapped Language-Image pretraining with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b81","series-title":"Proc. Int. Conf. Mach. Learn.","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b82","article-title":"Applications of large language models and multimodal large models in autonomous driving: A comprehensive review","author":"Li","year":"2025","journal-title":"Drones"},{"key":"10.1016\/j.engappai.2026.114805_b83","series-title":"Europ. Conf. Comput. Vis.","first-page":"280","article-title":"Exploring plain vision transformer backbones for object detection","author":"Li","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b84","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b85","article-title":"Multi-Modal attention perception for intelligent vehicle navigation using deep reinforcement learning","author":"Li","year":"2025","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b86","series-title":"Eyes can deceive: Benchmarking counterfactual reasoning abilities of Multi-modal large language models","author":"Li","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b87","series-title":"Automated evaluation of large vision-language models on Self-driving corner cases","author":"Li","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b88","series-title":"VoxelFormer: Bird\u2019s-eye-view feature generation based on dual-view attention for multi-view 3D object detection","author":"Li","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b89","doi-asserted-by":"crossref","unstructured":"Liang, M., Su, J.-C., Schulter, S., Garg, S., Zhao, S., Wu, Y., Chandraker, M., 2024. AIDE: An Automatic Data Engine for Object Detection in Autonomous Driving. In: Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recogn.. CVPR, pp. 14695\u201314706.","DOI":"10.1109\/CVPR52733.2024.01392"},{"key":"10.1016\/j.engappai.2026.114805_b90","doi-asserted-by":"crossref","unstructured":"Liao, G., Li, J., Ye, X., 2024. VLM2Scene: Self-Supervised Image-Text-LiDAR Learning with Foundation Models for Autonomous Driving Scene Understanding. In: Proc. AAAI Conf. Artifi. Intelli.. pp. 3351\u20133359.","DOI":"10.1609\/aaai.v38i4.28121"},{"key":"10.1016\/j.engappai.2026.114805_b91","series-title":"Training language models to follow rules with reinforcement learning from human feedback","author":"Liu","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b92","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ning, J., Cao, Y., Wei, Y., Zhang, Z., Lin, S., Hu, H., 2022. Video swin transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 3202\u20133211.","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"10.1016\/j.engappai.2026.114805_b93","series-title":"ED-ViT: Splitting vision transformer for distributed inference on edge devices","author":"Liu","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b94","series-title":"IEEE Int. Conf. Robot. Automa.","first-page":"2774","article-title":"BEVFusion: Multi-task multi-sensor fusion with unified bird\u2019s-eye view representation","author":"Liu","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b95","article-title":"A survey on autonomous driving datasets: Statistics, annotation quality, and a future outlook","author":"Liu","year":"2024","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.engappai.2026.114805_b96","doi-asserted-by":"crossref","DOI":"10.1016\/j.asoc.2025.113081","article-title":"Rethinking the multi-scale feature hierarchy in object detection transformer (DETR)","volume":"175","author":"Liu","year":"2025","journal-title":"Appl. Soft Comput."},{"key":"10.1016\/j.engappai.2026.114805_b97","first-page":"1","article-title":"Delving into Multi-Modal Multi-Task foundation models for road scene understanding: From learning paradigm perspectives","author":"Luo","year":"2024","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.engappai.2026.114805_b98","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recogn.","first-page":"15141","article-title":"LaMPilot: An open benchmark dataset for autonomous driving with language model programs","author":"Ma","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b99","doi-asserted-by":"crossref","unstructured":"Ma, Y., Cui, C., Cao, X., Ye, W., Liu, P., Lu, J., Abdelraouf, A., Gupta, R., Han, K., Bera, A., et al., 2024b. Lampilot: An open benchmark dataset for autonomous driving with language model programs. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 15141\u201315151.","DOI":"10.1109\/CVPR52733.2024.01434"},{"key":"10.1016\/j.engappai.2026.114805_b100","series-title":"Video-ChatGPT: Towards detailed video understanding via large vision and language models","author":"Maaz","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b101","series-title":"Smart Urban Comput. Appl.","first-page":"111","article-title":"IoT and artificial intelligence techniques for public safety and security","author":"Mahor","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b102","doi-asserted-by":"crossref","unstructured":"Malla, S., Choi, C., Dwivedi, I., Choi, J.H., Li, J., 2023. Drama: Joint risk localization and captioning in driving. In: Proc. IEEE\/CVF Winter Conf. Appl. Comput. Vis.. pp. 1043\u20131052.","DOI":"10.1109\/WACV56688.2023.00110"},{"key":"10.1016\/j.engappai.2026.114805_b103","series-title":"One million scenes for autonomous driving: Once dataset","author":"Mao","year":"2021"},{"key":"10.1016\/j.engappai.2026.114805_b104","series-title":"GPT-driver: Learning to drive with GPT","author":"Mao","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b105","series-title":"A language agent for autonomous driving","author":"Mao","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b106","series-title":"Proc. Int. Conf. Data Intelli. Cogn. Informat.","first-page":"387","article-title":"Prompt engineering in large language models","author":"Marvin","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b107","series-title":"MobileViT: Light-weight, general-purpose, and mobile-friendly vision transformer","author":"MehtaS","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b108","series-title":"Proc. Europ. Conf. Comput. Vis.","first-page":"53","article-title":"Waymo open dataset: Panoramic video panoptic segmentation","author":"Mei","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b109","doi-asserted-by":"crossref","unstructured":"Meng, L., Li, H., Chen, B.-C., Lan, S., Wu, Z., Jiang, Y.-G., Lim, S.-N., 2022. AdaViT: Adaptive vision transformers for efficient image recognition. In: Proc. IEEE\/CVF Conf. Comput. Vis. Patt. Recogn.. pp. 12309\u201312318.","DOI":"10.1109\/CVPR52688.2022.01199"},{"key":"10.1016\/j.engappai.2026.114805_b110","series-title":"Dialogue-based generation of self-driving simulation scenarios using Large Language Models","author":"Miceli-Barone","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b111","first-page":"1","article-title":"Transforming driver education: A comparative analysis of LLM-Augmented training and conventional instruction for autonomous vehicle technologies","author":"Murtaza","year":"2024","journal-title":"Int. J. Artif. Intell. Educ."},{"key":"10.1016\/j.engappai.2026.114805_b112","series-title":"Reason2drive: Towards interpretable and chain-based reasoning for autonomous driving","author":"Nie","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b113","series-title":"Proc. IEEE Int. Requir. Engineer. Conf.","first-page":"218","article-title":"Engineering safety requirements for autonomous driving with large language models","author":"Nouri","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b114","series-title":"Jetson AGX orin series modules","author":"NVIDIA","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b115","series-title":"NVIDIA DRIVE ThorTM \u2014 Centralized car computer unifying cluster, infotainment, automated driving and parking in a single cost-saving system","author":"NVIDIA Corporation","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b116","first-page":"1","article-title":"The potential of generative artificial intelligence across disciplines: Perspectives and future directions","author":"Ooi","year":"2023","journal-title":"J. Comput. Inf. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b117","series-title":"Dinov2: Learning robust visual features without supervision","author":"Oquab","year":"2023"},{"issue":"14","key":"10.1016\/j.engappai.2026.114805_b118","doi-asserted-by":"crossref","first-page":"2162","DOI":"10.3390\/electronics11142162","article-title":"A review on autonomous vehicles: Progress, methods and challenges","volume":"11","author":"Parekh","year":"2022","journal-title":"Electron."},{"key":"10.1016\/j.engappai.2026.114805_b119","doi-asserted-by":"crossref","unstructured":"Park, S., Lee, M., Kang, J., Choi, H., Park, Y., Cho, J., Lee, A., Kim, D., 2024. VLAAD: Vision and language assistant for autonomous driving. In: Proc. IEEE\/CVF Winter Conf. Appl. Comput. Vis.. pp. 980\u2013987.","DOI":"10.1109\/WACVW60836.2024.00107"},{"key":"10.1016\/j.engappai.2026.114805_b120","series-title":"LC-LLM: Explainable Lane-Change intention and trajectory predictions with large language models","author":"Peng","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b121","series-title":"MLLM-Protector: Ensuring MLLM\u2019s safety without hurting performance","author":"Pi","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b122","series-title":"2024 International Conference on Advances in Computing, Communication and Applied Informatics","first-page":"1","article-title":"BEiT transformer models to aid in the early detection of parkinson illness","author":"Rajesh","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b123","series-title":"Hierarchical text-conditional image generation with clip latents","first-page":"3","author":"Ramesh","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b124","doi-asserted-by":"crossref","unstructured":"Ray, J.Z.Z.H.A., Ohn-Bar, E., 2024. Feedback-Guided Autonomous Driving. In: Proc. IEEE\/CVF Conf. Comput. Vis. Patt. Recogn. (CVPR). pp. 15000\u201315011.","DOI":"10.1109\/CVPR52733.2024.01421"},{"key":"10.1016\/j.engappai.2026.114805_b125","series-title":"LanguageMPC: Large language models as decision makers for autonomous driving","author":"Sha","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b126","doi-asserted-by":"crossref","unstructured":"Shukor, M., Dancette, C., Cord, M., 2023. EP-ALM: Efficient perceptual augmentation of language models. In: Proc. IEEE\/CVF Int. Conf. Comput. Vis.. pp. 22056\u201322069.","DOI":"10.1109\/ICCV51070.2023.02016"},{"key":"10.1016\/j.engappai.2026.114805_b127","doi-asserted-by":"crossref","unstructured":"Singh, A., 2023. Transformer-based sensor fusion for autonomous driving: A survey. In: Proc. IEEE\/CVF Int. Conf. Comput. Vis.. pp. 3312\u20133317.","DOI":"10.1109\/ICCVW60793.2023.00355"},{"key":"10.1016\/j.engappai.2026.114805_b128","article-title":"Synthetic datasets for autonomous driving: A survey","author":"Song","year":"2023","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.engappai.2026.114805_b129","series-title":"Probing multimodal LLMs as world models for driving","author":"Sreeram","year":"2024"},{"issue":"2023-01-0788","key":"10.1016\/j.engappai.2026.114805_b130","doi-asserted-by":"crossref","first-page":"2068","DOI":"10.4271\/2023-01-0788","article-title":"Development and evaluation of comfort assessment approaches for passengers in autonomous vehicles","volume":"5","author":"Su","year":"2023","journal-title":"SAE Int. J. Adv. Curr. Pr. Mob."},{"key":"10.1016\/j.engappai.2026.114805_b131","series-title":"Evaluating the zero-shot robustness of instruction-tuned language models","author":"Sun","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b132","series-title":"Evaluation of large language models for decision making in autonomous driving","author":"Tanahashi","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b133","series-title":"MobileSAM: Lightweight segment anything model with MobileViT backbone","author":"Tang","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b134","article-title":"Tesla finally releases FSD v12: its last hope for self-driving","author":"Tesla","year":"2024","journal-title":"Electrek"},{"key":"10.1016\/j.engappai.2026.114805_b135","series-title":"DriveVLM: The convergence of autonomous driving and large vision-language models","author":"Tian","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b136","series-title":"Enhancing autonomous vehicle training with language model integration and critical scenario generation","author":"Tian","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b137","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b138","series-title":"How Machine Learning is Innovatin Today\u2019s World: A Concise Technical Guide","first-page":"329","article-title":"GPT-3-and DALL-E-Powered applications: A complete survey","author":"Vayadande","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b139","doi-asserted-by":"crossref","unstructured":"Wang, W., Bao, H., Dong, L., Bjorck, J., Peng, Z., Liu, Q., Aggarwal, K., Mohammed, O.K., Singhal, S., Som, S., et al., 2023. Image as a foreign language: Beit pretraining for vision and vision-language tasks. In: Proc. IEEE\/CVF Conf. Comput. Vis. Patt. Recogn.. pp. 19175\u201319186.","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"10.1016\/j.engappai.2026.114805_b140","series-title":"Exploring the reasoning abilities of multimodal large language models (MLLMs): A comprehensive survey on emerging trends in multimodal reasoning","author":"Wang","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b141","series-title":"AccidentGPT: Accident analysis and prevention from V2X environmental perception with multi-modal large model","author":"Wang","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b142","series-title":"Empowering autonomous driving with large language models: A safety perspective","author":"Wang","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b143","series-title":"GIT: A generative image-to-text transformer for vision and language","author":"Wang","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b144","doi-asserted-by":"crossref","unstructured":"Wang, Y., Liu, Q., Jiang, Z., Wang, T., Jiao, J., Chu, H., Gao, B., Chen, H., 2025. RAD: Retrieval-Augmented Decision-Making of Meta-Actions with Vision-Language Models in Autonomous Driving. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 3838\u20133848.","DOI":"10.1109\/CVPRW67362.2025.00369"},{"key":"10.1016\/j.engappai.2026.114805_b145","series-title":"VLC-BERT: Vision-language Pre-training of BERT using large-scale weakly supervised data","author":"Wang","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b146","doi-asserted-by":"crossref","DOI":"10.1016\/j.metrad.2023.100047","article-title":"Review of large vision models and visual prompt engineering","author":"Wang","year":"2023","journal-title":"Meta-Radiology"},{"key":"10.1016\/j.engappai.2026.114805_b147","series-title":"DriveCoT: Integrating chain-of-thought reasoning with End-to-End driving","author":"Wang","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b148","series-title":"DriveMLM: Aligning multi-modal large language models with behavioral planning states for autonomous driving","author":"Wang","year":"2023"},{"issue":"3","key":"10.1016\/j.engappai.2026.114805_b149","doi-asserted-by":"crossref","first-page":"415","DOI":"10.1007\/s41095-022-0274-8","article-title":"Pvt v2: Improved baselines with pyramid vision transformer","volume":"8","author":"Wang","year":"2022","journal-title":"Comput. Vis. Media"},{"key":"10.1016\/j.engappai.2026.114805_b150","series-title":"OmniDrive: A holistic LLM-Agent framework for autonomous driving with 3D perception, reasoning and planning","author":"Wang","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b151","series-title":"SimVLM: Simple visual language model pretraining with weak supervision","author":"Wang","year":"2021"},{"key":"10.1016\/j.engappai.2026.114805_b152","series-title":"DriveDreamer: Towards real-world-driven world models for autonomous driving","author":"Wang","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b153","doi-asserted-by":"crossref","unstructured":"Wei, Y., Wang, Z., Lu, Y., Xu, C., Liu, C., Zhao, H., Chen, S., Wang, Y., 2024a. Editable Scene Simulation for Autonomous Driving via Collaborative LLM-Agents. In: Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recogn.. CVPR, pp. 15077\u201315087.","DOI":"10.1109\/CVPR52733.2024.01428"},{"key":"10.1016\/j.engappai.2026.114805_b154","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Patt. Recogn.","first-page":"15077","article-title":"Editable scene simulation for autonomous driving via collaborative LLM-Agents","author":"Wei","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b155","series-title":"DiLu: A knowledge-driven approach to autonomous driving with large language models","author":"Wen","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b156","article-title":"Prospective role of foundation models in advancing autonomous vehicles","author":"Wu","year":"2023","journal-title":"Research"},{"key":"10.1016\/j.engappai.2026.114805_b157","series-title":"Language prompt for autonomous driving","author":"Wu","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b158","series-title":"Multi-agent autonomous driving systems with large language models: A survey of recent advances","author":"Wu","year":"2025"},{"key":"10.1016\/j.engappai.2026.114805_b159","series-title":"AccidentGPT: Large Multi-Modal foundation model for traffic accident analysis","author":"Wu","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b160","series-title":"Europ. Conf. Comp. Vis.","first-page":"68","article-title":"TinyViT: Fast pretraining distillation for small vision transformers","author":"Wu","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b161","article-title":"Multi-sensor fusion and cooperative perception for autonomous driving: A review","author":"Xiang","year":"2023","journal-title":"IEEE Intell. Transp. Syst. Mag."},{"key":"10.1016\/j.engappai.2026.114805_b162","series-title":"Cohere3D: Exploiting temporal coherence for unsupervised representation learning of Vision-based autonomous driving","author":"Xie","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b163","first-page":"12077","article-title":"SegFormer: Simple and efficient design for semantic segmentation with transformers","volume":"34","author":"Xie","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b164","series-title":"Versal AI edge series","author":"Xilinx","year":"2023"},{"issue":"10","key":"10.1016\/j.engappai.2026.114805_b165","article-title":"The ApolloScape open dataset for autonomous driving and its application","volume":"42","author":"XinyuHuang","year":"2020","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2026.114805_b166","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Patt. Recogn.","first-page":"17261","article-title":"DriveGPT4-V2: Harnessing large language model capabilities for enhanced closed-loop autonomous driving","author":"Xu","year":"2025"},{"key":"10.1016\/j.engappai.2026.114805_b167","series-title":"FwdLLM: Efficient FedLLM using forward gradient","author":"Xu","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b168","unstructured":"Xu, Y., Dai, W., Wang, F., Wu, L., Liang, X., 2023. LiteVLM: Vision-Language Foundation Models at Light-Speed. In: Proc. IEEE\/CVF Conf. Comp. Vis. Patt. Recogn.. CVPR, pp. 1\u20136."},{"key":"10.1016\/j.engappai.2026.114805_b169","series-title":"LVLM-eHUB: A comprehensive evaluation benchmark for large vision-language models","author":"Xu","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b170","series-title":"A survey of resource-efficient LLM and multimodal foundation models","author":"Xu","year":"2024"},{"issue":"10","key":"10.1016\/j.engappai.2026.114805_b171","doi-asserted-by":"crossref","first-page":"8186","DOI":"10.1109\/LRA.2024.3440097","article-title":"DriveGPT4: Interpretable End-to-End autonomous driving via large language model","volume":"9","author":"Xu","year":"2024","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.engappai.2026.114805_b172","article-title":"A deep reinforcement learning method for autonomous driving integrating Multi-Modal fusion","author":"Xu","year":"2025","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b173","series-title":"FusedMM: A fused matrix multiplication kernel for Memory-Efficient attention","author":"Xu","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b174","series-title":"Forging vision foundation models for autonomous driving: Challenges, methodologies, and opportunities","author":"Yan","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b175","series-title":"LLM4Drive: A survey of large language models for autonomous driving","author":"Yang","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b176","series-title":"Driving style alignment for LLM-powered driver agent","author":"Yang","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b177","doi-asserted-by":"crossref","unstructured":"Yang, Y., Zhang, Q., Li, C., Marta, D.S., Batool, N., Folkesson, J., 2024. Human-centric autonomous systems with LLMs for user command reasoning. In: Proc. IEEE\/CVF Winter Conf. Appl. Comput. Vis.. pp. 988\u2013994.","DOI":"10.1109\/WACVW60836.2024.00108"},{"key":"10.1016\/j.engappai.2026.114805_b178","series-title":"Proc. IEEE Intelli. Veh. Symp.","first-page":"2269","article-title":"Collaborative perception datasets in autonomous driving: A survey","author":"Yazgan","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b179","series-title":"A survey on multimodal large language models","author":"Yin","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b180","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Patt. Recogn.","first-page":"2636","article-title":"BDD100k: A diverse driving dataset for heterogeneous multitask learning","author":"Yu","year":"2020"},{"key":"10.1016\/j.engappai.2026.114805_b181","series-title":"CoCa: Contrastive captioners are Image-Text foundation models","author":"Yu","year":"2022"},{"key":"10.1016\/j.engappai.2026.114805_b182","series-title":"RAG-Driver: Generalisable driving explanations with Retrieval-Augmented In-Context learning in Multi-Modal large language model","author":"Yuan","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b183","article-title":"Multi-modal prompt learning for visual grounding","author":"Zang","year":"2022","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.engappai.2026.114805_b184","doi-asserted-by":"crossref","first-page":"95","DOI":"10.1016\/j.tranpol.2024.03.006","article-title":"TrafficGPT: Viewing, processing and interacting with traffic foundation models","volume":"150","author":"Zhang","year":"2024","journal-title":"Transp. Policy"},{"key":"10.1016\/j.engappai.2026.114805_b185","doi-asserted-by":"crossref","unstructured":"Zhang, H., Li, X., Bing, L., 2023. Video-Llama: An instruction-tuned audio-visual language model for video understanding. In: Proc. Conf. Empir. Methods Nat. Lang. Process.: Syst. Demo.. pp. 1\u201311.","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"10.1016\/j.engappai.2026.114805_b186","series-title":"MM-LLMs: Recent advances in multimodal large language models","author":"Zhang","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b187","article-title":"Potential sources of sensor data anomalies for autonomous vehicles: An overview from road vehicle safety perspective","author":"Zhao","year":"2023","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.engappai.2026.114805_b188","article-title":"On evaluating adversarial robustness of large vision-language models","volume":"36","author":"Zhao","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b189","series-title":"DriveDreamer-2: LLM-Enhanced world models for diverse driving video generation","author":"Zhao","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b190","series-title":"A survey on the application of large language models in scenario-based testing of automated driving systems","author":"Zhao","year":"2025"},{"key":"10.1016\/j.engappai.2026.114805_b191","series-title":"A survey of large language models","author":"Zhao","year":"2023"},{"key":"10.1016\/j.engappai.2026.114805_b192","series-title":"TrafficSafetyGPT: Tuning a pre-trained large language model to a domain-specific expert in transportation safety","author":"Zheng","year":"2023"},{"issue":"8","key":"10.1016\/j.engappai.2026.114805_b193","first-page":"3675","article-title":"Dynamic early exit for efficient inference of Transformer-Based models","volume":"33","author":"Zhou","year":"2022","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.engappai.2026.114805_b194","article-title":"Vision language models in autonomous driving: A survey and outlook","author":"Zhou","year":"2024","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.engappai.2026.114805_b195","series-title":"In-context learning for automated driving scenarios","author":"Zhou","year":"2024"},{"key":"10.1016\/j.engappai.2026.114805_b196","series-title":"MiniGPT-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","year":"2023"}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626010870?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626010870?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T17:12:56Z","timestamp":1778778776000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197626010870"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":196,"alternative-id":["S0952197626010870"],"URL":"https:\/\/doi.org\/10.1016\/j.engappai.2026.114805","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Foundation models for autonomous driving: A comprehensive survey","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.engappai.2026.114805","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"114805"}}