{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,17]],"date-time":"2026-03-17T08:08:57Z","timestamp":1773734937333,"version":"3.50.1"},"reference-count":36,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T00:00:00Z","timestamp":1763424000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T00:00:00Z","timestamp":1763424000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Project of China","doi-asserted-by":"publisher","award":["2023YFB4301800"],"award-info":[{"award-number":["2023YFB4301800"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,11,18]]},"DOI":"10.1109\/itsc60802.2025.11423082","type":"proceedings-article","created":{"date-parts":[[2026,3,16]],"date-time":"2026-03-16T20:10:23Z","timestamp":1773691823000},"page":"1870-1877","source":"Crossref","is-referenced-by-count":0,"title":["LoCo-VLM: End-to-End Autonomous Driving with a Loosely Coupled Vision-Language Model"],"prefix":"10.1109","author":[{"given":"Jiandong","family":"Xing","sequence":"first","affiliation":[{"name":"Beihang University,State Key Laboratory of Intelligent Transportation System,Beijing,China,100191"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuai","family":"Min","sequence":"additional","affiliation":[{"name":"Beihang University,State Key Laboratory of Intelligent Transportation System,Beijing,China,100191"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Danmu","family":"Xie","sequence":"additional","affiliation":[{"name":"Beihang University,State Key Laboratory of Intelligent Transportation System,Beijing,China,100191"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinyu","family":"Wang","sequence":"additional","affiliation":[{"name":"Beihang University,State Key Laboratory of Intelligent Transportation System,Beijing,China,100191"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Letian","family":"Kang","sequence":"additional","affiliation":[{"name":"Beihang University,State Key Laboratory of Intelligent Transportation System,Beijing,China,100191"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yilong","family":"Ren","sequence":"additional","affiliation":[{"name":"Beihang University,State Key Laboratory of Intelligent Transportation System,Beijing,China,100191"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haiyang","family":"Yu","sequence":"additional","affiliation":[{"name":"Beihang University,State Key Laboratory of Intelligent Transportation System,Beijing,China,100191"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuesong","family":"Bai","sequence":"additional","affiliation":[{"name":"Beihang University,State Key Laboratory of Intelligent Transportation System,Beijing,China,100191"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"17853","article-title":"Planning-oriented autonomous driving","volume-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Hu","year":"2022"},{"key":"ref2","doi-asserted-by":"crossref","first-page":"8306","DOI":"10.1109\/ICCV51070.2023.00766","article-title":"Vad: Vectorized scene representation for efficient autonomous driving","volume-title":"2023 IEEE\/CVF International Conference on Computer Vision (ICCV)","author":"Jiang","year":"2023"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3435937"},{"key":"ref4","article-title":"Emma: End-to-end multimodal model for autonomous driving","volume":"abs\/2410.23262","author":"Hwang","year":"2024","journal-title":"ArXiv"},{"key":"ref5","article-title":"Gpt-driver: Learning to drive with gpt","volume":"abs\/2310.01415","author":"Mao","year":"2023","journal-title":"ArXiv"},{"key":"ref6","article-title":"Drivevlm: The convergence of autonomous driving and large vision-language models","volume":"abs\/2402.12289","author":"Tian","year":"2024","journal-title":"ArXiv"},{"key":"ref7","article-title":"Senna: Bridging large vision-language models and end-to-end autonomous driving","volume":"abs\/2410.22313","author":"Jiang","year":"2024","journal-title":"ArXiv"},{"key":"ref8","doi-asserted-by":"crossref","first-page":"8354","DOI":"10.1109\/LRA.2024.3444668","article-title":"Diverse controllable diffusion policy with signal temporal logic","volume":"9","author":"Meng","year":"2024","journal-title":"IEEE Robotics and Automation Letters"},{"key":"ref9","article-title":"Vadv2: End-to-end vectorized autonomous driving via probabilistic planning","volume":"abs\/2402.13243","author":"Chen","year":"2024","journal-title":"ArXiv"},{"key":"ref10","article-title":"Carla: An open urban driving simulator","volume-title":"Conference on Robot Learning","author":"Dosovitskiy","year":"2017"},{"key":"ref11","doi-asserted-by":"crossref","first-page":"7073","DOI":"10.1109\/CVPR46437.2021.00700","article-title":"Multi-modal fusion transformer for end-to-end autonomous driving","volume-title":"2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Prakash","year":"2021"},{"key":"ref12","doi-asserted-by":"crossref","first-page":"12878","DOI":"10.1109\/TPAMI.2022.3200245","article-title":"Transfuser: Imitation with transformer-based sensor fusion for autonomous driving","volume":"45","author":"Chitta","year":"2022","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"ref13","first-page":"14864","article-title":"Is ego status all you need for open-loop end-to-end autonomous driving?","volume-title":"2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Li","year":"2023"},{"key":"ref14","article-title":"Sparsedrive: End-to-end autonomous driving via sparse scene representation","volume":"abs\/2405.19620","author":"Sun","year":"2024","journal-title":"ArXiv"},{"key":"ref15","article-title":"Trajectory-guided control prediction for end-to-end autonomous driving: A simple yet strong baseline","volume":"abs\/2206.08129","author":"Wu","year":"2022","journal-title":"ArXiv"},{"key":"ref16","article-title":"Policy pre-training for autonomous driving via self-supervised geometric modeling","volume-title":"International Conference on Learning Representations","author":"Wu","year":"2023"},{"key":"ref17","article-title":"King: Generating safety-critical driving scenarios for robust imitation via kinematics gradients","volume":"abs\/2204.13683","author":"Hanselmann","year":"2022","journal-title":"ArXiv"},{"key":"ref18","article-title":"Dilu: A knowledgedriven approach to autonomous driving with large language models","volume":"abs\/2309.16292","author":"Wen","year":"2023","journal-title":"ArXiv"},{"key":"ref19","first-page":"16345","article-title":"Talk2bev: Language-enhanced bird\u2019s-eye view maps for autonomous driving","volume-title":"2024 IEEE International Conference on Robotics and Automation (ICRA)","author":"Dewangan","year":"2023s"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/wacvw60836.2024.00104"},{"key":"ref21","article-title":"Vision language models in autonomous driving: A survey and outlook","author":"Zhou","year":"2023","journal-title":"IEEE Transactions on Intelligent Vehicles"},{"key":"ref22","doi-asserted-by":"crossref","first-page":"6662","DOI":"10.1109\/CVPR52729.2023.00644","article-title":"Vlpd: Context-aware pedestrian detection via vision-language semantic self-supervision","volume-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Liu","year":"2023"},{"key":"ref23","article-title":"Languageguided 3d object detection in point cloud for autonomous driving","volume":"abs\/2305.15765","author":"Cheng","year":"2023","journal-title":"ArXiv"},{"key":"ref24","first-page":"910","article-title":"Drive like a human: Rethinking autonomous driving with large language models","volume-title":"2024 IEEE\/CVF Winter Conference on Applications of Computer Vision Workshop s (WACVW)","author":"Fu","year":"2023"},{"key":"ref25","article-title":"He-drive: Human-like end-to-end driving with vision language models","volume":"abs\/2410.05051","author":"Wang","year":"2024","journal-title":"ArXiv"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3703155"},{"key":"ref27","first-page":"10674","article-title":"High-resolution image synthesis with latent diffusion models","volume-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Rombach","year":"2021"},{"key":"ref28","article-title":"Reconstruction vs. generation: Taming optimization dilemma in latent diffusion models","volume":"abs\/2501.01423","author":"Yao","year":"2025","journal-title":"ArXiv"},{"key":"ref29","article-title":"Diffusion policy: Visuomotor policy learning via action diffusion","volume":"abs\/2303.04137","author":"Chi","year":"2023","journal-title":"ArXiv"},{"key":"ref30","doi-asserted-by":"crossref","first-page":"9644","DOI":"10.1109\/CVPR52729.2023.00930","article-title":"Motiondiffuser: Controllable multi-agent motion prediction using diffusion","volume-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Jiang","year":"2023"},{"key":"ref31","article-title":"Diffusion-based planning for autonomous driving with flexible guidance","volume":"abs\/2501.15564","author":"Zheng","year":"2025","journal-title":"ArXiv"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/IV55156.2024.10588486"},{"key":"ref33","article-title":"Diffusiondrive: Truncated diffusion model for end-to-end autonomous driving","volume":"abs\/2411.15139","author":"Liao","year":"2024","journal-title":"ArXiv"},{"key":"ref34","article-title":"A survey on hallucination in large vision-language models","volume":"abs\/2402.00253","author":"Liu","year":"2024","journal-title":"ArXiv"},{"key":"ref35","article-title":"Vlm-ad: End-to-end autonomous driving through vision-language model supervision","volume":"abs\/2412.14446","author":"Xu","year":"2024","journal-title":"ArXiv"},{"key":"ref36","volume-title":"Automatic differentiation in pytorch","author":"Paszke","year":"2017"}],"event":{"name":"2025 IEEE 28th International Conference on Intelligent Transportation Systems (ITSC)","location":"Gold Coast, Australia","start":{"date-parts":[[2025,11,18]]},"end":{"date-parts":[[2025,11,21]]}},"container-title":["2025 IEEE 28th International Conference on Intelligent Transportation Systems (ITSC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11422813\/11423000\/11423082.pdf?arnumber=11423082","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,17]],"date-time":"2026-03-17T05:55:09Z","timestamp":1773726909000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11423082\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,18]]},"references-count":36,"URL":"https:\/\/doi.org\/10.1109\/itsc60802.2025.11423082","relation":{},"subject":[],"published":{"date-parts":[[2025,11,18]]}}}