{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T20:27:08Z","timestamp":1782764828573,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62202063"],"award-info":[{"award-number":["62202063"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Beijing Natural Science Foundation","award":["L243027"],"award-info":[{"award-number":["L243027"]}]},{"name":"Beijing Major Science and Technology Project","award":["Z231100007423014"],"award-info":[{"award-number":["Z231100007423014"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755868","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:38:54Z","timestamp":1761377934000},"page":"11170-11178","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["SafeDriveRAG: Towards Safe Autonomous Driving with Knowledge Graph-based Retrieval-Augmented Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-4738-4139","authenticated-orcid":false,"given":"Hao","family":"Ye","sequence":"first","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6955-6635","authenticated-orcid":false,"given":"Mengshi","family":"Qi","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2066-0215","authenticated-orcid":false,"given":"Zhaohong","family":"Liu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2173-9226","authenticated-orcid":false,"given":"Liang","family":"Liu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7199-5047","authenticated-orcid":false,"given":"Huadong","family":"Ma","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Post and Telecommunication, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Semscore: Automated evaluation of instruction-tuned llms based on semantic textual similarity. arXiv preprint arXiv:2401.17072","author":"Aynetdinov Ansar","year":"2024","unstructured":"Ansar Aynetdinov and Alan Akbik. 2024. Semscore: Automated evaluation of instruction-tuned llms based on semantic textual similarity. arXiv preprint arXiv:2401.17072 (2024)."},{"key":"e_1_3_2_1_2_1","unstructured":"Shuai Bai Keqin Chen and Xuejing Liu et al. 2025. Qwen2.5-VL Technical Report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Jasmin Breitenstein Jan-Aike Term\u00f6hlen and Daniel Lipinski et al. 2021. Corner cases for visual perception in automated driving: Some guidance on detection approaches. arXiv preprint arXiv:2102.05897 (2021).","DOI":"10.1109\/IV47402.2020.9304789"},{"key":"e_1_3_2_1_4_1","unstructured":"Kai Chen Yanze Li and Wenhua Zhang et al. 2024. Automated Evaluation of Large Vision-Language Models on Self-driving Corner Cases. arXiv preprint arXiv:2404.10595 (2024)."},{"key":"e_1_3_2_1_5_1","first-page":"5558","article-title":"MuRAG: Multimodal Retrieval-Augmented Generator for Open Question Answering over Images and Text. In Proc. Empir. Methods","author":"Chen Wenhu","year":"2022","unstructured":"Wenhu Chen, Hexiang Hu, and Xi Chen et al., 2022. MuRAG: Multimodal Retrieval-Augmented Generator for Open Question Answering over Images and Text. In Proc. Empir. Methods Nat. Lang. Process., Association for Computational Linguistics, Abu Dhabi, United Arab Emirates, 5558-5570.","journal-title":"Nat. Lang. Process., Association for Computational Linguistics, Abu Dhabi, United Arab Emirates"},{"key":"e_1_3_2_1_6_1","first-page":"2088","article-title":"Talk2Car: Taking Control of Your Self-Driving Car. In Proc. Empir. Methods","author":"Deruyttere Thierry","year":"2019","unstructured":"Thierry Deruyttere, Simon Vandenhende, and Dusan Grujicic et al., 2019. Talk2Car: Taking Control of Your Self-Driving Car. In Proc. Empir. Methods Nat. Lang. Process. (EMNLP-IJCNLP). Association for Computational Linguistics, Hong Kong, China, 2088-2098.","journal-title":"Nat. Lang. Process. (EMNLP-IJCNLP). Association for Computational Linguistics, Hong Kong, China"},{"key":"e_1_3_2_1_7_1","unstructured":"Hanxing Ding Liang Pang and Zihao Wei et al. 2024. Retrieve Only When It Needs: Adaptive Retrieval Augmentation for Hallucination Mitigation in Large Language Models. arXiv preprint arXiv:2402.10612 (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"Global: A Graph RAG Approach to Query-Focused Summarization. arXiv:2404.16130 [cs.CL] https:\/\/arxiv.org\/abs\/2404.16130","author":"Edge Darren","year":"2025","unstructured":"Darren Edge, Ha Trinh, and Newman Cheng et al., 2025. From Local to Global: A Graph RAG Approach to Query-Focused Summarization. arXiv:2404.16130 [cs.CL] https:\/\/arxiv.org\/abs\/2404.16130"},{"key":"e_1_3_2_1_9_1","unstructured":"Tianyu Fan Jingyuan Wang and Xubin Ren et al. 2025. MiniRAG: Towards Extremely Simple Retrieval-Augmented Generation. arXiv preprint arXiv:2501.06713 (2025)."},{"key":"e_1_3_2_1_10_1","volume-title":"Driving Scenes: A Multimodality Benchmark. CoRR","author":"Fang Jianwu","year":"2022","unstructured":"Jianwu Fang, Lei-Lei Li, and Kuan Yang et al., 2022. Cognitive Accident Prediction in Driving Scenes: A Multimodality Benchmark. CoRR, Vol. abs\/2212.09381 (2022)."},{"key":"e_1_3_2_1_11_1","unstructured":"Yunfan Gao Yun Xiong and Xinyu Gao et al. 2024. Retrieval-Augmented Generation for Large Language Models: A Survey. arXiv preprint arXiv:2312.10997 (2024)."},{"key":"e_1_3_2_1_12_1","volume-title":"arXiv preprint arXiv:2403.19838","author":"Gopalkrishnan Akshay","year":"2024","unstructured":"Akshay Gopalkrishnan, Ross Greer, and Mohan Trivedi. 2024. Multi-Frame, Lightweight & Efficient Vision-Language Models for Question Answering in Autonomous Driving. arXiv preprint arXiv:2403.19838 (2024)."},{"key":"e_1_3_2_1_13_1","unstructured":"Zirui Guo Lianghao Xia and Yanhua Yu et al. 2024. LightRAG: Simple and Fast Retrieval-Augmented Generation. arXiv preprint arXiv:2410.05779 (2024)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02238"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIV.2023.3275543"},{"key":"e_1_3_2_1_16_1","volume-title":"Proc. Eur. Conf. Comput. Vis., 563-578","author":"Kim Jinkyu","unstructured":"Jinkyu Kim, Anna Rohrbach, and Trevor Darrell et al., 2018. Textual Explanations for Self-driving Vehicles. In Proc. Eur. Conf. Comput. Vis., 563-578."},{"key":"e_1_3_2_1_17_1","unstructured":"Bo Li Yuanhan Zhang and Dong Guo et al. 2024. LLaVA-OneVision: Easy Visual Task Transfer. arXiv preprint arXiv:2408.03326 (2024)."},{"key":"e_1_3_2_1_18_1","unstructured":"Chenxia Li Weiwei Liu and Ruoyu Guo et al. 2022b. PP-OCRv3: More Attempts for the Improvement of Ultra Lightweight OCR System. arXiv preprint arXiv:2206.03001 (2022)."},{"key":"e_1_3_2_1_19_1","volume-title":"Coda: A real-world road corner case dataset for object detection in autonomous driving. In Proc. Eur. Conf. Comput. Vis.","author":"Li Kaican","year":"2022","unstructured":"Kaican Li, Kai Chen, and Haoyu Wang et al., 2022a. Coda: A real-world road corner case dataset for object detection in autonomous driving. In Proc. Eur. Conf. Comput. Vis., Springer, 406-423."},{"key":"e_1_3_2_1_20_1","first-page":"74","article-title":"Rouge: A package for automatic evaluation of summaries","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Proc. Text Summarization Branches Out. 74-81.","journal-title":"Proc. Text Summarization Branches Out."},{"key":"e_1_3_2_1_21_1","first-page":"34892","volume":"36","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, and Qingyang Wu et al., 2023. Visual Instruction Tuning. In Proc. Adv. Neural Inf. Process. Syst., Vol. 36. 34892-34916.","journal-title":"Visual Instruction Tuning. In Proc. Adv. Neural Inf. Process. Syst."},{"key":"e_1_3_2_1_22_1","unstructured":"Yuhang Lu Yichen Yao and Jiadong Tu et al. 2024. Can LVLMs Obtain a Driver's License? A Benchmark Towards Reliable AGI for Autonomous Driving. arXiv preprint arXiv:2409.02914 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit., 17197-17206","author":"Lv Changsheng","unstructured":"Changsheng Lv, Mengshi Qi, and Liang Liu et al., 2025. T2sg: Traffic topology scene graph for topology reasoning in autonomous driving. In Proc. IEEE Conf. Comput. Vis. Pattern Recognit., 17197-17206."},{"key":"e_1_3_2_1_24_1","first-page":"4035","volume-title":"Proc. AAAI Conf. Artif. Intell.","volume":"38","author":"Lv Changsheng","unstructured":"Changsheng Lv, Mengshi Qi, and Xia Li et al., 2024. SGFormer: Semantic Graph Transformer for Point Cloud-Based 3D Scene Graph Generation. In Proc. AAAI Conf. Artif. Intell., Vol. 38. 4035-4043."},{"key":"e_1_3_2_1_25_1","first-page":"12476","article-title":"Disentangled counterfactual learning for physical audiovisual commonsense reasoning","volume":"36","author":"Lv Changsheng","year":"2023","unstructured":"Changsheng Lv, Shuai Zhang, and Yapeng Tian et al., 2023. Disentangled counterfactual learning for physical audiovisual commonsense reasoning. Proc. Adv. Neural Inf. Process. Syst., Vol. 36 (2023), 12476-12488.","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"e_1_3_2_1_26_1","volume-title":"Proc. IEEE Winter Conf. Appl. Comput. Vis., 1043-1052","author":"Malla Srikanth","unstructured":"Srikanth Malla, Chiho Choi, and Isht Dwivedi et al., 2023. DRAMA: Joint Risk Localization and Captioning in Driving. In Proc. IEEE Winter Conf. Appl. Comput. Vis., 1043-1052."},{"key":"e_1_3_2_1_27_1","unstructured":"Jiageng Mao Junjie Ye and Yuxi Qian et al. 2024. A Language Agent for Autonomous Driving. arXiv preprint arXiv:2311.10813 (2024)."},{"key":"e_1_3_2_1_28_1","unstructured":"Yuning Mao Pengcheng He and Xiaodong Liu et al. 2020. Generation-augmented retrieval for open-domain question answering. arXiv preprint arXiv:2009.08553 (2020)."},{"key":"e_1_3_2_1_29_1","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. Workshops. 3317-3326","author":"Marathe Aboli","unstructured":"Aboli Marathe, Deva Ramanan, and Rahee Walambe et al., 2023. WEDGE: A Multi-Weather Autonomous Driving Dataset Built From Generative Vision-Language Models. In Proc. IEEE Conf. Comput. Vis. Pattern Recognit. Workshops. 3317-3326."},{"key":"e_1_3_2_1_30_1","volume-title":"Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs. arXiv preprint arXiv:2503.01743","author":"Abouelenin Abdelrahman","year":"2025","unstructured":"Microsoft, Abdelrahman Abouelenin, and Atabak Ashfaq et al., 2025. Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs. arXiv preprint arXiv:2503.01743 (2025)."},{"key":"e_1_3_2_1_31_1","unstructured":"OpenAI Josh Achiam and Steven Adler et al. 2024. GPT-4 Technical Report. arXiv preprint arXiv:2303.08774 (2024)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACVW60836.2024.00107"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00408"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.3048680"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2983567"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00538"},{"key":"e_1_3_2_1_37_1","volume-title":"Memorag: Moving towards next-gen rag via memory-inspired knowledge discovery. arXiv preprint arXiv:2409.05591","author":"Qian Hongjin","year":"2024","unstructured":"Hongjin Qian, Peitian Zhang, and Zheng Liu et al., 2024b. Memorag: Moving towards next-gen rag via memory-inspired knowledge discovery. arXiv preprint arXiv:2409.05591 (2024)."},{"key":"e_1_3_2_1_38_1","first-page":"4542","volume-title":"Proc. AAAI Conf. Artif. Intell.","volume":"38","author":"Qian Tianwen","unstructured":"Tianwen Qian, Jingjing Chen, and Linhai Zhuo et al., 2024a. nuScenes-QA: A Multi-modal Visual Question Answering Benchmark for Autonomous Driving Scenario. In Proc. AAAI Conf. Artif. Intell., Vol. 38. 4542-4550."},{"key":"e_1_3_2_1_39_1","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit., 15120-15130","author":"Shao Hao","unstructured":"Hao Shao, Yuxuan Hu, and Letian Wang et al., 2024. LMDrive: Closed-Loop End-to-End Driving with Large Language Models. In Proc. IEEE Conf. Comput. Vis. Pattern Recognit., 15120-15130."},{"key":"e_1_3_2_1_40_1","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Sima Chonghao","unstructured":"Chonghao Sima, Katrin Renz, and Kashyap Chitta et al., 2025. DriveLM: Driving with Graph Visual Question Answering. In Proc. Eur. Conf. Comput. Vis."},{"key":"e_1_3_2_1_41_1","unstructured":"Xiaoyu Tian Junru Gu and Bailin Li et al. 2024. DriveVLM: The Convergence of Autonomous Driving and Large Vision-Language Models. arXiv preprint arXiv:2402.12289 (2024)."},{"key":"e_1_3_2_1_42_1","unstructured":"Peng Wang Shuai Bai and Sinan Tan et al. 2024. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_43_1","first-page":"24824","article-title":". Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","volume":"35","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, and Dale Schuurmans et al., 2022. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. Proc. Adv. Neural Inf. Process. Syst., Vol. 35 (2022), 24824-24837.","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"e_1_3_2_1_44_1","unstructured":"Licheng Wen Xuemeng Yang and Daocheng Fu et al. 2023. On the Road with GPT-4V(ision): Early Explorations of Visual-Language Model on Autonomous Driving. arXiv preprint arXiv:2311.05332 (2023)."},{"key":"e_1_3_2_1_45_1","volume-title":"Global Status Report on Road Safety","author":"World Health Organization","year":"2023","unstructured":"World Health Organization. 2023. Global Status Report on Road Safety 2023. https:\/\/www.who.int\/teams\/social-determinants-of-health\/safety-and-mobility\/global-status-report-on-road-safety-2023. Accessed: 2025-07-21."},{"key":"e_1_3_2_1_46_1","volume-title":"Explainable Object-Induced Action Decision for Autonomous Vehicles. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Xu Yiran","unstructured":"Yiran Xu, Xiaoyin Yang, and Lihang Gong et al., 2020. Explainable Object-Induced Action Decision for Autonomous Vehicles. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_1_47_1","unstructured":"Zhenhua Xu Yujia Zhang and Enze Xie et al. 2024. DriveGPT4: Interpretable End-to-End Autonomous Driving via Large Language Model. IEEE Robot. Autom. Lett. (2024)."},{"key":"e_1_3_2_1_48_1","unstructured":"Gus Ye. 2024. nano-vectordb. https:\/\/github.com\/gusye1234\/nano-vectordb."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Tianyu Yu Haoye Zhang and Qiming Li et al. 2024. RLAIF-V: Open-Source AI Feedback Leads to Super GPT-4V Trustworthiness. arXiv preprint arXiv:2405.17220 (2024).","DOI":"10.1109\/CVPR52734.2025.01861"},{"key":"e_1_3_2_1_50_1","volume-title":"Avoid: Autonomous vehicle operation incident dataset across the globe. arXiv preprint arXiv:2303.12889","author":"Zheng Ou","year":"2023","unstructured":"Ou Zheng, Mohamed Abdel-Aty, and Zijin Wang et al., 2023. Avoid: Autonomous vehicle operation incident dataset across the globe. arXiv preprint arXiv:2303.12889 (2023)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755868","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:11:47Z","timestamp":1765339907000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755868"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":50,"alternative-id":["10.1145\/3746027.3755868","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755868","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}