{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T12:39:58Z","timestamp":1766061598907,"version":"3.48.0"},"reference-count":37,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iros60139.2025.11245662","type":"proceedings-article","created":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T18:54:45Z","timestamp":1764269685000},"page":"20662-20668","source":"Crossref","is-referenced-by-count":0,"title":["DriveBLIP2: Attention-Guided Explanation Generation for Complex Driving Scenarios"],"prefix":"10.1109","author":[{"given":"Shihong","family":"Ling","sequence":"first","affiliation":[{"name":"University of Pittsburgh,School of Computing and Information,Pittsburgh,PA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yue","family":"Wan","sequence":"additional","affiliation":[{"name":"University of Pittsburgh,School of Computing and Information,Pittsburgh,PA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaowei","family":"Jia","sequence":"additional","affiliation":[{"name":"University of Pittsburgh,School of Computing and Information,Pittsburgh,PA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Na","family":"Du","sequence":"additional","affiliation":[{"name":"University of Pittsburgh,School of Computing and Information,Pittsburgh,PA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1877","article-title":"Language Models are Few-Shot Learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1423"},{"issue":"140","key":"ref3","first-page":"1","article-title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref4","first-page":"8748","article-title":"Learning Transferable Visual Models from Natural Language Supervision","volume-title":"Proceedings of the 38th International Conference on Machine Learning (ICML)","author":"Radford"},{"key":"ref5","first-page":"4904","article-title":"Scaling Up Visual and Vision-Language Representation Learning with Noisy Text Supervision","volume-title":"Proceedings of the 38th International Conference on Machine Learning (ICML)","author":"Jia"},{"key":"ref6","first-page":"347","article-title":"Florence: A New Foundation Model for Computer Vision","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","author":"Yuan"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-019-0048-x"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/icra57147.2024.10611018"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/icra57147.2024.10611485"},{"article-title":"GAIA-1: A Generative World Model for Autonomous Driving","year":"2023","author":"Hu","key":"ref10"},{"key":"ref11","first-page":"9118","article-title":"Language Models as Zero-Shot Planners: Extracting Actionable Knowledge for Embodied Agents","volume-title":"Proceedings of the International Conference on Machine Learning (ICML)","author":"Huang"},{"article-title":"Dilu: A Knowledge-driven Approach to Autonomous Driving with Large Language Models","year":"2023","author":"Wen","key":"ref12"},{"article-title":"GPT-Driver: Learning to Drive with GPT","year":"2023","author":"Mao","key":"ref13"},{"article-title":"SurrealDriver: Designing Generative Driver Agent Simulation Framework in Urban Contexts Based on Large Language Model","year":"2023","author":"Jin","key":"ref14"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3440097"},{"article-title":"OPT: Open Pre-trained Transformer Language Models","year":"2022","author":"Zhang","key":"ref16"},{"key":"ref17","first-page":"19730","article-title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","volume-title":"Proceedings of the 40th International Conference on Machine Learning (ICML)","author":"Li"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1215"},{"article-title":"LLaMA: Open and Efficient Foundation Language Models","volume-title":"Proceedings of the 36th Annual Conference on Neural Information Processing Systems (NeurIPS)","author":"Touvron","key":"ref19"},{"article-title":"Finetuned Language Models Are Zero-Shot Learners","volume-title":"Proceedings of the 35th Conference on Neural Information Processing Systems (NeurIPS)","author":"Wei","key":"ref20"},{"article-title":"GPT-4 Technical Report","volume-title":"OpenAI","year":"2023","key":"ref21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00110"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"ref24","first-page":"74","article-title":"ROUGE: A Package for Automatic Evaluation of Summaries","volume-title":"Proceedings of the ACL Workshop on Text Summarization Branches Out","author":"Lin"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.360"},{"article-title":"BuboGPT: Enabling Visual Grounding in Multi-Modal LLMs","year":"2023","author":"Zhao","key":"ref28"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"article-title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","year":"2024","author":"Cheng","key":"ref30"},{"key":"ref31","first-page":"12888","article-title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation","volume-title":"Proceedings of the 39th International Conference on Machine Learning (ICML)","author":"Li"},{"issue":"6","key":"ref32","first-page":"51","article-title":"Transformative Fusion: Vision Transformers and GPT-2 Unleashing New Frontiers in Image Captioning","volume":"10","author":"Vasireddy","year":"2023","journal-title":"Int. J. Innov. Res. Eng. Manag."},{"key":"ref33","first-page":"19730","article-title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","volume-title":"Proceedings of the 40th International Conference on Machine Learning (ICML)","author":"Li"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2019.05.025"},{"key":"ref35","first-page":"443","article-title":"Improving Explainable Object-induced Model through Uncertainty for Automated Vehicles","volume-title":"Proceedings of the 2024 ACM\/IEEE International Conference on Human-Robot Interaction","author":"Ling"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ITSC58415.2024.10920066"},{"article-title":"Multi-Frame, Lightweight & Efficient Vision-Language Models for Question Answering in Autonomous Driving","year":"2024","author":"Gopalkrishnan","key":"ref37"}],"event":{"name":"2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","start":{"date-parts":[[2025,10,19]]},"location":"Hangzhou, China","end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11245651\/11245652\/11245662.pdf?arnumber=11245662","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T12:35:14Z","timestamp":1766061314000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11245662\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":37,"URL":"https:\/\/doi.org\/10.1109\/iros60139.2025.11245662","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}