{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T15:02:46Z","timestamp":1782313366812,"version":"3.54.5"},"reference-count":69,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,5,13]]},"DOI":"10.1109\/icra57147.2024.10611614","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T17:51:05Z","timestamp":1723139465000},"page":"6798-6805","source":"Crossref","is-referenced-by-count":9,"title":["Prompting Multi-Modal Tokens to Enhance End-to-End Autonomous Driving Imitation Learning with LLMs"],"prefix":"10.1109","author":[{"given":"Yiqun","family":"Duan","sequence":"first","affiliation":[{"name":"University of Technology Sydney,HAI Centre, Australia Artificial Intelligence Institute, School of Computer Science,Ultimo,2007"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiang","family":"Zhang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou),School of Computer Science"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Renjing","family":"Xu","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou),School of Computer Science"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561747"},{"key":"ref2","article-title":"Hercules: An autonomous logistic vehicle for contact-less goods transportation during the covid-19 outbreak","author":"Liu","year":"2020"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561262"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3052442"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2020.2973615"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58589-1_36"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1186\/s13673-020-00231-z"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2019.2913998"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561663"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1002\/9781119434610.ch29"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-1529-6_16"},{"key":"ref12","first-page":"1403","article-title":"Transformer-fusion: Monocular rgb scene reconstruction using transformers","volume":"34","author":"Bozic","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref13","first-page":"1795","article-title":"Trip router with individualized preferences (trip): Incorporating personalization into route planning","volume-title":"AAAI","author":"Letchner"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18196\/jrc.v3i5.14683"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/41.303790"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/IVS.2015.7225830"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19839-7_31"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2020.3043505"},{"key":"ref19","article-title":"End to end learning for self-driving cars","author":"Bojarski","year":"2016"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01550"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00718"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00700"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9981775"},{"key":"ref24","article-title":"Bevdet: High-performance multi-camera 3d object detection in bird-eye-view","author":"Huang","year":"2021"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/icra48891.2023.10160968"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00335"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.691"},{"key":"ref28","first-page":"923","article-title":"End-to-end multi-view fusion for 3d object detection in lidar point clouds","volume-title":"Conference on Robot Learning","author":"Zhou"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i2.25233"},{"key":"ref30","article-title":"Safety-enhanced autonomous driving using interpretable sensor fusion transformer","author":"Shao","year":"2022"},{"key":"ref31","article-title":"Trajectory-guided control prediction for end-to-end autonomous driving: A simple yet strong baseline","volume-title":"NeurIPS","author":"Wu"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01712"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02105"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160326"},{"key":"ref35","article-title":"Palm-e: An embodied multimodal language model","author":"Driess","year":"2023"},{"key":"ref36","article-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","year":"2023"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.33540\/2168"},{"key":"ref38","first-page":"1","article-title":"Carla: An open urban driving simulator","volume-title":"Conference on robot learning","author":"Dosovitskiy"},{"key":"ref39","article-title":"Alvinn: An autonomous land vehicle in a neural network","volume-title":"Advances in Neural Information Processing Systems","volume":"1","author":"Pomerleau"},{"key":"ref40","article-title":"End to end learning for self-driving cars","author":"Bojarski","year":"2016"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8460487"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2019.xv.031"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8793742"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_27"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01530"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00718"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1989.1.4.541"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.4324\/9780203978948-13"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00766"},{"key":"ref51","article-title":"Deep reinforcement learning from human preferences","volume-title":"Advances in neural information processing systems","volume":"30","author":"Christiano"},{"key":"ref52","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume-title":"Advances in Neural Information Processing Systems","volume":"35","author":"Ouyang"},{"key":"ref53","article-title":"Palm: Scaling language modeling with pathways","author":"Chowdhery","year":"2022"},{"key":"ref54","article-title":"Scaling vision transformers to 22 billion parameters","author":"Dehghani","year":"2023"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01298"},{"key":"ref56","first-page":"4","article-title":"Precog: Predictions conditioned on goals in visual multi-agent scenarios","volume-title":"Proceedings of International Conference on Computer Vision","volume":"2","author":"Rhinehart"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01154"},{"key":"ref58","first-page":"66","article-title":"Learning by cheating","volume-title":"Conference on Robot Learning","author":"Chen"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/s11431-020-1582-8"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01499"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00376"},{"key":"ref62","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1145\/3410439"},{"key":"ref64","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref65","first-page":"726","article-title":"Safety-enhanced autonomous driving using interpretable sensor fusion transformer","volume-title":"Conference on Robot Learning","author":"Shao"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9811901"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01530"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.3390\/robotics12050127"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01671"}],"event":{"name":"2024 IEEE International Conference on Robotics and Automation (ICRA)","location":"Yokohama, Japan","start":{"date-parts":[[2024,5,13]]},"end":{"date-parts":[[2024,5,17]]}},"container-title":["2024 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10609961\/10609862\/10611614.pdf?arnumber=10611614","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,11]],"date-time":"2024-08-11T04:12:31Z","timestamp":1723349551000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10611614\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,13]]},"references-count":69,"URL":"https:\/\/doi.org\/10.1109\/icra57147.2024.10611614","relation":{},"subject":[],"published":{"date-parts":[[2024,5,13]]}}}