{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T15:01:46Z","timestamp":1786978906245,"version":"build-2736575974"},"publisher-location":"Cham","reference-count":43,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726972","type":"print"},{"value":"9783031726989","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72698-9_10","type":"book-chapter","created":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T00:45:57Z","timestamp":1729817157000},"page":"165-181","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Improving Agent Behaviors with\u00a0RL Fine-Tuning for\u00a0Autonomous Driving"],"prefix":"10.1007","author":[{"given":"Zhenghao","family":"Peng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenjie","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiren","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianyi","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cole","family":"Gulino","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ari","family":"Seff","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Justin","family":"Fu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,26]]},"reference":[{"issue":"5","key":"10_CR1","doi-asserted-by":"publisher","first-page":"469","DOI":"10.1016\/j.robot.2008.10.024","volume":"57","author":"BD Argall","year":"2009","unstructured":"Argall, B.D., Chernova, S., Veloso, M., Browning, B.: A survey of robot learning from demonstration. Robot. Auton. Syst. 57(5), 469\u2013483 (2009)","journal-title":"Robot. Auton. Syst."},{"key":"10_CR2","doi-asserted-by":"crossref","unstructured":"Bergamini, L., et al.: SimNet: learning reactive self-driving simulations from real-world observations. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 5119\u20135125. IEEE (2021)","DOI":"10.1109\/ICRA48506.2021.9561666"},{"key":"10_CR3","unstructured":"Black, K., Janner, M., Du, Y., Kostrikov, I., Levine, S.: Training diffusion models with reinforcement learning. arXiv preprint arXiv:2305.13301 (2023)"},{"key":"10_CR4","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"10_CR5","doi-asserted-by":"crossref","unstructured":"Codevilla, F., M\u00fcller, M., L\u00f3pez, A., Koltun, V., Dosovitskiy, A.: End-to-end driving via conditional imitation learning. In: 2018 IEEE International Conference on Robotics and Automation (ICRA), pp. 4693\u20134700. IEEE (2018)","DOI":"10.1109\/ICRA.2018.8460487"},{"key":"10_CR6","unstructured":"Dosovitskiy, A., Ros, G., Codevilla, F., Lopez, A., Koltun, V.: CARLA: an open urban driving simulator. In: Proceedings of the 1st Annual Conference on Robot Learning, pp. 1\u201316 (2017)"},{"key":"10_CR7","doi-asserted-by":"crossref","unstructured":"Feng, L., Li, Q., Peng, Z., Tan, S., Zhou, B.: TrafficGen: learning to generate diverse and realistic traffic scenarios. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 3567\u20133575. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10160296"},{"key":"10_CR8","unstructured":"Fu, J., et\u00a0al.: Benchmarks for deep off-policy evaluation. In: International Conference on Learning Representations (2020)"},{"key":"10_CR9","unstructured":"Gulino, C., et\u00a0al.: Waymax: an accelerated, data-driven simulator for large-scale autonomous driving research. arXiv preprint arXiv:2310.08710 (2023)"},{"key":"10_CR10","doi-asserted-by":"crossref","unstructured":"Hu, Y., et\u00a0al.: Planning-oriented autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17853\u201317862 (2023)","DOI":"10.1109\/CVPR52729.2023.01712"},{"key":"10_CR11","doi-asserted-by":"crossref","unstructured":"Kamenev, A., et al.: PredictionNet: real-time joint probabilistic traffic prediction for planning, control, and simulation. In: 2022 International Conference on Robotics and Automation (ICRA), pp. 8936\u20138942. IEEE (2022)","DOI":"10.1109\/ICRA46639.2022.9812223"},{"key":"10_CR12","unstructured":"Leurent, E.: An environment for autonomous driving decision-making. https:\/\/github.com\/eleurent\/highway-env (2018)"},{"key":"10_CR13","unstructured":"Li, Q., Peng, Z., Feng, L., Liu, Z., Duan, C., Mo, W., Zhou, B.: ScenarioNet: open-source platform for large-scale traffic scenario simulation and modeling. In: Advances in Neural Information Processing Systems (2023)"},{"key":"10_CR14","doi-asserted-by":"crossref","unstructured":"Li, Q., Peng, Z., Feng, L., Zhang, Q., Xue, Z., Zhou, B.: MetaDrive: composing diverse driving scenarios for generalizable reinforcement learning. IEEE Trans. Pattern Analysis Mach. Intell. (2022)","DOI":"10.1109\/TPAMI.2022.3190471"},{"key":"10_CR15","unstructured":"LLC, W.: Waymo open dataset: an autonomous driving dataset (2019)"},{"key":"10_CR16","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: ViLBERT: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"10_CR17","doi-asserted-by":"crossref","unstructured":"Lu, Y., et\u00a0al.: Imitation is not enough: Robustifying imitation with reinforcement learning for challenging driving scenarios. In: 2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 7553\u20137560. IEEE (2023)","DOI":"10.1109\/IROS55552.2023.10342038"},{"key":"10_CR18","unstructured":"Mendez-Lucio, O., Nicolaou, C.A., Earnshaw, B.: MolE: a molecular foundation model for drug discovery. In: NeurIPS 2022 Workshop on Learning Meaningful Representations of Life (2022)"},{"key":"10_CR19","unstructured":"Min, C., Zhao, D., Xiao, L., Nie, Y., Dai, B.: UniWorld: autonomous driving pre-training via world models. arXiv preprint arXiv:2308.07234 (2023)"},{"key":"10_CR20","unstructured":"Montali, N., et\u00a0al.: The Waymo open sim agents challenge. arXiv preprint arXiv:2305.12032 (2023)"},{"key":"10_CR21","unstructured":"Montali, N., et\u00a0al.: The Waymo open sim agents challenge. arXiv preprint arXiv:2305.12032 (2023)"},{"issue":"7956","key":"10_CR22","doi-asserted-by":"publisher","first-page":"259","DOI":"10.1038\/s41586-023-05881-4","volume":"616","author":"M Moor","year":"2023","unstructured":"Moor, M., et al.: Foundation models for generalist medical artificial intelligence. Nature 616(7956), 259\u2013265 (2023)","journal-title":"Nature"},{"key":"10_CR23","doi-asserted-by":"crossref","unstructured":"Nayakanti, N., Al-Rfou, R., Zhou, A., Goel, K., Refaat, K.S., Sapp, B.: Wayformer: motion forecasting via simple & efficient attention networks. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 2980\u20132987. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10160609"},{"key":"10_CR24","unstructured":"OpenAI: GPT-4 technical report (2023)"},{"key":"10_CR25","first-page":"27730","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback. Adv. Neural. Inf. Process. Syst. 35, 27730\u201327744 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"10_CR26","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.061251(2), 3 (2022)"},{"key":"10_CR27","unstructured":"Ross, S., Gordon, G., Bagnell, D.: A reduction of imitation learning and structured prediction to no-regret online learning. In: Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics, pp. 627\u2013635. JMLR Workshop and Conference Proceedings (2011)"},{"key":"10_CR28","doi-asserted-by":"crossref","unstructured":"Seff, A., et al.: MotionLM: multi-agent motion forecasting as language modeling. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8579\u20138590 (2023)","DOI":"10.1109\/ICCV51070.2023.00788"},{"key":"10_CR29","first-page":"6531","volume":"35","author":"S Shi","year":"2022","unstructured":"Shi, S., Jiang, L., Dai, D., Schiele, B.: Motion transformer with global intention localization and local movement refinement. Adv. Neural. Inf. Process. Syst. 35, 6531\u20136543 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"10_CR30","doi-asserted-by":"crossref","unstructured":"Shi, S., Jiang, L., Dai, D., Schiele, B.: MTR++: multi-agent motion prediction with symmetric scene modeling and guided intention querying. arXiv preprint arXiv:2306.17770 (2023)","DOI":"10.1109\/TPAMI.2024.3352811"},{"key":"10_CR31","doi-asserted-by":"crossref","unstructured":"Suo, S., Regalado, S., Casas, S., Urtasun, R.: TrafficSim: learning to simulate realistic multi-agent behaviors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10400\u201310409 (2021)","DOI":"10.1109\/CVPR46437.2021.01026"},{"key":"10_CR32","unstructured":"Uehara, M., Shi, C., Kallus, N.: A review of off-policy evaluation in reinforcement learning. arXiv preprint arXiv:2212.06355 (2022)"},{"key":"10_CR33","first-page":"3962","volume":"35","author":"E Vinitsky","year":"2022","unstructured":"Vinitsky, E., Lichtl\u00e9, N., Yang, X., Amos, B., Foerster, J.: Nocturne: a scalable driving benchmark for bringing multi-agent learning one step closer to the real world. Adv. Neural. Inf. Process. Syst. 35, 3962\u20133974 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"10_CR34","unstructured":"Wang, Y., Zhao, T., Yi, F.: Multiverse transformer: 1st place solution for Waymo open sim agents challenge (2023). arXiv preprint arXiv:2306.11868 (2023)"},{"key":"10_CR35","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1007\/BF00992696","volume":"8","author":"RJ Williams","year":"1992","unstructured":"Williams, R.J.: Simple statistical gradient-following algorithms for connectionist reinforcement learning. Mach. Learn. 8, 229\u2013256 (1992)","journal-title":"Mach. Learn."},{"key":"10_CR36","unstructured":"Zhang, C., Tu, J., Zhang, L., Wong, K., Suo, S., Urtasun, R.: Learning realistic traffic agents in closed-loop. In: 7th Annual Conference on Robot Learning (2023)"},{"issue":"12","key":"10_CR37","doi-asserted-by":"publisher","first-page":"24474","DOI":"10.1109\/TITS.2022.3202185","volume":"23","author":"Q Zhang","year":"2022","unstructured":"Zhang, Q., et al.: TrajGen: generating realistic and diverse trajectories with reactive and feasible agent behaviors for autonomous driving. IEEE Trans. Intell. Transp. Syst. 23(12), 24474\u201324487 (2022)","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10_CR38","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Liniger, A., Dai, D., Yu, F., Van\u00a0Gool, L.: TrafficBots: towards world models for autonomous driving simulation and motion prediction. arXiv preprint arXiv:2303.04116 (2023)","DOI":"10.1109\/ICRA48891.2023.10161243"},{"key":"10_CR39","doi-asserted-by":"crossref","unstructured":"Zhong, Z., Rempe, D., Xu, D., Chen, Y., Veer, S., Che, T., Ray, B., Pavone, M.: Guided conditional diffusion for controllable traffic simulation. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 3560\u20133566. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10161463"},{"key":"10_CR40","unstructured":"Zhou, C., et\u00a0al.: A comprehensive survey on pretrained foundation models: a history from BERT to ChatGPT. arXiv preprint arXiv:2302.09419 (2023)"},{"key":"10_CR41","unstructured":"Zhou, M., et al.: Smarts: scalable multi-agent reinforcement learning training school for autonomous driving (2020)"},{"key":"10_CR42","unstructured":"Zhou, Y., et\u00a0al.: A foundation model for generalizable disease detection from retinal images. Nature, 1\u20138 (2023)"},{"key":"10_CR43","unstructured":"Zitkovich, B., et\u00a0al.: RT-2: vision-language-action models transfer web knowledge to robotic control. In: 7th Annual Conference on Robot Learning (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72698-9_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T00:49:12Z","timestamp":1729817352000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72698-9_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,26]]},"ISBN":["9783031726972","9783031726989"],"references-count":43,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72698-9_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,26]]},"assertion":[{"value":"26 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}