{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T14:32:09Z","timestamp":1784817129174,"version":"3.55.0"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,2,23]],"date-time":"2023-02-23T00:00:00Z","timestamp":1677110400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,2,23]],"date-time":"2023-02-23T00:00:00Z","timestamp":1677110400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["52102460"],"award-info":[{"award-number":["52102460"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61903220"],"award-info":[{"award-number":["61903220"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U1864203"],"award-info":[{"award-number":["U1864203"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2021M701883"],"award-info":[{"award-number":["2021M701883"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100009592","name":"Beijing Municipal Science and Technology Commission","doi-asserted-by":"publisher","award":["Z221100008122011"],"award-info":[{"award-number":["Z221100008122011"]}],"id":[{"id":"10.13039\/501100009592","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Nat Mach Intell"],"DOI":"10.1038\/s42256-023-00610-y","type":"journal-article","created":{"date-parts":[[2023,2,23]],"date-time":"2023-02-23T17:03:30Z","timestamp":1677171810000},"page":"145-158","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":86,"title":["Continuous improvement of self-driving cars using dynamic confidence-aware reinforcement learning"],"prefix":"10.1038","volume":"5","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2243-5705","authenticated-orcid":false,"given":"Zhong","family":"Cao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kun","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1266-3843","authenticated-orcid":false,"given":"Weitao","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shaobing","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huei","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0825-5609","authenticated-orcid":false,"given":"Diange","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,2,23]]},"reference":[{"key":"610_CR1","unstructured":"Sutton, R. S. & Barto, A. G. Reinforcement Learning: An Introduction (MIT Press, 2018)."},{"key":"610_CR2","doi-asserted-by":"publisher","first-page":"1140","DOI":"10.1126\/science.aar6404","volume":"362","author":"D Silver","year":"2018","unstructured":"Silver, D. et al. A general reinforcement learning algorithm that masters chess, shogi, and Go through self-play. Science 362, 1140\u20131144 (2018).","journal-title":"Science"},{"key":"610_CR3","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D. et al. Mastering the game of Go with deep neural networks and tree search. Nature 529, 484\u2013489 (2016).","journal-title":"Nature"},{"key":"610_CR4","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V. et al. Human-level control through deep reinforcement learning. Nature 518, 529\u2013533 (2015).","journal-title":"Nature"},{"key":"610_CR5","doi-asserted-by":"crossref","unstructured":"Ye, F., Zhang, S., Wang, P. & Chan, C.-Y. A survey of deep reinforcement learning algorithms for motion planning and control of autonomous vehicles. In 2021 IEEE Intelligent Vehicles Symposium (IV) 1073\u20131080 (IEEE, 2021).","DOI":"10.1109\/IV48863.2021.9575880"},{"key":"610_CR6","doi-asserted-by":"crossref","unstructured":"Zhu, Z. & Zhao, H. A survey of deep RL and IL for autonomous driving policy learning. IEEE Trans. Intell. Transp. Syst. 23, 14043\u201314065 (2022).","DOI":"10.1109\/TITS.2021.3134702"},{"key":"610_CR7","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1109\/TITS.2020.3024655","volume":"23","author":"S Aradi","year":"2022","unstructured":"Aradi, S. Survey of deep reinforcement learning for motion planning of autonomous vehicles. IEEE Trans. Intell. Transp. Syst. 23, 740\u2013759 (2022).","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"610_CR8","doi-asserted-by":"publisher","first-page":"990","DOI":"10.1109\/TITS.2019.2961739","volume":"22","author":"Z Cao","year":"2020","unstructured":"Cao, Z. et al. Highway exiting planner for automated vehicles using reinforcement learning. IEEE Trans. Intell. Transp. Syst. 22, 990\u20131000 (2020).","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"610_CR9","doi-asserted-by":"publisher","first-page":"202","DOI":"10.1038\/s42256-019-0046-z","volume":"1","author":"J Stilgoe","year":"2019","unstructured":"Stilgoe, J. Self-driving cars will take a while to get right. Nat. Mach. Intell. 1, 202\u2013203 (2019).","journal-title":"Nat. Mach. Intell."},{"key":"610_CR10","first-page":"182","volume":"94","author":"N Kalra","year":"2016","unstructured":"Kalra, N. & Paddock, S. M. Driving to safety: How many miles of driving would it take to demonstrate autonomous vehicle reliability? Transp. Res. Part A 94, 182\u2013193 (2016).","journal-title":"Transp. Res. Part A"},{"key":"610_CR11","unstructured":"Disengagement reports. California DMV https:\/\/www.dmv.ca.gov\/portal\/vehicle-industry-services\/autonomous-vehicles\/disengagement-reports\/ (2021)."},{"key":"610_CR12","doi-asserted-by":"publisher","first-page":"103452","DOI":"10.1016\/j.trc.2021.103452","volume":"134","author":"G Li","year":"2022","unstructured":"Li, G. et al. Decision making of autonomous vehicles in lane change scenarios: deep reinforcement learning approaches with risk awareness. Transp. Res. Part C 134, 103452 (2022).","journal-title":"Transp. Res. Part C"},{"key":"610_CR13","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1109\/TVT.2021.3121985","volume":"71","author":"H Shu","year":"2021","unstructured":"Shu, H., Liu, T., Mu, X. & Cao, D. Driving tasks transfer using deep reinforcement learning for decision-making of autonomous vehicles in unsignalized intersection. IEEE Trans. Veh. Technol. 71, 41\u201352 (2021).","journal-title":"IEEE Trans. Veh. Technol."},{"key":"610_CR14","doi-asserted-by":"publisher","first-page":"518","DOI":"10.1038\/s42256-020-0225-y","volume":"2","author":"C Pek","year":"2020","unstructured":"Pek, C., Manzinger, S., Koschi, M. & Althoff, M. Using online verification to prevent autonomous vehicles from causing accidents. Nat. Mach. Intell. 2, 518\u2013528 (2020).","journal-title":"Nat. Mach. Intell."},{"key":"610_CR15","doi-asserted-by":"publisher","first-page":"29944","DOI":"10.1109\/ACCESS.2020.2972329","volume":"8","author":"S Xu","year":"2020","unstructured":"Xu, S., Peng, H., Lu, P., Zhu, M. & Tang, Y. Design and experiments of safeguard protected preview lane keeping control for autonomous vehicles. IEEE Access 8, 29944\u201329953 (2020).","journal-title":"IEEE Access"},{"key":"610_CR16","doi-asserted-by":"publisher","unstructured":"Yang, J., Zhang, J., Xi, M., Lei, Y. & Sun, Y. A deep reinforcement learning algorithm suitable for autonomous vehicles: double bootstrapped soft-actor-critic-discrete. IEEE Trans. Cogn. Dev. Syst. https:\/\/doi.org\/10.1109\/TCDS.2021.3092715 (2021).","DOI":"10.1109\/TCDS.2021.3092715"},{"key":"610_CR17","doi-asserted-by":"publisher","unstructured":"Schwall, M., Daniel, T., Victor, T., Favaro, F. & Hohnhold, H. Waymo public road safety performance data. Preprint at arXiv https:\/\/doi.org\/10.48550\/arXiv.2011.00038 (2020).","DOI":"10.48550\/arXiv.2011.00038"},{"key":"610_CR18","doi-asserted-by":"publisher","unstructured":"Fan, H. et al. Baidu Apollo EM motion planner. Preprint at arXiv https:\/\/doi.org\/10.48550\/arXiv.1807.08048 (2018).","DOI":"10.48550\/arXiv.1807.08048"},{"key":"610_CR19","doi-asserted-by":"crossref","unstructured":"Kato, S. et al. Autoware on board: enabling autonomous vehicles with embedded systems. In 2018 ACM\/IEEE 9th International Conference on Cyber-Physical Systems 287\u2013296 (IEEE, 2018).","DOI":"10.1109\/ICCPS.2018.00035"},{"key":"610_CR20","doi-asserted-by":"publisher","first-page":"7419","DOI":"10.1109\/TITS.2021.3069497","volume":"23","author":"Z Cao","year":"2022","unstructured":"Cao, Z., Xu, S., Peng, H., Yang, D. & Zidek, R. Confidence-aware reinforcement learning for self-driving cars. IEEE Trans. Intell. Transp. Syst. 23, 7419\u20137430 (2022).","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"610_CR21","doi-asserted-by":"publisher","first-page":"999","DOI":"10.1126\/science.aag3311","volume":"366","author":"PS Thomas","year":"2019","unstructured":"Thomas, P. S. et al. Preventing undesirable behavior of intelligent machines. Science 366, 999\u20131004 (2019).","journal-title":"Science"},{"key":"610_CR22","doi-asserted-by":"publisher","unstructured":"Levine, S., Kumar, A., Tucker, G. & Fu, J. Offline reinforcement learning: tutorial, review, and perspectives on open problems. Preprint at arXiv https:\/\/doi.org\/10.48550\/arXiv.2005.01643 (2020).","DOI":"10.48550\/arXiv.2005.01643"},{"key":"610_CR23","first-page":"1437","volume":"16","author":"J Garc\u0131a","year":"2015","unstructured":"Garc\u0131a, J. & Fern\u00e1ndez, F. A comprehensive survey on safe reinforcement learning. J. Mach. Learn. Res. 16, 1437\u20131480 (2015).","journal-title":"J. Mach. Learn. Res."},{"key":"610_CR24","unstructured":"Achiam, J., Held, D., Tamar, A. & Abbeel, P. Constrained policy optimization. In International Conference on Machine Learning 22\u201331 (JMLR, 2017)."},{"key":"610_CR25","unstructured":"Berkenkamp, F., Turchetta, M., Schoellig, A. & Krause, A. Safe model-based reinforcement learning with stability guarantees. Adv. Neural Inf. Process. Syst. 30, 908-919 (2017)."},{"key":"610_CR26","doi-asserted-by":"crossref","unstructured":"Ghadirzadeh, A., Maki, A., Kragic, D. & Bj\u00f6rkman, M. Deep predictive policy training using reinforcement learning. In 2017 IEEE\/RSJ International Conference on Intelligent Robots and Systems 2351\u20132358 (IEEE, 2017).","DOI":"10.1109\/IROS.2017.8206046"},{"key":"610_CR27","doi-asserted-by":"crossref","unstructured":"Abbeel, P. & Ng, A. Y. Apprenticeship learning via inverse reinforcement learning. In Proc. Twenty-first International Conference on Machine Learning, 1 (Association for Computing Machinery, 2004).","DOI":"10.1145\/1015330.1015430"},{"key":"610_CR28","doi-asserted-by":"crossref","unstructured":"Abbeel, P. & Ng, A. Y. Exploration and apprenticeship learning in reinforcement learning. In Proc. 22nd International Conference on Machine Learning 1\u20138 (Association for Computing Machinery, 2005).","DOI":"10.1145\/1102351.1102352"},{"key":"610_CR29","unstructured":"Ross, S., Gordon, G. & Bagnell, D. A reduction of imitation learning and structured prediction to no-regret online learning. In Gordon, G., Dunson, D. & Dud\u00edk, M. (eds) Proc. Fourteenth International Conference on Artificial Intelligence and Statistics, 627\u2013635 (JMLR, 2011)."},{"key":"610_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, J. & Cho, K. Query-efficient imitation learning for end-to-end autonomous driving. In Thirty-First AAAI Conference on Artificial Intelligence (AAAI), 2891\u20132897 (AAAI Press, 2017).","DOI":"10.1609\/aaai.v31i1.10857"},{"key":"610_CR31","doi-asserted-by":"crossref","unstructured":"Bicer, Y., Alizadeh, A., Ure, N. K., Erdogan, A. & Kizilirmak, O. Sample efficient interactive end-to-end deep learning for self-driving cars with selective multi-class safe dataset aggregation. In 2019 IEEE\/RSJ International Conference on Intelligent Robots and Systems 2629\u20132634 (IEEE, 2019).","DOI":"10.1109\/IROS40897.2019.8967948"},{"key":"610_CR32","doi-asserted-by":"crossref","unstructured":"Alshiekh, M. et al Safe reinforcement learning via shielding. In Proc. Thirty-Second AAAI Conference on Artificial Intelligence Vol. 32, 2669-2678 (AAAI Press, 2018).","DOI":"10.1609\/aaai.v32i1.11797"},{"key":"610_CR33","unstructured":"Brun, W., Keren, G., Kirkeboen, G. & Montgomery, H. Perspectives on Thinking, Judging, and Decision Making (Universitetsforlaget, 2011)."},{"key":"610_CR34","doi-asserted-by":"publisher","first-page":"671","DOI":"10.1038\/s41586-019-1924-6","volume":"577","author":"W Dabney","year":"2020","unstructured":"Dabney, W. et al. A distributional code for value in dopamine-based reinforcement learning. Nature 577, 671\u2013675 (2020).","journal-title":"Nature"},{"key":"610_CR35","doi-asserted-by":"publisher","first-page":"274","DOI":"10.1016\/j.trc.2019.03.009","volume":"102","author":"Z Cao","year":"2019","unstructured":"Cao, Z. et al. A geometry-driven car-following distance estimation algorithm robust to road slopes. Transp. Res. Part C 102, 274\u2013288 (2019).","journal-title":"Transp. Res. Part C"},{"key":"610_CR36","doi-asserted-by":"publisher","first-page":"5975","DOI":"10.1109\/TSMC.2021.3131141","volume":"52","author":"S Xu","year":"2022","unstructured":"Xu, S. et al. System and experiments of model-driven motion planning and control for autonomous vehicles. IEEE Trans. Syst. Man. Cybern. Syst. 52, 5975\u20135988 (2022).","journal-title":"IEEE Trans. Syst. Man. Cybern. Syst."},{"key":"610_CR37","unstructured":"Cao, Z. Codes and data for dynamic confidence-aware reinforcement learning. DCARL. Zenodo https:\/\/zenodo.org\/badge\/latestdoi\/578512035 (2022)."},{"key":"610_CR38","doi-asserted-by":"crossref","unstructured":"Kochenderfer, M. J. Decision Making Under Uncertainty: Theory and Application (MIT Press, 2015).","DOI":"10.7551\/mitpress\/10187.001.0001"},{"key":"610_CR39","doi-asserted-by":"crossref","unstructured":"Ivanovic, B. et al. Heterogeneous-agent trajectory forecasting incorporating class uncertainty. In 2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), 12196\u201312203 (IEEE, 2022).","DOI":"10.1109\/IROS47612.2022.9982283"},{"key":"610_CR40","unstructured":"Yang, Y., Zha, K., Chen, Y., Wang, H. & Katabi, D. Delving into deep imbalanced regression. In International Conference on Machine Learning 11842\u201311851 (PMLR, 2021)."},{"key":"610_CR41","doi-asserted-by":"crossref","unstructured":"Efron, B. & Tibshirani, R. J. An Introduction to the Bootstrap (CRC Press, 1994).","DOI":"10.1201\/9780429246593"},{"key":"610_CR42","unstructured":"Dosovitskiy, A., Ros, G., Codevilla, F., Lopez, A. & Koltun, V. CARLA: An open urban driving simulator. In Proceedings of the 1st Annual Conference on Robot Learning, 1\u201316 (PMLR, 2017)."}],"container-title":["Nature Machine Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.nature.com\/articles\/s42256-023-00610-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s42256-023-00610-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s42256-023-00610-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,23]],"date-time":"2023-02-23T17:17:26Z","timestamp":1677172646000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.nature.com\/articles\/s42256-023-00610-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,23]]},"references-count":42,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2023,2]]}},"alternative-id":["610"],"URL":"https:\/\/doi.org\/10.1038\/s42256-023-00610-y","relation":{},"ISSN":["2522-5839"],"issn-type":[{"value":"2522-5839","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,2,23]]},"assertion":[{"value":"14 May 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 January 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 February 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}