{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T02:57:40Z","timestamp":1764212260736,"version":"3.37.3"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"23","license":[{"start":{"date-parts":[[2022,12,23]],"date-time":"2022-12-23T00:00:00Z","timestamp":1671753600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,12,23]],"date-time":"2022-12-23T00:00:00Z","timestamp":1671753600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100006245","name":"Ministry of Science and Technology, Israel","doi-asserted-by":"publisher","award":["1627\/17"],"award-info":[{"award-number":["1627\/17"]}],"id":[{"id":"10.13039\/501100006245","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007028","name":"Leona M. and Harry B. Helmsley Charitable Trust","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100007028","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2023,8]]},"DOI":"10.1007\/s00521-022-07947-2","type":"journal-article","created":{"date-parts":[[2022,12,23]],"date-time":"2022-12-23T07:02:46Z","timestamp":1671778966000},"page":"16791-16804","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Example-guided learning of stochastic human driving policies using deep reinforcement learning"],"prefix":"10.1007","volume":"35","author":[{"given":"Ran","family":"Emuna","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rotem","family":"Duffney","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Avinoam","family":"Borowsky","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0087-3675","authenticated-orcid":false,"given":"Armin","family":"Biess","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,12,23]]},"reference":[{"key":"7947_CR1","unstructured":"Li Y (2017) Deep reinforcement learning: an overview. arXiv preprint arXiv:1701.07274."},{"issue":"3\u20134","key":"7947_CR2","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1561\/2200000071","volume":"11","author":"V Fran\u00e7ois-Lavet","year":"2018","unstructured":"Fran\u00e7ois-Lavet V, Henderson P, Islam R, Bellemare MG, Pineau J et al (2018) An introduction to deep reinforcement learning. Found Trends Mach Learn 11(3\u20134):219\u2013354","journal-title":"Found Trends Mach Learn"},{"key":"7947_CR3","unstructured":"Heess N, TB D, Sriram S, Lemmon J, Merel J, Wayne G, Tassa Y, Erez T, Wang Z, Eslami S et al. (2017) Emergence of locomotion behaviours in rich environments. arXiv preprint arXiv:1707.02286"},{"issue":"50","key":"7947_CR4","doi-asserted-by":"publisher","first-page":"24972","DOI":"10.1073\/pnas.1820676116","volume":"116","author":"W Schwarting","year":"2019","unstructured":"Schwarting W, Pierson A, Alonso-Mora J, Karaman S, Rus D (2019) Social behavior for autonomous vehicles. Proc Natl Acad Sci 116(50):24972\u201324978","journal-title":"Proc Natl Acad Sci"},{"key":"7947_CR5","doi-asserted-by":"crossref","unstructured":"Puterman ML (1994) Markov decision processes: discrete stochastic dynamic programming. Wiley, New York","DOI":"10.1002\/9780470316887"},{"key":"7947_CR6","unstructured":"Schulman J, Levine S, Abbeel P, Jordan M, Moritz P (2015) Trust region policy optimization. ICML'15: Proceedings of the 32nd International Conference on International Conference on Machine Learning 37:1889\u20131897"},{"key":"7947_CR7","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O (2017) Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347."},{"key":"7947_CR8","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y (2014) Generative adversarial nets. In: Advances in neural information processing systems. p 2672\u20132680"},{"key":"7947_CR9","unstructured":"Ho J, Ermon S (2016) Generative adversarial imitation learning. In: Advances in neural information processing systems, vol 26. p 4565\u20134573"},{"issue":"6","key":"7947_CR10","doi-asserted-by":"publisher","first-page":"733","DOI":"10.1016\/0001-4575(94)90051-5","volume":"26","author":"TA Ranney","year":"1994","unstructured":"Ranney TA (1994) Models of driving behavior: a review of their evolution. Accid Anal Prev 26(6):733\u2013750","journal-title":"Accid Anal Prev"},{"issue":"3","key":"7947_CR11","doi-asserted-by":"publisher","first-page":"461","DOI":"10.1016\/j.aap.2004.11.003","volume":"37","author":"R Fuller","year":"2005","unstructured":"Fuller R (2005) Towards a general theory of driver behaviour. Accid Anal Prev 37(3):461\u2013472","journal-title":"Accid Anal Prev"},{"issue":"7\u20138","key":"7947_CR12","doi-asserted-by":"publisher","first-page":"699","DOI":"10.1080\/00423110701432482","volume":"45","author":"M Pl\u00f6chl","year":"2007","unstructured":"Pl\u00f6chl M, Edelmann J (2007) Driver models in automobile dynamics application. Veh Syst Dyn 45(7\u20138):699\u2013741","journal-title":"Veh Syst Dyn"},{"key":"7947_CR13","doi-asserted-by":"crossref","unstructured":"Grigorescu S, Trasnea B, Cocias T, Macesanu G (2020) A survey of deep learning techniques for autonomous driving. J Field Robot 37:362\u2013386","DOI":"10.1002\/rob.21918"},{"key":"7947_CR14","doi-asserted-by":"publisher","first-page":"102021","DOI":"10.1109\/ACCESS.2019.2926040","volume":"7","author":"L Fridman","year":"2019","unstructured":"Fridman L, Brown DE, Glazer M, Angell W, Dodd S, Jenik B, Terwilliger J, Patsekin A, Kindelsberger J, Ding L et al (2019) MIT advanced vehicle technology study: large-scale naturalistic driving study of driver behavior and interaction with automation. IEEE Access 7:102021\u2013102038","journal-title":"IEEE Access"},{"key":"7947_CR15","doi-asserted-by":"crossref","unstructured":"Kiran BR, Sobh I, Talpaert V, Mannion P, Al Sallab AA, Yogamani S, P\u00e9rez P (2021) Deep reinforcement learning for autonomous driving: a survey. IEEE Trans Intell Transp Syst","DOI":"10.1109\/TITS.2021.3054625"},{"key":"7947_CR16","doi-asserted-by":"crossref","unstructured":"Kuutti S, Bowden R, Jin Y, Barber P, Fallah S (2020) A survey of deep learning applications to autonomous vehicle control. IEEE Trans Intell Transp Syst 22(2):712\u2013733","DOI":"10.1109\/TITS.2019.2962338"},{"key":"7947_CR17","doi-asserted-by":"crossref","unstructured":"Zhu Z, Zhao H (2021) A survey of deep rl and il for autonomous driving policy learning. IEEE Trans Intell Transp Syst","DOI":"10.1109\/TITS.2021.3134702"},{"issue":"4","key":"7947_CR18","first-page":"143","volume":"37","author":"XB Peng","year":"2018","unstructured":"Peng XB, Abbeel P, Levine S, van de Panne M (2018) Deepmimic: Example-guided deep reinforcement learning of physics-based character skills. ACM Graph (TOG) 37(4):143","journal-title":"ACM Graph (TOG)"},{"issue":"8","key":"7947_CR19","doi-asserted-by":"publisher","first-page":"6788","DOI":"10.1109\/TVT.2018.2820002","volume":"67","author":"C Lu","year":"2018","unstructured":"Lu C, Wang H, Lv C, Gong J, Xi J, Cao D (2018) Learning driver-specific behavior for overtaking: a combined learning framework. IEEE Trans Veh Technol 67(8):6788\u20136802","journal-title":"IEEE Trans Veh Technol"},{"key":"7947_CR20","doi-asserted-by":"publisher","first-page":"348","DOI":"10.1016\/j.trc.2018.10.024","volume":"97","author":"M Zhu","year":"2018","unstructured":"Zhu M, Wang X, Wang Y (2018) Human-like autonomous car-following model with deep reinforcement learning. Transport Res Part C 97:348\u2013368","journal-title":"Transport Res Part C"},{"issue":"1\u20132","key":"7947_CR21","first-page":"1","volume":"7","author":"T Osa","year":"2018","unstructured":"Osa T, Pajarinen J, Neumann G, Bagnell JA, Abbeel P, Peters J et al (2018) An algorithmic perspective on imitation learning. Founda Trends Robot 7(1\u20132):1\u2013179","journal-title":"Founda Trends Robot"},{"key":"7947_CR22","unstructured":"Ng A.Y, Russell SJ (2000) et al. (2000) Algorithms for inverse reinforcement learning. ICML '00: Proceedings of the Seventeenth International Conference on Machine Learning, 663\u2013670"},{"key":"7947_CR23","doi-asserted-by":"crossref","unstructured":"Abbeel P, Ng AY (2004) Apprenticeship learning via inverse reinforcement learning. ICML '04: Proceedings of the twenty-first International Conference on Machine Learning, 2004","DOI":"10.1145\/1015330.1015430"},{"key":"7947_CR24","doi-asserted-by":"publisher","unstructured":"Kuderer M, Gulati S, Burgard W (2015) Learning driving styles for autonomous vehicles from demonstration. In: 2015 IEEE International Conference on Robotics and Automation (ICRA). p 2641\u20132646.   https:\/\/doi.org\/10.1109\/ICRA.2015.7139555","DOI":"10.1109\/ICRA.2015.7139555"},{"key":"7947_CR25","unstructured":"Levine S, Popovic Z, Koltun V (2011) Nonlinear inverse reinforcement learning with gaussian processes. In: Advances in Neural Information Processing Systems vol 24. p 19\u201327"},{"key":"7947_CR26","unstructured":"Levine S, Koltun V (2012) Continuous inverse optimal control with locally optimal examples. arXiv preprint arXiv:1206.4617"},{"key":"7947_CR27","unstructured":"Udacity: (2017) Udacity\u2019s self-driving car simulator. https:\/\/github.com\/udacity\/self-driving-car-sim"},{"key":"7947_CR28","unstructured":"Udacity: (2017) Self-driving car engineer nanodegree program. https:\/\/github.com\/udacity\/CarND-Path-Planning-Project"},{"key":"7947_CR29","volume-title":"Distributional prediction of human driving behaviours using mixture density networks","author":"K Leung","year":"2016","unstructured":"Leung K, Schmerling E, Pavone M (2016) Distributional prediction of human driving behaviours using mixture density networks. Stanford University, Stanford"},{"issue":"2","key":"7947_CR30","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1504\/IJVAS.2005.008237","volume":"3","author":"F Borrelli","year":"2005","unstructured":"Borrelli F, Falcone P, Keviczky T, Asgari J, Hrovat D (2005) MPC-based approach to active steering for autonomous vehicle systems. Int J Veh Auton Syst 3(2):265\u2013291","journal-title":"Int J Veh Auton Syst"},{"key":"7947_CR31","doi-asserted-by":"crossref","unstructured":"Kong J, Pfeiffer M, Schildbach G, Borrelli F (2015) Kinematic and dynamic vehicle models for autonomous driving control design. In: 2015 IEEE Intelligent vehicles symposium (IV), 1094\u20131099","DOI":"10.1109\/IVS.2015.7225830"},{"key":"7947_CR32","volume-title":"Machine learning: a probabilistic perspective","author":"KP Murphy","year":"2012","unstructured":"Murphy KP (2012) Machine learning: a probabilistic perspective. MIT Press, Cambridge"},{"key":"7947_CR33","unstructured":"Bishop C.M (1994) Mixture density networks. Neural Computing Research Group Report: NCRG\/94\/004"},{"key":"7947_CR34","unstructured":"Zolna K, Reed S, Novikov A, Colmenarej SG, Budden D, Cabi S, Denil M, de Freitas N, Wang Z (2019) Task-relevant adversarial imitation learning. arXiv preprint arXiv:1910.01077"},{"key":"7947_CR35","unstructured":"Peng XB, Kanazawa A, Toyer S, Abbeel P, Levine S (2018) Variational discriminator bottleneck: Improving imitation learning, inverse RL, and GANs by constraining information flow. arXiv preprint arXiv:1810.00821"},{"key":"7947_CR36","unstructured":"Wang R, Ciliberto C, Amadori PV, Demiris Y (2019) Random expert distillation: Imitation learning via expert policy support estimation. In: International Conference on Machine Learning, PMLR Vol 97. p 6536\u20136544"},{"key":"7947_CR37","unstructured":"Cobbe K, Klimov O, Hesse C, Kim T, Schulman J (2018) Quantifying generalization in reinforcement learning. arXiv preprint arXiv:1812.02341"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-022-07947-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-022-07947-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-022-07947-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,12]],"date-time":"2023-07-12T19:06:27Z","timestamp":1689188787000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-022-07947-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,23]]},"references-count":37,"journal-issue":{"issue":"23","published-print":{"date-parts":[[2023,8]]}},"alternative-id":["7947"],"URL":"https:\/\/doi.org\/10.1007\/s00521-022-07947-2","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"type":"print","value":"0941-0643"},{"type":"electronic","value":"1433-3058"}],"subject":[],"published":{"date-parts":[[2022,12,23]]},"assertion":[{"value":"19 January 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 October 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 December 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}