{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T08:55:34Z","timestamp":1730278534092,"version":"3.28.0"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,9,24]],"date-time":"2023-09-24T00:00:00Z","timestamp":1695513600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,9,24]],"date-time":"2023-09-24T00:00:00Z","timestamp":1695513600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"publisher","award":["2021YFB2501201,2018AAA0102801"],"award-info":[{"award-number":["2021YFB2501201,2018AAA0102801"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010822","name":"Chengdu Science and Technology Bureau","doi-asserted-by":"publisher","award":["2021-YF08-00140-GX"],"award-info":[{"award-number":["2021-YF08-00140-GX"]}],"id":[{"id":"10.13039\/501100010822","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,9,24]]},"DOI":"10.1109\/itsc57777.2023.10422026","type":"proceedings-article","created":{"date-parts":[[2024,2,13]],"date-time":"2024-02-13T23:32:39Z","timestamp":1707867159000},"page":"1103-1109","source":"Crossref","is-referenced-by-count":0,"title":["MEERL: Maximum Experience Entropy Reinforcement Learning Method for Navigation and Control of Automated Vehicles"],"prefix":"10.1109","author":[{"given":"Xin","family":"Bi","sequence":"first","affiliation":[{"name":"School of Automotive Studies, Tongji University,Shanghai,China,201804"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Caien","family":"Weng","sequence":"additional","affiliation":[{"name":"School of Automotive Studies, Tongji University,Shanghai,China,201804"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Panpan","family":"Tong","sequence":"additional","affiliation":[{"name":"School of Automotive Studies, Tongji University,Shanghai,China,201804"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Automotive Studies, Tongji University,Shanghai,China,201804"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhichao","family":"Li","sequence":"additional","affiliation":[{"name":"School of Automotive Studies, Tongji University,Shanghai,China,201804"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv preprint"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref3","first-page":"465","article-title":"PILCO: A model-based and data-efficient approach to policy search","volume-title":"Proceedings of the 28th International Conference on machine learning (ICML-11)","author":"Deisenroth","year":"2011"},{"key":"ref4","article-title":"Deterministic policy gradient algorithms","author":"Silver","year":"2014","journal-title":"ICML"},{"key":"ref5","first-page":"1433","volume-title":"Maximum entropy inverse reinforcement learning","volume":"8","author":"Ziebart","year":"2008"},{"key":"ref6","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"Haarnoja","year":"2018","journal-title":"arXiv preprint"},{"key":"ref7","first-page":"7120","article-title":"Virel: A variational inference framework for reinforcement learning","author":"Fellows","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref8","article-title":"Making Sense of Reinforcement Learning and Probabilistic Inference","author":"ODonoghue","year":"2020","journal-title":"arXiv preprint"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1103\/PhysRev.106.620"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/9816.003.0050"},{"issue":"1","key":"ref11","first-page":"1334","article-title":"End-to-end training of deep visuomotor policies","volume":"17","author":"Levine","year":"2016","journal-title":"The Journal of Machine Learning Research"},{"key":"ref12","first-page":"2775","article-title":"Bridging the gap between value and policy based reinforcement learning","author":"Nachum","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref13","first-page":"1352","article-title":"Reinforcement learning with deep energy-based policies","volume-title":"Proceedings of the 34th International Conference on Machine Learning","volume":"70","author":"Haarnoja","year":"2017"},{"key":"ref14","first-page":"7120","article-title":"A distributional perspective on reinforcement learning","author":"Bellemare","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref15","article-title":"Provably efficient maximum entropy exploration","author":"Hazan","year":"2018","journal-title":"arXiv preprint"},{"key":"ref16","first-page":"10489","article-title":"Diversity-driven exploration strategy for deep reinforcement learning","author":"Hong","year":"2018","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref17","article-title":"Auto-encoding variational bayes","author":"Kingma","year":"2013","journal-title":"arXiv preprint"},{"key":"ref18","first-page":"3483","article-title":"Learning structured output representation using deep conditional generative models","author":"Sohn","year":"2015","journal-title":"Advances in neural information processing systems"},{"key":"ref19","article-title":"Pixel recurrent neural networks","author":"Oord","year":"2016","journal-title":"arXiv preprint"},{"key":"ref20","first-page":"2378","article-title":"Stein variational gradient descent: A general purpose bayesian inference algorithm","author":"Liu","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref21","article-title":"Implicit quantile networks for distributional reinforcement learning","author":"Dabney","year":"2018","journal-title":"arXiv preprint"},{"key":"ref22","first-page":"1350","article-title":"Distributional policy optimization: An alternative approach for continuous control","author":"Tessler","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/tnn.1998.712192"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1257\/jep.15.4.143"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11791"}],"event":{"name":"2023 IEEE 26th International Conference on Intelligent Transportation Systems (ITSC)","start":{"date-parts":[[2023,9,24]]},"location":"Bilbao, Spain","end":{"date-parts":[[2023,9,28]]}},"container-title":["2023 IEEE 26th International Conference on Intelligent Transportation Systems (ITSC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10420842\/10420843\/10422026.pdf?arnumber=10422026","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,14]],"date-time":"2024-03-14T07:58:27Z","timestamp":1710403107000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10422026\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,24]]},"references-count":26,"URL":"https:\/\/doi.org\/10.1109\/itsc57777.2023.10422026","relation":{},"subject":[],"published":{"date-parts":[[2023,9,24]]}}}