{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,20]],"date-time":"2024-10-20T04:02:42Z","timestamp":1729396962490,"version":"3.27.0"},"reference-count":24,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,7,1]],"date-time":"2024-07-01T00:00:00Z","timestamp":1719792000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,7,1]],"date-time":"2024-07-01T00:00:00Z","timestamp":1719792000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,7,1]]},"DOI":"10.1109\/codit62066.2024.10708603","type":"proceedings-article","created":{"date-parts":[[2024,10,18]],"date-time":"2024-10-18T17:27:18Z","timestamp":1729272438000},"page":"1518-1523","source":"Crossref","is-referenced-by-count":0,"title":["Reward Planning For Underactuated Robotic Systems With Parameters Uncertainty: Greedy-Divide and Conquer"],"prefix":"10.1109","volume":"33","author":[{"given":"Sinan","family":"Ibrahim","sequence":"first","affiliation":[{"name":"Skolkovo Institute for Science and Technology,Center for Digital Engineering,Moscow,Russia,121205"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"S. M. Ahsan","family":"Kazmi","sequence":"additional","affiliation":[{"name":"University of the West of England (UWE),School of Computing and Creative Technologies,Bristol"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dmitrii","family":"Dobriborsci","sequence":"additional","affiliation":[{"name":"Technology Campus Cham,Deggendorf Institute of Technology,Cham,93413"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Roman","family":"Zashchitin","sequence":"additional","affiliation":[{"name":"Technology Campus Cham,Deggendorf Institute of Technology,Cham,93413"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mostafa","family":"Mostafa","sequence":"additional","affiliation":[{"name":"Skolkovo Institute for Science and Technology,Center for Digital Engineering,Moscow,Russia,121205"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pavel","family":"Osinenko","sequence":"additional","affiliation":[{"name":"Skolkovo Institute for Science and Technology,Center for Digital Engineering,Moscow,Russia,121205"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2021.103535"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/41.969389"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.01.096"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CAC48633.2019.8996875"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3169309"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/DCNA59899.2023.10290178"},{"key":"ref7","first-page":"15 931","article-title":"Learning to utilize shaping rewards: A new approach of reward shaping","volume":"33","author":"Hu","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref8","first-page":"15 281","article-title":"Unpacking reward shaping: Understanding the benefits of reward engineering on sample complexity","volume":"35","author":"Gupta","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.1995.478951"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/IECON.1991.239008"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/37.341864"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ACC.2007.4282176"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.2991\/eame-15.2015.108"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2020.12.2426"},{"key":"ref15","first-page":"145154","article-title":"Discrete-time online learning control for a class of unknown nonaffine nonlinear systems using reinforcement learning","volume":"438","author":"Xiong","journal-title":"Neurocomputing"},{"article-title":"Combining model-predictive control and predictive reinforcement learning for stable quadrupedal robot locomotion","year":"2023","author":"Kovalev","key":"ref16"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2015.2402512"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1108\/EC-08-2018-0356"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/s0005-1098(02)00046-8"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8793634"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.1997.657637"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1080\/00207179.2014.893450"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.3182\/20110828-6-IT-1002.01985"},{"article-title":"Proximal policy optimization algorithms","year":"2017","author":"Schulman","key":"ref24"}],"event":{"name":"2024 10th International Conference on Control, Decision and Information Technologies (CoDIT)","start":{"date-parts":[[2024,7,1]]},"location":"Vallette, Malta","end":{"date-parts":[[2024,7,4]]}},"container-title":["2024 10th International Conference on Control, Decision and Information Technologies (CoDIT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10708054\/10708053\/10708603.pdf?arnumber=10708603","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,19]],"date-time":"2024-10-19T04:50:45Z","timestamp":1729313445000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10708603\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,1]]},"references-count":24,"URL":"https:\/\/doi.org\/10.1109\/codit62066.2024.10708603","relation":{},"subject":[],"published":{"date-parts":[[2024,7,1]]}}}