{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T01:27:43Z","timestamp":1740101263434,"version":"3.37.3"},"reference-count":30,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,11,28]],"date-time":"2022-11-28T00:00:00Z","timestamp":1669593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,11,28]],"date-time":"2022-11-28T00:00:00Z","timestamp":1669593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001659","name":"German Research Foundation (DFG)","doi-asserted-by":"publisher","award":["SE 1042\/41-1,PE 2315\/14-1"],"award-info":[{"award-number":["SE 1042\/41-1,PE 2315\/14-1"]}],"id":[{"id":"10.13039\/501100001659","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,11,28]]},"DOI":"10.1109\/humanoids53995.2022.10000068","type":"proceedings-article","created":{"date-parts":[[2023,1,5]],"date-time":"2023-01-05T19:08:26Z","timestamp":1672945706000},"page":"587-593","source":"Crossref","is-referenced-by-count":0,"title":["Improving Sample Efficiency of Example-Guided Deep Reinforcement Learning for Bipedal Walking"],"prefix":"10.1109","author":[{"given":"Rustam","family":"Galljamov","sequence":"first","affiliation":[{"name":"Intelligent Autonomous Systems Lab,Department of Computer Science,TU Darmstadt,Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guoping","family":"Zhao","sequence":"additional","affiliation":[{"name":"Institute of Sport Science,Lauflabor Locomotion Laboratory,TU Darmstadt,Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Boris","family":"Belousov","sequence":"additional","affiliation":[{"name":"Intelligent Autonomous Systems Lab,Department of Computer Science,TU Darmstadt,Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andre","family":"Seyfarth","sequence":"additional","affiliation":[{"name":"Institute of Sport Science,Lauflabor Locomotion Laboratory,TU Darmstadt,Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jan","family":"Peters","sequence":"additional","affiliation":[{"name":"Intelligent Autonomous Systems Lab,Department of Computer Science,TU Darmstadt,Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"doi-asserted-by":"publisher","key":"ref1","DOI":"10.1109\/MEX.1986.4307016"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1098\/rsta.2006.1917"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1109\/HUMANOIDS.2014.7041347"},{"year":"2017","author":"Heess","journal-title":"Emergence of locomotion behaviours in rich environments","key":"ref4"},{"year":"2017","author":"Merel","journal-title":"Learning human behaviors from motion capture by adversarial imitation","key":"ref5"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1145\/3197517.3201311"},{"doi-asserted-by":"publisher","key":"ref7","DOI":"10.1109\/humanoids43949.2019.9035034"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1145\/3306346.3322972"},{"key":"ref9","article-title":"Deep reinforcement learning for modeling human locomotion control in neuromechanical simulation","author":"Song","year":"2020","journal-title":"bioRxiv"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1109\/IROS.2018.8593722"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.1126\/scirobotics.aau5872"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.15607\/rss.2020.xvi.064"},{"doi-asserted-by":"publisher","key":"ref13","DOI":"10.1007\/978-3-031-21090-7_31"},{"doi-asserted-by":"publisher","key":"ref14","DOI":"10.1109\/ICRA48506.2021.9561705"},{"key":"ref15","article-title":"Solving rubiks cube with a robot hand","author":"Akkaya","year":"2019","journal-title":"arXiv preprint"},{"doi-asserted-by":"publisher","key":"ref16","DOI":"10.1145\/3099564.3099567"},{"doi-asserted-by":"publisher","key":"ref17","DOI":"10.1145\/3424636.3426907"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1145\/3197517.3201397"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1145\/3359566.3360070"},{"doi-asserted-by":"publisher","key":"ref20","DOI":"10.1109\/LRA.2020.2972879"},{"year":"2017","author":"Schulman","journal-title":"Proximal Policy Optimization Algorithms","key":"ref21"},{"key":"ref22","article-title":"Soft actor-critic algorithms and applications","author":"Haarnoja","year":"2018","journal-title":"arXiv preprint"},{"key":"ref23","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","author":"Levine","year":"2020","journal-title":"arXiv preprint"},{"volume-title":"Stable baselines","year":"2018","author":"Hill","key":"ref24"},{"key":"ref25","article-title":"Implementation matters in deep RL: A case study on PPO and TRPO","volume-title":"8th International Conference on Learning Representations, ICLR 2020","author":"Engstrom","year":"2020"},{"key":"ref26","article-title":"What matters in on-policy reinforcement learning? a large-scale empirical study","author":"Andrychowicz","year":"2020","journal-title":"arXiv preprint"},{"doi-asserted-by":"publisher","key":"ref27","DOI":"10.1109\/HUMANOIDS.2014.7041473"},{"doi-asserted-by":"publisher","key":"ref28","DOI":"10.1088\/1748-3190\/ab6ed8"},{"doi-asserted-by":"publisher","key":"ref29","DOI":"10.1109\/LRA.2020.2976639"},{"doi-asserted-by":"publisher","key":"ref30","DOI":"10.1186\/s40537-019-0197-0"}],"event":{"name":"2022 IEEE-RAS 21st International Conference on Humanoid Robots (Humanoids)","start":{"date-parts":[[2022,11,28]]},"location":"Ginowan, Japan","end":{"date-parts":[[2022,11,30]]}},"container-title":["2022 IEEE-RAS 21st International Conference on Humanoid Robots (Humanoids)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9999736\/9999739\/10000068.pdf?arnumber=10000068","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,2]],"date-time":"2024-03-02T13:39:36Z","timestamp":1709386776000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10000068\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,28]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/humanoids53995.2022.10000068","relation":{},"subject":[],"published":{"date-parts":[[2022,11,28]]}}}