{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,5]],"date-time":"2024-09-05T04:39:11Z","timestamp":1725511151852},"reference-count":14,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2011,4]]},"DOI":"10.1109\/adprl.2011.5967352","type":"proceedings-article","created":{"date-parts":[[2011,8,4]],"date-time":"2011-08-04T01:40:00Z","timestamp":1312422000000},"page":"156-163","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing the episodic natural actor-critic algorithm by a regularisation term to stabilize learning of control structures"],"prefix":"10.1109","author":[{"given":"Andreas","family":"Witsch","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Roland","family":"Reichle","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kurt","family":"Geihs","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sascha","family":"Lange","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Martin","family":"Riedmiller","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"journal-title":"Pattern Recognition and Machine Learning (Information Science and Statistics)","year":"2007","author":"bishop","key":"ref10"},{"journal-title":"CLS2 Closed Loop Simulation System","year":"2004","author":"riedmiller","key":"ref11"},{"key":"ref12","article-title":"Applying policy gradient reinforcement learning to optimise robot behaviours","author":"witsch","year":"2010","journal-title":"Master's thesis"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ADPRL.2007.368196"},{"key":"ref14","article-title":"Carpe Noctem 2009","author":"baer","year":"2009","journal-title":"RoboCup 2009 International Symposium TU Graz"},{"journal-title":"Policy Gradient Methods for Reinforcement Learning with Function Approximation","year":"1999","author":"sutton","key":"ref4"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2008.02.003"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2007.11.026"},{"key":"ref8","article-title":"Machine learning of motor skills for robotics","author":"peters","year":"2007","journal-title":"Ph D Dissertation"},{"journal-title":"Matrix Algebra from a Statistician's Perspective","year":"2008","author":"harville","key":"ref7"},{"journal-title":"Reinforcement Learning An Introduction","year":"1998","author":"sutton","key":"ref2"},{"key":"ref1","article-title":"The &#x201C;echo state&#x201D; approach to analysing and training recurrent neural networks","author":"jaeger","year":"2001","journal-title":"Tech Rep"},{"journal-title":"Adaptive IIR Filtering in Signal Processing and Control (Electrical and Computer Engineering)","year":"1994","author":"regalia","key":"ref9"}],"event":{"name":"2011 Ieee Symposium On Adaptive Dynamic Programming And Reinforcement Learning","start":{"date-parts":[[2011,4,11]]},"location":"Paris, France","end":{"date-parts":[[2011,4,15]]}},"container-title":["2011 IEEE Symposium on Adaptive Dynamic Programming and Reinforcement Learning (ADPRL)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx5\/5958170\/5967347\/05967352.pdf?arnumber=5967352","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2017,3,21]],"date-time":"2017-03-21T09:16:04Z","timestamp":1490087764000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/5967352\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2011,4]]},"references-count":14,"URL":"https:\/\/doi.org\/10.1109\/adprl.2011.5967352","relation":{},"subject":[],"published":{"date-parts":[[2011,4]]}}}