{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T11:45:39Z","timestamp":1784547939821,"version":"3.55.0"},"reference-count":25,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016,12]]},"DOI":"10.1109\/cdc.2016.7798980","type":"proceedings-article","created":{"date-parts":[[2017,1,5]],"date-time":"2017-01-05T12:11:18Z","timestamp":1483618278000},"page":"4667-4673","source":"Crossref","is-referenced-by-count":33,"title":["Learning state representation for deep actor-critic control"],"prefix":"10.1109","author":[{"given":"Jelle","family":"Munk","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jens","family":"Kober","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Robert","family":"Babuska","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","article-title":"Learning grounded relational symbols from continuous data for abstract reasoning","author":"jetchev","year":"2013","journal-title":"Workshop on Autonomous Learning Int Conf on Robotics and Automation (ICRA)"},{"key":"ref11","article-title":"Offroad obstacle avoidance through end-to-end learning","author":"muller","year":"2005","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.2174\/1573399812666160613113556"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/s13218-015-0356-1"},{"key":"ref14","volume":"39","author":"bu?oniu","year":"2010","journal-title":"Reinforcement Learning and Dynamic Programming Using Function Approximators"},{"key":"ref15","article-title":"Policy gradient methods for reinforcement learning with function approximation","author":"sutton","year":"2000","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref16","article-title":"Deterministic policy gradient algorithms","author":"silver","year":"2014","journal-title":"Proc of the International Conference on Machine Learning (ICML)"},{"key":"ref17","article-title":"The importance of experience replay database composition in deep reinforcement learning","author":"de bruin","year":"2015","journal-title":"Deep Reinforcement Learning Workshop Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref18","article-title":"Learning visual feature spaces for robotic manipulation with deep spatial autoencoders","author":"finn","year":"2015"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2015.12.271"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.1983.6313077"},{"key":"ref3","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"2015"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1038\/nrn3112"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2009.05.011"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2012.6252823"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2014.X.019"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1162\/089976602317318938"},{"key":"ref1","article-title":"Neural fitted Q iteration - first experiences with a data efficient neural reinforcement learning method","author":"riedmiller","year":"2005","journal-title":"Euro Conf on Machine Learning (ECML)"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pcbi.1000894"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCB.2011.2170565"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2013.50"},{"key":"ref24","article-title":"Adam: a method for stochastic optimization","author":"kingma","year":"2015","journal-title":"Proc of the Int Conf on Learning Representations (ICLR)"},{"key":"ref23","article-title":"Learning to control an octopus arm with Gaussian process temporal difference methods","author":"engel","year":"2005","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1103\/PhysRev.36.823"}],"event":{"name":"2016 IEEE 55th Conference on Decision and Control (CDC)","location":"Las Vegas, NV, USA","start":{"date-parts":[[2016,12,12]]},"end":{"date-parts":[[2016,12,14]]}},"container-title":["2016 IEEE 55th Conference on Decision and Control (CDC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7786694\/7798233\/07798980.pdf?arnumber=7798980","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2017,12,13]],"date-time":"2017-12-13T16:09:34Z","timestamp":1513181374000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7798980\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,12]]},"references-count":25,"URL":"https:\/\/doi.org\/10.1109\/cdc.2016.7798980","relation":{},"subject":[],"published":{"date-parts":[[2016,12]]}}}