{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T13:18:46Z","timestamp":1765545526325},"publisher-location":"Berlin, Heidelberg","reference-count":16,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540691594"},{"type":"electronic","value":"9783540691624"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.1007\/978-3-540-69162-4_25","type":"book-chapter","created":{"date-parts":[[2008,7,31]],"date-time":"2008-07-31T06:38:20Z","timestamp":1217486300000},"page":"233-242","source":"Crossref","is-referenced-by-count":7,"title":["Policy Learning for Motor Skills"],"prefix":"10.1007","author":[{"given":"Jan","family":"Peters","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stefan","family":"Schaal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"unstructured":"Aberdeen, D.: POMDPs and policy gradients. In: Proceedings of the Machine Learning Summer School (MLSS), Canberra, Australia (2006)","key":"25_CR1"},{"unstructured":"Aberdeen, D.A.: Policy-Gradient Algorithms for Partially Observable Markov Decision Processes. PhD thesis, Australian National Unversity (2003)","key":"25_CR2"},{"issue":"2","key":"25_CR3","doi-asserted-by":"publisher","first-page":"271","DOI":"10.1162\/neco.1997.9.2.271","volume":"9","author":"P. Dayan","year":"1997","unstructured":"Dayan, P., Hinton, G.E.: Using expectation-maximization for reinforcement learning. Neural Computation\u00a09(2), 271\u2013278 (1997)","journal-title":"Neural Computation"},{"key":"25_CR4","first-page":"1547","volume-title":"Advances in Neural Information Processing Systems","author":"A. Ijspeert","year":"2003","unstructured":"Ijspeert, A., Nakanishi, J., Schaal, S.: Learning attractor landscapes for learning motor primitives. In: Becker, S., Thrun, S., Obermayer, K. (eds.) Advances in Neural Information Processing Systems, vol.\u00a015, pp. 1547\u20131554. MIT Press, Cambridge (2003)"},{"unstructured":"Kakade, S.A.: Natural policy gradient. In: Advances in Neural Information Processing Systems, Vancouver, CA, vol.\u00a014 (2002)","key":"25_CR5"},{"unstructured":"Konda, V., Tsitsiklis, J.: Actor-critic algorithms. Advances in Neural Information Processing Systems 12 (2000)","key":"25_CR6"},{"unstructured":"Peters, J.: The bias of the greedy update. Technical report, University of Southern California (2007)","key":"25_CR7"},{"doi-asserted-by":"crossref","unstructured":"Peters, J., Mistry, M., Udwadia, F., Cory, R., Nakanishi, J., Schaal, S.: A unifying methodology for the control of robotic systems. In: Proceedings of the IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), Edmonton, Canada (2005)","key":"25_CR8","DOI":"10.1109\/IROS.2005.1545516"},{"doi-asserted-by":"crossref","unstructured":"Peters, J., Schaal, S.: Learning operational space control. In: Proceedings of Robotics: Science and Systems (RSS), Philadelphia, PA (2006)","key":"25_CR9","DOI":"10.15607\/RSS.2006.II.033"},{"unstructured":"Peters, J., Vijayakumar, S., Schaal, S.: Reinforcement learning for humanoid robotics. In: Proceedings of the IEEE-RAS International Conference on Humanoid Robots (HUMANOIDS), Karlsruhe, Germany (September 2003)","key":"25_CR10"},{"key":"25_CR11","series-title":"Lecture Notes in Artificial Intelligence","doi-asserted-by":"publisher","first-page":"280","DOI":"10.1007\/11564096_29","volume-title":"Machine Learning: ECML 2005","author":"J. Peters","year":"2005","unstructured":"Peters, J., Vijayakumar, S., Schaal, S.: Natural actor-critic. In: Gama, J., Camacho, R., Brazdil, P.B., Jorge, A.M., Torgo, L. (eds.) ECML 2005. LNCS (LNAI), vol.\u00a03720, pp. 280\u2013291. Springer, Heidelberg (2005)"},{"key":"25_CR12","volume-title":"Advances in Neural Information Processing Systems","author":"S. Richter","year":"2007","unstructured":"Richter, S., Aberdeen, D., Yu, J.: Natural actor-critic for road traffic optimisation. In: Schoelkopf, B., Platt, J.C., Hofmann, T. (eds.) Advances in Neural Information Processing Systems, vol.\u00a019, MIT Press, Cambridge (2007)"},{"unstructured":"Schaal, S.: Dynamic movement primitives - a framework for motor control in humans and humanoid robots. In: Proceedings of the International Symposium on Adaptive Motion of Animals and Machines (2003)","key":"25_CR13"},{"key":"25_CR14","doi-asserted-by":"crossref","first-page":"199","DOI":"10.1093\/oso\/9780198529255.003.0009","volume-title":"The Neuroscience of Social Interaction","author":"S. Schaal","year":"2004","unstructured":"Schaal, S., Ijspeert, A., Billard, A.: Computational approaches to motor learning by imitation. In: Frith, C.D., Wolpert, D. (eds.) The Neuroscience of Social Interaction, pp. 199\u2013218. Oxford University Press, Oxford (2004)"},{"key":"25_CR15","volume-title":"Modeling and control of robot manipulators","author":"L. Sciavicco","year":"2007","unstructured":"Sciavicco, L., Siciliano, B.: Modeling and control of robot manipulators. MacGraw-Hill, Heidelberg (2007)"},{"key":"25_CR16","volume-title":"Advances in Neural Information Processing Systems (NIPS)","author":"R.S. Sutton","year":"2000","unstructured":"Sutton, R.S., McAllester, D., Singh, S., Mansour, Y.: Policy gradient methods for reinforcement learning with function approximation. In: Solla, S.A., Leen, T.K., Mueller, K.-R. (eds.) Advances in Neural Information Processing Systems (NIPS), Denver, CO, MIT Press, Cambridge (2000)"}],"container-title":["Lecture Notes in Computer Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-540-69162-4_25.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,28]],"date-time":"2024-02-28T21:32:30Z","timestamp":1709155950000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-540-69162-4_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[null]]},"ISBN":["9783540691594","9783540691624"],"references-count":16,"URL":"https:\/\/doi.org\/10.1007\/978-3-540-69162-4_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[]}}