{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,12]],"date-time":"2026-02-12T07:28:56Z","timestamp":1770881336565,"version":"3.50.1"},"publisher-location":"Berlin, Heidelberg","reference-count":15,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783540292432","type":"print"},{"value":"9783540316923","type":"electronic"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2005]]},"DOI":"10.1007\/11564096_29","type":"book-chapter","created":{"date-parts":[[2005,11,9]],"date-time":"2005-11-09T11:54:27Z","timestamp":1131537267000},"page":"280-291","source":"Crossref","is-referenced-by-count":82,"title":["Natural Actor-Critic"],"prefix":"10.1007","author":[{"given":"Jan","family":"Peters","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sethu","family":"Vijayakumar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stefan","family":"Schaal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"29_CR1","doi-asserted-by":"publisher","first-page":"251","DOI":"10.1162\/089976698300017746","volume":"10","author":"S. Amari","year":"1998","unstructured":"Amari, S.: Natural gradient works efficiently in learning. Neural Computation\u00a010, 251\u2013276 (1998)","journal-title":"Neural Computation"},{"key":"29_CR2","unstructured":"Bagnell, J., Schneider, J.: Covariant policy search. In: International Joint Conference on Artificial Intelligence (2003)"},{"key":"29_CR3","doi-asserted-by":"crossref","unstructured":"Baird, L.C.: Advantage Updating. Wright Lab. Tech. Rep. WL-TR-93-1146 (1993)","DOI":"10.21236\/ADA280862"},{"key":"29_CR4","unstructured":"Baird, L.C., Moore, A.W.: Gradient descent for general reinforcement learning. In: Advances in Neural Information Processing Systems 11 (1999)"},{"key":"29_CR5","series-title":"Lecture Notes in Artificial Intelligence","doi-asserted-by":"publisher","first-page":"184","DOI":"10.1007\/3-540-36434-X_5","volume-title":"Advanced Lectures on Machine Learning","author":"P. Bartlett","year":"2003","unstructured":"Bartlett, P.: An introduction to reinforcement learning theory: Value function methods. In: Mendelson, S., Smola, A.J. (eds.) Advanced Lectures on Machine Learning. LNCS (LNAI), vol.\u00a02600, pp. 184\u2013202. Springer, Heidelberg (2003)"},{"key":"29_CR6","volume-title":"Neuro-Dynamic Programming","author":"D.P. Bertsekas","year":"1996","unstructured":"Bertsekas, D.P., Tsitsiklis, J.N.: Neuro-Dynamic Programming. Athena Scientific, Belmont (1996)"},{"key":"29_CR7","unstructured":"Boyan, J.: Least-squares temporal difference learning. In: Machine Learning: Proceedings of the Sixteenth International Conference, pp. 49\u201356 (1999)"},{"key":"29_CR8","doi-asserted-by":"crossref","unstructured":"Bradtke, S., Ydstie, E., Barto, A.G.: Adaptive Linear Quadratic Control Using Policy Iteration. University of Massachusetts, Amherst, MA (1994)","DOI":"10.1109\/ACC.1994.735224"},{"key":"29_CR9","doi-asserted-by":"crossref","unstructured":"Ijspeert, A., Nakanishi, J., Schaal, S.: Learning rhythmic movements by demonstration using nonlinear oscillators. In: IEEE International Conference on Intelligent Robots and Systems (IROS 2002), pp. 958\u2013963 (2002)","DOI":"10.1109\/IRDS.2002.1041514"},{"key":"29_CR10","unstructured":"Kakade, S.A.: Natural policy gradient. In: Advances in Neural Information Processing Systems 14 (2002)"},{"key":"29_CR11","unstructured":"Konda, V., Tsitsiklis, J.: Actor-critic algorithms. In: Advances in Neural Information Processing Systems 12 (2000)"},{"key":"29_CR12","volume-title":"Mathematical Methods and Algorithms for Signal Processing","author":"T. Moon","year":"2000","unstructured":"Moon, T., Stirling, W.: Mathematical Methods and Algorithms for Signal Processing. Prentice Hall, Englewood Cliffs (2000)"},{"key":"29_CR13","unstructured":"Peters, J., Vijaykumar, S., Schaal, S.: Reinforcement learning for humanoid robotics. In: IEEE International Conference on Humandoid Robots (2003)"},{"key":"29_CR14","volume-title":"Reinforcement Learning","author":"R.S. Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning. MIT Press, Cambridge (1998)"},{"key":"29_CR15","unstructured":"Sutton, R.S., McAllester, D., Singh, S., Mansour, Y.: Policy gradient methods for reinforcement learning with function approximation. In: Advances in Neural Information Processing Systems 12 (2000)"}],"container-title":["Lecture Notes in Computer Science","Machine Learning: ECML 2005"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/11564096_29.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,17]],"date-time":"2020-11-17T19:52:49Z","timestamp":1605642769000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/11564096_29"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2005]]},"ISBN":["9783540292432","9783540316923"],"references-count":15,"URL":"https:\/\/doi.org\/10.1007\/11564096_29","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2005]]}}}