{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,13]],"date-time":"2026-05-13T17:45:31Z","timestamp":1778694331984,"version":"3.51.4"},"publisher-location":"Cham","reference-count":285,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319325507","type":"print"},{"value":"9783319325521","type":"electronic"}],"license":[{"start":{"date-parts":[[2016,1,1]],"date-time":"2016-01-01T00:00:00Z","timestamp":1451606400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2016,1,1]],"date-time":"2016-01-01T00:00:00Z","timestamp":1451606400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016]]},"DOI":"10.1007\/978-3-319-32552-1_15","type":"book-chapter","created":{"date-parts":[[2016,7,27]],"date-time":"2016-07-27T19:03:33Z","timestamp":1469646213000},"page":"357-398","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":21,"title":["Robot Learning"],"prefix":"10.1007","author":[{"given":"Jan","family":"Peters","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daniel D.","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jens","family":"Kober","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Duy","family":"Nguyen-Tuong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"J. Andrew","family":"Bagnell","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stefan","family":"Schaal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,7,27]]},"reference":[{"issue":"2","key":"15_CR1","first-page":"115","volume":"1","author":"S. Schaal","year":"2007","unstructured":"S. Schaal: The new robotics\u00a0\u2013 Towards human-centered machines, HFSP J. Front. Interdiscip. Res, Life Sci. 1(2), 115\u2013126 (2007)","journal-title":"Life Sci."},{"key":"15_CR2","volume-title":"AAAI Conf. Artif. Intell.","author":"B.D. Ziebart","year":"2008","unstructured":"B.D. Ziebart, A. Maas, J.A. Bagnell, A.K. Dey: Maximum entropy inverse reinforcement learning, AAAI Conf. Artif. Intell. (2008)"},{"key":"15_CR3","volume-title":"Probabilistic Robotics","author":"S. Thrun","year":"2005","unstructured":"S. Thrun, W. Burgard, D. Fox: Probabilistic Robotics (MIT, Cambridge 2005)"},{"key":"15_CR4","series-title":"Stud. Comput. Intell.","volume-title":"Machine Learning and Robot Perception","year":"2005","unstructured":"B. Apolloni, A. Ghosh, F. Alpaslan, L.C. Jain, S. Patnaik (Eds.): Machine Learning and Robot Perception, Stud. Comput. Intell., Vol. 7 (Springer, Berlin, Heidelberg 2005)"},{"key":"15_CR5","volume-title":"Int. Conf. Dev. Learn.","author":"O. Jenkins","year":"2006","unstructured":"O. Jenkins, R. Bodenheimer, R. Peters: Manipulation manifolds: Explorations into uncovering manifolds in sensory-motor spaces, Int. Conf. Dev. Learn. (2006)"},{"key":"15_CR6","volume-title":"Tutor. Conf. Mach. Learn.","author":"M. Toussaint","year":"2011","unstructured":"M. Toussaint: Machine learning and robotics, Tutor. Conf. Mach. Learn. (2011)"},{"key":"15_CR7","volume-title":"Dynamic Programming and Optimal Control","author":"D.P. Bertsekas","year":"1995","unstructured":"D.P. Bertsekas: Dynamic Programming and Optimal Control (Athena Scientific, Nashua 1995)"},{"issue":"1","key":"15_CR8","doi-asserted-by":"publisher","first-page":"51","DOI":"10.1115\/1.3653115","volume":"86","author":"R.E. Kalman","year":"1964","unstructured":"R.E. Kalman: When is a\u00a0linear control system optimal?, J. Basic Eng. 86(1), 51\u201360 (1964)","journal-title":"J. Basic Eng."},{"issue":"4","key":"15_CR9","doi-asserted-by":"publisher","first-page":"319","DOI":"10.1007\/s10339-011-0404-1","volume":"12","author":"D. Nguyen-Tuong","year":"2011","unstructured":"D. Nguyen-Tuong, J. Peters: Model learning in robotics: A\u00a0survey, Cogn. Process. 12(4), 319\u2013340 (2011)","journal-title":"Cogn. Process."},{"issue":"11","key":"15_CR10","doi-asserted-by":"publisher","first-page":"1238","DOI":"10.1177\/0278364913495721","volume":"32","author":"J. Kober","year":"2013","unstructured":"J. Kober, D. Bagnell, J. Peters: Reinforcement learning in robotics: A\u00a0survey, Int. J. Robotics Res. 32(11), 1238\u20131274 (2013)","journal-title":"Int. J. Robotics Res."},{"key":"15_CR11","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-3184-5","volume-title":"Robot Learning","author":"J.H. Connell","year":"1993","unstructured":"J.H. Connell, S. Mahadevan: Robot Learning (Kluwer Academic, Dordrecht 1993)"},{"key":"15_CR12","volume-title":"Int. Conf. Intell. Robots Syst.","author":"J. Ham","year":"2005","unstructured":"J. Ham, Y. Lin, D.D. Lee: Learning nonlinear appearance manifolds for robot localization, Int. Conf. Intell. Robots Syst. (2005)"},{"key":"15_CR13","volume-title":"Reinforcement Learning","author":"R.S. Sutton","year":"1998","unstructured":"R.S. Sutton, A.G. Barto: Reinforcement Learning (MIT, Cambridge 1998)"},{"issue":"15","key":"15_CR14","doi-asserted-by":"publisher","first-page":"2015","DOI":"10.1163\/016918609X12529286896877","volume":"23","author":"D. Nguyen-Tuong","year":"2009","unstructured":"D. Nguyen-Tuong, J. Peters: Model learning with local Gaussian process regression, Adv. Robotics 23(15), 2015\u20132034 (2009)","journal-title":"Adv. Robotics"},{"issue":"6","key":"15_CR15","doi-asserted-by":"publisher","first-page":"737","DOI":"10.1177\/0278364908091463","volume":"27","author":"J. Nakanishi","year":"2008","unstructured":"J. Nakanishi, R. Cory, M. Mistry, J. Peters, S. Schaal: Operational space control: A\u00a0theoretical and emprical comparison, Int. J. Robotics Res. 27(6), 737\u2013757 (2008)","journal-title":"Int. J. Robotics Res."},{"key":"15_CR16","volume-title":"Proc. Eur. Symp. Artif. Neural Netw.","author":"F.R. Reinhart","year":"2009","unstructured":"F.R. Reinhart, J.J. Steil: Attractor-based computation with reservoirs for online learning of inverse kinematics, Proc. Eur. Symp. Artif. Neural Netw. (2009)"},{"key":"15_CR17","first-page":"1673","volume-title":"Adv. Neural Inform. Process. Syst.","author":"J. Ting","year":"2008","unstructured":"J. Ting, M. Kalakrishnan, S. Vijayakumar, S. Schaal: Bayesian kernel shaping for learning control, Adv. Neural Inform. Process. Syst., Vol. 21 (2008) pp. 1673\u20131680"},{"key":"15_CR18","volume-title":"ICRA 2009 Workshop: Approaches Sens. Learn. Humanoid Robots, Kobe","author":"J. Steffen","year":"2009","unstructured":"J. Steffen, S. Klanke, S. Vijayakumar, H.J. Ritter: Realising dextrous manipulation with structured manifolds using unsupervised kernel regression with structural hints, ICRA 2009 Workshop: Approaches Sens. Learn. Humanoid Robots, Kobe (2009)"},{"key":"15_CR19","volume-title":"Proc. 2009 IEEE Int. Conf. Intell. Robots Syst.","author":"S. Klanke","year":"2006","unstructured":"S. Klanke, D. Lebedev, R. Haschke, J.J. Steil, H. Ritter: Dynamic path planning for a\u00a07-dof robot arm, Proc. 2009 IEEE Int. Conf. Intell. Robots Syst. (2006)"},{"key":"15_CR20","volume-title":"Proc. Robotics Sci. Syst., Philadelphia","author":"A. Angelova","year":"2006","unstructured":"A. Angelova, L. Matthies, D. Helmick, P. Perona: Slip prediction using visual information, Proc. Robotics Sci. Syst., Philadelphia (2006)"},{"key":"15_CR21","volume-title":"IEEE Int. Conf. Intell. Robots Syst.","author":"M. Kalakrishnan","year":"2009","unstructured":"M. Kalakrishnan, J. Buchli, P. Pastor, S. Schaal: Learning locomotion over rough terrain using terrain templates, IEEE Int. Conf. Intell. Robots Syst. (2009)"},{"key":"15_CR22","doi-asserted-by":"publisher","first-page":"367","DOI":"10.1007\/978-3-642-11694-0_9","volume":"8","author":"N. Hawes","year":"2010","unstructured":"N. Hawes, J.L. Wyatt, M. Sridharan, M. Kopicki, S. Hongeng, I. Calvert, A. Sloman, G.-J. Kruijff, H. Jacobsson, M. Brenner, D. Sko\u010daj, A. Vre\u010dko, N. Majer, M. Zillich: The playmate system, Cognit. Syst. 8, 367\u2013393 (2010)","journal-title":"Cognit. Syst."},{"key":"15_CR23","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1007\/978-3-642-11694-0_7","volume":"8","author":"D. Sko\u010daj","year":"2010","unstructured":"D. Sko\u010daj, M. Kristan, A. Vre\u010dko, A. Leonardis, M. Fritz, M. Stark, B. Schiele, S. Hongeng, J.L. Wyatt: Multi-modal learning, Cogn. Syst. 8, 265\u2013309 (2010)","journal-title":"Cogn. Syst."},{"key":"15_CR24","first-page":"28","volume":"6","author":"O.J. Smith","year":"1959","unstructured":"O.J. Smith: A\u00a0controller to overcome dead-time, Instrum. Soc. Am. J. 6, 28\u201333 (1959)","journal-title":"Instrum. Soc. Am. J."},{"key":"15_CR25","volume-title":"Stable Adaptive Systems","author":"K.S. Narendra","year":"1989","unstructured":"K.S. Narendra, A.M. Annaswamy: Stable Adaptive Systems (Prentice Hall, New Jersey 1989)"},{"key":"15_CR26","doi-asserted-by":"publisher","first-page":"635","DOI":"10.1016\/0005-1098(84)90013-X","volume":"20","author":"S. Nicosia","year":"1984","unstructured":"S. Nicosia, P. Tomei: Model reference adaptive control algorithms for industrial robots, Automatica 20, 635\u2013644 (1984)","journal-title":"Automatica"},{"key":"15_CR27","volume-title":"Predictive Control with Constraints","author":"J.M. Maciejowski","year":"2002","unstructured":"J.M. Maciejowski: Predictive Control with Constraints (Prentice Hall, New Jersey 2002)"},{"issue":"4","key":"15_CR28","doi-asserted-by":"publisher","first-page":"160","DOI":"10.1145\/122344.122377","volume":"2","author":"R.S. Sutton","year":"1991","unstructured":"R.S. Sutton: Dyna, an integrated architecture for learning, planning, and reacting, SIGART Bulletin 2(4), 160\u2013163 (1991)","journal-title":"SIGART Bulletin"},{"key":"15_CR29","volume-title":"Adv. Neural Inform. Process. Syst.","author":"C.G. Atkeson","year":"2002","unstructured":"C.G. Atkeson, J. Morimoto: Nonparametric representation of policies and value functions: A\u00a0trajectory-based approach, Adv. Neural Inform. Process. Syst., Vol. 15 (2002)"},{"key":"15_CR30","volume-title":"Proc. 11th Int. Symp. Exp. Robotics","author":"A.Y. Ng","year":"2004","unstructured":"A.Y. Ng, A. Coates, M. Diel, V. Ganapathi, J. Schulte, B. Tse, E. Berger, E. Liang: Autonomous inverted helicopter flight via reinforcement learning, Proc. 11th Int. Symp. Exp. Robotics (2004)"},{"key":"15_CR31","first-page":"751","volume-title":"Adv. Neural Inform. Process. Syst.","author":"C.E. Rasmussen","year":"2003","unstructured":"C.E. Rasmussen, M. Kuss: Gaussian processes in reinforcement learning, Adv. Neural Inform. Process. Syst., Vol. 16 (2003) pp. 751\u2013758"},{"key":"15_CR32","volume-title":"Proc. IEEE Int. Conf. Robotics Autom.","author":"A. Rottmann","year":"2009","unstructured":"A. Rottmann, W. Burgard: Adaptive autonomous control using online value iteration with Gaussian processes, Proc. IEEE Int. Conf. Robotics Autom. (2009)"},{"key":"15_CR33","volume-title":"Applied Nonlinear Control","author":"J.-J.E. Slotine","year":"1991","unstructured":"J.-J.E. Slotine, W. Li: Applied Nonlinear Control (Prentice Hall, Upper Saddle River 1991)"},{"key":"15_CR34","volume-title":"Proc. IEEE Int. Conf. Robotics Autom.","author":"A. De Luca","year":"1998","unstructured":"A. De Luca, P. Lucibello: A\u00a0general algorithm for dynamic feedback linearization of robots with elastic joints, Proc. IEEE Int. Conf. Robotics Autom. (1998)"},{"key":"15_CR35","doi-asserted-by":"publisher","first-page":"307","DOI":"10.1207\/s15516709cog1603_1","volume":"16","author":"I. Jordan","year":"1992","unstructured":"I. Jordan, D. Rumelhart: Forward models: Supervised learning with a\u00a0distal teacher, Cognit. Sci. 16, 307\u2013354 (1992)","journal-title":"Cognit. Sci."},{"key":"15_CR36","doi-asserted-by":"publisher","first-page":"1317","DOI":"10.1016\/S0893-6080(98)00066-5","volume":"11","author":"D.M. Wolpert","year":"1998","unstructured":"D.M. Wolpert, M. Kawato: Multiple paired forward and inverse models for motor control, Neural Netw. 11, 1317\u20131329 (1998)","journal-title":"Neural Netw."},{"issue":"6","key":"15_CR37","doi-asserted-by":"publisher","first-page":"718","DOI":"10.1016\/S0959-4388(99)00028-8","volume":"9","author":"M. Kawato","year":"1999","unstructured":"M. Kawato: Internal models for motor control and trajectory planning, Curr. Opin. Neurobiol. 9(6), 718\u2013727 (1999)","journal-title":"Curr. Opin. Neurobiol."},{"issue":"9","key":"15_CR38","doi-asserted-by":"publisher","first-page":"338","DOI":"10.1016\/S1364-6613(98)01221-2","volume":"2","author":"D.M. Wolpert","year":"1998","unstructured":"D.M. Wolpert, R.C. Miall, M. Kawato: Internal models in the cerebellum, Trends Cogn. Sci. 2(9), 338\u2013347 (1998)","journal-title":"Trends Cogn. Sci."},{"key":"15_CR39","first-page":"3","volume-title":"Adv. Neural Inform. Process. Syst.","author":"N. Bhushan","year":"1999","unstructured":"N. Bhushan, R. Shadmehr: Evidence for a\u00a0forward dynamics model in human adaptive motor control, Adv. Neural Inform. Process. Syst., Vol. 11 (1999) pp. 3\u20139"},{"issue":"3","key":"15_CR40","first-page":"37","volume":"15","author":"K. Narendra","year":"1995","unstructured":"K. Narendra, J. Balakrishnan, M. Ciliz: Adaptation and learning using multiple models, switching and tuning, IEEE Control Syst, Mag. 15(3), 37\u201351 (1995)","journal-title":"Mag."},{"issue":"2","key":"15_CR41","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1109\/9.554398","volume":"42","author":"K. Narendra","year":"1997","unstructured":"K. Narendra, J. Balakrishnan: Adaptive control using multiple models, IEEE Trans. Autom. Control 42(2), 171\u2013187 (1997)","journal-title":"IEEE Trans. Autom. Control"},{"issue":"10","key":"15_CR42","doi-asserted-by":"publisher","first-page":"2201","DOI":"10.1162\/089976601750541778","volume":"13","author":"M. Haruno","year":"2001","unstructured":"M. Haruno, D.M. Wolpert, M. Kawato: Mosaic model for sensorimotor learning and control, Neural Comput. 13(10), 2201\u20132220 (2001)","journal-title":"Neural Comput."},{"issue":"2","key":"15_CR43","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1177\/0278364907087548","volume":"27","author":"J. Peters","year":"2008","unstructured":"J. Peters, S. Schaal: Learning to control in operational space, Int. J. Robotics Res. 27(2), 197\u2013212 (2008)","journal-title":"Int. J. Robotics Res."},{"key":"15_CR44","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1007\/BF02479221","volume":"23","author":"H. Akaike","year":"1970","unstructured":"H. Akaike: Autoregressive model fitting for control, Ann. Inst. Stat. Math. 23, 163\u2013180 (1970)","journal-title":"Ann. Inst. Stat. Math."},{"key":"15_CR45","doi-asserted-by":"publisher","first-page":"167","DOI":"10.1016\/0005-1098(81)90092-3","volume":"17","author":"R.M.C. De Keyser","year":"1980","unstructured":"R.M.C. De Keyser, A.R.V. Cauwenberghe: A\u00a0self-tuning multistep predictor application, Automatica 17, 167\u2013174 (1980)","journal-title":"Automatica"},{"key":"15_CR46","doi-asserted-by":"publisher","first-page":"2157","DOI":"10.1080\/00207178908559767","volume":"49","author":"S.S. Billings","year":"1989","unstructured":"S.S. Billings, S. Chen, G. Korenberg: Identification of mimo nonlinear systems using a\u00a0forward-regression orthogonal estimator, Int. J. Control 49, 2157\u20132189 (1989)","journal-title":"Int. J. Control"},{"key":"15_CR47","doi-asserted-by":"publisher","first-page":"521","DOI":"10.1016\/0005-1098(89)90095-2","volume":"25","author":"E. Mosca","year":"1989","unstructured":"E. Mosca, G. Zappa, J.M. Lemos: Robustness of multipredictor adaptive regulators: MUSMAR, Automatica 25, 521\u2013529 (1989)","journal-title":"Automatica"},{"key":"15_CR48","volume-title":"Proc. Am. Control Conf.","author":"J. Kocijan","year":"2004","unstructured":"J. Kocijan, R. Murray-Smith, C. Rasmussen, A. Girard: Gaussian process model based predictive control, Proc. Am. Control Conf. (2004)"},{"key":"15_CR49","first-page":"545","volume-title":"Adv. Neural Inform. Process. Syst.","author":"A. Girard","year":"2002","unstructured":"A. Girard, C.E. Rasmussen, J.Q. Candela, R.M. Smith: Gaussian process priors with uncertain inputs application to multiple-step ahead time series forecasting, Adv. Neural Inform. Process. Syst., Vol. 15 (2002) pp. 545\u2013552"},{"key":"15_CR50","first-page":"75","volume":"11","author":"C.G. Atkeson","year":"1997","unstructured":"C.G. Atkeson, A. Moore, S. Stefan: Locally weighted learning for control, AI Review 11, 75\u2013113 (1997)","journal-title":"AI Review"},{"key":"15_CR51","volume-title":"System Identification\u00a0\u2013 Theory for the User","author":"L. Ljung","year":"2004","unstructured":"L. Ljung: System Identification\u00a0\u2013 Theory for the User (Prentice-Hall, New Jersey 2004)"},{"key":"15_CR52","volume-title":"Neural Networks: A\u00a0Comprehensive Foundation","author":"S. Haykin","year":"1999","unstructured":"S. Haykin: Neural Networks: A\u00a0Comprehensive Foundation (Prentice Hall, New Jersey 1999)"},{"key":"15_CR53","volume-title":"Proc. Int. Jt. Conf. Neural Netw.","author":"J.J. Steil","year":"2004","unstructured":"J.J. Steil: Backpropagation-decorrelation: Online recurrent learning with O(N) complexity, Proc. Int. Jt. Conf. Neural Netw. (2004)"},{"key":"15_CR54","volume-title":"Gaussian Processes for Machine Learning","author":"C.E. Rasmussen","year":"2006","unstructured":"C.E. Rasmussen, C.K. Williams: Gaussian Processes for Machine Learning (MIT, Cambridge 2006)"},{"key":"15_CR55","volume-title":"Learning with Kernels: Support Vector Machines, Regularization, Optimization and Beyond","author":"B. Sch\u00f6lkopf","year":"2002","unstructured":"B. Sch\u00f6lkopf, A. Smola: Learning with Kernels: Support Vector Machines, Regularization, Optimization and Beyond (MIT, Cambridge 2002)"},{"key":"15_CR56","volume-title":"Adaptive Control","author":"K.J. Astr\u00f6m","year":"1995","unstructured":"K.J. Astr\u00f6m, B. Wittenmark: Adaptive Control (Addison Wesley, Boston 1995)"},{"key":"15_CR57","doi-asserted-by":"publisher","first-page":"684","DOI":"10.1177\/027836499101000607","volume":"10","author":"F.J. Coito","year":"1991","unstructured":"F.J. Coito, J.M. Lemos: A\u00a0long-range adaptive controller for robot manipulators, Int. J. Robotics Res. 10, 684\u2013707 (1991)","journal-title":"Int. J. Robotics Res."},{"key":"15_CR58","volume-title":"Proc. World Congr. Eng. Comput. Sci.","author":"P. Vempaty","year":"2009","unstructured":"P. Vempaty, K. Cheok, R. Loh: Model reference adaptive control for actuators of a\u00a0biped robot locomotion, Proc. World Congr. Eng. Comput. Sci. (2009)"},{"key":"15_CR59","doi-asserted-by":"publisher","first-page":"33","DOI":"10.3233\/IFS-1996-4103","volume":"4","author":"J.R. Layne","year":"1996","unstructured":"J.R. Layne, K.M. Passino: Fuzzy model reference learning control, J. Intell. Fuzzy Syst. 4, 33\u201347 (1996)","journal-title":"J. Intell. Fuzzy Syst."},{"issue":"1","key":"15_CR60","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1016\/j.neunet.2004.08.009","volume":"18","author":"J. Nakanishi","year":"2005","unstructured":"J. Nakanishi, J.A. Farrell, S. Schaal: Composite adaptive control with locally weighted statistical learning, Neural Netw. 18(1), 71\u201390 (2005)","journal-title":"Neural Netw."},{"key":"15_CR61","volume-title":"Introduction to Robotics: Mechanics and Control","author":"J.J. Craig","year":"2004","unstructured":"J.J. Craig: Introduction to Robotics: Mechanics and Control (Prentice Hall, Upper Saddle River 2004)"},{"key":"15_CR62","volume-title":"Robot Dynamics and Control","author":"M.W. Spong","year":"2006","unstructured":"M.W. Spong, S. Hutchinson, M. Vidyasagar: Robot Dynamics and Control (Wiley, New York 2006)"},{"issue":"1","key":"15_CR63","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1023\/A:1015727715131","volume":"17","author":"S. Schaal","year":"2002","unstructured":"S. Schaal, C.G. Atkeson, S. Vijayakumar: Scalable techniques from nonparametric statistics for real-time robot learning, Appl. Intell. 17(1), 49\u201360 (2002)","journal-title":"Appl. Intell."},{"key":"15_CR64","volume-title":"13th Int. Conf. Neural Inform. Process.","author":"H. Cao","year":"2006","unstructured":"H. Cao, Y. Yin, D. Du, L. Lin, W. Gu, Z. Yang: Neural network inverse dynamic online learning control on physical exoskeleton, 13th Int. Conf. Neural Inform. Process. (2006)"},{"issue":"3","key":"15_CR65","doi-asserted-by":"publisher","first-page":"101","DOI":"10.1177\/027836498600500306","volume":"5","author":"C.G. Atkeson","year":"1986","unstructured":"C.G. Atkeson, C.H. An, J.M. Hollerbach: Estimation of inertial parameters of manipulator loads and links, Int. J. Robotics Res. 5(3), 101\u2013119 (1986)","journal-title":"Int. J. Robotics Res."},{"key":"15_CR66","doi-asserted-by":"publisher","first-page":"537","DOI":"10.1109\/ROBOT.1997.620092","volume":"1","author":"E. Burdet","year":"1997","unstructured":"E. Burdet, B. Sprenger, A. Codourey: Experiments in nonlinear adaptive control, Int. Conf. Robotics Autom. 1, 537\u2013542 (1997)","journal-title":"Int. Conf. Robotics Autom."},{"issue":"1","key":"15_CR67","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1017\/S0263574798000150","volume":"16","author":"E. Burdet","year":"1998","unstructured":"E. Burdet, A. Codourey: Evaluation of parametric and nonparametric nonlinear adaptive controllers, Robotica 16(1), 59\u201373 (1998)","journal-title":"Robotica"},{"key":"15_CR68","doi-asserted-by":"publisher","first-page":"127","DOI":"10.1080\/00207178708933715","volume":"45","author":"K.S. Narendra","year":"1987","unstructured":"K.S. Narendra, A.M. Annaswamy: Persistent excitation in adaptive systems, Int. J. Control 45, 127\u2013160 (1987)","journal-title":"Int. J. Control"},{"issue":"2","key":"15_CR69","doi-asserted-by":"publisher","first-page":"343","DOI":"10.1109\/72.991420","volume":"13","author":"H.D. Patino","year":"2002","unstructured":"H.D. Patino, R. Carelli, B.R. Kuchen: Neural networks for advanced control of robot manipulators, IEEE Trans. Neural Netw. 13(2), 343\u2013354 (2002)","journal-title":"IEEE Trans. Neural Netw."},{"issue":"11","key":"15_CR70","doi-asserted-by":"publisher","first-page":"1859","DOI":"10.1016\/j.neucom.2010.06.033","volume":"74","author":"D. Nguyen-Tuong","year":"2011","unstructured":"D. Nguyen-Tuong, J. Peters: Incremental sparsification for real-time online model learning, Neurocomputing 74(11), 1859\u20131867 (2011)","journal-title":"Neurocomputing"},{"key":"15_CR71","volume-title":"Proc. IEEE Int. Conf. Robotics Autom.","author":"D. Nguyen-Tuong","year":"2010","unstructured":"D. Nguyen-Tuong, J. Peters: Using model knowledge for learning inverse dynamics, Proc. IEEE Int. Conf. Robotics Autom. (2010)"},{"issue":"6","key":"15_CR72","doi-asserted-by":"publisher","first-page":"623","DOI":"10.1080\/00207729808929555","volume":"29","author":"S.S. Ge","year":"1998","unstructured":"S.S. Ge, T.H. Lee, E.G. Tan: Adaptive neural network control of flexible joint robots based on feedback linearization, Int. J. Syst. Sci. 29(6), 623\u2013635 (1998)","journal-title":"Int. J. Syst. Sci."},{"key":"15_CR73","first-page":"971","volume":"29","author":"C.M. Chow","year":"1998","unstructured":"C.M. Chow, A.G. Kuznetsov, D.W. Clarke: Successive one-step-ahead predictions in multiple model predictive control, Int. J. Control 29, 971\u2013979 (1998)","journal-title":"Int. J. Control"},{"key":"15_CR74","first-page":"365","volume-title":"Advanced Neural Computers","author":"M. Kawato","year":"1990","unstructured":"M. Kawato: Feedback error learning neural network for supervised motor learning. In: Advanced Neural Computers, ed. by R. Eckmiller (Elsevier, North-Holland, Amsterdam 1990) pp. 365\u2013372"},{"issue":"10","key":"15_CR75","doi-asserted-by":"publisher","first-page":"1453","DOI":"10.1016\/j.neunet.2004.05.003","volume":"17","author":"J. Nakanishi","year":"2004","unstructured":"J. Nakanishi, S. Schaal: Feedback error learning and nonlinear adaptive control, Neural Netw. 17(10), 1453\u20131465 (2004)","journal-title":"Neural Netw."},{"issue":"2","key":"15_CR76","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1016\/S0893-6080(00)00084-8","volume":"14","author":"T. Shibata","year":"2001","unstructured":"T. Shibata, C. Schaal: Biomimetic gaze stabilization based on feedback-error learning with nonparametric regression networks, Neural Netw. 14(2), 201\u2013216 (2001)","journal-title":"Neural Netw."},{"issue":"3","key":"15_CR77","doi-asserted-by":"publisher","first-page":"251","DOI":"10.1016\/0893-6080(88)90030-5","volume":"1","author":"H. Miyamoto","year":"1988","unstructured":"H. Miyamoto, M. Kawato, T. Setoyama, R. Suzuki: Feedback-error-learning neural network for trajectory control of a\u00a0robotic manipulator, Neural Netw. 1(3), 251\u2013265 (1988)","journal-title":"Neural Netw."},{"issue":"4","key":"15_CR78","doi-asserted-by":"publisher","first-page":"485","DOI":"10.1016\/S0893-6080(05)80053-X","volume":"6","author":"H. Gomi","year":"1993","unstructured":"H. Gomi, M. Kawato: Recognition of manipulated objects by motor learning with modular architecture networks, Neural Netw. 6(4), 485\u2013497 (1993)","journal-title":"Neural Netw."},{"key":"15_CR79","volume-title":"IEEE Int. Conf. Intell. Robots Syst.","author":"A. D'Souza","year":"2001","unstructured":"A. D'Souza, S. Vijayakumar, S. Schaal: Learning inverse kinematics, IEEE Int. Conf. Intell. Robots Syst. (2001)"},{"key":"15_CR80","volume-title":"Proc. 16th Int. Conf. Mach. Learn.","author":"S. Vijayakumar","year":"2000","unstructured":"S. Vijayakumar, S. Schaal: Locally weighted projection regression: An O(N) algorithm for incremental real time learning in high dimensional space, Proc. 16th Int. Conf. Mach. Learn. (2000)"},{"key":"15_CR81","volume-title":"Proc. 22nd Int. Conf. Mach. Learn.","author":"M. Toussaint","year":"2005","unstructured":"M. Toussaint, S. Vijayakumar: Learning discontinuities with products-of-sigmoids for switching between local models, Proc. 22nd Int. Conf. Mach. Learn. (2005)"},{"key":"15_CR82","doi-asserted-by":"publisher","first-page":"2319","DOI":"10.1126\/science.290.5500.2319","volume":"290","author":"J. Tenenbaum","year":"2000","unstructured":"J. Tenenbaum, V. de Silva, J. Langford: A\u00a0global geometric framework for nonlinear dimensionality reduction, Science 290, 2319\u20132323 (2000)","journal-title":"Science"},{"key":"15_CR83","doi-asserted-by":"publisher","first-page":"2323","DOI":"10.1126\/science.290.5500.2323","volume":"290","author":"S. Roweis","year":"2000","unstructured":"S. Roweis, L. Saul: Nonlinear dimensionality reduction by locally linear embedding, Science 290, 2323 (2000)","journal-title":"Science"},{"issue":"2","key":"15_CR84","doi-asserted-by":"publisher","first-page":"109","DOI":"10.1007\/s11063-009-9098-0","volume":"29","author":"H. Hoffman","year":"2009","unstructured":"H. Hoffman, S. Schaal, S. Vijayakumar: Local dimensionality reduction for non-parametric regression, Neural Process. Lett. 29(2), 109\u2013131 (2009)","journal-title":"Neural Process. Lett."},{"key":"15_CR85","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1016\/0921-8890(95)00004-Y","volume":"15","author":"S. Thrun","year":"1995","unstructured":"S. Thrun, T. Mitchell: Lifelong robot learning, Robotics Auton. Syst. 15, 25\u201346 (1995)","journal-title":"Robotics Auton. Syst."},{"key":"15_CR86","volume-title":"Eur. Conf. Mach. Learn.","author":"Y. Engel","year":"2002","unstructured":"Y. Engel, S. Mannor, R. Meir: Sparse online greedy support vector regression, Eur. Conf. Mach. Learn. (2002)"},{"issue":"3","key":"15_CR87","doi-asserted-by":"publisher","first-page":"199","DOI":"10.1023\/B:STCO.0000035301.49549.88","volume":"14","author":"A.J. Smola","year":"2004","unstructured":"A.J. Smola, B. Sch\u00f6lkopf: A\u00a0tutorial on support vector regression, Stat. Comput. 14(3), 199\u2013222 (2004)","journal-title":"Stat. Comput."},{"key":"15_CR88","volume-title":"Evaluation of Gaussian Processes and Other Methods for Non-Linear Regression","author":"C.E. Rasmussen","year":"1996","unstructured":"C.E. Rasmussen: Evaluation of Gaussian Processes and Other Methods for Non-Linear Regression (University of Toronto, Toronto 1996)"},{"key":"15_CR89","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/7496.001.0001","volume-title":"Large-Scale Kernel Machines","author":"L. Bottou","year":"2007","unstructured":"L. Bottou, O. Chapelle, D. DeCoste, J. Weston: Large-Scale Kernel Machines (MIT, Cambridge 2007)"},{"key":"15_CR90","first-page":"1939","volume":"6","author":"J.Q. Candela","year":"2005","unstructured":"J.Q. Candela, C.E. Rasmussen: A\u00a0unifying view of sparse approximate Gaussian process regression, J. Mach. Learn. Res. 6, 1939\u20131959 (2005)","journal-title":"J. Mach. Learn. Res."},{"key":"15_CR91","doi-asserted-by":"publisher","first-page":"385","DOI":"10.1142\/S0218001403002472","volume":"17","author":"R. Genov","year":"2003","unstructured":"R. Genov, S. Chakrabartty, G. Cauwenberghs: Silicon support vector machine with online learning, Int. J. Pattern Recognit. Articial Intell. 17, 385\u2013404 (2003)","journal-title":"Int. J. Pattern Recognit. Articial Intell."},{"issue":"11","key":"15_CR92","doi-asserted-by":"publisher","first-page":"2602","DOI":"10.1162\/089976605774320557","volume":"12","author":"S. Vijayakumar","year":"2005","unstructured":"S. Vijayakumar, A. D'Souza, S. Schaal: Incremental online learning in high dimensions, Neural Comput. 12(11), 2602\u20132634 (2005)","journal-title":"Neural Comput."},{"key":"15_CR93","first-page":"640","volume-title":"Adv. Neural Inform. Process. Syst.","author":"B. Sch\u00f6lkopf","year":"1998","unstructured":"B. Sch\u00f6lkopf, P. Simard, A. Smola, V. Vapnik: Prior knowledge in support vector kernel, Adv. Neural Inform. Process. Syst., Vol. 10 (1998) pp. 640\u2013646"},{"key":"15_CR94","volume-title":"Int. Conf. Artif. Intell. Stat.","author":"E. Krupka","year":"2007","unstructured":"E. Krupka, N. Tishby: Incorporating prior knowledge on features into learning, Int. Conf. Artif. Intell. Stat. (San Juan, Puerto Rico 2007)"},{"key":"15_CR95","first-page":"585","volume-title":"Adv. Neural Inform. Process. Syst.","author":"A. Smola","year":"1999","unstructured":"A. Smola, T. Friess, B. Schoelkopf: Semiparametric support vector and linear programming machines, Adv. Neural Inform. Process. Syst., Vol. 11 (1999) pp. 585\u2013591"},{"key":"15_CR96","doi-asserted-by":"publisher","first-page":"381","DOI":"10.1016\/S0262-8856(00)00086-X","volume":"19","author":"B.J. Kr\u00f6se","year":"2001","unstructured":"B.J. Kr\u00f6se, N. Vlassis, R. Bunschoten, Y. Motomura: A\u00a0probabilistic model for appearance-based robot localization, Image Vis. Comput. 19, 381\u2013391 (2001)","journal-title":"Image Vis. Comput."},{"key":"15_CR97","volume-title":"Proc. 13th Int. Conf. Artif. Intell. Stat.","author":"M.K. Titsias","year":"2010","unstructured":"M.K. Titsias, N.D. Lawrence: Bayesian Gaussian process latent variable model, Proc. 13th Int. Conf. Artif. Intell. Stat. (2010)"},{"key":"15_CR98","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1162\/neco.1991.3.1.79","volume":"3","author":"R. Jacobs","year":"1991","unstructured":"R. Jacobs, M. Jordan, S. Nowlan, G.E. Hinton: Adaptive mixtures of local experts, Neural Comput. 3, 79\u201387 (1991)","journal-title":"Neural Comput."},{"key":"15_CR99","doi-asserted-by":"publisher","first-page":"44","DOI":"10.1109\/MRA.2010.936947","volume":"17","author":"S. Calinon","year":"2010","unstructured":"S. Calinon, F. D'halluin, E. Sauser, D. Caldwell, A. Billard: A\u00a0probabilistic approach based on dynamical systems to learn and reproduce gestures by imitation, IEEE Robotics Autom. Mag. 17, 44\u201354 (2010)","journal-title":"IEEE Robotics Autom. Mag."},{"issue":"11","key":"15_CR100","doi-asserted-by":"publisher","first-page":"2719","DOI":"10.1162\/089976600300014908","volume":"12","author":"V. Treps","year":"2000","unstructured":"V. Treps: A\u00a0bayesian committee machine, Neural Comput. 12(11), 2719\u20132741 (2000)","journal-title":"Neural Comput."},{"issue":"3","key":"15_CR101","doi-asserted-by":"publisher","first-page":"641","DOI":"10.1162\/089976602317250933","volume":"14","author":"L. Csato","year":"2002","unstructured":"L. Csato, M. Opper: Sparse online Gaussian processes, Neural Comput. 14(3), 641\u2013668 (2002)","journal-title":"Neural Comput."},{"key":"15_CR102","volume-title":"IEEE Int. Conf. Robotics Autom., Pasadena","author":"D.H. Grollman","year":"2008","unstructured":"D.H. Grollman, O.C. Jenkins: Sparse incremental learning for interactive robot control policy estimation, IEEE Int. Conf. Robotics Autom., Pasadena (2008)"},{"issue":"2","key":"15_CR103","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1142\/S0129065704001899","volume":"14","author":"M. Seeger","year":"2004","unstructured":"M. Seeger: Gaussian processes for machine learning, Int. J. Neural Syst. 14(2), 69\u2013106 (2004)","journal-title":"Int. J. Neural Syst."},{"key":"15_CR104","volume-title":"Proc. IEEE Int. Conf. Intell. Robots Syst.","author":"C. Plagemann","year":"2008","unstructured":"C. Plagemann, S. Mischke, S. Prentice, K. Kersting, N. Roy, W. Burgard: Learning predictive terrain models for legged robot locomotion, Proc. IEEE Int. Conf. Intell. Robots Syst. (2008)"},{"issue":"1","key":"15_CR105","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1007\/s10514-009-9119-x","volume":"27","author":"J. Ko","year":"2009","unstructured":"J. Ko, D. Fox: GP-bayesfilters: Bayesian filtering using Gaussian process prediction and observation models, Auton. Robots 27(1), 75\u201390 (2009)","journal-title":"Auton. Robots"},{"key":"15_CR106","volume-title":"IEEE Int. Symp. Intell. Signal Process.","author":"J.P. Ferreira","year":"2007","unstructured":"J.P. Ferreira, M. Crisostomo, A.P. Coimbra, B. Ribeiro: Simulation control of a\u00a0biped robot with support vector regression, IEEE Int. Symp. Intell. Signal Process. (2007)"},{"key":"15_CR107","volume-title":"IEEE Int. Conf. Robotics Autom.","author":"R. Pelossof","year":"2004","unstructured":"R. Pelossof, A. Miller, P. Allen, T. Jebara: An SVM learning approach to robotic grasping, IEEE Int. Conf. Robotics Autom. (2004)"},{"key":"15_CR108","doi-asserted-by":"publisher","first-page":"2683","DOI":"10.1162\/089976603322385117","volume":"15","author":"J. Ma","year":"2005","unstructured":"J. Ma, J. Theiler, S. Perkins: Accurate on-line support vector regression, Neural Comput. 15, 2683\u20132703 (2005)","journal-title":"Neural Comput."},{"key":"15_CR109","volume-title":"Proc. IEEE Int. Symp. Comput. Intell. Robotics Autom.","author":"Y. Choi","year":"2007","unstructured":"Y. Choi, S.Y. Cheong, N. Schweighofer: Local online support vector regression for learning control, Proc. IEEE Int. Symp. Comput. Intell. Robotics Autom. (2007)"},{"issue":"1","key":"15_CR110","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1016\/j.neunet.2010.08.011","volume":"24","author":"J.-A. Ting","year":"2011","unstructured":"J.-A. Ting, A. D'Souza, S. Schaal: Bayesian robot system identification with input and output noise, Neural Netw. 24(1), 99\u2013108 (2011)","journal-title":"Neural Netw."},{"key":"15_CR111","first-page":"774","volume-title":"Adv. Neural Inform. Process. Syst.","author":"S. Nowlan","year":"1991","unstructured":"S. Nowlan, G.E. Hinton: Evaluation of adaptive mixtures of competing experts, Adv. Neural Inform. Process. Syst., Vol. 3 (1991) pp. 774\u2013780"},{"key":"15_CR112","first-page":"654","volume-title":"Adv. Neural Inform. Process. Syst.","author":"V. Treps","year":"2001","unstructured":"V. Treps: Mixtures of Gaussian processes, Adv. Neural Inform. Process. Syst., Vol. 13 (2001) pp. 654\u2013660"},{"key":"15_CR113","first-page":"881","volume-title":"Adv. Neural Inform. Process. Syst.","author":"C.E. Rasmussen","year":"2002","unstructured":"C.E. Rasmussen, Z. Ghahramani: Infinite mixtures of Gaussian process experts, Adv. Neural Inform. Process. Syst., Vol. 14 (2002) pp. 881\u2013888"},{"key":"15_CR114","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-387-21606-5","volume-title":"The Elements of Statistical Learning","author":"T. Hastie","year":"2001","unstructured":"T. Hastie, R. Tibshirani, J. Friedman: The Elements of Statistical Learning (Springer, New York, 2001)"},{"key":"15_CR115","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-17146-8","volume-title":"Nonparametric and Semiparametric Models","author":"W.K. Haerdle","year":"2004","unstructured":"W.K. Haerdle, M. Mueller, S. Sperlich, A. Werwatz: Nonparametric and Semiparametric Models (Springer, New York 2004)"},{"issue":"3","key":"15_CR116","first-page":"448","volume":"4","author":"D.J. MacKay","year":"1992","unstructured":"D.J. MacKay: A\u00a0practical Bayesian framework for back-propagation networks, Computation 4(3), 448\u2013472 (1992)","journal-title":"Computation"},{"key":"15_CR117","series-title":"Lecture Notes in Statistics","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-0745-0","volume-title":"Bayesian Learning for Neural Networks","author":"R.M. Neal","year":"1996","unstructured":"R.M. Neal: Bayesian Learning for Neural Networks, Lecture Notes in Statistics, Vol. 118 (Springer, New York 1996)"},{"issue":"5","key":"15_CR118","doi-asserted-by":"publisher","first-page":"1207","DOI":"10.1162\/089976600300015565","volume":"12","author":"B. Sch\u00f6lkopf","year":"2000","unstructured":"B. Sch\u00f6lkopf, A.J. Smola, R. Williamson, P.L. Bartlett: New support vector algorithms, Neural Comput. 12(5), 1207\u20131245 (2000)","journal-title":"Neural Comput."},{"key":"15_CR119","volume-title":"Snowbird Learn. Workshop","author":"C. Plagemann","year":"2007","unstructured":"C. Plagemann, K. Kersting, P. Pfaff, W. Burgard: Heteroscedastic Gaussian process regression for modeling range sensors in mobile robotics, Snowbird Learn. Workshop (2007)"},{"key":"15_CR120","volume-title":"Statistical Theory and Computational Aspects of Smoothing","author":"W.S. Cleveland","year":"1996","unstructured":"W.S. Cleveland, C.L. Loader: Smoothing by local regression: Principles and methods. In: Statistical Theory and Computational Aspects of Smoothing, ed. by W. H\u00e4rdle, M.G. Schimele (Physica, Heidelberg 1996)"},{"key":"15_CR121","volume-title":"Local Polynomial Modelling and Its Applications","author":"J. Fan","year":"1996","unstructured":"J. Fan, I. Gijbels: Local Polynomial Modelling and Its Applications (Chapman Hall, New York 1996)"},{"issue":"2","key":"15_CR122","doi-asserted-by":"crossref","first-page":"371","DOI":"10.1111\/j.2517-6161.1995.tb02034.x","volume":"57","author":"J. Fan","year":"1995","unstructured":"J. Fan, I. Gijbels: Data driven bandwidth selection in local polynomial fitting, J. R. Stat. Soc. 57(2), 371\u2013394 (1995)","journal-title":"J. R. Stat. Soc."},{"key":"15_CR123","volume-title":"Proc. 11th Int. Conf. Mach. Learn.","author":"A. Moore","year":"1994","unstructured":"A. Moore, M.S. Lee: Efficient algorithms for minimizing cross validation error, Proc. 11th Int. Conf. Mach. Learn. (1994)"},{"key":"15_CR124","first-page":"571","volume-title":"Adv. Neural Inform. Process. Syst.","author":"A. Moore","year":"1992","unstructured":"A. Moore: Fast, robust adaptive control by learning only forward models, Adv. Neural Inform. Process. Syst., Vol. 4 (1992) pp. 571\u2013578"},{"key":"15_CR125","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1023\/A:1006511328852","volume":"11","author":"C.G. Atkeson","year":"1997","unstructured":"C.G. Atkeson, A.W. Moore, S. Schaal: Locally weighted learning for control, Artif. Intell. Rev. 11, 75\u2013113 (1997)","journal-title":"Artif. Intell. Rev."},{"key":"15_CR126","volume-title":"Efficient Inverse Kinematics Algorithms for High-Dimensional Movement Systems","author":"G. Tevatia","year":"2008","unstructured":"G. Tevatia, S. Schaal: Efficient Inverse Kinematics Algorithms for High-Dimensional Movement Systems (University of Southern California, Los Angeles 2008)"},{"issue":"1\u20135","key":"15_CR127","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1023\/A:1006559212014","volume":"11","author":"C.G. Atkeson","year":"1997","unstructured":"C.G. Atkeson, A.W. Moore, S. Schaal: Locally weighted learning, Artif. Intell. Rev. 11(1\u20135), 11\u201373 (1997)","journal-title":"Artif. Intell. Rev."},{"key":"15_CR128","volume-title":"Proc. 20th Int. Jt. Conf. Artif. Intell.","author":"N.U. Edakunni","year":"2007","unstructured":"N.U. Edakunni, S. Schaal, S. Vijayakumar: Kernel carpentry for online regression using randomly varying coefficient model, Proc. 20th Int. Jt. Conf. Artif. Intell. (2007)"},{"key":"15_CR129","volume-title":"Differential Dynamic Programming","author":"D.H. Jacobson","year":"1973","unstructured":"D.H. Jacobson, D.Q. Mayne: Differential Dynamic Programming (American Elsevier, New York 1973)"},{"key":"15_CR130","volume-title":"Proc. 14th Int. Conf. Mach. Learn.","author":"C.G. Atkeson","year":"1997","unstructured":"C.G. Atkeson, S. Schaal: Robot learning from demonstration, Proc. 14th Int. Conf. Mach. Learn. (1997)"},{"key":"15_CR131","volume-title":"Proc. 2009 IEEE Int. Conf. Intell. Robots Syst.","author":"J. Morimoto","year":"2003","unstructured":"J. Morimoto, G. Zeglin, C.G. Atkeson: Minimax differential dynamic programming: Application to a\u00a0biped walking robot, Proc. 2009 IEEE Int. Conf. Intell. Robots Syst. (2003)"},{"key":"15_CR132","first-page":"1","volume-title":"Adv. Neural Inform. Process. Syst.","author":"P. Abbeel","year":"2007","unstructured":"P. Abbeel, A. Coates, M. Quigley, A.Y. Ng: An application of reinforcement learning to aerobatic helicopter flight, Adv. Neural Inform. Process. Syst., Vol. 19 (2007) pp. 1\u20138"},{"key":"15_CR133","volume-title":"Proc. Winter Simul. Conf.","author":"P.W. Glynn","year":"1987","unstructured":"P.W. Glynn: Likelihood ratio gradient estimation: An overview, Proc. Winter Simul. Conf. (1987)"},{"key":"15_CR134","volume-title":"Proc. 16th Conf. Uncertain. Artif. Intell.","author":"A.Y. Ng","year":"2000","unstructured":"A.Y. Ng, M. Jordan: Pegasus: A\u00a0policy search method for large MDPs and POMDPs, Proc. 16th Conf. Uncertain. Artif. Intell. (2000)"},{"issue":"9","key":"15_CR135","doi-asserted-by":"publisher","first-page":"937","DOI":"10.1016\/j.jprocont.2006.06.001","volume":"16","author":"B.M. Akesson","year":"2006","unstructured":"B.M. Akesson, H.T. Toivonen: A\u00a0neural network model predictive controller, J. Process Control 16(9), 937\u2013946 (2006)","journal-title":"J. Process Control"},{"key":"15_CR136","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1016\/S0921-8890(02)00172-0","volume":"39","author":"D. Gu","year":"2002","unstructured":"D. Gu, H. Hu: Predictive control for a\u00a0car-like mobile robot, Robotics Auton. Syst. 39, 73\u201386 (2002)","journal-title":"Robotics Auton. Syst."},{"key":"15_CR137","volume-title":"Proc. Am. Control Conf.","author":"E.A. Wan","year":"2001","unstructured":"E.A. Wan, A.A. Bogdanov: Model predictive neural control with applications to a\u00a06 DOF helicopter model, Proc. Am. Control Conf. (2001)"},{"issue":"1","key":"15_CR138","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1109\/JRA.1987.1087068","volume":"3","author":"O. Khatib","year":"1987","unstructured":"O. Khatib: A\u00a0unified approach for motion and force control of robot manipulators: The operational space formulation, J. Robotics Autom. 3(1), 43\u201353 (1987)","journal-title":"J. Robotics Autom."},{"issue":"1","key":"15_CR139","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s10514-007-9051-x","volume":"24","author":"J. Peters","year":"2008","unstructured":"J. Peters, M. Mistry, F.E. Udwadia, J. Nakanishi, S. Schaal: A\u00a0unifying methodology for robot control with redundant dofs, Auton. Robots 24(1), 1\u201312 (2008)","journal-title":"Auton. Robots"},{"key":"15_CR140","volume-title":"Proc. IEEE Int. Conf. Intell. Robots Syst.","author":"C. Salaun","year":"2009","unstructured":"C. Salaun, V. Padois, O. Sigaud: Control of redundant robots using learned models: An operational space control approach, Proc. IEEE Int. Conf. Intell. Robots Syst. (2009)"},{"key":"15_CR141","volume-title":"Symp. Learn. Adapt. Behav. Robotics Syst.","author":"F.R. Reinhart","year":"2008","unstructured":"F.R. Reinhart, J.J. Steil: Recurrent neural associative learning of forward and inverse kinematics for movement generation of the redundant PA-10 robot, Symp. Learn. Adapt. Behav. Robotics Syst. (2008)"},{"key":"15_CR142","volume-title":"Large Scale Kernel Machines","author":"J.Q. Candela","year":"2007","unstructured":"J.Q. Candela, C.E. Rasmussen, C.K. Williams: Large Scale Kernel Machines (MIT, Cambridge 2007)"},{"key":"15_CR143","volume-title":"Proc. Conf. Learn. Theory","author":"S. Ben-David","year":"2003","unstructured":"S. Ben-David, R. Schuller: Exploiting task relatedness for multiple task learning, Proc. Conf. Learn. Theory (2003)"},{"key":"15_CR144","first-page":"1453","volume":"6","author":"I. Tsochantaridis","year":"2005","unstructured":"I. Tsochantaridis, T. Joachims, T. Hofmann, Y. Altun: Large margin methods for structured and interdependent output variables, J. Mach. Learn. Res. 6, 1453\u20131484 (2005)","journal-title":"J. Mach. Learn. Res."},{"key":"15_CR145","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/9780262033589.001.0001","volume-title":"Semi-Supervised Learning","author":"O. Chapelle","year":"2006","unstructured":"O. Chapelle, B. Sch\u00f6lkopf, A. Zien: Semi-Supervised Learning (MIT, Cambridge 2006)"},{"key":"15_CR146","volume-title":"Proc. 18th Int. Conf. Mach. Learn.","author":"J.D. Lafferty","year":"2001","unstructured":"J.D. Lafferty, A. McCallum, F.C.N. Pereira: Conditional random fields: Probabilistic models for segmenting and labeling sequence data, Proc. 18th Int. Conf. Mach. Learn. (2001)"},{"issue":"3","key":"15_CR147","doi-asserted-by":"publisher","first-page":"263","DOI":"10.1177\/0278364912472380","volume":"32","author":"K. Muelling","year":"2012","unstructured":"K. Muelling, J. Kober, O. Kroemer, J. Peters: Learning to select and generalize striking movements in robot table tennis, Int. J. Robotics Res. 32(3), 263\u2013279 (2012)","journal-title":"Int. J. Robotics Res."},{"issue":"2\/3","key":"15_CR148","doi-asserted-by":"publisher","first-page":"311","DOI":"10.1016\/0004-3702(92)90058-6","volume":"55","author":"S. Mahadevan","year":"1992","unstructured":"S. Mahadevan, J. Connell: Automatic programming of behavior-based robots using reinforcement learning, Artif. Intell. 55(2\/3), 311\u2013365 (1992)","journal-title":"Artif. Intell."},{"issue":"1","key":"15_CR149","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1109\/37.257890","volume":"14","author":"V. Gullapalli","year":"1994","unstructured":"V. Gullapalli, J.A. Franklin, H. Benbrahim: Acquiring robot skills via reinforcement learning, IEEE Control Syst. Mag. 14(1), 13\u201324 (1994)","journal-title":"IEEE Control Syst. Mag."},{"key":"15_CR150","volume-title":"IEEE Int. Conf. Robotics Autom.","author":"J.A. Bagnell","year":"2001","unstructured":"J.A. Bagnell, J.C. Schneider: Autonomous helicopter control using reinforcement learning policy search methods, IEEE Int. Conf. Robotics Autom. (2001)"},{"key":"15_CR151","first-page":"1040","volume-title":"Adv. Neural Inform. Process. Syst.","author":"S. Schaal","year":"1996","unstructured":"S. Schaal: Learning from demonstration, Adv. Neural Inform. Process. Syst., Vol. 9 (1996) pp. 1040\u20131046"},{"key":"15_CR152","unstructured":"W. B. Powell: AI, OR and Control Theory: A\u00a0Rosetta Stone for Stochastic Optimization, Tech. Rep. (Princeton University, Princeton 2012)"},{"key":"15_CR153","first-page":"1008","volume-title":"Adv. Neural Inform. Process. Syst.","author":"C.G. Atkeson","year":"1998","unstructured":"C.G. Atkeson: Nonparametric model-based reinforcement learning, Adv. Neural Inform. Process. Syst., Vol. 10 (1998) pp. 1008\u20131014"},{"issue":"7","key":"15_CR154","doi-asserted-by":"publisher","first-page":"97","DOI":"10.1145\/1538788.1538812","volume":"52","author":"A. Coates","year":"2009","unstructured":"A. Coates, P. Abbeel, A.Y. Ng: Apprenticeship learning for helicopter control, Communication ACM 52(7), 97\u2013105 (2009)","journal-title":"Communication ACM"},{"key":"15_CR155","volume-title":"Am. Control Conf.","author":"R.S. Sutton","year":"1991","unstructured":"R.S. Sutton, A.G. Barto, R.J. Williams: Reinforcement learning is direct adaptive optimal control, Am. Control Conf. (1991)"},{"key":"15_CR156","volume-title":"Theory and Application of Reward Shaping in Reinforcement Learning","author":"A.D. Laud","year":"2004","unstructured":"A.D. Laud: Theory and Application of Reward Shaping in Reinforcement Learning (University of Illinois, Urbana-Champaign 2004)"},{"key":"15_CR157","volume-title":"28th Int. Conf. Mach. Learn.","author":"M.P. Deisenrot","year":"2011","unstructured":"M.P. Deisenrot, C.E. Rasmussen: PILCO: A\u00a0model-based and data-efficient approach to policy search, 28th Int. Conf. Mach. Learn. (2011)"},{"issue":"8","key":"15_CR158","doi-asserted-by":"publisher","first-page":"1281","DOI":"10.1016\/S0893-6080(96)00043-3","volume":"9","author":"H. Miyamoto","year":"1996","unstructured":"H. Miyamoto, S. Schaal, F. Gandolfo, H. Gomi, Y. Koike, R. Osu, E. Nakano, Y. Wada, M. Kawato: A\u00a0Kendama learning robot based on bidirectional theory, Neural Netw. 9(8), 1281\u20131302 (1996)","journal-title":"Neural Netw."},{"key":"15_CR159","volume-title":"IEEE Int. Conf. Robotics Autom.","author":"N. Kohl","year":"2004","unstructured":"N. Kohl, P. Stone: Policy gradient reinforcement learning for fast quadrupedal locomotion, IEEE Int. Conf. Robotics Autom. (2004)"},{"key":"15_CR160","volume-title":"Yale Workshop Adapt. Learn. Syst.","author":"R. Tedrake","year":"2005","unstructured":"R. Tedrake, T.W. Zhang, H.S. Seung: Learning to walk in 20\u00a0minutes, Yale Workshop Adapt. Learn. Syst. (2005)"},{"issue":"4","key":"15_CR161","doi-asserted-by":"publisher","first-page":"682","DOI":"10.1016\/j.neunet.2008.02.003","volume":"21","author":"J. Peters","year":"2008","unstructured":"J. Peters, S. Schaal: Reinforcement learning of motor skills with policy gradients, Neural Netw. 21(4), 682\u2013697 (2008)","journal-title":"Neural Netw."},{"issue":"7\u20139","key":"15_CR162","doi-asserted-by":"publisher","first-page":"1180","DOI":"10.1016\/j.neucom.2007.11.026","volume":"71","author":"J. Peters","year":"2008","unstructured":"J. Peters, S. Schaal: Natural actor-critic, Neurocomputing 71(7\u20139), 1180\u20131190 (2008)","journal-title":"Neurocomputing"},{"key":"15_CR163","first-page":"849","volume-title":"Adv. Neural Inform. Process. Syst.","author":"J. Kober","year":"2009","unstructured":"J. Kober, J. Peters: Policy search for motor primitives in robotics, Adv. Neural Inform. Process. Syst., Vol. 21 (2009) pp. 849\u2013856"},{"key":"15_CR164","volume-title":"Robotics: Science and Systems VII","author":"M.P. Deisenroth","year":"2011","unstructured":"M.P. Deisenroth, C.E. Rasmussen, D. Fox: Learning to control a\u00a0low-cost manipulator using data-efficient reinforcement learning. In: Robotics: Science and Systems VII, ed. by H. Durrand-Whyte, N. Roy, P. Abbeel (MIT, Cambridge 2011)"},{"key":"15_CR165","doi-asserted-by":"publisher","first-page":"237","DOI":"10.1613\/jair.301","volume":"4","author":"L.P. Kaelbling","year":"1996","unstructured":"L.P. Kaelbling, M.L. Littman, A.W. Moore: Reinforcement learning: A\u00a0survey, J. Artif. Intell. Res. 4, 237\u2013285 (1996)","journal-title":"J. Artif. Intell. Res."},{"key":"15_CR166","first-page":"89","volume-title":"The Handbook of Markov Decision Processes: Methods and Applications","author":"M.E. Lewis","year":"2001","unstructured":"M.E. Lewis, M.L. Puterman: The Handbook of Markov Decision Processes: Methods and Applications (Kluwer, Dordrecht 2001) pp. 89\u2013111"},{"key":"15_CR167","series-title":"Technical Report","volume-title":"Linear Quadratic Regulation as Benchmark for Policy Gradient Methods","author":"J. Peters","year":"2004","unstructured":"J. Peters, S. Vijayakumar, S. Schaal: Linear Quadratic Regulation as Benchmark for Policy Gradient Methods, Technical Report (University of Southern California, Los Angeles 2004)"},{"key":"15_CR168","volume-title":"Dynamic Programming","author":"R.E. Bellman","year":"1957","unstructured":"R.E. Bellman: Dynamic Programming (Princeton Univ. Press, Princeton 1957)"},{"key":"15_CR169","first-page":"1057","volume-title":"Adv. Neural Inform. Process. Syst.","author":"R.S. Sutton","year":"1999","unstructured":"R.S. Sutton, D. McAllester, S.P. Singh, Y. Mansour: Policy gradient methods for reinforcement learning with function approximation, Adv. Neural Inform. Process. Syst., Vol. 12 (1999) pp. 1057\u20131063"},{"key":"15_CR170","first-page":"703","volume-title":"Adv. Neural Inform. Process. Syst.","author":"T. Jaakkola","year":"1993","unstructured":"T. Jaakkola, M.I. Jordan, S.P. Singh: Convergence of stochastic iterative dynamic programming algorithms, Adv. Neural Inform. Process. Syst., Vol. 6 (1993) pp. 703\u2013710"},{"issue":"3","key":"15_CR171","doi-asserted-by":"publisher","first-page":"487","DOI":"10.2307\/2171751","volume":"65","author":"J. Rust","year":"1997","unstructured":"J. Rust: Using randomization to break the curse of dimensionality, Econometrica 65(3), 487\u2013516 (1997)","journal-title":"Econometrica"},{"key":"15_CR172","volume-title":"Optimal Control Theory","author":"D.E. Kirk","year":"1970","unstructured":"D.E. Kirk: Optimal Control Theory (Prentice-Hall, Englewood Cliffs 1970)"},{"key":"15_CR173","volume-title":"Int. Conf. Mach. Learn.","author":"A. Schwartz","year":"1993","unstructured":"A. Schwartz: A\u00a0reinforcement learning method for maximizing undiscounted rewards, Int. Conf. Mach. Learn. (1993)"},{"key":"15_CR174","volume-title":"Int. Conf. Mach. Learn.","author":"C.G. Atkeson","year":"1997","unstructured":"C.G. Atkeson, S. Schaal: Robot learning from demonstration, Int. Conf. Mach. Learn. (1997)"},{"key":"15_CR175","volume-title":"Natl. Conf. Artif. Intell.","author":"J. Peters","year":"2010","unstructured":"J. Peters, K. Muelling, Y. Altun: Relative entropy policy search, Natl. Conf. Artif. Intell. (2010)"},{"issue":"2","key":"15_CR176","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1177\/0278364907084980","volume":"27","author":"G. Endo","year":"2008","unstructured":"G. Endo, J. Morimoto, T. Matsubara, J. Nakanishi, G. Cheng: Learning CPG-based biped locomotion with a\u00a0policy gradient method: Application to a\u00a0humanoid robot, Int. J. Robotics Res. 27(2), 213\u2013228 (2008)","journal-title":"Int. J. Robotics Res."},{"issue":"13","key":"15_CR177","doi-asserted-by":"publisher","first-page":"1521","DOI":"10.1163\/156855307782148550","volume":"21","author":"F. Guenter","year":"2007","unstructured":"F. Guenter, M. Hersch, S. Calinon, A. Billard: Reinforcement learning for imitating constrained reaching movements, Adv. Robotics 21(13), 1521\u20131544 (2007)","journal-title":"Adv. Robotics"},{"key":"15_CR178","volume-title":"Robotics Sci. Syst. V, Seattle","author":"J.Z. Kolter","year":"2009","unstructured":"J.Z. Kolter, A.Y. Ng: Policy search via the signed derivative, Robotics Sci. Syst. V, Seattle (2009)"},{"key":"15_CR179","first-page":"799","volume-title":"Adv. Neural Inform. Process. Syst.","author":"A.Y. Ng","year":"2004","unstructured":"A.Y. Ng, H.J. Kim, M.I. Jordan, S. Sastry: Autonomous helicopter flight via reinforcement learning, Adv. Neural Inform. Process. Syst., Vol. 16 (2004) pp. 799\u2013806"},{"key":"15_CR180","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1007\/978-3-642-05181-4_13","volume":"264","author":"J.W. Roberts","year":"2010","unstructured":"J.W. Roberts, L. Moret, J. Zhang, R. Tedrake: From motor to interaction learning in robots, Stud. Comput. Intell. 264, 293\u2013309 (2010)","journal-title":"Stud. Comput. Intell."},{"key":"15_CR181","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"R. Tedrake","year":"2004","unstructured":"R. Tedrake: Stochastic policy gradient reinforcement learning on a\u00a0simple 3D biped, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2004)"},{"key":"15_CR182","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"F. Stulp","year":"2011","unstructured":"F. Stulp, E. Theodorou, M. Kalakrishnan, P. Pastor, L. Righetti, S. Schaal: Learning motion primitive goals for robust manipulation, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2011)"},{"key":"15_CR183","volume-title":"Int. Conf. Mach. Learn.","author":"M. Strens","year":"2001","unstructured":"M. Strens, A. Moore: Direct policy search using paired statistical tests, Int. Conf. Mach. Learn. (2001)"},{"key":"15_CR184","volume-title":"Int. Symp. Exp. Robotics","author":"A.Y. Ng","year":"2004","unstructured":"A.Y. Ng, A. Coates, M. Diel, V. Ganapathi, J. Schulte, B. Tse, E. Berger, E. Liang: Autonomous inverted helicopter flight via reinforcement learning, Int. Symp. Exp. Robotics (2004)"},{"key":"15_CR185","first-page":"427","volume-title":"Adv. Neural Inform. Process. Syst.","author":"T. Geng","year":"2006","unstructured":"T. Geng, B. Porr, F. W\u00f6rg\u00f6tter: Fast biped walking with a\u00a0reflexive controller and real-time policy searching, Adv. Neural Inform. Process. Syst., Vol. 18 (2006) pp. 427\u2013434"},{"key":"15_CR186","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"N. Mitsunaga","year":"2005","unstructured":"N. Mitsunaga, C. Smith, T. Kanda, H. Ishiguro, N. Hagita: Robot behavior adaptation for human-robot interaction based on policy gradient reinforcement learning, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2005)"},{"key":"15_CR187","volume-title":"Int. Conf. Artif. Neural Netw.","author":"M. Sato","year":"2002","unstructured":"M. Sato, Y. Nakamura, S. Ishii: Reinforcement learning for biped locomotion, Int. Conf. Artif. Neural Netw. (2002)"},{"key":"15_CR188","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4757-4321-0","volume-title":"The Cross Entropy Method: A\u00a0Unified Approach to Combinatorial Optimization, Monte-Carlo Simulation","author":"R.Y. Rubinstein","year":"2004","unstructured":"R.Y. Rubinstein, D.P. Kroese: The Cross Entropy Method: A\u00a0Unified Approach to Combinatorial Optimization, Monte-Carlo Simulation (Springer, New York 2004)"},{"key":"15_CR189","volume-title":"Genetic Algorithms","author":"D.E. Goldberg","year":"1989","unstructured":"D.E. Goldberg: Genetic Algorithms (Addision Wesley, New York 1989)"},{"key":"15_CR190","series-title":"Adv. Design Control","volume-title":"Practical Methods for Optimal Control Using Nonlinear Programming","author":"J.T. Betts","year":"2001","unstructured":"J.T. Betts: Practical Methods for Optimal Control Using Nonlinear Programming, Adv. Design Control, Vol. 3 (SIAM, Philadelphia 2001)"},{"key":"15_CR191","first-page":"229","volume":"8","author":"R.J. Williams","year":"1992","unstructured":"R.J. Williams: Simple statistical gradient-following algorithms for connectionist reinforcement learning, Mach. Learn. 8, 229\u2013256 (1992)","journal-title":"Mach. Learn."},{"issue":"2","key":"15_CR192","doi-asserted-by":"publisher","first-page":"271","DOI":"10.1162\/neco.1997.9.2.271","volume":"9","author":"P. Dayan","year":"1997","unstructured":"P. Dayan, G.E. Hinton: Using expectation-maximization for reinforcement learning, Neural Comput. 9(2), 271\u2013278 (1997)","journal-title":"Neural Comput."},{"issue":"2","key":"15_CR193","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1007\/s10514-009-9132-0","volume":"27","author":"N. Vlassis","year":"2009","unstructured":"N. Vlassis, M. Toussaint, G. Kontes, S. Piperidis: Learning model-free robot control by a\u00a0Monte Carlo EM algorithm, Auton. Robots 27(2), 123\u2013130 (2009)","journal-title":"Auton. Robots"},{"key":"15_CR194","volume-title":"Proc. Robotics Sci. Syst. Conf.","author":"J. Kober","year":"2010","unstructured":"J. Kober, E. Oztop, J. Peters: Reinforcement learning to adjust robot movements to new situations, Proc. Robotics Sci. Syst. Conf. (2010)"},{"key":"15_CR195","volume-title":"IEEE Int. Conf. Robotics Autom.","author":"E.A. Theodorou","year":"2010","unstructured":"E.A. Theodorou, J. Buchli, S. Schaal: Reinforcement learning of motor skills in high dimensions: A\u00a0path integral approach, IEEE Int. Conf. Robotics Autom. (2010)"},{"key":"15_CR196","first-page":"831","volume-title":"Adv. Neural Inform. Process. Syst.","author":"J.A. Bagnell","year":"2003","unstructured":"J.A. Bagnell, A.Y. Ng, S. Kakade, J. Schneider: Policy search by dynamic programming, Adv. Neural Inform. Process. Syst., Vol. 16 (2003) pp. 831\u2013838"},{"issue":"2","key":"15_CR197","doi-asserted-by":"publisher","first-page":"175","DOI":"10.1177\/0278364907087426","volume":"27","author":"T. Kollar","year":"2008","unstructured":"T. Kollar, N. Roy: Trajectory optimization using reinforcement learning for map exploration, Int. J. Robotics Res. 27(2), 175\u2013197 (2008)","journal-title":"Int. J. Robotics Res."},{"key":"15_CR198","volume-title":"Int. Jt. Conf. Artif. Intell.","author":"D. Lizotte","year":"2007","unstructured":"D. Lizotte, T. Wang, M. Bowling, D. Schuurmans: Automatic gait optimization with Gaussian process regression, Int. Jt. Conf. Artif. Intell. (2007)"},{"key":"15_CR199","volume-title":"IEEE-RAS Int. Conf. Humanoid Robots","author":"S. Kuindersma","year":"2011","unstructured":"S. Kuindersma, R. Grupen, A.G. Barto: Learning dynamic arm motions for postural recovery, IEEE-RAS Int. Conf. Humanoid Robots (2011)"},{"key":"15_CR200","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"M. Tesch","year":"2011","unstructured":"M. Tesch, J.G. Schneider, H. Choset: Using response surfaces and expected improvement to optimize snake robot gait parameters, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2011)"},{"key":"15_CR201","volume-title":"IEEE Proc. Int. Conf. Robotics Autom.","author":"S.-J. Yi","year":"2011","unstructured":"S.-J. Yi, B.-T. Zhang, D. Hong, D.D. Lee: Learning full body push recovery control for small humanoid robots, IEEE Proc. Int. Conf. Robotics Autom. (2011)"},{"key":"15_CR202","first-page":"369","volume-title":"Adv. Neural Inform. Process. Syst.","author":"J.A. Boyan","year":"1995","unstructured":"J.A. Boyan, A.W. Moore: Generalization in reinforcement learning: Safely approximating the value function, Adv. Neural Inform. Process. Syst., Vol. 7 (1995) pp. 369\u2013376"},{"key":"15_CR203","volume-title":"Int. Conf. Mach. Learn.","author":"S. Kakade","year":"2002","unstructured":"S. Kakade, J. Langford: Approximately optimal approximate reinforcement learning, Int. Conf. Mach. Learn. (2002)"},{"key":"15_CR204","first-page":"1471","volume":"5","author":"E. Greensmith","year":"2004","unstructured":"E. Greensmith, P.L. Bartlett, J. Baxter: Variance reduction techniques for gradient estimates in reinforcement learning, J. Mach. Learn. Res. 5, 1471\u20131530 (2004)","journal-title":"J. Mach. Learn. Res."},{"key":"15_CR205","volume-title":"Am. Control Conf.","author":"M.T. Rosenstein","year":"2004","unstructured":"M.T. Rosenstein, A.G. Barto: Reinforcement learning with supervision by a\u00a0stable controller, Am. Control Conf. (2004)"},{"issue":"5","key":"15_CR206","doi-asserted-by":"publisher","first-page":"674","DOI":"10.1109\/9.580874","volume":"42","author":"J.N. Tsitsiklis","year":"1997","unstructured":"J.N. Tsitsiklis, B. Van Roy: An analysis of temporal-difference learning with function approximation, IEEE Trans. Autom. Control 42(5), 674\u2013690 (1997)","journal-title":"IEEE Trans. Autom. Control"},{"key":"15_CR207","volume-title":"Int. Conf. Mach. Learn.","author":"J.Z. Kolter","year":"2009","unstructured":"J.Z. Kolter, A.Y. Ng: Regularization and feature selection in least-squares temporal difference learning, Int. Conf. Mach. Learn. (2009)"},{"key":"15_CR208","series-title":"Technical Report WL-TR-93-1147","doi-asserted-by":"publisher","DOI":"10.21236\/ADA280844","volume-title":"Reinforcement Learning with High-Dimensional Continuous Actions","author":"L.C. Baird","year":"1993","unstructured":"L.C. Baird, H. Klopf: Reinforcement Learning with High-Dimensional Continuous Actions, Technical Report WL-TR-93-1147 (Wright-Patterson Air Force Base, Dayton 1993)"},{"key":"15_CR209","volume-title":"AAAI Conf. Artif. Intell.","author":"G.D. Konidaris","year":"2011","unstructured":"G.D. Konidaris, S. Osentoski, P. Thomas: Value function approximation in reinforcement learning using the Fourier basis, AAAI Conf. Artif. Intell. (2011)"},{"key":"15_CR210","volume-title":"Int. Symp. Robotics Res.","author":"J. Peters","year":"2010","unstructured":"J. Peters, K. Muelling, J. Kober, D. Nguyen-Tuong, O. Kroemer: Towards motor skill learning for robotics, Int. Symp. Robotics Res. (2010)"},{"key":"15_CR211","volume-title":"Reinforcement Learning and Dynamic Programming Using Function Approximators","author":"L. Bu\u015foniu","year":"2010","unstructured":"L. Bu\u015foniu, R. Babu\u0161ka, B. de Schutter, D. Ernst: Reinforcement Learning and Dynamic Programming Using Function Approximators (CRC, Boca Raton 2010)"},{"issue":"4","key":"15_CR212","doi-asserted-by":"publisher","first-page":"341","DOI":"10.1023\/A:1025696116075","volume":"13","author":"A.G. Barto","year":"2003","unstructured":"A.G. Barto, S. Mahadevan: Recent advances in hierarchical reinforcement learning, Discret. Event Dyn. Syst. 13(4), 341\u2013379 (2003)","journal-title":"Discret. Event Dyn. Syst."},{"issue":"3","key":"15_CR213","doi-asserted-by":"publisher","first-page":"216","DOI":"10.1109\/TAMD.2010.2103311","volume":"3","author":"S. Hart","year":"2011","unstructured":"S. Hart, R. Grupen: Learning generalizable control programs, IEEE Trans. Auton. Mental Dev. 3(3), 216\u2013231 (2011)","journal-title":"IEEE Trans. Auton. Mental Dev."},{"key":"15_CR214","first-page":"1047","volume-title":"Adv. Neural Inform. Process. Syst.","author":"J.G. Schneider","year":"1997","unstructured":"J.G. Schneider: Exploiting model uncertainty estimates for safe dynamic control learning, Adv. Neural Inform. Process. Syst., Vol. 9 (1997) pp. 1047\u20131053"},{"key":"15_CR215","volume-title":"Learning Decisions: Robustness, Uncertainty, and Approximation. Dissertation","author":"J.A. Bagnell","year":"2004","unstructured":"J.A. Bagnell: Learning Decisions: Robustness, Uncertainty, and Approximation. Dissertation (Robotics Institute, Carnegie Mellon University, Pittsburgh 2004)"},{"key":"15_CR216","volume-title":"29th Int. Conf. Mach. Learn.","author":"T.M. Moldovan","year":"2012","unstructured":"T.M. Moldovan, P. Abbeel: Safe exploration in markov decision processes, 29th Int. Conf. Mach. Learn. (2012)"},{"key":"15_CR217","volume-title":"IEEE Int. Conf. Robotics Autom.","author":"T. Hester","year":"2012","unstructured":"T. Hester, M. Quinlan, P. Stone: RTMBA: A\u00a0real-time model-based reinforcement learning architecture for robot control, IEEE Int. Conf. Robotics Autom. (2012)"},{"key":"15_CR218","first-page":"663","volume-title":"Adv. Neural Inform. Process. Syst.","author":"C.G. Atkeson","year":"1994","unstructured":"C.G. Atkeson: Using local trajectory optimizers to speed up global optimization in dynamic programming, Adv. Neural Inform. Process. Syst., Vol. 6 (1994) pp. 663\u2013670"},{"issue":"1\/2","key":"15_CR219","first-page":"171","volume":"84","author":"J. Kober","year":"2010","unstructured":"J. Kober, J. Peters: Policy search for motor primitives in robotics, Mach. Learn. 84(1\/2), 171\u2013203 (2010)","journal-title":"Mach. Learn."},{"key":"15_CR220","volume-title":"Conf. Comput. Learn. Theory","author":"S. Russell","year":"1989","unstructured":"S. Russell: Learning agents for uncertain environments (extended abstract), Conf. Comput. Learn. Theory (1989)"},{"key":"15_CR221","volume-title":"Int. Conf. Mach. Learn.","author":"P. Abbeel","year":"2004","unstructured":"P. Abbeel, A.Y. Ng: Apprenticeship learning via inverse reinforcement learning, Int. Conf. Mach. Learn. (2004)"},{"key":"15_CR222","volume-title":"Int. Conf. Mach. Learn.","author":"N.D. Ratliff","year":"2006","unstructured":"N.D. Ratliff, J.A. Bagnell, M.A. Zinkevich: Maximum margin planning, Int. Conf. Mach. Learn. (2006)"},{"key":"15_CR223","volume-title":"Decisions with Multiple Objectives: Preferences and Value Tradeoffs","author":"R.L. Keeney","year":"1976","unstructured":"R.L. Keeney, H. Raiffa: Decisions with Multiple Objectives: Preferences and Value Tradeoffs (Wiley, New York 1976)"},{"key":"15_CR224","first-page":"1153","volume-title":"Adv. Neural Inform. Process. Syst.","author":"N. Ratliff","year":"2006","unstructured":"N. Ratliff, D. Bradley, J.A. Bagnell, J. Chestnutt: Boosting structured prediction for imitation learning, Adv. Neural Inform. Process. Syst., Vol. 19 (2006) pp. 1153\u20131160"},{"key":"15_CR225","volume-title":"Robotics: Science and Systems","author":"D. Silver","year":"2008","unstructured":"D. Silver, J.A. Bagnell, A. Stentz: High performance outdoor navigation from overhead data using imitation learning. In: Robotics: Science and Systems, Vol. IV, ed. by O. Brock, J. Trinkle, F. Ramos (MIT, Cambridge 2008)"},{"issue":"12","key":"15_CR226","doi-asserted-by":"publisher","first-page":"1565","DOI":"10.1177\/0278364910369715","volume":"29","author":"D. Silver","year":"2010","unstructured":"D. Silver, J.A. Bagnell, A. Stentz: Learning from demonstration for autonomous navigation in complex unstructured terrain, Int. J. Robotics Res. 29(12), 1565\u20131592 (2010)","journal-title":"Int. J. Robotics Res."},{"key":"15_CR227","volume-title":"IEEE-RAS Int. Conf. Humanoid Robots","author":"N. Ratliff","year":"2007","unstructured":"N. Ratliff, J.A. Bagnell, S. Srinivasa: Imitation learning for locomotion and manipulation, IEEE-RAS Int. Conf. Humanoid Robots (2007)"},{"key":"15_CR228","first-page":"769","volume-title":"Adv. Neural Inform. Process. Syst.","author":"J.Z. Kolter","year":"2007","unstructured":"J.Z. Kolter, P. Abbeel, A.Y. Ng: Hierarchical apprenticeship learning with application to quadruped locomotion, Adv. Neural Inform. Process. Syst., Vol. 20 (2007) pp. 769\u2013776"},{"key":"15_CR229","first-page":"2190","volume-title":"Adv. Neural Inform. Process. Syst.","author":"J. Sorg","year":"2010","unstructured":"J. Sorg, S.P. Singh, R.L. Lewis: Reward design via online gradient ascent, Adv. Neural Inform. Process. Syst., Vol. 23 (2010) pp. 2190\u20132198"},{"key":"15_CR230","volume-title":"IEEE Proc. Int. Conf. Robotics Autom.","author":"M. Zucker","year":"2012","unstructured":"M. Zucker, J.A. Bagnell: Reinforcement planning: RL for optimal planners, IEEE Proc. Int. Conf. Robotics Autom. (2012)"},{"key":"15_CR231","volume-title":"Int. Jt. Conf. Neural Netw.","author":"H. Benbrahim","year":"1992","unstructured":"H. Benbrahim, J.S. Doleac, J.A. Franklin, O.G. Selfridge: Real-time learning: A\u00a0ball on a\u00a0beam, Int. Jt. Conf. Neural Netw. (1992)"},{"key":"15_CR232","volume-title":"Int. Workshop Robotics, Alpe-Adria-Danube Region","author":"B. Nemec","year":"2010","unstructured":"B. Nemec, M. Zorko, L. Zlajpah: Learning of a\u00a0ball-in-a-cup playing robot, Int. Workshop Robotics, Alpe-Adria-Danube Region (2010)"},{"key":"15_CR233","volume-title":"Int. Fla. Artif. Intell. Res. Soc. Conf.","author":"M. Tokic","year":"2009","unstructured":"M. Tokic, W. Ertel, J. Fessler: The crawler, a\u00a0class room demonstrator for reinforcement learning, Int. Fla. Artif. Intell. Res. Soc. Conf. (2009)"},{"key":"15_CR234","volume-title":"IEEE Conf. Decis. Control","author":"H. Kimura","year":"2001","unstructured":"H. Kimura, T. Yamashita, S. Kobayashi: Reinforcement learning of walking behavior for a\u00a0four-legged robot, IEEE Conf. Decis. Control (2001)"},{"key":"15_CR235","volume-title":"Aust. Conf. Robotics Autom.","author":"R.A. Willgoss","year":"1999","unstructured":"R.A. Willgoss, J. Iqbal: Reinforcement learning of behaviors in mobile robots using noisy infrared sensing, Aust. Conf. Robotics Autom. (1999)"},{"key":"15_CR236","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1007\/978-3-540-74565-5_19","volume":"4667","author":"L. Paletta","year":"2007","unstructured":"L. Paletta, G. Fritz, F. Kintzler, J. Irran, G. Dorffner: Perception and developmental learning of affordances in autonomous robots, Lect. Notes Comput. Sci. 4667, 235\u2013250 (2007)","journal-title":"Lect. Notes Comput. Sci."},{"key":"15_CR237","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"C. Kwok","year":"2004","unstructured":"C. Kwok, D. Fox: Reinforcement learning for sensing strategies, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2004)"},{"key":"15_CR238","volume-title":"Int. Conf. Simul. Adapt. Behav.","author":"T. Yasuda","year":"2008","unstructured":"T. Yasuda, K. Ohkura: A\u00a0reinforcement learning technique with an adaptive action generator for a\u00a0multi-robot system, Int. Conf. Simul. Adapt. Behav. (2008)"},{"issue":"3","key":"15_CR239","doi-asserted-by":"publisher","first-page":"294","DOI":"10.1177\/0278364910382464","volume":"30","author":"J.H. Piater","year":"2011","unstructured":"J.H. Piater, S. Jodogne, R. Detry, D. Kraft, N. Kr\u00fcger, O. Kroemer, J. Peters: Learning visual representations for perception-action systems, Int. J. Robotics Res. 30(3), 294\u2013307 (2011)","journal-title":"Int. J. Robotics Res."},{"issue":"2\/3","key":"15_CR240","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1023\/A:1018237008823","volume":"23","author":"M. Asada","year":"1996","unstructured":"M. Asada, S. Noda, S. Tawaratsumida, K. Hosoda: Purposive behavior acquisition for a\u00a0real robot by vision-based reinforcement learning, Mach. Learn. 23(2\/3), 279\u2013303 (1996)","journal-title":"Mach. Learn."},{"issue":"3\/4","key":"15_CR241","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1016\/S0921-8890(97)00044-4","volume":"22","author":"M. Huber","year":"1997","unstructured":"M. Huber, R.A. Grupen: A\u00a0feedback control structure for on-line learning tasks, Robotics Auton. Syst. 22(3\/4), 303\u2013315 (1997)","journal-title":"Robotics Auton. Syst."},{"key":"15_CR242","volume-title":"Int. Symp. Robotics Autom.","author":"P. Fidelman","year":"2004","unstructured":"P. Fidelman, P. Stone: Learning ball acquisition on a\u00a0physical robot, Int. Symp. Robotics Autom. (2004)"},{"key":"15_CR243","volume-title":"Int. Conf. Dev. Learn.","author":"V. Soni","year":"2006","unstructured":"V. Soni, S.P. Singh: Reinforcement learning of hierarchical skills on the Sony AIBO robot, Int. Conf. Dev. Learn. (2006)"},{"key":"15_CR244","volume-title":"IEEE-RAS Int. Conf. Humanoid Robots","author":"B. Nemec","year":"2009","unstructured":"B. Nemec, M. Tamo\u0161iunait\u0117, F. W\u00f6rg\u00f6tter, A. Ude: Task adaptation through exploration and action sequencing, IEEE-RAS Int. Conf. Humanoid Robots (2009)"},{"key":"15_CR245","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1023\/A:1008819414322","volume":"4","author":"M.J. Matari\u0107","year":"1997","unstructured":"M.J. Matari\u0107: Reinforcement learning in the multi-robot domain, Auton. Robots 4, 73\u201383 (1997)","journal-title":"Auton. Robots"},{"key":"15_CR246","volume-title":"Int. Conf. Mach. Learn. (ICML)","author":"M.J. Matari\u0107","year":"1994","unstructured":"M.J. Matari\u0107: Reward functions for accelerated learning, Int. Conf. Mach. Learn. (ICML) (1994)"},{"key":"15_CR247","volume-title":"Int. Conf. Dev. Learn.","author":"R. Platt","year":"2006","unstructured":"R. Platt, R.A. Grupen, A.H. Fagg: Improving grasp skills using schema structured learning, Int. Conf. Dev. Learn. (2006)"},{"key":"15_CR248","series-title":"Technical Report","volume-title":"Robot Shaping: Developing Situated Agents Through Learning","author":"M. Dorigo","year":"1993","unstructured":"M. Dorigo, M. Colombetti: Robot Shaping: Developing Situated Agents Through Learning, Technical Report (International Computer Science Institute, Berkeley 1993)"},{"key":"15_CR249","volume-title":"AAAI Conf. Artif. Intell.","author":"G.D. Konidaris","year":"2011","unstructured":"G.D. Konidaris, S. Kuindersma, R. Grupen, A.G. Barto: Autonomous skill acquisition on a\u00a0mobile manipulator, AAAI Conf. Artif. Intell. (2011)"},{"issue":"3","key":"15_CR250","doi-asserted-by":"publisher","first-page":"360","DOI":"10.1177\/0278364911428653","volume":"31","author":"G.D. Konidaris","year":"2012","unstructured":"G.D. Konidaris, S. Kuindersma, R. Grupen, A.G. Barto: Robot learning from demonstration by constructing skill trees, Int. J. Robotics Res. 31(3), 360\u2013375 (2012)","journal-title":"Int. J. Robotics Res."},{"key":"15_CR251","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"A. Cocora","year":"2006","unstructured":"A. Cocora, K. Kersting, C. Plagemann, W. Burgard, L. de Raedt: Learning relational navigation policies, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2006)"},{"key":"15_CR252","volume-title":"Robotics: Science and Systems","author":"D. Katz","year":"2008","unstructured":"D. Katz, Y. Pyuro, O. Brock: Learning to manipulate articulated objects in unstructured environments using a\u00a0grounded relational representation. In: Robotics: Science and Systems, Vol. IV, ed. by O. Brock, J. Trinkle, F. Ramos (MIT, Cambridge 2008)"},{"key":"15_CR253","volume-title":"Model-Based Control of a\u00a0Robot Manipulator","author":"C.H. An","year":"1988","unstructured":"C.H. An, C.G. Atkeson, J.M. Hollerbach: Model-Based Control of a\u00a0Robot Manipulator (MIT, Press, Cambridge 1988)"},{"key":"15_CR254","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"C. Gaskett","year":"2000","unstructured":"C. Gaskett, L. Fletcher, A. Zelinsky: Reinforcement learning for a\u00a0vision based mobile robot, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2000)"},{"key":"15_CR255","volume-title":"Int. Symp. Neural Netw.","author":"Y. Duan","year":"2008","unstructured":"Y. Duan, B. Cui, H. Yang: Robot navigation based on fuzzy RL algorithm, Int. Symp. Neural Netw. (2008)"},{"issue":"3\/4","key":"15_CR256","doi-asserted-by":"publisher","first-page":"283","DOI":"10.1016\/S0921-8890(97)00043-2","volume":"22","author":"H. Benbrahim","year":"1997","unstructured":"H. Benbrahim, J.A. Franklin: Biped dynamic walking using reinforcement learning, Robotics Auton. Syst. 22(3\/4), 283\u2013302 (1997)","journal-title":"Robotics Auton. Syst."},{"key":"15_CR257","volume-title":"Natl. Conf. Artif. Intell.\/Innov. Appl. Artif. Intell.","author":"W.D. Smart","year":"1989","unstructured":"W.D. Smart, L. Pack Kaelbling: A\u00a0framework for reinforcement learning on real robots, Natl. Conf. Artif. Intell.\/Innov. Appl. Artif. Intell. (1989)"},{"key":"15_CR258","volume-title":"Learning from Observation Using Primitives","author":"D.C. Bentivegna","year":"2004","unstructured":"D.C. Bentivegna: Learning from Observation Using Primitives (Georgia Institute of Technology, Atlanta 2004)"},{"key":"15_CR259","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"A. Rottmann","year":"2007","unstructured":"A. Rottmann, C. Plagemann, P. Hilgers, W. Burgard: Autonomous blimp control using model-free reinforcement learning in a\u00a0continuous state and action space, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2007)"},{"key":"15_CR260","volume-title":"Jt. Int. Symp. Robotics (ISR) Ger. Conf. Robotics (ROBOTIK)","author":"K. Gr\u00e4ve","year":"2010","unstructured":"K. Gr\u00e4ve, J. St\u00fcckler, S. Behnke: Learning motion skills from expert demonstrations and own experience using Gaussian process regression, Jt. Int. Symp. Robotics (ISR) Ger. Conf. Robotics (ROBOTIK) (2010)"},{"key":"15_CR261","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"O. Kroemer","year":"2009","unstructured":"O. Kroemer, R. Detry, J. Piater, J. Peters: Active learning using mean shift optimization for robot grasping, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2009)"},{"issue":"9","key":"15_CR262","doi-asserted-by":"publisher","first-page":"1105","DOI":"10.1016\/j.robot.2010.06.001","volume":"58","author":"O. Kroemer","year":"2010","unstructured":"O. Kroemer, R. Detry, J. Piater, J. Peters: Combining active learning and reactive control for robot grasping, Robotics Auton. Syst. 58(9), 1105\u20131116 (2010)","journal-title":"Robotics Auton. Syst."},{"key":"15_CR263","volume-title":"Int. Conf. Neural Inf. Process.","author":"T. Tamei","year":"2009","unstructured":"T. Tamei, T. Shibata: Policy gradient learning of cooperative interaction with a\u00a0robot using user's biological signals, Int. Conf. Neural Inf. Process. (2009)"},{"key":"15_CR264","first-page":"1547","volume-title":"Adv. Neural Inform. Process. Syst.","author":"A.J. Ijspeert","year":"2003","unstructured":"A.J. Ijspeert, J. Nakanishi, S. Schaal: Learning attractor landscapes for learning motor primitives, Adv. Neural Inform. Process. Syst., Vol. 15 (2003) pp. 1547\u20131554"},{"issue":"1","key":"15_CR265","doi-asserted-by":"publisher","first-page":"425","DOI":"10.1016\/S0079-6123(06)65027-9","volume":"165","author":"S. Schaal","year":"2007","unstructured":"S. Schaal, P. Mohajerian, A.J. Ijspeert: Dynamics systems vs. optimal control\u00a0\u2013 A\u00a0unifying view, Prog. Brain Res. 165(1), 425\u2013445 (2007)","journal-title":"Prog. Brain Res."},{"key":"15_CR266","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"H.-I. Lin","year":"2012","unstructured":"H.-I. Lin, C.-C. Lai: Learning collision-free reaching skill from primitives, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2012)"},{"key":"15_CR267","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"J. Kober","year":"2008","unstructured":"J. Kober, B. Mohler, J. Peters: Learning perceptual coupling for motor primitives, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2008)"},{"key":"15_CR268","volume-title":"Proc. IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"S. Bitzer","year":"2010","unstructured":"S. Bitzer, M. Howard, S. Vijayakumar: Using dimensionality reduction to exploit constraints in reinforcement learning, Proc. IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2010)"},{"issue":"7","key":"15_CR269","doi-asserted-by":"publisher","first-page":"820","DOI":"10.1177\/0278364911402527","volume":"30","author":"J. Buchli","year":"2011","unstructured":"J. Buchli, F. Stulp, E. Theodorou, S. Schaal: Learning variable impedance control, Int. J. Robotics Res. 30(7), 820\u2013833 (2011)","journal-title":"Int. J. Robotics Res."},{"key":"15_CR270","volume-title":"IEEE Int. Conf. Robotics Autom.","author":"P. Pastor","year":"2011","unstructured":"P. Pastor, M. Kalakrishnan, S. Chitta, E. Theodorou, S. Schaal: Skill learning and task outcome prediction for manipulation, IEEE Int. Conf. Robotics Autom. (2011)"},{"key":"15_CR271","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"M. Kalakrishnan","year":"2011","unstructured":"M. Kalakrishnan, L. Righetti, P. Pastor, S. Schaal: Learning force control policies for compliant manipulation, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2011)"},{"key":"15_CR272","volume-title":"11th Int. Symp. Robotics Res.","author":"D.C. Bentivegna","year":"2004","unstructured":"D.C. Bentivegna, C.G. Atkeson, G. Cheng: Learning from observation and practice using behavioral primitives: Marble maze, 11th Int. Symp. Robotics Res. (2004)"},{"key":"15_CR273","volume-title":"EUROMICRO Workshop Adv. Mobile Robots","author":"F. Kirchner","year":"1997","unstructured":"F. Kirchner: Q-learning of complex behaviours on a\u00a0six-legged walking machine, EUROMICRO Workshop Adv. Mobile Robots (1997)"},{"issue":"1","key":"15_CR274","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1016\/S0921-8890(01)00113-0","volume":"36","author":"J. Morimoto","year":"2001","unstructured":"J. Morimoto, K. Doya: Acquisition of stand-up behavior by a\u00a0real robot using hierarchical reinforcement learning, Robotics Auton. Syst. 36(1), 37\u201351 (2001)","journal-title":"Robotics Auton. Syst."},{"issue":"3","key":"15_CR275","doi-asserted-by":"publisher","first-page":"381","DOI":"10.1109\/3477.499790","volume":"26","author":"J.-Y. Donnart","year":"1996","unstructured":"J.-Y. Donnart, J.-A. Meyer: Learning reactive and planning rules in a\u00a0motivationally autonomous animat, Syst. Man Cybern. B 26(3), 381\u2013395 (1996)","journal-title":"Syst. Man Cybern. B"},{"key":"15_CR276","volume-title":"IEEE\/RSJ Int. Conf. Intell. Robots Syst.","author":"C. Daniel","year":"2012","unstructured":"C. Daniel, G. Neumann, J. Peters: Learning concurrent motor skills in versatile solution spaces, IEEE\/RSJ Int. Conf. Intell. Robots Syst. (2012)"},{"key":"15_CR277","volume-title":"IEEE-RAS Int. Conf. Humanoid Robots","author":"E.C. Whitman","year":"2010","unstructured":"E.C. Whitman, C.G. Atkeson: Control of instantaneously coupled systems applied to humanoid walking, IEEE-RAS Int. Conf. Humanoid Robots (2010)"},{"key":"15_CR278","volume-title":"2nd Int. Workshop Epigenetic Robotics Model. Cognit. Dev. Robotic Syst.","author":"X. Huang","year":"2002","unstructured":"X. Huang, J. Weng: Novelty and reinforcement learning in the value system of developmental robots, 2nd Int. Workshop Epigenetic Robotics Model. Cognit. Dev. Robotic Syst. (2002)"},{"key":"15_CR279","volume-title":"Eur. Workshop Learn. Robots","author":"M. Pendrith","year":"1999","unstructured":"M. Pendrith: Reinforcement learning in situated agents: Some theoretical problems and practical solutions, Eur. Workshop Learn. Robots (1999)"},{"key":"15_CR280","volume-title":"IEEE Conf. Robotics Autom. Mechatron.","author":"B. Wang","year":"2006","unstructured":"B. Wang, J.W. Li, H. Liu: A\u00a0heuristic reinforcement learning for robot approaching objects, IEEE Conf. Robotics Autom. Mechatron. (2006)"},{"key":"15_CR281","volume-title":"Learning in Embedded Systems","author":"L.P. Kaelbling","year":"1990","unstructured":"L.P. Kaelbling: Learning in Embedded Systems (Stanford University, Stanford 1990)"},{"key":"15_CR282","volume-title":"Int. Conf. Mach. Learn.","author":"R.S. Sutton","year":"1990","unstructured":"R.S. Sutton: Integrated architectures for learning, planning, and reacting based on approximating dynamic programming, Int. Conf. Mach. Learn. (1990)"},{"issue":"1","key":"15_CR283","first-page":"103","volume":"13","author":"A.W. Moore","year":"1993","unstructured":"A.W. Moore, C.G. Atkeson: Prioritized sweeping: Reinforcement learning with less data and less time, Mach. Learn. 13(1), 103\u2013130 (1993)","journal-title":"Mach. Learn."},{"issue":"1","key":"15_CR284","first-page":"283","volume":"22","author":"J. Peng","year":"1996","unstructured":"J. Peng, R.J. Williams: Incremental multi-step Q-learning, Mach. Learn. 22(1), 283\u2013290 (1996)","journal-title":"Mach. Learn."},{"key":"15_CR285","volume-title":"3rd Eur. Conf. Artif. Life","author":"N. Jakobi","year":"1995","unstructured":"N. Jakobi, P. Husbands, I. Harvey: Noise and the reality gap: The use of simulation in evolutionary robotics, 3rd Eur. Conf. Artif. Life (1995)"}],"container-title":["Springer Handbooks","Springer Handbook of Robotics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-32552-1_15","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,4]],"date-time":"2025-06-04T06:36:08Z","timestamp":1749018968000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-32552-1_15"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016]]},"ISBN":["9783319325507","9783319325521"],"references-count":285,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-32552-1_15","relation":{},"ISSN":["2522-8692","2522-8706"],"issn-type":[{"value":"2522-8692","type":"print"},{"value":"2522-8706","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016]]},"assertion":[{"value":"27 July 2016","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}