{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,14]],"date-time":"2026-01-14T19:05:28Z","timestamp":1768417528470,"version":"3.49.0"},"reference-count":32,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016,5]]},"DOI":"10.1109\/icra.2016.7487172","type":"proceedings-article","created":{"date-parts":[[2016,6,9]],"date-time":"2016-06-09T17:33:24Z","timestamp":1465493604000},"page":"504-511","source":"Crossref","is-referenced-by-count":19,"title":["Model-based reinforcement learning with parametrized physical models and optimism-driven exploration"],"prefix":"10.1109","author":[{"given":"Chris","family":"Xie","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sachin","family":"Patil","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Teodor","family":"Moldovan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sergey","family":"Levine","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pieter","family":"Abbeel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1016\/j.rcim.2010.03.013"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2014.6942749"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2014.6907001"},{"key":"ref10","author":"camacho","year":"2013","journal-title":"Model Predictive Control"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2014.6907423"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2015.7139550"},{"key":"ref13","first-page":"465","article-title":"PILCO: A model-based and data-efficient approach to policy search","author":"deisenroth","year":"2011","journal-title":"Proceedings of the 28th International Conference on Machine Learning"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1177\/027836499201100408"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/AIM.2013.6584295"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30301-5_15"},{"key":"ref17","author":"jacobson","year":"1970","journal-title":"Differential dynamic programming ser Modern analytic and computational methods in science and mathematics American Elsevier Pub Co"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913495721"},{"key":"ref19","article-title":"Variational bayesian optimization for runtime risk-sensitive control","author":"kuindersma","year":"2013","journal-title":"Robotics Science and Systems (RSS)"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/70.631234"},{"key":"ref4","first-page":"97","article-title":"Near-optimal BRL using optimistic local transitions","author":"araya","year":"2012","journal-title":"Proc Int Conf on Machine Learning (ICML) ser ICML '12"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/37.126844"},{"key":"ref3","article-title":"Optimistic linear programming gives logarithmic regret for irreducible MDPs","author":"tewari","year":"2007","journal-title":"Proc of Neural Information Processing Systems Conference (NIPS)"},{"key":"ref6","author":"astrom","year":"1994","journal-title":"Adaptive Control"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386025"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1177\/027836498900800603"},{"key":"ref8","article-title":"Approx-imate real-time optimal control based on sparse gaussian process models","author":"boedecker","year":"2014","journal-title":"Proc Adaptive Dynamic Programming and Reinforcement Learning (ADPRL)"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.tcs.2009.01.016"},{"key":"ref2","article-title":"An application of reinforcement learning to aerobatic helicopter flight","author":"abbeel","year":"2006","journal-title":"Advances in neural information processing systems"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ADPRL.2014.7010608"},{"key":"ref1","article-title":"Regret bounds for the adaptive control of linear quadratic systems","author":"abbasi-yadkori","year":"2011","journal-title":"Proc 24th Annu Conf Learn Theory"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-1768-8_11"},{"key":"ref22","first-page":"16","article-title":"Conjugate bayesian analysis of the gaussian distribution","volume":"1","author":"murphy","year":"2007","journal-title":"Def"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2015.7139645"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2014.6907444"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2010.5509858"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913514870"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2012.6225279"}],"event":{"name":"2016 IEEE International Conference on Robotics and Automation (ICRA)","location":"Stockholm, Sweden","start":{"date-parts":[[2016,5,16]]},"end":{"date-parts":[[2016,5,21]]}},"container-title":["2016 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7478842\/7487087\/07487172.pdf?arnumber=7487172","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2016,9,29]],"date-time":"2016-09-29T18:22:29Z","timestamp":1475173349000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7487172\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,5]]},"references-count":32,"URL":"https:\/\/doi.org\/10.1109\/icra.2016.7487172","relation":{},"subject":[],"published":{"date-parts":[[2016,5]]}}}