{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T15:06:33Z","timestamp":1771340793552,"version":"3.50.1"},"reference-count":44,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,10,23]],"date-time":"2022-10-23T00:00:00Z","timestamp":1666483200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,23]],"date-time":"2022-10-23T00:00:00Z","timestamp":1666483200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100002347","name":"BMBF","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002347","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100004807","name":"DFG","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100004807","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,10,23]]},"DOI":"10.1109\/iros47612.2022.9982061","type":"proceedings-article","created":{"date-parts":[[2022,12,26]],"date-time":"2022-12-26T19:38:15Z","timestamp":1672083495000},"page":"9355-9362","source":"Crossref","is-referenced-by-count":4,"title":["Active Exploration for Robotic Manipulation"],"prefix":"10.1109","author":[{"given":"Tim","family":"Schneider","sequence":"first","affiliation":[{"name":"Technical University of Darmstadt,Intelligent Autonomous Systems Lab,Darmstadt,Germany,64289"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Boris","family":"Belousov","sequence":"additional","affiliation":[{"name":"Technical University of Darmstadt,Intelligent Autonomous Systems Lab,Darmstadt,Germany,64289"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Georgia","family":"Chalvatzaki","sequence":"additional","affiliation":[{"name":"Technical University of Darmstadt,Intelligent Autonomous Systems Lab,Darmstadt,Germany,64289"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Diego","family":"Romeres","sequence":"additional","affiliation":[{"name":"Mitsubishi Electric Research Laboratories (MERL),Cambridge,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Devesh K.","family":"Jha","sequence":"additional","affiliation":[{"name":"Mitsubishi Electric Research Laboratories (MERL),Cambridge,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jan","family":"Peters","sequence":"additional","affiliation":[{"name":"Technical University of Darmstadt,Intelligent Autonomous Systems Lab,Darmstadt,Germany,64289"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.2307\/1412130"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1046\/j.1469-7580.2003.00144.x"},{"issue":"30","key":"ref3","article-title":"A review of robot learning for manipulation: Challenges, representations, and algorithms","volume":"22","author":"Kroemer","year":"2021","journal-title":"Journal of Machine Learning Research"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1080\/713754275"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/B978-012240530-3\/50013-4"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.1989.100078"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/s41315-019-00103-5"},{"key":"ref8","first-page":"651","article-title":"Scalable deep reinforcement learning for vision-based robotic manipulation","volume-title":"Conference on Robot Learning","author":"Kalashnikov"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/12441.001.0001"},{"key":"ref10","article-title":"When to trust your model: Model-based policy optimization","author":"Janner","year":"2019","journal-title":"arXiv preprint"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561298"},{"key":"ref12","article-title":"Dream to control: Learning behaviors by latent imagination","author":"Hafner","year":"2019","journal-title":"arXiv preprint"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"ref14","first-page":"5062","article-title":"Self-supervised exploration via disagreement","volume-title":"International conference on machine learning","author":"Pathak"},{"key":"ref15","first-page":"465","article-title":"Pilco: A model-based and data-efficient approach to policy search","volume-title":"Proceedings of the 28th International Conference on machine learning (ICML-11)","author":"Deisenroth"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/s10846-020-01183-3"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2019.2929996"},{"key":"ref18","first-page":"09081","article-title":"Temporal difference models: Model-free deep rl for model-based control","author":"Pong","year":"2018","journal-title":"arXiv preprint"},{"key":"ref19","first-page":"703","article-title":"Combining model-based and model-free updates for trajectory-centric reinforcement learning","volume-title":"International conference on machine learning","author":"Chebotar"},{"key":"ref20","article-title":"Deep rein-forcement learning in a handful of trials using probabilistic dynamics models","author":"Chua","year":"2018","journal-title":"arXiv preprint"},{"key":"ref21","first-page":"2555","article-title":"Learning latent dynamics for planning from pixels","volume-title":"International Conference on Machine Learning","author":"Hafner"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TAMD.2010.2051031"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2015.05.002"},{"key":"ref24","article-title":"Exploration in model-based reinforcement learning by empirically estimating learning progress","volume":"25","author":"Lopes","year":"2012","journal-title":"Advances in neural information processing systems"},{"key":"ref25","first-page":"5779","article-title":"Model-based active exploration","volume-title":"International conference on machine learning","author":"Shyam"},{"key":"ref26","first-page":"8583","article-title":"Planning to explore via self-supervised world models","volume-title":"International Conference on Machine Learning","author":"Sekar"},{"key":"ref27","article-title":"Curious iLQR: Resolving Uncertainty in Model-based RL","author":"Bechtle","year":"2019","journal-title":"arXiv"},{"key":"ref28","article-title":"Go-Explore: a New Approach for Hard-Exploration Problems","author":"Ecoffet","year":"2019","journal-title":"arXiv"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177728069"},{"issue":"320","key":"ref30","first-page":"201","article-title":"The im algorithm: A variational approach to in-formation maximization","volume":"16","author":"Agakov","year":"2004","journal-title":"Advances in neural information processing systems"},{"key":"ref31","article-title":"Variational bayesian optimal experimental design","author":"Foster","year":"2019","journal-title":"arXiv preprint"},{"key":"ref32","article-title":"Mine: Mutual information neural estimation","author":"Belghazi","year":"2018","journal-title":"arXiv preprint"},{"key":"ref33","first-page":"5171","article-title":"On variational bounds of mutual information","volume-title":"International Conference on Machine Learning","author":"Poole"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1613\/jair.614"},{"key":"ref35","article-title":"Re-inforcement learning through active inference","author":"Tschantz","year":"2020","journal-title":"arXiv preprint"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1016\/S0893-6080(00)00098-8"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/MCI.2022.3155327"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2007.915715"},{"key":"ref39","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"International conference on machine learning","author":"Haarnoja"},{"key":"ref40","article-title":"A Survey of Exploration Methods in Reinforcement Learning","author":"Amin","year":"2021","journal-title":"arXiv"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1007\/s00422-010-0364-z"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1080\/17588928.2015.1020053"},{"key":"ref43","article-title":"Openai gym","author":"Brockman","year":"2016","journal-title":"arXiv preprint"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1016\/j.simpa.2020.100022"}],"event":{"name":"2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","location":"Kyoto, Japan","start":{"date-parts":[[2022,10,23]]},"end":{"date-parts":[[2022,10,27]]}},"container-title":["2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9981026\/9981028\/09982061.pdf?arnumber=9982061","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T04:17:20Z","timestamp":1706761040000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9982061\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,23]]},"references-count":44,"URL":"https:\/\/doi.org\/10.1109\/iros47612.2022.9982061","relation":{},"subject":[],"published":{"date-parts":[[2022,10,23]]}}}