{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T02:27:28Z","timestamp":1730255248760,"version":"3.28.0"},"reference-count":39,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,5]]},"DOI":"10.1109\/icra.2019.8793650","type":"proceedings-article","created":{"date-parts":[[2019,8,13]],"date-time":"2019-08-13T01:26:12Z","timestamp":1565659572000},"page":"3210-3216","source":"Crossref","is-referenced-by-count":0,"title":["Adaptive Variance for Changing Sparse-Reward Environments"],"prefix":"10.1109","author":[{"given":"Xingyu","family":"Lin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pengsheng","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Carlos","family":"Florensa","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David","family":"Held","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-28645-5_29"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ACC.2012.6315022"},{"key":"ref32","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","author":"mnih","year":"2016","journal-title":"International Conference on Machine Learning"},{"key":"ref31","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"2015","journal-title":"International Conference on Machine Learning"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(90)90056-Q"},{"key":"ref37","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2014","journal-title":"arXiv preprint arXiv 1412 6980"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.2307\/2684423"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1137\/S0363012992237273"},{"key":"ref10","first-page":"4026","article-title":"Deep exploration via bootstrapped DQN","author":"osband","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref11","article-title":"Reinforcement learning with deep Energy-Based policies","author":"haarnoja","year":"2017","journal-title":"International Conference on Machine Learning"},{"key":"ref12","article-title":"Taming the noise in reinforcement learning via soft updates","author":"fox","year":"2016","journal-title":"Conference on Uncertainty in Artificial Intelligence"},{"key":"ref13","article-title":"Bridging the gap between value and policy based reinforcement learning","author":"nachum","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref14","article-title":"#exploration: A study of Count-Based exploration for deep reinforcement learning","author":"tang","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref15","article-title":"Unifying Count-Based exploration and intrinsic motivation","author":"bellemare","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref16","article-title":"Curiosity-driven exploration in deep reinforcement learning via bayesian neural networks","author":"houthooft","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref17","article-title":"Surprise-Based intrinsic motivation for deep reinforcement learning","author":"achiam","year":"2017","journal-title":"ArXiv preprint ArXiv 1703 0173"},{"key":"ref18","article-title":"The uncertainty bellman equation and exploration","author":"o\u2019donoghue","year":"2017","journal-title":"arXiv preprint arXiv 1709 01922"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-16111-7_23"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-44565-X_12"},{"key":"ref4","article-title":"Concrete problems in AI safety","author":"amodei","year":"2016","journal-title":"arXiv preprint arXiv 1606 06565"},{"key":"ref27","article-title":"Continuous adaptation via Meta-Learning in nonstationary and competitive environments","author":"al-shedivat","year":"2018","journal-title":"International Conference on Learning Representations"},{"key":"ref3","article-title":"Learning complex dexterous manipulation with deep reinforcement learning and demonstrations","author":"rajeswaran","year":"2017","journal-title":"arXiv preprint arXiv 1709 10119"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/9780262017091.001.0001"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143872"},{"journal-title":"Dataset Shift in Machine Learning","year":"2009","author":"quionero-candela","key":"ref5"},{"key":"ref8","first-page":"2079","article-title":"Active learning with a drifting distribution","author":"yang","year":"2011","journal-title":"Advances in Neural Information Processing Systems 24"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/BF00116827"},{"key":"ref2","article-title":"Learning dexterous in-hand manipulation","author":"andrychowicz","year":"2018","journal-title":"arXiv preprint arXiv 1808 02194"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390228"},{"key":"ref1","article-title":"Proximal policy optimization algorithms","author":"schulman","year":"2017","journal-title":"arXiv preprint arXiv 1707 07816"},{"key":"ref20","first-page":"1","article-title":"Learning the variance of the reward-to-go","volume":"17","author":"tamar","year":"2016","journal-title":"Journal of Machine Learning Research"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2017.XIII.048"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2004.05.004"},{"key":"ref24","article-title":"EPOpt: Learning robust neural network policies using model ensembles","author":"rajeswaran","year":"2016","journal-title":"International Conference on Learning Representations"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8460528"},{"key":"ref26","article-title":"RL2: Fast reinforcement learning via slow reinforcement learning","author":"duan","year":"2016","journal-title":"arXiv preprint arXiv 1611 02779"},{"key":"ref25","article-title":"Model-Agnostic Meta-Learning for fast adaptation of deep networks","author":"finn","year":"2017","journal-title":"International Conference on Machine Learning"}],"event":{"name":"2019 International Conference on Robotics and Automation (ICRA)","start":{"date-parts":[[2019,5,20]]},"location":"Montreal, QC, Canada","end":{"date-parts":[[2019,5,24]]}},"container-title":["2019 International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8780387\/8793254\/08793650.pdf?arnumber=8793650","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,15]],"date-time":"2022-07-15T03:13:25Z","timestamp":1657854805000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8793650\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,5]]},"references-count":39,"URL":"https:\/\/doi.org\/10.1109\/icra.2019.8793650","relation":{},"subject":[],"published":{"date-parts":[[2019,5]]}}}