{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,23]],"date-time":"2024-10-23T05:23:52Z","timestamp":1729661032640,"version":"3.28.0"},"reference-count":21,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2011,4]]},"DOI":"10.1109\/adprl.2011.5967364","type":"proceedings-article","created":{"date-parts":[[2011,8,3]],"date-time":"2011-08-03T21:40:00Z","timestamp":1312407600000},"page":"40-47","source":"Crossref","is-referenced-by-count":1,"title":["Active exploration by searching for experiments that falsify the computed control policy"],"prefix":"10.1109","author":[{"given":"Raphael","family":"Fonteneau","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Susan A.","family":"Murphy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Louis","family":"Wehenkel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Damien","family":"Ernst","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"journal-title":"Towards min max generalization in reinforcement learning","year":"2010","author":"fonteneau","key":"ref10"},{"key":"ref11","article-title":"Generating informative trajectories by using bounds on the return of control policies","author":"fonteneau","year":"2010","journal-title":"Proceedings of the Workshop on Active Learning and Experimental Design 2010 (in conjunction with AISTATS 2010)"},{"journal-title":"Theory of Financial Decision Making","year":"1987","author":"ingersoll","key":"ref12"},{"key":"ref13","doi-asserted-by":"crossref","DOI":"10.7551\/mitpress\/4168.001.0001","author":"kaelbling","year":"1993","journal-title":"Learning in embedded systems"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1023\/A:1017992615625"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1111\/1467-9868.00389"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1002\/sim.2022"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1023\/A:1017928328829"},{"key":"ref18","article-title":"Optimal sample selection for batch-mode reinforcement learning","author":"rachelson","year":"2011","journal-title":"International Conference on Agents and Artificial Intelligence (ICAART)"},{"key":"ref19","first-page":"317","article-title":"Neural fitted Q iteration - first experiences with a data efficient neural reinforcement learning method","author":"riedmiller","year":"2005","journal-title":"Proceedings of the Sixteenth European Conference on Machine Learning (ECML'05)"},{"key":"ref4","doi-asserted-by":"crossref","DOI":"10.1201\/9781439821091","author":"busoniu","year":"2010","journal-title":"Reinforcement Learning and Dynamic Programming Using Function Approximators"},{"key":"ref3","first-page":"201","article-title":"Online optimization in X-armed bandits","author":"bubeck","year":"2009","journal-title":"Advances in Neural Information Processing Systems 21"},{"key":"ref6","article-title":"Active reinforcement learning","volume":"307","author":"ephsteyn","year":"2008","journal-title":"Proceedings of the 25th International Conference on Machine Learning (ICML 2008)"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1098\/rstb.2007.2098"},{"key":"ref8","first-page":"503","article-title":"Tree-based batch mode reinforcement learning","volume":"6","author":"ernst","year":"2005","journal-title":"Journal of Machine Learning Research"},{"key":"ref7","article-title":"Selecting concise sets of samples for a reinforcement learning agent","author":"ernst","year":"2005","journal-title":"Proceedings of the third International Conference on Computational Intelligence Robotics and Autonomous Systems (CIRAS 2005)"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/116873.116880"},{"key":"ref1","first-page":"397","article-title":"Using confidence bounds for exploitation-exploration trade-offs","volume":"3","author":"auer","year":"2003","journal-title":"Journal of Machine Learning Research"},{"key":"ref9","article-title":"Voronoi model learning for batch mode reinforcement learning","author":"fonteneau","year":"2010","journal-title":"Technical report University of Li&#x00E8;ge"},{"journal-title":"Reinforcement Learning","year":"1998","author":"sutton","key":"ref20"},{"key":"ref21","article-title":"The role of exploration in learning control","author":"thrun","year":"1992","journal-title":"Handbook for Intelligent Control Neural Fuzzy and Adaptive Approaches"}],"event":{"name":"2011 Ieee Symposium On Adaptive Dynamic Programming And Reinforcement Learning","start":{"date-parts":[[2011,4,11]]},"location":"Paris, France","end":{"date-parts":[[2011,4,15]]}},"container-title":["2011 IEEE Symposium on Adaptive Dynamic Programming and Reinforcement Learning (ADPRL)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx5\/5958170\/5967347\/05967364.pdf?arnumber=5967364","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,6,13]],"date-time":"2019-06-13T14:53:37Z","timestamp":1560437617000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/5967364\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2011,4]]},"references-count":21,"URL":"https:\/\/doi.org\/10.1109\/adprl.2011.5967364","relation":{},"subject":[],"published":{"date-parts":[[2011,4]]}}}