{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T05:26:05Z","timestamp":1730265965746,"version":"3.28.0"},"reference-count":37,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,7,18]]},"DOI":"10.1109\/ijcnn52387.2021.9533317","type":"proceedings-article","created":{"date-parts":[[2021,9,20]],"date-time":"2021-09-20T17:27:41Z","timestamp":1632158861000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["Structure and Randomness in Planning and Reinforcement Learning"],"prefix":"10.1109","author":[{"given":"Konrad","family":"Czechowski","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Piotr","family":"Januszewski","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Piotr","family":"Kozakowski","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lukasz","family":"Kucinski","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Piotr","family":"Milos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref33","article-title":"Learning model-based planning from scratch","author":"pascanu","year":"2017","journal-title":"CoRR vol abs\/1707 06170"},{"key":"ref32","article-title":"Continuous deep q-learning with model-based acceleration","author":"gu","year":"2016","journal-title":"ICML"},{"key":"ref31","article-title":"Model-based reinforcement learning for atari","author":"kaiser","year":"2019","journal-title":"CoRR vol abs\/1903 00374"},{"key":"ref30","article-title":"Treeqn and atreec: Differentiable tree planning for deep reinforcement learning","author":"farquhar","year":"2017","journal-title":"CoRR vol abs\/1710 11417"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1016\/S0925-7721(99)00017-6"},{"key":"ref36","article-title":"What matters in on-policy reinforcement learning? A large-scale empirical study","author":"andrychowicz","year":"2020","journal-title":"CoRR vol abs\/2006 05990"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/BF00993306"},{"key":"ref34","first-page":"2464","article-title":"An investigation of model-free planning","author":"guez","year":"2019","journal-title":"Proceedings of the 36th International Conference on Machine Learning ICML 2019"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-020-03051-4"},{"key":"ref11","article-title":"Combining q-learning and search with amortized value estimates","author":"hamrick","year":"2020","journal-title":"8th International Conference on Learning Representations ICLR 2020"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2016.7860448"},{"key":"ref13","first-page":"72","article-title":"Efficient selectivity and backup operators in monte-carlo tree search","volume":"4630","author":"coulom","year":"2006","journal-title":"Computers and Games 5th International Conference CG 2006"},{"journal-title":"Introduction to Algorithms","year":"2009","author":"cormen","key":"ref14"},{"journal-title":"Artificial Intelligence A Modern Approach","year":"2002","author":"russell","key":"ref15"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TSSC.1968.300136"},{"key":"ref17","first-page":"235","article-title":"Experiments with the graph traverser program","volume":"294","author":"doran","year":"0","journal-title":"Proceedings of the Royal Society of London Series A Mathematical and Physical Sciences"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2012.2186810"},{"key":"ref19","first-page":"3205","article-title":"Single-agent policy tree search with guarantees","author":"orseau","year":"2018","journal-title":"Advances in Neural Information Processing Systems 31 Annual Conference on Neural Information Processing Systems 2018 NeurIPS 2018"},{"journal-title":"Model predictive path integral control using covariance variable importance sampling","year":"2015","author":"williams","key":"ref28"},{"key":"ref4","article-title":"Deep learning for real-time atari game play using offline monte-carlo tree search planning","author":"guo","year":"2014","journal-title":"NIPS"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487277"},{"journal-title":"Reinforcement Learning An Introduction","year":"2018","author":"sutton","key":"ref3"},{"key":"ref6","doi-asserted-by":"crossref","DOI":"10.1038\/nature24270","article-title":"Mastering the game of Go without human knowledge","author":"silver","year":"2017","journal-title":"Nature"},{"key":"ref29","article-title":"Value prediction network","author":"oh","year":"2017","journal-title":"NIPS"},{"key":"ref5","article-title":"Reinforcement and imitation learning via interactive no-regret learning","author":"ross","year":"2014","journal-title":"CoRR vol abs\/1406 5979"},{"key":"ref8","article-title":"Thinking fast and slow with deep learning and tree search","author":"anthony","year":"2017","journal-title":"NIPS"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6404"},{"key":"ref2","article-title":"Google research football: A novel reinforcement learning environment","author":"kurach","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref9","article-title":"Uncertainty-sensitive learning and planning with ensembles","author":"mi?o?","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref1","article-title":"Imagination-augmented agents for deep reinforcement learning","author":"racani\u00e8re","year":"2017","journal-title":"NIPS"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-019-0070-z"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1613\/jair.820"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/34.44404"},{"journal-title":"Model Predictive Control","year":"2013","author":"camacho","key":"ref24"},{"key":"ref23","article-title":"World-championship-caliber Scrabble","author":"sheppard","year":"2002","journal-title":"Artificial Intelligence"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8463189"},{"key":"ref25","first-page":"4754","article-title":"Deep reinforcement learning in a handful of trials using probabilistic dynamics models","author":"chua","year":"2018","journal-title":"Advances in neural information processing systems"}],"event":{"name":"2021 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2021,7,18]]},"location":"Shenzhen, China","end":{"date-parts":[[2021,7,22]]}},"container-title":["2021 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9533266\/9533267\/09533317.pdf?arnumber=9533317","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T11:45:51Z","timestamp":1652183151000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9533317\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,18]]},"references-count":37,"URL":"https:\/\/doi.org\/10.1109\/ijcnn52387.2021.9533317","relation":{},"subject":[],"published":{"date-parts":[[2021,7,18]]}}}