{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T05:31:22Z","timestamp":1730266282304,"version":"3.28.0"},"reference-count":33,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T00:00:00Z","timestamp":1658102400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T00:00:00Z","timestamp":1658102400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,7,18]]},"DOI":"10.1109\/ijcnn55064.2022.9892556","type":"proceedings-article","created":{"date-parts":[[2022,9,30]],"date-time":"2022-09-30T19:56:04Z","timestamp":1664567764000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["Planning and Learning using Adaptive Entropy Tree Search"],"prefix":"10.1109","author":[{"given":"Piotr","family":"Kozakowski","sequence":"first","affiliation":[{"name":"University of Warsaw,Faculty of Mathematics, Informatics and Mechanics,Warsaw,Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mikolaj","family":"Pacek","sequence":"additional","affiliation":[{"name":"University of Warsaw,Faculty of Mathematics, Informatics and Mechanics,Warsaw,Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Piotr","family":"Milos","sequence":"additional","affiliation":[{"name":"Institute of Mathematics of the Polish Academy of Sciences,Warsaw,Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref33","first-page":"9","volume":"97","author":"ahmed","year":"0","journal-title":"Proceedings of the 36th International Conference on Machine Learning ser Proceedings of Machine Learning Research"},{"key":"ref32","article-title":"Bridging the gap between value and policy based reinforcement learning","volume":"30","author":"nachum","year":"2017","journal-title":"Advances in neural information processing systems"},{"journal-title":"Reinforcement Learning and Control as Probabilistic Inference Tutorial and Review","year":"2018","author":"levine","key":"ref31"},{"key":"ref30","first-page":"13","volume":"119","author":"badia","year":"0","journal-title":"Proceedings of the 37th International Conference on Machine Learning ser Proceedings of Machine Learning Research"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1609\/icaps.v21i1.13484"},{"key":"ref11","article-title":"Monte-carlo tree search in production management problems","author":"chaslot","year":"0","journal-title":"BelgiumNetherlands Conference on Artificial Intelligence"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-20525-5_10"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-020-03051-4"},{"key":"ref14","first-page":"13","article-title":"Monte-Carlo tree search as regularized policy optimization","volume":"119","author":"grill","year":"0","journal-title":"Proceedings of the 37th International Conference on Machine Learning ser Proceedings of Machine Learning Research"},{"key":"ref15","article-title":"Mastering atari games with limited data","author":"ye","year":"2021","journal-title":"NeurIPS"},{"journal-title":"Modeling Purposeful Adaptive Behavior with the Principle of Maximum Causal Entropy","year":"2018","author":"ziebart","key":"ref16"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1098\/rsif.2012.0758"},{"journal-title":"If MaxEnt RL is the Answer What is the Question?","year":"2019","author":"eysenbach","key":"ref18"},{"key":"ref19","first-page":"6","article-title":"Reinforcement learning with deep energy-based policies","volume":"70","author":"haarnoja","year":"2017","journal-title":"Proceedings of the 34th International Conference on Machine Learning ser Proceedings of Machine Learning Research"},{"key":"ref28","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6404"},{"journal-title":"Dopamine A research framework for deep reinforcement learning","year":"2018","author":"castro","key":"ref27"},{"key":"ref3","doi-asserted-by":"crossref","first-page":"405","DOI":"10.1016\/j.ejor.2020.07.063","volume":"290","author":"bengio","year":"2021","journal-title":"European Journal of Operational Research"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2012.2186810"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-019-0070-z"},{"key":"ref8","doi-asserted-by":"crossref","first-page":"73","DOI":"10.1109\/TCIAIG.2009.2018703","article-title":"The computational intelligence of mogo revealed in taiwan's computer go tournaments","volume":"1","author":"lee","year":"2009","journal-title":"Computational Intelligence and AI in Games IEEE Transactions on"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/11871842_29"},{"journal-title":"Artificial Intelligence A Modern Approach","year":"2009","author":"russell","key":"ref2"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/BTAS.2009.5339016"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuron.2017.06.011"},{"key":"ref20","first-page":"10","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume":"80","author":"haarnoja","year":"2018","journal-title":"Proceedings of the 35th International Conference on Machine Learning ser Proceedings of Machine Learning Research"},{"journal-title":"Convex regularization in monte-carlo tree search","year":"2020","author":"dam","key":"ref22"},{"key":"ref21","first-page":"9520","article-title":"Maximum entropy monte-carlo planning","volume":"32","author":"xiao","year":"2019","journal-title":"Advances in neural information processing systems"},{"year":"0","author":"jurafsky","key":"ref24"},{"key":"ref23","doi-asserted-by":"crossref","first-page":"253","DOI":"10.1613\/jair.3912","volume":"47","author":"bellemare","year":"2012","journal-title":"Journal of Artificial Intelligence Research vol"},{"journal-title":"Algorithms for Minimization Without Derivatives","year":"1973","author":"brent","key":"ref26"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2018.2800085"}],"event":{"name":"2022 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2022,7,18]]},"location":"Padua, Italy","end":{"date-parts":[[2022,7,23]]}},"container-title":["2022 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9891857\/9889787\/09892556.pdf?arnumber=9892556","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,4]],"date-time":"2022-11-04T01:27:04Z","timestamp":1667525224000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9892556\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,7,18]]},"references-count":33,"URL":"https:\/\/doi.org\/10.1109\/ijcnn55064.2022.9892556","relation":{},"subject":[],"published":{"date-parts":[[2022,7,18]]}}}