{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:02:04Z","timestamp":1750309324464,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,2]],"date-time":"2024-08-02T00:00:00Z","timestamp":1722556800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,2]]},"DOI":"10.1145\/3696271.3696296","type":"proceedings-article","created":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T10:47:56Z","timestamp":1733136476000},"page":"151-156","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Entropy-Guided Exploration in AlphaZero: Enhancing MCTS with Information Gain for Strategic Decision Making"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9909-5449","authenticated-orcid":false,"given":"Shuqin","family":"Li","sequence":"first","affiliation":[{"name":"School of Computer Science, Hangzhou Dianzi University Information Engineering College, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-8731-006X","authenticated-orcid":false,"given":"Xueli","family":"Jia","sequence":"additional","affiliation":[{"name":"School of Computer Science, Hangzhou Dianzi University Information Engineering College, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-4639-9449","authenticated-orcid":false,"given":"Yadong","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Computer Science, Hangzhou Dianzi University Information Engineering College, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9887-643X","authenticated-orcid":false,"given":"Dong","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Computer Science, Hangzhou Dianzi University Information Engineering College, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-0038-0771","authenticated-orcid":false,"given":"Haiping","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science, Hangzhou Dianzi University Information Engineering College, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8843-4318","authenticated-orcid":false,"given":"Miao","family":"Hu","sequence":"additional","affiliation":[{"name":"School of Communication Engineering, Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,12,2]]},"reference":[{"doi-asserted-by":"publisher","key":"e_1_3_3_1_1_2","DOI":"10.1007\/s10462-022-10228-y"},{"key":"e_1_3_3_1_2_2","volume-title":"Berlin","author":"Coulom R\u00e9mi","year":"2006","unstructured":"Coulom, R\u00e9mi. \"Efficient selectivity and backup operators in Monte-Carlo tree search.\" International conference on computers and games. Berlin, Heidelberg: Springer Berlin Heidelberg, 2006."},{"key":"e_1_3_3_1_3_2","volume-title":"Vime: Variational information maximizing exploration.\"\u00a0Advances in neural information processing systems\u00a029","author":"Houthooft Rein","year":"2016","unstructured":"Houthooft, Rein, et al. \"Vime: Variational information maximizing exploration.\"\u00a0Advances in neural information processing systems\u00a029 (2016)."},{"key":"e_1_3_3_1_4_2","volume-title":"Berlin","author":"Kocsis Levente","year":"2006","unstructured":"Kocsis, Levente, and Csaba Szepesv\u00e1ri. \"Bandit based monte-carlo planning.\" European conference on machine learning. Berlin, Heidelberg: Springer Berlin Heidelberg, 2006."},{"doi-asserted-by":"crossref","unstructured":"Silver David et al. \"Mastering the game of go without human knowledge.\" nature 550.7676 (2017): 354-359.","key":"e_1_3_3_1_5_2","DOI":"10.1038\/nature24270"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_6_2","DOI":"10.1038\/s41586-020-03051-4"},{"key":"e_1_3_3_1_7_2","first-page":"36","article-title":"LightZero: A Unified Benchmark for Monte Carlo Tree Search in General Sequential Decision Scenarios","author":"Niu Yazhe","year":"2024","unstructured":"Niu, Yazhe, et al. \"LightZero: A Unified Benchmark for Monte Carlo Tree Search in General Sequential Decision Scenarios.\" Advances in Neural Information Processing Systems 36 (2024).","journal-title":"Advances in Neural Information Processing Systems"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_8_2","DOI":"10.1007\/s10462-022-10228-y"},{"key":"e_1_3_3_1_9_2","volume-title":"\"Reinforcement learning: An introduction.\" MIT press","author":"Sutton Richard S","year":"2018","unstructured":"Sutton, Richard S., and Andrew G. Barto. \"Reinforcement learning: An introduction.\" MIT press, 2018."},{"doi-asserted-by":"crossref","unstructured":"Gelly Sylvain and David Silver. \"Combining online and offline knowledge in UCT.\" Proceedings of the 24th international conference on machine learning. 2007.","key":"e_1_3_3_1_10_2","DOI":"10.1145\/1273496.1273531"},{"volume-title":"Monte-carlo tree search: A new framework for game ai.\" Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","author":"Chaslot Guillaume","unstructured":"Chaslot, Guillaume, et al. \"Monte-carlo tree search: A new framework for game ai.\" Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment. Vol. 4. No. 1. 2008.","key":"e_1_3_3_1_11_2"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_12_2","DOI":"10.1109\/TCIAIG.2012.2186810"},{"key":"e_1_3_3_1_13_2","volume-title":"A unified game-theoretic approach to multiagent reinforcement learning.\" Advances in Neural Information Processing Systems","author":"Lanctot Marc","year":"2017","unstructured":"Lanctot, Marc, et al. \"A unified game-theoretic approach to multiagent reinforcement learning.\" Advances in Neural Information Processing Systems. 2017."},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_14_2","DOI":"10.1038\/nature14236"},{"key":"e_1_3_3_1_15_2","volume-title":"\"Self-Improvement for Neural Combinatorial Optimization: Sample without Replacement, but Improvement.\" arXiv preprint arXiv:2403.15180","author":"Pirnay Jonathan","year":"2024","unstructured":"Pirnay, Jonathan, and Dominik G. Grimm. \"Self-Improvement for Neural Combinatorial Optimization: Sample without Replacement, but Improvement.\" arXiv preprint arXiv:2403.15180 (2024)."},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_16_2","DOI":"10.3233\/JIFS-179052"},{"key":"e_1_3_3_1_17_2","first-page":"26","article-title":"Convergence of Monte Carlo tree search in simultaneous move games","author":"Lisy Viliam","year":"2013","unstructured":"Lisy, Viliam, et al. \"Convergence of Monte Carlo tree search in simultaneous move games.\" Advances in Neural Information Processing Systems 26 (2013).","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_3_1_18_2","volume-title":"Monte Carlo tree search for games with hidden information and uncertainty. Diss","author":"Whitehouse Daniel","year":"2014","unstructured":"Whitehouse, Daniel. Monte Carlo tree search for games with hidden information and uncertainty. Diss. University of York, 2014."},{"key":"e_1_3_3_1_19_2","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence.","volume":"38","author":"Phan Thomy","year":"2024","unstructured":"Phan, Thomy, et al. \"Adaptive Anytime Multi-Agent Path Finding Using Bandit-Based Large Neighborhood Search.\" Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 38. No. 16. 2024."},{"key":"e_1_3_3_1_20_2","volume-title":"Language agent tree search unifies reasoning acting and planning in language models.\" arXiv preprint arXiv:2310.04406","author":"Zhou Andy","year":"2023","unstructured":"Zhou, Andy, et al. \"Language agent tree search unifies reasoning acting and planning in language models.\" arXiv preprint arXiv:2310.04406 (2023)."},{"key":"e_1_3_3_1_21_2","volume-title":"Bayesian active learning for classification and preference learning.\" arXiv preprint arXiv:1112.5745","author":"Houlsby Neil","year":"2011","unstructured":"Houlsby, Neil, et al. \"Bayesian active learning for classification and preference learning.\" arXiv preprint arXiv:1112.5745 (2011)."},{"unstructured":"Hern\u00e1ndez-Lobato Jos\u00e9 Miguel Matthew W. Hoffman and Zoubin Ghahramani. \"Predictive entropy search for efficient global optimization of black-box functions.\" Advances in neural information processing systems 27 (2014).","key":"e_1_3_3_1_22_2"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_23_2","DOI":"10.1002\/j.1538-7305.1948.tb01338.x"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_24_2","DOI":"10.1109\/5.726791"},{"key":"e_1_3_3_1_25_2","volume-title":"Deep residual learning for image recognition.\" Proceedings of the IEEE conference on computer vision and pattern recognition","author":"He Kaiming","year":"2016","unstructured":"He, Kaiming, et al. \"Deep residual learning for image recognition.\" Proceedings of the IEEE conference on computer vision and pattern recognition. 2016."},{"key":"e_1_3_3_1_26_2","volume-title":"IEEE","author":"Matsuzaki Kiminori","year":"2018","unstructured":"Matsuzaki, Kiminori. \"Empirical analysis of PUCT algorithm with evaluation functions of different quality.\" 2018 Conference on Technologies and Applications of Artificial Intelligence (TAAI). IEEE, 2018."}],"event":{"acronym":"MLMI 2024","name":"MLMI 2024: 2024 The 7th International Conference on Machine Learning and Machine Intelligence (MLMI)","location":"Osaka Japan"},"container-title":["Proceedings of the 2024 7th International Conference on Machine Learning and Machine Intelligence (MLMI)"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696271.3696296","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696271.3696296","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:04:01Z","timestamp":1750291441000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696271.3696296"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,2]]},"references-count":26,"alternative-id":["10.1145\/3696271.3696296","10.1145\/3696271"],"URL":"https:\/\/doi.org\/10.1145\/3696271.3696296","relation":{},"subject":[],"published":{"date-parts":[[2024,8,2]]},"assertion":[{"value":"2024-12-02","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}