{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:28:25Z","timestamp":1763191705046,"version":"3.45.0"},"reference-count":33,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/ijcnn64981.2025.11227788","type":"proceedings-article","created":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T18:46:15Z","timestamp":1763145975000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["A2CHoldem: An Intelligent Agent for Texas Hold\u2019em Based on Deep Reinforcement Learning"],"prefix":"10.1109","author":[{"given":"Qilong","family":"Guo","sequence":"first","affiliation":[{"name":"Shenyang Aerospace University,College of Computer Science,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yajie","family":"Wang","sequence":"additional","affiliation":[{"name":"Shenyang Aerospace University,College of Computer Science,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shipeng","family":"Wang","sequence":"additional","affiliation":[{"name":"Shenyang Aerospace University,College of Computer Science,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tianyu","family":"Zhao","sequence":"additional","affiliation":[{"name":"Shenyang Aerospace University,College of Computer Science,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Polynomial-time linear-swap regret minimization in imperfect-information sequential games","volume":"36","author":"Farina","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2743240"},{"key":"ref3","first-page":"507","article-title":"Agent57: Outperforming the atari human benchmark","volume-title":"International conference on machine learning","author":"Badia"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-020-03051-4"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN55064.2022.9892417"},{"key":"ref8","first-page":"1951","article-title":"Learning rewards to optimize global performance metrics in deep reinforcement learning","volume-title":"Proceedings of the 2023 International Conference on Autonomous Agents and Multiagent Systems","author":"Qian"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3236361"},{"key":"ref10","article-title":"Regret minimization in games with incomplete information","volume":"20","author":"Zinkevich","year":"2007","journal-title":"Advances in neural information processing systems"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1126\/science.aao1733"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1126\/science.aam6960"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i4.20394"},{"article-title":"On bonus based exploration methods in the arcade learning environment","volume-title":"International Conference on Learning Representations","author":"Taiga","key":"ref14"},{"key":"ref15","first-page":"1","article-title":"Reinforcement learning through asynchronous advantage actor-critic on a gpu","volume-title":"ICLR: proceedings","author":"Babaeizadeh"},{"article-title":"Exploration by random network distillation","year":"2018","author":"Burda","key":"ref16"},{"key":"ref17","article-title":"Monte carlo sampling for regret minimization in extensive games","volume":"22","author":"Lanctot","year":"2009","journal-title":"Advances in neural information processing systems"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1016\/s0927-0507(03)10006-0"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1126\/science.1259433"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v28i1.8810"},{"key":"ref21","article-title":"Safe and nested subgame solving for imperfect-information games","volume":"30","author":"Brown","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1126\/science.aay2400"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10056"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33011829"},{"article-title":"Reduced space and faster convergence in imperfect-information games via regret-based pruning","volume-title":"Workshops at the thirty-first AAAI conference on artificial intelligence","author":"Brown","key":"ref25"},{"key":"ref26","first-page":"17 057","article-title":"Combining deep reinforcement learning and search for imperfect-information games","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Deep reinforcement learning from self-play in imperfect-information games","year":"2016","author":"Heinrich","key":"ref27"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10013"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1126\/sciadv.adg3256"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00041"},{"article-title":"Adam: A method for stochastic optimization","year":"2014","author":"Kingma","key":"ref33"}],"event":{"name":"2025 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2025,6,30]]},"location":"Rome, Italy","end":{"date-parts":[[2025,7,5]]}},"container-title":["2025 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11227166\/11227148\/11227788.pdf?arnumber=11227788","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:25:41Z","timestamp":1763191541000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11227788\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":33,"URL":"https:\/\/doi.org\/10.1109\/ijcnn64981.2025.11227788","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}