{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T15:10:21Z","timestamp":1779203421565,"version":"3.51.4"},"reference-count":46,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2023,12,1]],"date-time":"2023-12-01T00:00:00Z","timestamp":1701388800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,12,1]],"date-time":"2023-12-01T00:00:00Z","timestamp":1701388800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,12,1]],"date-time":"2023-12-01T00:00:00Z","timestamp":1701388800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Chongqing Key Laboratory of Mobile Communication Technology","award":["cqupt-mct-202101"],"award-info":[{"award-number":["cqupt-mct-202101"]}]},{"DOI":"10.13039\/100018915","name":"Shanghai Automotive Industry Science and Technology Development Foundation","doi-asserted-by":"publisher","award":["2207"],"award-info":[{"award-number":["2207"]}],"id":[{"id":"10.13039\/100018915","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1109\/tai.2023.3240674","type":"journal-article","created":{"date-parts":[[2023,1,31]],"date-time":"2023-01-31T14:42:44Z","timestamp":1675176164000},"page":"1449-1460","source":"Crossref","is-referenced-by-count":2,"title":["Learning to Play <i>Koi-Koi<\/i> Hanafuda Card Games With Transformers"],"prefix":"10.1109","volume":"4","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4574-5682","authenticated-orcid":false,"given":"Sanghai","family":"Guan","sequence":"first","affiliation":[{"name":"iFLYTEK Research, iFLYTEK Co., Ltd., Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3170-8952","authenticated-orcid":false,"given":"Jingjing","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Cyber Science and Technology, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1210-546X","authenticated-orcid":false,"given":"Ruijie","family":"Zhu","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence, Zhengzhou University, Zhengzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8017-725X","authenticated-orcid":false,"given":"Junhui","family":"Qian","sequence":"additional","affiliation":[{"name":"School of Microelectronic and Communication Engineering, Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0584-5566","authenticated-orcid":false,"given":"Zhongxiang","family":"Wei","sequence":"additional","affiliation":[{"name":"College of Electronic and Information Engineering, Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"key":"ref2","first-page":"1","article-title":"Playing atari with deep reinforcement learning","volume-title":"Proc. NIPS Deep Learn. Workshop","author":"Mnih","year":"2013"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2743240"},{"key":"ref5","first-page":"387","article-title":"Deterministic policy gradient algorithms","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Silver","year":"2014"},{"key":"ref6","first-page":"1995","article-title":"Dueling network architectures for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wang","year":"2016"},{"key":"ref7","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mnih","year":"2016"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1126\/science.aam6960"},{"key":"ref9","article-title":"Suphx: Mastering mahjong with deep reinforcement learning","author":"Li"},{"key":"ref10","first-page":"12333","article-title":"DouZero: Mastering DouDizhu with self-play deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zha","year":"2021"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2021.3133846"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00349"},{"key":"ref15","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3152247"},{"key":"ref17","first-page":"1","article-title":"An image is worth 16  16 words: Transformers for image recognition at scale","volume-title":"Proc. 9th Int. Conf. Learn. Representations","author":"Dosovitskiy","year":"2021"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.23919\/mipro.2018.8400040"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00084"},{"key":"ref21","volume-title":"Game Theory","author":"Fudenberg","year":"1991"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/764"},{"key":"ref23","first-page":"1","article-title":"Regret minimization in games with incomplete information","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"20","author":"Zinkevich","year":"2007"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1126\/science.aao1733"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1126\/science.aay2400"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2019.103216"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TG.2020.3009359"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2017.2743347"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2018.8490368"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CoG52621.2021.9619134"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6404"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2012.2186810"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TG.2018.2866036"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.10.013"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/SII.2017.8279334"},{"issue":"2","key":"ref38","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1023\/A:1013689704352","article-title":"Finite-time analysis of the multiarmed bandit problem","volume":"47","author":"Auer","year":"2002","journal-title":"Mach. Learn."},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/2695664.2695734"},{"key":"ref40","first-page":"64","article-title":"Applying policy gradient method and neural fitted Q iteration for hanafuda Koi-Koi game player","volume-title":"Proc. 22nd Game Program. Workshop. Inform. Process. Soc. Japan","author":"Sato","year":"2017"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref42","first-page":"10524","article-title":"On layer normalization in the transformer architecture","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Xiong","year":"2020"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/tnn.1998.712192"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2021.3111142"},{"key":"ref45","first-page":"565","article-title":"Reward shaping in episodic reinforcement learning","volume-title":"Proc. 16th Conf. Auton. Agents MultiAgent Syst.","author":"Grze","year":"2017"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9078688\/10330793\/10032777.pdf?arnumber=10032777","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,20]],"date-time":"2025-10-20T17:59:43Z","timestamp":1760983183000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10032777\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12]]},"references-count":46,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/tai.2023.3240674","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,12]]}}}