{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,14]],"date-time":"2026-08-14T16:17:47Z","timestamp":1786724267057,"version":"3.56.0"},"reference-count":44,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"7","license":[{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U20B2070"],"award-info":[{"award-number":["U20B2070"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61976199"],"award-info":[{"award-number":["61976199"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1109\/tnnls.2024.3454477","type":"journal-article","created":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T13:29:10Z","timestamp":1727789350000},"page":"12389-12399","source":"Crossref","is-referenced-by-count":2,"title":["COPSRO: An Offline Empirical Game Theoretic Method With Conservative Critic"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0025-8306","authenticated-orcid":false,"given":"Zhengdao","family":"Shao","sequence":"first","affiliation":[{"name":"School of Information Science and Technology, University of Science and Technology of China (USTC), Hefei, Anhui, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4345-856X","authenticated-orcid":false,"given":"Liansheng","family":"Zhuang","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, University of Science and Technology of China (USTC), Hefei, Anhui, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2188-3028","authenticated-orcid":false,"given":"Houqiang","family":"Li","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, University of Science and Technology of China (USTC), Hefei, Anhui, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shafei","family":"Wang","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Empirical game-theoretic analysis for mean field games","volume-title":"arXiv:2112.00900","author":"Wang","year":"2021"},{"key":"ref2","first-page":"4933","article-title":"Offline RL without off-policy evaluation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brandfonbrener"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/tit.2022.3185139"},{"key":"ref4","first-page":"1","article-title":"Offline reinforcement learning with implicit Q-learning","volume-title":"Int. Conf. Learn. Represent.","author":"Kostrikov"},{"key":"ref5","article-title":"Conservative Q-learning for offline reinforcement learning","volume-title":"arXiv:2006.04779","author":"Kumar","year":"2020"},{"key":"ref6","article-title":"When should we prefer offline reinforcement learning over behavioral cloning?","volume-title":"arXiv:2204.05618","author":"Kumar","year":"2022"},{"key":"ref7","article-title":"Model-based reinforcement learning for offline zero-sum Markov games","volume-title":"arXiv:2206.04044","author":"Yan","year":"2022"},{"key":"ref8","first-page":"1","article-title":"When are offline two-player zero-sum Markov games solvable?","volume-title":"Proc. 36th Int. Conf. Neural Inf. Process. Syst","author":"Cui"},{"key":"ref9","first-page":"28954","article-title":"Combo: Conservative offline model-based policy optimization","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Yu"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/1461928.1461951"},{"key":"ref11","article-title":"Evaluating strategy exploration in empirical game-theoretic analysis","volume-title":"arXiv:2105.10423","author":"Wang","year":"2021"},{"key":"ref12","first-page":"536","article-title":"Planning in the presence of cost functions controlled by an adversary","volume-title":"Proc. 20th Int. Conf. Mach. Learn. (ICML)","author":"McMahan"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.ejor.2017.04.058"},{"key":"ref14","article-title":"Online double Oracle","volume-title":"arXiv:2103.07780","author":"Dinh","year":"2021"},{"key":"ref15","article-title":"A generalized training approach for multiagent learning","volume-title":"arXiv:1909.12823","author":"M\u00fcller","year":"2020"},{"key":"ref16","article-title":"Neural auto-curricula","volume-title":"arXiv:2106.02745","author":"Feng","year":"2021"},{"key":"ref17","article-title":"Anytime PSRO for two-player zero-sum games","volume-title":"arXiv:2201.07700","author":"McAleer","year":"2022"},{"key":"ref18","article-title":"Open-ended learning in symmetric zero-sum games","volume-title":"arXiv:1901.08106","author":"Balduzzi","year":"2019"},{"key":"ref19","article-title":"Pipeline PSRO: A scalable approach for finding approximate nash equilibria in large games","volume-title":"arXiv:2006.08555","author":"McAleer","year":"2021"},{"key":"ref20","first-page":"1","article-title":"Policy space diversity for non-transitive games","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yao"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref22","article-title":"Dota 2 with large scale deep reinforcement learning","volume-title":"arXiv:1912.06680","author":"Berner","year":"2019"},{"key":"ref23","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","volume-title":"arXiv:2005.01643","author":"Levine","year":"2020"},{"key":"ref24","first-page":"1","article-title":"Stabilizing off-policy Q-learning via bootstrapping error reduction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Kumar"},{"key":"ref25","first-page":"2052","article-title":"Off-policy deep reinforcement learning without exploration","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref26","article-title":"Offline reinforcement learning with value-based episodic memory","volume-title":"arXiv:2110.09796","author":"Ma","year":"2021"},{"key":"ref27","first-page":"1","article-title":"Pessimistic bootstrapping for uncertainty-driven offline reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Bai"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1007\/BF00122574"},{"key":"ref29","first-page":"5774","article-title":"Offline reinforcement learning with Fisher divergence critic regularization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kostrikov"},{"key":"ref30","first-page":"20132","article-title":"A minimalist approach to offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Fujimoto"},{"key":"ref31","first-page":"7436","article-title":"Uncertainty-based offline reinforcement learning with diversified Q-ensemble","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"An"},{"key":"ref32","article-title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","volume-title":"arXiv:1712.01815","author":"Silver","year":"2017"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i4.20394"},{"key":"ref34","first-page":"12333","article-title":"DouZero: Mastering DouDizhu with self-play deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zha"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1103\/physrevfluids.5.054401"},{"key":"ref36","article-title":"Survey of matrix completion algorithms","volume-title":"arXiv:2204.01532","author":"Jafarov","year":"2022"},{"key":"ref37","first-page":"4820","article-title":"Conformalized matrix completion","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Gui"},{"key":"ref38","first-page":"941","article-title":"Towards unifying behavioral and response diversity for open-ended learning in zero-sum games","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Liu"},{"key":"ref39","first-page":"8514","article-title":"Modelling behavioural diversity for learning in open-ended games","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Perez-Nieves"},{"key":"ref40","first-page":"17443","article-title":"Real world games look like spinning Tops","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Czarnecki"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/66"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/484"},{"key":"ref43","first-page":"1","article-title":"A unified game-theoretic approach to multiagent reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lanctot"},{"key":"ref44","first-page":"805","article-title":"Fictitious self-play in extensive-form games","volume-title":"Proc. 32nd Int. Conf. Mach. Learn.","author":"Heinrich"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/5962385\/11073756\/10701058.pdf?arnumber=10701058","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,9]],"date-time":"2025-07-09T23:20:30Z","timestamp":1752103230000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10701058\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7]]},"references-count":44,"journal-issue":{"issue":"7"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2024.3454477","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7]]}}}