{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,4]],"date-time":"2025-11-04T23:53:39Z","timestamp":1762300419086,"version":"3.37.3"},"reference-count":62,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,8,11]],"date-time":"2023-08-11T00:00:00Z","timestamp":1691712000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,8,11]],"date-time":"2023-08-11T00:00:00Z","timestamp":1691712000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["Grant No.51935005"],"award-info":[{"award-number":["Grant No.51935005"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"China Academy of Launch Vehicle Technology","award":["CALT2022-18","CALT2022-18"],"award-info":[{"award-number":["CALT2022-18","CALT2022-18"]}]},{"name":"Basic Research Project","award":["Grant No.JCKY20200603C010"],"award-info":[{"award-number":["Grant No.JCKY20200603C010"]}]},{"DOI":"10.13039\/501100005046","name":"Natural Science Foundation of Heilongjiang Province of China","doi-asserted-by":"crossref","award":["Grant No.LH2021F023"],"award-info":[{"award-number":["Grant No.LH2021F023"]}],"id":[{"id":"10.13039\/501100005046","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Science and Technology Planning Project of Heilongjiang Province of China","award":["Grant No.GA21C031"],"award-info":[{"award-number":["Grant No.GA21C031"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Agent Multi-Agent Syst"],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1007\/s10458-023-09620-x","type":"journal-article","created":{"date-parts":[[2023,8,11]],"date-time":"2023-08-11T02:01:17Z","timestamp":1691719277000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Classifying ambiguous identities in hidden-role Stochastic games with multi-agent reinforcement learning"],"prefix":"10.1007","volume":"37","author":[{"given":"Shijie","family":"Han","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siyuan","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"An","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,8,11]]},"reference":[{"key":"9620_CR1","doi-asserted-by":"publisher","first-page":"387","DOI":"10.1007\/s10458-005-2631-2","volume":"11","author":"L Panait","year":"2005","unstructured":"Panait, L., & Luke, S. (2005). Cooperative multi-agent learning: The state of the art. Autonomous Agents and Multi-Agent Systems, 11, 387\u2013434.","journal-title":"Autonomous Agents and Multi-Agent Systems"},{"key":"9620_CR2","first-page":"8","volume":"5","author":"ZH Ismail","year":"2018","unstructured":"Ismail, Z. H., Sariff, N., & Hurtado, E. (2018). A survey and analysis of cooperative multi-agent robot systems: Challenges and directions. Applications of Mobile Robots, 5, 8\u201314.","journal-title":"Applications of Mobile Robots"},{"issue":"7857","key":"9620_CR3","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1038\/d41586-021-01170-0","volume":"593","author":"A Dafoe","year":"2021","unstructured":"Dafoe, A., Bachrach, Y., Hadfield, G., Horvitz, E., Larson, K., & Graepel, T. (2021). Cooperative ai: Machines must learn to find common ground. Nature, 593(7857), 33\u201336.","journal-title":"Nature"},{"key":"9620_CR4","doi-asserted-by":"crossref","unstructured":"Carta, S. (2022). Machine Learning and the City: Applications in Architecture and Urban Design, pp. 143\u2013166.","DOI":"10.1002\/9781119815075"},{"issue":"7587","key":"9620_CR5","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D., Huang, A., Maddison, C. J., Guez, A., Sifre, L., Van Den Driessche, G., Schrittwieser, J., Antonoglou, I., Panneershelvam, V., Lanctot, M., et al. (2016). Mastering the game of go with deep neural networks and tree search. Nature, 529(7587), 484\u2013489.","journal-title":"Nature"},{"issue":"6419","key":"9620_CR6","doi-asserted-by":"publisher","first-page":"1140","DOI":"10.1126\/science.aar6404","volume":"362","author":"D Silver","year":"2018","unstructured":"Silver, D., Hubert, T., Schrittwieser, J., Antonoglou, I., Lai, M., Guez, A., Lanctot, M., Sifre, L., Kumaran, D., Graepel, T., et al. (2018). A general reinforcement learning algorithm that masters chess, shogi, and go through self-play. Science, 362(6419), 1140\u20131144.","journal-title":"Science"},{"issue":"7839","key":"9620_CR7","doi-asserted-by":"publisher","first-page":"604","DOI":"10.1038\/s41586-020-03051-4","volume":"588","author":"J Schrittwieser","year":"2020","unstructured":"Schrittwieser, J., Antonoglou, I., Hubert, T., Simonyan, K., Sifre, L., Schmitt, S., Guez, A., Lockhart, E., Hassabis, D., Graepel, T., et al. (2020). Mastering atari, go, chess and shogi by planning with a learned model. Nature, 588(7839), 604\u2013609.","journal-title":"Nature"},{"key":"9620_CR8","doi-asserted-by":"crossref","unstructured":"Ye, D., Liu, Z., Sun, M., Shi, B., Zhao, P., Wu, H., Yu, H., Yang, S., Wu, X., Guo, Q., & et al. (2020). Mastering complex control in moba games with deep reinforcement learning. In Proceedings of the AAAI Conference on Artificial Intelligence (vol. 34, pp. 6672\u20136679).","DOI":"10.1609\/aaai.v34i04.6144"},{"key":"9620_CR9","unstructured":"Berner, C., Brockman, G., Chan, B., Cheung, V., Debiak, P., Dennison, C., Farhi, D., Fischer, Q., Hashme, S., & Hesse, C., et al. (2019). Dota 2 with large scale deep reinforcement learning. arXiv preprint arXiv:1912.06680."},{"issue":"6374","key":"9620_CR10","doi-asserted-by":"publisher","first-page":"418","DOI":"10.1126\/science.aao1733","volume":"359","author":"N Brown","year":"2018","unstructured":"Brown, N., & Sandholm, T. (2018). Superhuman ai for heads-up no-limit poker: Libratus beats top professionals. Science, 359(6374), 418\u2013424.","journal-title":"Science"},{"key":"9620_CR11","unstructured":"Li, J., Koyamada, S., Ye, Q., Liu, G., Wang, C., Yang, R., Zhao, L., Qin, T., Liu, T.-Y., & Hon, H.-W. (2020). Suphx: Mastering mahjong with deep reinforcement learning. arXiv preprint arXiv:2003.13590."},{"key":"9620_CR12","unstructured":"Zha, D., Xie, J., Ma, W., Zhang, S., Lian, X., Hu, X., & Liu, J. (2021). Douzero: Mastering doudizhu with self-play deep reinforcement learning. In International conference on machine learning (pp. 12333\u201312344). PMLR."},{"key":"9620_CR13","doi-asserted-by":"crossref","unstructured":"Kurach, K., Raichuk, A., Sta\u0144czyk, P., Zajac, M., Bachem, O., Espeholt, L., Riquelme, C., Vincent, D., Michalski, M., Bousquet, O., et al. (2020). Google research football: A novel reinforcement learning environment. In Proceedings of the AAAI conference on artificial intelligence (vol. 34, pp. 4501\u20134510).","DOI":"10.1609\/aaai.v34i04.5878"},{"key":"9620_CR14","first-page":"3991","volume":"34","author":"L Chenghao","year":"2021","unstructured":"Chenghao, L., Wang, T., Wu, C., Zhao, Q., Yang, J., & Zhang, C. (2021). Celebrating diversity in shared multi-agent reinforcement learning. Advances in Neural Information Processing Systems, 34, 3991\u20134002.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR15","doi-asserted-by":"publisher","first-page":"183","DOI":"10.1007\/978-3-642-14435-6_7","volume":"1","author":"L Bu\u015foniu","year":"2010","unstructured":"Bu\u015foniu, L., Babu\u0161ka, R., & Schutter, B. D. (2010). Multi-agent reinforcement learning: An overview. Innovations in Multi-Agent Systems and Applications, 1, 183\u2013221.","journal-title":"Innovations in Multi-Agent Systems and Applications"},{"key":"9620_CR16","unstructured":"Sunehag, P., Lever, G., Gruslys, A., Czarnecki, W.M., Zambaldi, V., Jaderberg, M., Lanctot, M., Sonnerat, N., Leibo, J.Z., & Tuyls, K., et al. (2017). Value-decomposition networks for cooperative multi-agent learning. arXiv preprint arXiv:1706.05296."},{"key":"9620_CR17","unstructured":"Rashid, T., Samvelyan, M., Schroeder, C., Farquhar, G., Foerster, J., & Whiteson, S. (2018). Qmix: Monotonic value function factorisation for deep multi-agent reinforcement learning. In International conference on machine learning (pp. 4295\u20134304). PMLR."},{"key":"9620_CR18","first-page":"558","volume":"32","author":"Y Du","year":"2019","unstructured":"Du, Y., Han, L., Fang, M., Liu, J., Dai, T., & Tao, D. (2019). Liir: Learning individual intrinsic reward in multi-agent reinforcement learning. Advances in Neural Information Processing Systems, 32, 558.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR19","unstructured":"Xiao, B., Ramasubramanian, B., & Poovendran, R. (2022). Agent-temporal attention for reward redistribution in episodic multi-agent reinforcement learning. arXiv preprint arXiv:2201.04612."},{"key":"9620_CR20","first-page":"12208","volume":"34","author":"B Peng","year":"2021","unstructured":"Peng, B., Rashid, T., Schroeder de Witt, C., Kamienny, P.-A., Torr, P., B\u00f6hmer, W., & Whiteson, S. (2021). Facmac: Factored multi-agent centralised policy gradients. Advances in Neural Information Processing Systems, 34, 12208\u201312221.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR21","first-page":"552","volume":"29","author":"J Foerster","year":"2016","unstructured":"Foerster, J., Assael, I. A., De Freitas, N., & Whiteson, S. (2016). Learning to communicate with deep multi-agent reinforcement learning. Advances in Neural Information Processing Systems, 29, 552.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR22","doi-asserted-by":"crossref","unstructured":"Peng, Z., Zhang, L., & Luo, T. (2018). Learning to communicate via supervised attentional message processing. In Proceedings of the 31st international conference on computer animation and social agents (pp. 11\u201316).","DOI":"10.1145\/3205326.3205346"},{"key":"9620_CR23","first-page":"15230","volume":"19","author":"T Lin","year":"2021","unstructured":"Lin, T., Huh, M., Stauffer, C., Lim, S. N., & Isola, P. (2021). Learning to ground multi-agent communication with autoencoders. Advances in Neural Information Processing Systems, 19, 15230\u201315242.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR24","unstructured":"Vanneste, S., Vanneste, A., Mets, K., Anwar, A., Mercelis, S., Latr\u00e9, S., & Hellinckx, P (2020) .Learning to communicate using counterfactual reasoning. arXiv preprint arXiv:2006.07200."},{"key":"9620_CR25","unstructured":"Heinrich, J., & Silver, D. (2016). Deep reinforcement learning from self-play in imperfect-information games. arXiv preprint arXiv:1603.01121."},{"issue":"7782","key":"9620_CR26","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals, O., Babuschkin, I., Czarnecki, W. M., Mathieu, M., Dudzik, A., Chung, J., Choi, D. H., Powell, R., Ewalds, T., Georgiev, P., et al. (2019). Grandmaster level in starcraft ii using multi-agent reinforcement learning. Nature, 575(7782), 350\u2013354.","journal-title":"Nature"},{"key":"9620_CR27","unstructured":"Foerster, J.N., Chen, R.Y., Al-Shedivat, M., Whiteson, S., Abbeel, P., & Mordatch, I. (2017). Learning with opponent-learning awareness. arXiv preprint arXiv:1709.04326."},{"key":"9620_CR28","first-page":"17987","volume":"33","author":"T Anthony","year":"2020","unstructured":"Anthony, T., Eccles, T., Tacchetti, A., Kram\u00e1r, J., Gemp, I., Hudson, T., Porcel, N., Lanctot, M., P\u00e9rolat, J., Everett, R., et al. (2020). Learning to play no-press diplomacy with best response policy iteration. Advances in Neural Information Processing Systems, 33, 17987\u201318003.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR29","first-page":"569","volume":"32","author":"P Paquette","year":"2019","unstructured":"Paquette, P., Lu, Y., Bocco, S. S., Smith, M., & O-G, S., Kummerfeld, J.K., Pineau, J., Singh, S., & Courville, A.C. (2019). No-press diplomacy: Modeling multi-agent gameplay. Advances in Neural Information Processing Systems, 32, 569.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR30","doi-asserted-by":"crossref","unstructured":"(FAIR)\u2020, M.F.A.R.D.T., Bakhtin, A., Brown, N., Dinan, E., Farina, G., Flaherty, C., Fried, D., Goff, A., Gray, J., & Hu, H., et al. (2022). Human-level play in the game of diplomacy by combining language models with strategic reasoning. Science, 378(6624), 1067\u20131074.","DOI":"10.1126\/science.ade9097"},{"key":"9620_CR31","first-page":"669","volume":"32","author":"J Serrino","year":"2019","unstructured":"Serrino, J., Kleiman-Weiner, M., Parkes, D. C., & Tenenbaum, J. (2019). Finding friend and foe in multi-agent games. Advances in Neural Information Processing Systems, 32, 669.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR32","doi-asserted-by":"crossref","unstructured":"Wang, T., & Kaneko, T. (2018). Application of deep reinforcement learning in werewolf game agents. In 2018 conference on technologies and applications of artificial intelligence (TAAI) (pp. 28\u201333). IEEE.","DOI":"10.1109\/TAAI.2018.00016"},{"key":"9620_CR33","unstructured":"Sutton, R. S., & Barto, A.G. (2018). Reinforcement Learning: An Introduction."},{"key":"9620_CR34","unstructured":"Yang, Y., Luo, R., Li, M., Zhou, M., Zhang, W., & Wang, J. (2018). Mean field multi-agent reinforcement learning. In International conference on machine learning (pp. 5571\u20135580). PMLR."},{"key":"9620_CR35","doi-asserted-by":"crossref","unstructured":"Wang, B., Xie, J., & Atanasov, N. (2022). Darl1n: Distributed multi-agent reinforcement learning with one-hop neighbors. arXiv preprint arXiv:2202.09019.","DOI":"10.1109\/IROS47612.2022.9981441"},{"key":"9620_CR36","first-page":"5689","volume":"30","author":"R Lowe","year":"2017","unstructured":"Lowe, R., Wu, Y. I., Tamar, A., Harb, J., Pieter Abbeel, O., & Mordatch, I. (2017). Multi-agent actor-critic for mixed cooperative-competitive environments. Advances in Neural Information Processing Systems, 30, 5689.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR37","unstructured":"P\u00e9rolat, J., Strub, F., Piot, B., & Pietquin, O. (2017). Learning nash equilibrium for general-sum markov games from batch data. In Artificial intelligence and statistics (pp. 232\u2013241). PMLR."},{"key":"9620_CR38","doi-asserted-by":"crossref","unstructured":"uz Zaman, M.A., Zhang, K., Miehling, E., & Ba\u015far, T. (2020). Approximate equilibrium computation for discrete-time linear-quadratic mean-field games. In 2020 American control conference (ACC) (pp. 333\u2013339). IEEE.","DOI":"10.23919\/ACC45564.2020.9147474"},{"key":"9620_CR39","unstructured":"Fu, Z., Yang, Z., Chen, Y., & Wang, Z. (2019). Actor-critic provably finds nash equilibria of linear-quadratic mean-field games. arXiv preprint arXiv:1910.07498."},{"key":"9620_CR40","unstructured":"Nair, A., Srinivasan, P., Blackwell, S., Alcicek, C., Fearon, R., De\u00a0Maria, A., Panneershelvam, V., Suleyman, M., Beattie, C., & Petersen, S. et al. (2015). Massively parallel methods for deep reinforcement learning. arXiv preprint arXiv:1507.04296."},{"key":"9620_CR41","unstructured":"Wang, T., Wang, J., Wu, Y., & Zhang, C. (2019). Influence-based multi-agent exploration. arXiv preprint arXiv:1910.05512."},{"key":"9620_CR42","unstructured":"Liu, I.-J., Jain, U., Yeh, R.A., & Schwing, A. (2021). Cooperative exploration for multi-agent deep reinforcement learning. In International conference on machine learning (pp. 6826\u20136836). PMLR."},{"key":"9620_CR43","doi-asserted-by":"crossref","unstructured":"Viseras, A., Wiedemann, T., Manss, C., Magel, L., Mueller, J., Shutin, D., & Merino, L. (2016). Decentralized multi-agent exploration with online-learning of gaussian processes. In 2016 IEEE international conference on robotics and automation (ICRA) (pp. 4222\u20134229). IEEE.","DOI":"10.1109\/ICRA.2016.7487617"},{"key":"9620_CR44","first-page":"556","volume":"29","author":"D Hadfield-Menell","year":"2016","unstructured":"Hadfield-Menell, D., Russell, S. J., Abbeel, P., & Dragan, A. (2016). Cooperative inverse reinforcement learning. Advances in Neural Information Processing Systems, 29, 556.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9620_CR45","unstructured":"Wu, H., Sequeira, P., & Pynadath, D. V. (2023). Multiagent inverse reinforcement learning via theory of mind reasoning. arXiv preprint arXiv:2302.10238."},{"key":"9620_CR46","unstructured":"He, H., Boyd-Graber, J., Kwok, K., & Daum\u00e9\u00a0III, H. (2016). Opponent modeling in deep reinforcement learning. In International conference on machine learning (pp. 1804\u20131813). PMLR."},{"key":"9620_CR47","doi-asserted-by":"publisher","first-page":"66","DOI":"10.1016\/j.artint.2018.01.002","volume":"258","author":"SV Albrecht","year":"2018","unstructured":"Albrecht, S. V., & Stone, P. (2018). Autonomous agents modelling other agents: A comprehensive survey and open problems. Artificial Intelligence, 258, 66\u201395.","journal-title":"Artificial Intelligence"},{"issue":"6623","key":"9620_CR48","doi-asserted-by":"publisher","first-page":"990","DOI":"10.1126\/science.add4679","volume":"378","author":"J Perolat","year":"2022","unstructured":"Perolat, J., De Vylder, B., Hennes, D., Tarassov, E., Strub, F., de Boer, V., Muller, P., Connor, J. T., Burch, N., Anthony, T., et al. (2022). Mastering the game of stratego with model-free multiagent reinforcement learning. Science, 378(6623), 990\u2013996.","journal-title":"Science"},{"key":"9620_CR49","unstructured":"Rabinowitz, N., Perbet, F., Song, F., Zhang, C., Eslami, S. A., & Botvinick, M. (2018). Machine theory of mind. In International conference on machine learning (pp. 4218\u20134227). PMLR."},{"issue":"7","key":"9620_CR50","doi-asserted-by":"publisher","first-page":"1057","DOI":"10.1017\/S0033291720000835","volume":"50","author":"F Cuzzolin","year":"2020","unstructured":"Cuzzolin, F., Morelli, A., Cirstea, B., & Sahakian, B. J. (2020). Knowing me, knowing you: Theory of mind in ai. Psychological Medicine, 50(7), 1057\u20131061.","journal-title":"Psychological Medicine"},{"key":"9620_CR51","doi-asserted-by":"crossref","unstructured":"Stone, P., Kaminka, G.A., Kraus, S., & Rosenschein, J.S. (2010). Ad hoc autonomous agent teams: Collaboration without pre-coordination. In Twenty-fourth AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v24i1.7529"},{"key":"9620_CR52","doi-asserted-by":"crossref","unstructured":"Mirsky, R., Carlucho, I., Rahman, A., Fosong, E., Macke, W., Sridharan, M., Stone, P., & Albrecht, S. V. (2022). A survey of ad hoc teamwork research. In European conference on multi-agent systems (pp. 275\u2013293). Springer.","DOI":"10.1007\/978-3-031-20614-6_16"},{"key":"9620_CR53","doi-asserted-by":"crossref","unstructured":"Barrett, S., & Stone, P. (2015). Cooperating with unknown teammates in complex domains: A robot soccer case study of ad hoc teamwork. In Twenty-ninth AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v29i1.9428"},{"key":"9620_CR54","doi-asserted-by":"crossref","unstructured":"Ravula, M., Alkoby, S., & Stone, P. (2019). Ad hoc teamwork with behavior switching agents. In Proceedings of the 28th international joint conference on artificial intelligence (pp. 550\u2013556).","DOI":"10.24963\/ijcai.2019\/78"},{"key":"9620_CR55","doi-asserted-by":"crossref","unstructured":"Chen, S., Andrejczuk, E., Cao, Z., & Zhang, J. (2020). Aateam: Achieving the ad hoc teamwork by employing the attention mechanism. In Proceedings of the AAAI conference on artificial intelligence (vol. 34, pp. 7095\u20137102).","DOI":"10.1609\/aaai.v34i05.6196"},{"key":"9620_CR56","unstructured":"Gu, P., Zhao, M., Hao, J., & An, B. (2021). Online ad hoc teamwork under partial observability. In International conference on learning representations."},{"key":"9620_CR57","unstructured":"Rahman, M.A., Hopner, N., Christianos, F., & Albrecht, S.V. (2021). Towards open ad hoc teamwork using graph-based policy learning. In International conference on machine learning (pp. 8776\u20138786). PMLR."},{"key":"9620_CR58","doi-asserted-by":"crossref","unstructured":"Zha, D., Lai, K.-H., Huang, S., Cao, Y., Reddy, K., Vargas, J., Nguyen, A., Wei, R., Guo, J., & Hu, X. (2021). Rlcard: a platform for reinforcement learning in card games. In Proceedings of the twenty-ninth international conference on international joint conferences on artificial intelligence (pp. 5264\u20135266).","DOI":"10.24963\/ijcai.2020\/764"},{"key":"9620_CR59","doi-asserted-by":"crossref","unstructured":"Jiang, Q., Li, K., Du, B., Chen, H., & Fang, H. (2019). Deltadou: Expert-level doudizhu ai through self-play. In IJCAI (pp. 1265\u20131271).","DOI":"10.24963\/ijcai.2019\/176"},{"key":"9620_CR60","unstructured":"You, Y., Li, L., Guo, B., Wang, W., & Lu, C. (2019). Combinational q-learning for dou di zhu. arXiv preprint arXiv:1901.08925."},{"key":"9620_CR61","unstructured":"Arnob, S.Y. (2020). Off-policy adversarial inverse reinforcement learning. arXiv preprint arXiv:2005.01138."},{"key":"9620_CR62","doi-asserted-by":"crossref","unstructured":"Singh, S., Soni, V., & Wellman, M. (2004). Computing approximate bayes-nash equilibria in tree-games of incomplete information. In Proceedings of the 5th ACM conference on electronic commerce (pp. 81\u201390).","DOI":"10.1145\/988772.988785"}],"container-title":["Autonomous Agents and Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-023-09620-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10458-023-09620-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-023-09620-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,18]],"date-time":"2023-12-18T04:54:34Z","timestamp":1702875274000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10458-023-09620-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,11]]},"references-count":62,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,12]]}},"alternative-id":["9620"],"URL":"https:\/\/doi.org\/10.1007\/s10458-023-09620-x","relation":{},"ISSN":["1387-2532","1573-7454"],"issn-type":[{"type":"print","value":"1387-2532"},{"type":"electronic","value":"1573-7454"}],"subject":[],"published":{"date-parts":[[2023,8,11]]},"assertion":[{"value":"25 July 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 August 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"35"}}