{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,10]],"date-time":"2026-01-10T19:21:03Z","timestamp":1768072863294,"version":"3.49.0"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2019,5,15]],"date-time":"2019-05-15T00:00:00Z","timestamp":1557878400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,5,15]],"date-time":"2019-05-15T00:00:00Z","timestamp":1557878400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61572349"],"award-info":[{"award-number":["61572349"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61272106"],"award-info":[{"award-number":["61272106"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61702362"],"award-info":[{"award-number":["61702362"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61602391"],"award-info":[{"award-number":["61602391"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["3132019207"],"award-info":[{"award-number":["3132019207"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Agent Multi-Agent Syst"],"published-print":{"date-parts":[[2019,7]]},"DOI":"10.1007\/s10458-019-09411-3","type":"journal-article","created":{"date-parts":[[2019,5,15]],"date-time":"2019-05-15T22:51:58Z","timestamp":1557960718000},"page":"403-429","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":18,"title":["SA-IGA: a multiagent reinforcement learning method towards socially optimal outcomes"],"prefix":"10.1007","volume":"33","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9157-6050","authenticated-orcid":false,"given":"Chengwei","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaohong","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianye","family":"Hao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siqi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Karl","family":"Tuyls","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wanli","family":"Xue","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiyong","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,5,15]]},"reference":[{"issue":"1","key":"9411_CR1","doi-asserted-by":"publisher","first-page":"521","DOI":"10.1613\/jair.2628","volume":"33","author":"S Abdallah","year":"2008","unstructured":"Abdallah, S., & Lesser, V. (2008). A multiagent reinforcement learning algorithm with non-linear dynamics. Journal of Artificial Intelligence Research, 33(1), 521\u2013549.","journal-title":"Journal of Artificial Intelligence Research"},{"key":"9411_CR2","doi-asserted-by":"crossref","unstructured":"Alvard, M. S. (2004) The ultimatum game, fairness, and cooperation among big game hunters. In Foundations of human sociality (pp. 413\u2013435).","DOI":"10.1093\/0199262055.003.0014"},{"key":"9411_CR3","volume-title":"Partners versus strangers: Random rematching in public goods experiments","author":"J Andreoni","year":"1998","unstructured":"Andreoni, J., & Croson, R. (1998). Partners versus strangers: Random rematching in public goods experiments. Amsterdam: Elsevier."},{"key":"9411_CR4","doi-asserted-by":"crossref","unstructured":"Banerjee, B., & Peng, J. (2003). Adaptive policy gradient in multiagent learning. In International joint conference on autonomous agents and multiagent systems (pp. 686\u2013692).","DOI":"10.1145\/860575.860686"},{"key":"9411_CR5","unstructured":"Banerjee, B., & Peng, J. (2004). The role of reactivity in multiagent learning. In International joint conference on autonomous agents and multiagent systems (pp. 538\u2013545)."},{"key":"9411_CR6","doi-asserted-by":"crossref","unstructured":"Banerjee, B., & Peng, J. (2005). Efficient learning of multi-step best response. In Proceedings of the fourth international joint conference on autonomous agents and multiagent systems (pp. 60\u201366).","DOI":"10.1145\/1082473.1082483"},{"key":"9411_CR7","doi-asserted-by":"crossref","unstructured":"Banerjee, D., & Sen, S. (2007). Reaching pareto optimality in Prisoner\u2019s Dilemma using conditional joint action learning. In AAMAS\u201907 (pp. 91\u2013108).","DOI":"10.1007\/s10458-007-0020-8"},{"key":"9411_CR8","doi-asserted-by":"publisher","first-page":"659","DOI":"10.1613\/jair.4818","volume":"53","author":"D Bloembergen","year":"2015","unstructured":"Bloembergen, D., Tuyls, K., Hennes, D., & Kaisers, M. (2015). Evolutionary dynamics of multi-agent learning: A survey. Journal of Artificial Intelligence Research, 53, 659\u2013697.","journal-title":"Journal of Artificial Intelligence Research"},{"key":"9411_CR9","unstructured":"Bowling, M. (2004). Convergence and no-regret in multiagent learning. In International conference on neural information processing systems (pp. 209\u2013216)."},{"key":"9411_CR10","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1016\/S0004-3702(02)00121-2","volume":"136","author":"MH Bowling","year":"2003","unstructured":"Bowling, M. H., & Veloso, M. M. (2003). Multiagent learning using a variable learning rate. Artificial Intelligence, 136, 215\u2013250.","journal-title":"Artificial Intelligence"},{"issue":"2","key":"9411_CR11","doi-asserted-by":"publisher","first-page":"156","DOI":"10.1109\/TSMCC.2007.913919","volume":"38","author":"L Busoniu","year":"2008","unstructured":"Busoniu, L., Babuska, R., & De Schutter, B. (2008). A comprehensive survey of multiagent reinforcement learning. IEEE Transactions on Systems, Man, and Cybernetics, Part C: Applications and Reviews, 38(2), 156\u2013172.","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics, Part C: Applications and Reviews"},{"issue":"2","key":"9411_CR12","doi-asserted-by":"publisher","first-page":"182","DOI":"10.1007\/s10458-013-9222-4","volume":"28","author":"D Chakraborty","year":"2014","unstructured":"Chakraborty, D., & Stone, P. (2014). Multiagent learning in the presence of memory-bounded agents. Autonomous Agents and Multi-agent Systems, 28(2), 182\u2013213.","journal-title":"Autonomous Agents and Multi-agent Systems"},{"key":"9411_CR13","volume-title":"Theory of ordinary differential equations","author":"EA Coddington","year":"1955","unstructured":"Coddington, E. A., & Levinson, N. (1955). Theory of ordinary differential equations. New York: McGraw-Hill."},{"issue":"1\u20132","key":"9411_CR14","doi-asserted-by":"publisher","first-page":"23","DOI":"10.1007\/s10994-006-0143-1","volume":"67","author":"V Conitzer","year":"2007","unstructured":"Conitzer, V., & Sandholm, T. (2007). Awesome: A general multiagent learning algorithm that converges in self-play and learns a best response against stationary opponents. Machine Learning, 67(1\u20132), 23\u201343.","journal-title":"Machine Learning"},{"key":"9411_CR15","unstructured":"Crandall, J. W. (2013). Just add pepper: Extending learning algorithms for repeated matrix games to repeated Markov games. In International conference on autonomous agents and multiagent systems (pp. 399\u2013406)."},{"key":"9411_CR16","unstructured":"Foerster, J. N., Chen, R. Y., Al-Shedivat, M., Whiteson, S., Abbeel, P., & Mordatch, I. (2017). Learning with opponent-learning awareness. CoRR arXiv:1709.04326 ."},{"issue":"4","key":"9411_CR17","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1002\/cplx.10092","volume":"8","author":"C Hauert","year":"2003","unstructured":"Hauert, C., & Szab, G. (2003). Prisoner\u2019s Dilemma and public goods games in different geometries: Compulsory versus voluntary interactions. Complexity, 8(4), 31\u201338.","journal-title":"Complexity"},{"key":"9411_CR18","first-page":"1039","volume":"4","author":"J Hu","year":"2003","unstructured":"Hu, J., & Wellman, M. P. (2003). Nash q-learning for general-sum stochastic games. The Journal of Machine Learning Research, 4, 1039\u20131069.","journal-title":"The Journal of Machine Learning Research"},{"key":"9411_CR19","unstructured":"Hughes, E., Leibo, J. Z., Phillips, M., Tuyls, K., Due\u00f1ez-Guzman, E., Casta\u00f1eda, A.G., et\u00a0al. (2018). Inequity aversion improves cooperation in intertemporal social dilemmas. In Advances in neural information processing systems (pp. 3330\u20133340)."},{"key":"9411_CR20","unstructured":"Lauer, M., & Rienmiller, M. (2000). An algorithm for distributed reinforcement learning in cooperative multi-agent systems. In ICML\u201900 (pp. 535\u2013542)."},{"key":"9411_CR21","doi-asserted-by":"crossref","unstructured":"Littman, M. (1994). Markov games as a framework for multi-agent reinforcement learning. In Proceedings of the 11th international conference on machine learning, (pp. 322\u2013328).","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"9411_CR22","unstructured":"Littman, M. L. (2001). Friend-or-foe q-learning in general-sum games. In ICML (Vol. 1, pp. 322\u2013328)."},{"issue":"01","key":"9411_CR23","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1017\/S0269888912000057","volume":"27","author":"L Matignon","year":"2012","unstructured":"Matignon, L., Laurent, G. J., & Le Fort-Piat, N. (2012). Independent reinforcement learners in cooperative Markov games: A survey regarding coordination problems. The Knowledge Engineering Review, 27(01), 1\u201331.","journal-title":"The Knowledge Engineering Review"},{"key":"9411_CR24","unstructured":"Peysakhovich, A., & Lerer, A. (2017). Prosocial learning agents solve generalized stag hunts better than selfish ones. CoRR arXiv:1709.02865 ."},{"key":"9411_CR25","unstructured":"Powers, R., & Shoham, Y. (2005). Learning against opponents with bounded memory. In IJCAI (Vol. 5, pp. 817\u2013822)."},{"key":"9411_CR26","unstructured":"Rodrigues\u00a0Gomes, E., & Kowalczyk, R. (2009). Dynamic analysis of multiagent q-learning with $$\\varepsilon $$-greedy exploration. In Proceedings of the 26th annual international conference on machine learning (pp. 369\u2013376). ACM."},{"key":"9411_CR27","doi-asserted-by":"publisher","DOI":"10.1142\/4221","volume-title":"Methods of qualitative theory in nonlinear dynamics","author":"LP Shilnikov","year":"2001","unstructured":"Shilnikov, L. P., Shilnikov, A. L., Turaev, D. V., & Chua, L. O. (2001). Methods of qualitative theory in nonlinear dynamics (Vol. 5). Singapore: World Scientific."},{"issue":"5","key":"9411_CR28","doi-asserted-by":"publisher","first-page":"2015","DOI":"10.1109\/TVT.2014.2334655","volume":"64","author":"S Shivshankar","year":"2015","unstructured":"Shivshankar, S., & Jamalipour, A. (2015). An evolutionary game theory-based approach to cooperation in vanets under different network conditions. IEEE Transactions on Vehicular Technology, 64(5), 2015\u20132022.","journal-title":"IEEE Transactions on Vehicular Technology"},{"key":"9411_CR29","unstructured":"Singh, S., Kearns, M., & Mansour, Y. (2000). Nash convergence of gradient dynamics in general-sum games. In Proceedings of the sixteenth conference on uncertainty in artificial intelligence (pp. 541\u2013548). Morgan Kaufmann."},{"issue":"1","key":"9411_CR30","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1007\/s10458-005-3783-9","volume":"12","author":"K Tuyls","year":"2006","unstructured":"Tuyls, K., Hoen, P. J., & Vanschoenwinkel, B. (2006). An evolutionary dynamical analysis of multi-agent learning in iterated games. Autonomous Agents and Multi-agent Systems, 12(1), 115\u2013153.","journal-title":"Autonomous Agents and Multi-agent Systems"},{"key":"9411_CR31","doi-asserted-by":"crossref","unstructured":"Tuyls, K., Verbeeck, K., & Lenaerts, T. (2003). A selection-mutation model for q-learning in multi-agent systems. In Proceedings of the second international joint conference on autonomous agents and multiagent systems (pp. 693\u2013700). ACM.","DOI":"10.1145\/860575.860687"},{"issue":"7","key":"9411_CR32","doi-asserted-by":"publisher","first-page":"363","DOI":"10.1016\/j.artint.2007.05.002","volume":"171","author":"RV Vohra","year":"2007","unstructured":"Vohra, R. V., & Wellman, M. P. (2007). Foundations of multi-agent learning. Artificial Intelligence, 171(7), 363\u2013452.","journal-title":"Artificial Intelligence"},{"issue":"4","key":"9411_CR33","first-page":"233","volume":"15","author":"CJCH Watkins","year":"1989","unstructured":"Watkins, C. J. C. H. (1989). Learning from delayed rewards. Robotics & Autonomous Systems, 15(4), 233\u2013235.","journal-title":"Robotics & Autonomous Systems"},{"key":"9411_CR34","first-page":"279","volume":"8","author":"CJCH Watkins","year":"1992","unstructured":"Watkins, C. J. C. H., & Dayan, P. D. (1992). Q-learning. Machine Learning, 8, 279\u2013292.","journal-title":"Machine Learning"},{"issue":"6","key":"9411_CR35","doi-asserted-by":"publisher","first-page":"1135","DOI":"10.1109\/JSAC.2013.130615","volume":"31","author":"G Wei","year":"2013","unstructured":"Wei, G., Zhu, P., Vasilakos, A. V., & Mao, Y. (2013). Cooperation dynamics on collaborative social networks of heterogeneous population. IEEE Journal on Selected Areas in Communications, 31(6), 1135\u20131146.","journal-title":"IEEE Journal on Selected Areas in Communications"},{"key":"9411_CR36","doi-asserted-by":"crossref","unstructured":"Zhang, C., & Lesser, V. R. (2010). Multi-agent learning with policy prediction. In Proceedings of the twenty-fourth AAAI conference on artificial intelligence (pp. 927\u2013934).","DOI":"10.1609\/aaai.v24i1.7639"},{"issue":"6","key":"9411_CR37","doi-asserted-by":"publisher","first-page":"1367","DOI":"10.1109\/TCYB.2016.2544866","volume":"47","author":"Z Zhang","year":"2017","unstructured":"Zhang, Z., Zhao, D., Gao, J., Wang, D., & Dai, Y. (2017). Fmrq\u2014A multiagent reinforcement learning algorithm for fully cooperative tasks. IEEE Transactions on Cybernetics, 47(6), 1367\u20131379.","journal-title":"IEEE Transactions on Cybernetics"},{"key":"9411_CR38","unstructured":"Zinkevich, M. (2003). Online convex programming and generalized infinitesimal gradient ascent. In ICML (pp. 928\u2013936)."}],"container-title":["Autonomous Agents and Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-019-09411-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10458-019-09411-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-019-09411-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,18]],"date-time":"2022-09-18T02:44:32Z","timestamp":1663469072000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10458-019-09411-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,5,15]]},"references-count":38,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2019,7]]}},"alternative-id":["9411"],"URL":"https:\/\/doi.org\/10.1007\/s10458-019-09411-3","relation":{},"ISSN":["1387-2532","1573-7454"],"issn-type":[{"value":"1387-2532","type":"print"},{"value":"1573-7454","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,5,15]]},"assertion":[{"value":"15 May 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}