{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T04:54:12Z","timestamp":1755838452218},"reference-count":36,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"7","license":[{"start":{"date-parts":[[2016,7,1]],"date-time":"2016-07-01T00:00:00Z","timestamp":1467331200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Cybern."],"published-print":{"date-parts":[[2016,7]]},"DOI":"10.1109\/tcyb.2015.2453165","type":"journal-article","created":{"date-parts":[[2015,8,13]],"date-time":"2015-08-13T16:08:54Z","timestamp":1439482134000},"page":"1640-1654","source":"Crossref","is-referenced-by-count":16,"title":["Learning Stationary Correlated Equilibria in Constrained General-Sum Stochastic Games"],"prefix":"10.1109","volume":"46","author":[{"given":"Vesal","family":"Hakami","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mehdi","family":"Dehghan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref33","doi-asserted-by":"crossref","DOI":"10.1007\/978-93-86279-38-5","author":"borkar","year":"2008","journal-title":"Stochastic Approximation A Dynamical Systems Viewpoint"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1186\/1687-1499-2012-371"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511841224"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.2012.6211486"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-60756-1"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysconle.2004.08.007"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4471-4285-0"},{"key":"ref10","first-page":"322","article-title":"Friend-or-Foe Q-learning in general-sum Markov games","author":"littman","year":"2001","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992698"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2013.2253094"},{"key":"ref13","first-page":"964","article-title":"Model learning and knowledge sharing for a multiagent system with Dyna-Q learning","volume":"45","author":"hwang","year":"2015","journal-title":"IEEE Trans Cybern"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2014.2319733"},{"key":"ref15","first-page":"1021","article-title":"Rational and convergent learning in stochastic games","volume":"2","author":"bowling","year":"2001","journal-title":"Proc Int Joint Conf AI"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCB.2008.920998"},{"key":"ref17","first-page":"1371","article-title":"Two-timescale algorithms for learning Nash equilibria in general-sum stochastic games","author":"prasad","year":"2015","journal-title":"Proc 14th AAMAS"},{"key":"ref18","article-title":"A direct proof of the existence of correlated equilibrium policies in general-sum Markov games","author":"greenwald","year":"2005"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1111\/1468-0262.00153"},{"key":"ref28","author":"altman","year":"1999","journal-title":"Constrained Markov Decision Processes"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-1336-9_11"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1214\/11-SSY056"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/0304-4068(74)90037-8"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.orl.2013.11.007"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/MWC.2012.6189405"},{"key":"ref5","author":"filar","year":"1997","journal-title":"Competitive Markov Decision Processes"},{"key":"ref8","first-page":"1039","article-title":"Nash Q-learning for general-sum stochastic games","volume":"4","author":"hu","year":"2003","journal-title":"J Mach Learn Res"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.2307\/1969529"},{"key":"ref9","article-title":"Learning average reward irreducible stochastic games: Analysis and applications","author":"li","year":"2003"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.39.10.1095"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1287\/moor.1060.0213"},{"key":"ref22","first-page":"242","article-title":"Correlated Q-learning","author":"greenwald","year":"2003","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780199269181.001.0001"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/BF00993306"},{"key":"ref23","author":"gondek","year":"2015","journal-title":"QnR-Learning in Markov Games"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177700285"},{"key":"ref25","author":"bertsekas","year":"1999","journal-title":"Nonlinear Programming"}],"container-title":["IEEE Transactions on Cybernetics"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6221036\/7491394\/07182300.pdf?arnumber=7182300","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T16:45:32Z","timestamp":1642005932000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7182300\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,7]]},"references-count":36,"journal-issue":{"issue":"7"},"URL":"https:\/\/doi.org\/10.1109\/tcyb.2015.2453165","relation":{},"ISSN":["2168-2267","2168-2275"],"issn-type":[{"value":"2168-2267","type":"print"},{"value":"2168-2275","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016,7]]}}}