{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T16:32:07Z","timestamp":1783009927792,"version":"3.54.5"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2015,10,19]],"date-time":"2015-10-19T00:00:00Z","timestamp":1445212800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"published-print":{"date-parts":[[2016,3]]},"DOI":"10.1007\/s10462-015-9447-5","type":"journal-article","created":{"date-parts":[[2015,10,20]],"date-time":"2015-10-20T08:37:16Z","timestamp":1445330236000},"page":"299-332","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Exponential moving average based multiagent reinforcement learning algorithms"],"prefix":"10.1007","volume":"45","author":[{"given":"Mostafa D.","family":"Awheda","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Howard M.","family":"Schwartz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2015,10,19]]},"reference":[{"key":"9447_CR1","doi-asserted-by":"crossref","first-page":"521","DOI":"10.1613\/jair.2628","volume":"33","author":"S Abdallah","year":"2008","unstructured":"Abdallah S, Lesser V (2008) A multiagent reinforcement learning algorithm with non-linear dynamics. J Artif Intell Res 33:521\u2013549","journal-title":"J Artif Intell Res"},{"key":"9447_CR2","doi-asserted-by":"crossref","unstructured":"Awheda MD, Schwartz HM (2013) Exponential moving average Q-learning algorithm. In: Adaptive dynamic programming and reinforcement learning (ADPRL), 2013 IEEE symposium on, IEEE, pp 31\u201338. IEEE","DOI":"10.1109\/ADPRL.2013.6614986"},{"key":"9447_CR3","doi-asserted-by":"crossref","unstructured":"Awheda MD, Schwartz HM (2015) The residual gradient FACL algorithm for differential games. In Electrical and computer engineering (CCECE). 2015 IEEE 28th Canadian conference on, IEEE, pp 1006\u20131011. IEEE","DOI":"10.1109\/CCECE.2015.7129412"},{"issue":"3","key":"9447_CR4","doi-asserted-by":"crossref","first-page":"281","DOI":"10.1007\/s10458-007-9013-x","volume":"15","author":"B Banerjee","year":"2007","unstructured":"Banerjee B, Peng J (2007) Generalized multiagent learning with performance bound. Auton Agents Multi-Agent Syst 15(3):281\u2013312","journal-title":"Auton Agents Multi-Agent Syst"},{"key":"9447_CR5","volume-title":"Dynamic programming","author":"R Bellman","year":"1957","unstructured":"Bellman R (1957) Dynamic programming. Princeton University Press, Princeton"},{"key":"9447_CR6","first-page":"209","volume":"17","author":"M Bowling","year":"2005","unstructured":"Bowling M (2005) Convergence and no-regret in multiagent learning. Adv Neural Inf Process Syst 17:209\u2013216","journal-title":"Adv Neural Inf Process Syst"},{"key":"9447_CR7","unstructured":"Bowling M, Veloso M (2001a) Convergence of gradient dynamics with a variable learning rate. In: ICML, pp 27\u201334"},{"key":"9447_CR8","unstructured":"Bowling M, Veloso M (2001b) Rational and convergent learning in stochastic games. In: International joint conference on artificial intelligence, vol. 17. Lawrence Erlbaum Associates Ltd, pp 1021\u20131026"},{"issue":"2","key":"9447_CR9","doi-asserted-by":"crossref","first-page":"215","DOI":"10.1016\/S0004-3702(02)00121-2","volume":"136","author":"M Bowling","year":"2002","unstructured":"Bowling M, Veloso M (2002) Multiagent learning using a variable learning rate. Artif Intell 136(2):215\u2013250","journal-title":"Artif Intell"},{"issue":"4","key":"9447_CR10","doi-asserted-by":"crossref","first-page":"127","DOI":"10.1016\/j.jalgor.2009.04.003","volume":"64","author":"A Burkov","year":"2009","unstructured":"Burkov A, Chaib-draa B (2009) Effective learning in the presence of adaptive counterparts. J Algorithms 64(4):127\u2013138","journal-title":"J Algorithms"},{"key":"9447_CR11","doi-asserted-by":"crossref","unstructured":"Busoniu L, Babuska R, De\u00a0Schutter B (2006) Multi-agent reinforcement learning: A survey. In: Control, automation, robotics and vision, 2006. ICARCV\u201906. 9th international conference on, IEEE, pp 1\u20136. IEEE","DOI":"10.1109\/ICARCV.2006.345353"},{"issue":"2","key":"9447_CR12","doi-asserted-by":"crossref","first-page":"156","DOI":"10.1109\/TSMCC.2007.913919","volume":"38","author":"L Busoniu","year":"2008","unstructured":"Busoniu L, Babuska R, De Schutter B (2008) A comprehensive survey of multiagent reinforcement learning. Syst Man Cybern Part C: Appl Rev, IEEE Trans 38(2):156\u2013172","journal-title":"Syst Man Cybern Part C: Appl Rev, IEEE Trans"},{"key":"9447_CR13","unstructured":"Claus C, Boutilier C (1998) The dynamics of reinforcement learning in cooperative multiagent systems. In: AAAI\/IAAI, pp 746\u2013752"},{"issue":"1\u20132","key":"9447_CR14","doi-asserted-by":"crossref","first-page":"23","DOI":"10.1007\/s10994-006-0143-1","volume":"67","author":"V Conitzer","year":"2007","unstructured":"Conitzer V, Sandholm T (2007) Awesome: A general multiagent learning algorithm that converges in self-play and learns a best response against stationary opponents. Mach Learn 67(1\u20132):23\u201343","journal-title":"Mach Learn"},{"issue":"3","key":"9447_CR15","doi-asserted-by":"crossref","first-page":"285","DOI":"10.1109\/TITS.2005.853698","volume":"6","author":"X Dai","year":"2005","unstructured":"Dai X, Li C-K, Rad AB (2005) An approach to tune fuzzy controllers based on reinforcement learning for autonomous vehicle control. Intell Transp Syst, IEEE Trans 6(3):285\u2013293","journal-title":"Intell Transp Syst, IEEE Trans"},{"key":"9447_CR16","volume-title":"Linear time-varying systems: analysis and synthesis","author":"H D\u2019Angelo","year":"1970","unstructured":"D\u2019Angelo H (1970) Linear time-varying systems: analysis and synthesis. Allyn & Bacon, Newton"},{"key":"9447_CR17","volume-title":"Linear systems: a state variable approach with numerical implementation","author":"RA DeCarlo","year":"1989","unstructured":"DeCarlo RA (1989) Linear systems: a state variable approach with numerical implementation. Prentice-Hall Inc, Upper Saddle River"},{"issue":"3","key":"9447_CR18","doi-asserted-by":"crossref","first-page":"1048","DOI":"10.2514\/1.G000173","volume":"37","author":"W Dixon","year":"2014","unstructured":"Dixon W (2014) Optimal adaptive control and differential games by reinforcement learning principles. J Guid Control Dyn 37(3):1048\u20131049","journal-title":"J Guid Control Dyn"},{"key":"9447_CR19","unstructured":"Fulda N, Ventura D (2007) Predicting and preventing coordination problems in cooperative Q-learning systems. In: IJCAI, vol. 2007, pp 780\u2013785"},{"issue":"1","key":"9447_CR20","doi-asserted-by":"crossref","first-page":"65","DOI":"10.1162\/106454604322875913","volume":"10","author":"DA Gutnisky","year":"2004","unstructured":"Gutnisky DA, Zanutto BS (2004) Learning obstacle avoidance with an operant behavior model. Artif Life 10(1):65\u201381","journal-title":"Artif Life"},{"issue":"1","key":"9447_CR21","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1109\/TFUZZ.2010.2081994","volume":"19","author":"W Hinojosa","year":"2011","unstructured":"Hinojosa W, Nefti S, Kaymak U (2011) Systems control with generalized probabilistic fuzzy-reinforcement learning. Fuzzy Syst, IEEE Trans 19(1):51\u201364","journal-title":"Fuzzy Syst, IEEE Trans"},{"key":"9447_CR22","unstructured":"Howard RA (1960) Dynamic programming and markov processes. MIT Press, Cambridge"},{"key":"9447_CR23","first-page":"1039","volume":"4","author":"J Hu","year":"2003","unstructured":"Hu J, Wellman MP (2003) Nash Q-learning for general-sum stochastic games. J Mach Learn Res 4:1039\u20131069","journal-title":"J Mach Learn Res"},{"key":"9447_CR24","unstructured":"Hu J, Wellman MP, et al (1998) Multiagent reinforcement learning: theoretical framework and an algorithm. In: ICML, vol. 98, Citeseer, pp 242\u2013250"},{"key":"9447_CR25","doi-asserted-by":"crossref","unstructured":"Kaelbling LP, Littman ML, Moore AW (1996) Reinforcement learning: a survey. J Artif Intell Res 4:237\u2013285","DOI":"10.1613\/jair.301"},{"issue":"2","key":"9447_CR26","doi-asserted-by":"crossref","first-page":"111","DOI":"10.1016\/j.robot.2003.11.006","volume":"46","author":"T Kondo","year":"2004","unstructured":"Kondo T, Ito K (2004) A reinforcement learning with evolutionary state recruitment strategy for autonomous mobile robots control. Robot Auton Syst 46(2):111\u2013124","journal-title":"Robot Auton Syst"},{"issue":"19","key":"9447_CR27","doi-asserted-by":"crossref","first-page":"8106","DOI":"10.1021\/ie4031743","volume":"53","author":"B Luo","year":"2014","unstructured":"Luo B, Wu H-N, Li H-X (2014a) Data-based suboptimal neuro-control design with reinforcement learning for dissipative spatially distributed processes. Ind Eng Chem Res 53(19):8106\u20138119","journal-title":"Ind Eng Chem Res"},{"issue":"12","key":"9447_CR28","doi-asserted-by":"crossref","first-page":"3281","DOI":"10.1016\/j.automatica.2014.10.056","volume":"50","author":"B Luo","year":"2014","unstructured":"Luo B, Wu H-N, Huang T, Liu D (2014b) Data-based approximate policy iteration for nonlinear continuous-time optimal control design. Automatica 50(12):3281\u20133290","journal-title":"Automatica"},{"issue":"1","key":"9447_CR29","doi-asserted-by":"crossref","first-page":"65","DOI":"10.1109\/TCYB.2014.2319577","volume":"45","author":"B Luo","year":"2015","unstructured":"Luo B, Wu H-N, Huang T (2015a) Off-policy reinforcement learning for $$H_{\\infty }$$ H \u221e control design. Cybern, IEEE Trans 45(1):65\u201376","journal-title":"Cybern, IEEE Trans"},{"issue":"4","key":"9447_CR30","doi-asserted-by":"crossref","first-page":"684","DOI":"10.1109\/TNNLS.2014.2320744","volume":"26","author":"B Luo","year":"2015","unstructured":"Luo B, Wu H-N, Li H-X (2015b) Adaptive optimal control of highly dissipative nonlinear spatially distributed processes with neuro-dynamic programming. Neural Netw Learn Syst, IEEE Trans 26(4):684\u2013696","journal-title":"Neural Netw Learn Syst, IEEE Trans"},{"key":"9447_CR31","doi-asserted-by":"crossref","unstructured":"Luo B, Huang T, Wu H-N, Yang X (2015c) Data-driven $$H_{\\infty }$$ H \u221e control for nonlinear distributed parameter systems. Neural Netw Learn Syst, IEEE Trans 26(11):2949\u20132961","DOI":"10.1109\/TNNLS.2015.2461023"},{"key":"9447_CR32","doi-asserted-by":"crossref","first-page":"150","DOI":"10.1016\/j.neunet.2015.08.007","volume":"71","author":"B Luo","year":"2015","unstructured":"Luo B, Wu H-N, Huang T, Liu D (2015d) Reinforcement learning solution for HJB equation arising in constrained optimal control problem. Neural Netw 71:150\u2013158","journal-title":"Neural Netw"},{"issue":"1","key":"9447_CR33","doi-asserted-by":"crossref","first-page":"193","DOI":"10.1016\/j.automatica.2013.09.043","volume":"50","author":"H Modares","year":"2014","unstructured":"Modares H, Lewis FL, Naghibi-Sistani M-B (2014) Integral reinforcement learning and experience replay for adaptive optimal control of partially-unknown constrained-input continuous-time systems. Automatica 50(1):193\u2013202","journal-title":"Automatica"},{"issue":"9","key":"9447_CR34","doi-asserted-by":"crossref","first-page":"735","DOI":"10.1016\/j.robot.2007.05.005","volume":"55","author":"M Rodr\u00edguez","year":"2007","unstructured":"Rodr\u00edguez M, Iglesias R, Regueiro CV, Correa J, Barro S (2007) Autonomous and fast robot learning through motivation. Robot Auton Syst 55(9):735\u2013740","journal-title":"Robot Auton Syst"},{"key":"9447_CR35","doi-asserted-by":"crossref","DOI":"10.1002\/9781118884614","volume-title":"Multi-Agent Machine Learning: A Reinforcement Approach","author":"HM Schwartz","year":"2014","unstructured":"Schwartz HM (2014) Multi-Agent Machine Learning: A Reinforcement Approach. Wiley, New York"},{"key":"9447_CR36","unstructured":"Sen S, Sekaran M, Hale J (1994) Learning to coordinate without sharing information. In: AAAI, pp 426\u2013431"},{"key":"9447_CR37","unstructured":"Singh S, Kearns M, Mansour Y (2000) Nash convergence of gradient dynamics in general-sum games. In: Proceedings of the sixteenth conference on Uncertainty in artificial intelligence, Morgan Kaufmann Publishers Inc., pp 541\u2013548"},{"key":"9447_CR38","doi-asserted-by":"crossref","unstructured":"Smart WD, Kaelbling LP (2002) Effective reinforcement learning for mobile robots. In: Robotics and automation. Proceedings. ICRA\u201902. IEEE international conference on, vol. 4, IEEE, 2002, pp. 3404\u20133410. IEEE","DOI":"10.1109\/ROBOT.2002.1014237"},{"key":"9447_CR39","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton RS, Barto AG (1998) Reinforcement learning: an introduction. The MIT Press, Cambridge"},{"key":"9447_CR40","doi-asserted-by":"crossref","unstructured":"Tan M (1993) Multi-agent reinforcement learning: Independent vs. cooperative agents. In: Proceedings of the tenth international conference on machine learning, pp 330\u2013337","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"9447_CR41","unstructured":"Tesauro G (2004) Extending Q-learning to general adaptive multi-agent systems. In: Advances in neural information processing systems, vol. 16. MIT press, pp 871\u2013878"},{"key":"9447_CR42","volume-title":"Networks of learning automata: techniques for online stochastic optimization","author":"MA Thathachar","year":"2011","unstructured":"Thathachar MA, Sastry PS (2011) Networks of learning automata: techniques for online stochastic optimization. Springer, Boston"},{"issue":"3\u20134","key":"9447_CR43","first-page":"279","volume":"8","author":"CJ Watkins","year":"1992","unstructured":"Watkins CJ, Dayan P (1992) Q-learning. Mach Learn 8(3\u20134):279\u2013292","journal-title":"Mach Learn"},{"key":"9447_CR44","unstructured":"Watkins CJCH (1989) Learning from delayed rewards, Ph.D. thesis, University of Cambridge England"},{"key":"9447_CR45","unstructured":"Weiss G (1999) Multiagent systems: a modern approach to distributed artificial intelligence. MIT Press"},{"issue":"12","key":"9447_CR46","doi-asserted-by":"crossref","first-page":"1884","DOI":"10.1109\/TNNLS.2012.2217349","volume":"23","author":"H-N Wu","year":"2012","unstructured":"Wu H-N, Luo B (2012) Neural network based online simultaneous policy update algorithm for solving the HJI equation in nonlinear control. Neural Netw Learn Syst, IEEE Trans 23(12):1884\u20131895","journal-title":"Neural Netw Learn Syst, IEEE Trans"},{"issue":"1","key":"9447_CR47","doi-asserted-by":"crossref","first-page":"17","DOI":"10.1109\/TSMCB.2003.808179","volume":"33","author":"C Ye","year":"2003","unstructured":"Ye C, Yung NH, Wang D (2003) A fuzzy controller with supervised learning assisted reinforcement learning algorithm for obstacle avoidance. Syst Man Cybern Part B: Cybern, IEEE Trans 33(1):17\u201327","journal-title":"Syst Man Cybern Part B: Cybern, IEEE Trans"},{"key":"9447_CR48","doi-asserted-by":"crossref","unstructured":"Zhang C, Lesser VR (2010) Multi-agent learning with policy prediction. In: AAAI","DOI":"10.1609\/aaai.v24i1.7639"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-015-9447-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10462-015-9447-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-015-9447-5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,15]],"date-time":"2023-08-15T14:26:30Z","timestamp":1692109590000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10462-015-9447-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,10,19]]},"references-count":48,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2016,3]]}},"alternative-id":["9447"],"URL":"https:\/\/doi.org\/10.1007\/s10462-015-9447-5","relation":{},"ISSN":["0269-2821","1573-7462"],"issn-type":[{"value":"0269-2821","type":"print"},{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,10,19]]}}}