{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,25]],"date-time":"2026-02-25T17:15:49Z","timestamp":1772039749959,"version":"3.50.1"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2022,1,8]],"date-time":"2022-01-08T00:00:00Z","timestamp":1641600000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,8]],"date-time":"2022-01-08T00:00:00Z","timestamp":1641600000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2022,7]]},"DOI":"10.1007\/s10489-021-02873-7","type":"journal-article","created":{"date-parts":[[2022,1,8]],"date-time":"2022-01-08T00:03:13Z","timestamp":1641600193000},"page":"9701-9716","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Enhancing cooperation by cognition differences and consistent representation in multi-agent reinforcement learning"],"prefix":"10.1007","volume":"52","author":[{"given":"Hongwei","family":"Ge","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhixin","family":"Ge","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2794-8654","authenticated-orcid":false,"given":"Liang","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,1,8]]},"reference":[{"issue":"4","key":"2873_CR1","doi-asserted-by":"publisher","first-page":"819","DOI":"10.1287\/moor.27.4.819.297","volume":"27","author":"DS Bernstein","year":"2002","unstructured":"Bernstein D S, Givan R, Immerman N, Zilberstein S (2002) The complexity of decentralized control of markov decision processes. Math Oper Res 27(4):819\u2013840","journal-title":"Math Oper Res"},{"issue":"1","key":"2873_CR2","doi-asserted-by":"publisher","first-page":"427","DOI":"10.1109\/TII.2012.2219061","volume":"9","author":"Y Cao","year":"2012","unstructured":"Cao Y, Yu W, Ren W, Chen G (2012) An overview of recent progress in the study of distributed multi-agent coordination. IEEE Trans Ind Inf 9(1):427\u2013438","journal-title":"IEEE Trans Ind Inf"},{"issue":"12","key":"2873_CR3","doi-asserted-by":"publisher","first-page":"4195","DOI":"10.1007\/s10489-020-01755-8","volume":"50","author":"H Chen","year":"2020","unstructured":"Chen H, Liu Y, Zhou Z, Hu D, Zhang M (2020) Gama: Graph attention multi-agent reinforcement learning algorithm for cooperation. Appl Intell 50(12):4195\u20134205","journal-title":"Appl Intell"},{"key":"2873_CR4","unstructured":"Das A, Gervet T, Romoff J, Batra D, Parikh D, Rabbat M, Pineau J (2019) Tarmac: Targeted multi-agent communication. In: International Conference on Machine Learning, pp 1538\u20131546"},{"key":"2873_CR5","unstructured":"Foerster J, Nardelli N, Farquhar G, Afouras T, Torr PH, Kohli P, Whiteson S (2017) Stabilising experience replay for deep multi-agent reinforcement learning. In: International Conference on Machine Learning, pp 1146\u20131155"},{"key":"2873_CR6","doi-asserted-by":"crossref","unstructured":"Foerster JN, Farquhar G, Afouras T, Nardelli N, Whiteson S (2018) Counterfactual multi-agent policy gradients. In: Association for the Advancement of Artificial Intelligence, pp 2974\u20132982","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"2873_CR7","doi-asserted-by":"publisher","first-page":"40797","DOI":"10.1109\/ACCESS.2019.2907618","volume":"7","author":"H Ge","year":"2019","unstructured":"Ge H, Song Y, Wu C, Ren J, Tan G (2019) Cooperative deep q-learning with q-value transfer for multi-intersection signal control. IEEE Access 7:40797\u201340809","journal-title":"IEEE Access"},{"key":"2873_CR8","doi-asserted-by":"crossref","unstructured":"Gu S, Holly E, Lillicrap T, Levine S (2017) Deep reinforcement learning for robotic manipulation with asynchronous off-policy updates. In: 2017 IEEE International Conference on Robotics and Automation, pp 3389\u20133396","DOI":"10.1109\/ICRA.2017.7989385"},{"key":"2873_CR9","unstructured":"Iqbal S, Sha F (2019) Actor-attention-critic for multi-agent reinforcement learning. In: International Conference on Machine Learning, pp 2961\u20132970"},{"issue":"8","key":"2873_CR10","doi-asserted-by":"publisher","first-page":"5793","DOI":"10.1007\/s10489-020-02065-9","volume":"51","author":"H Jiang","year":"2021","unstructured":"Jiang H, Shi D, Xue C, Wang Y, Zhang Y (2021) Multi-agent deep reinforcement learning with type-based hierarchical group communication. Appl Intell 51(8):5793\u20135808","journal-title":"Appl Intell"},{"key":"2873_CR11","unstructured":"Jiang J, Lu Z (2018) Learning attentional communication for multi-agent cooperation. In: Advances in Neural Information Processing Systems, pp 7254\u20137264"},{"key":"2873_CR12","unstructured":"Kim D, Moon S, Hostallero D, Kang W J, Lee T, Son K, Yi Y (2018) Learning to schedule communication in multi-agent reinforcement learning. In: International Conference on Learning Representations"},{"key":"2873_CR13","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1016\/j.neucom.2016.01.031","volume":"190","author":"L Kraemer","year":"2016","unstructured":"Kraemer L, Banerjee B (2016) Multi-agent reinforcement learning as a rehearsal for decentralized planning. Neurocomputing 190:82\u201394","journal-title":"Neurocomputing"},{"key":"2873_CR14","doi-asserted-by":"crossref","unstructured":"Lakkaraju K, Speed A (2019) A cognitive-consistency based model of population wide attitude change. In: Complex Adaptive Systems. Springer, pp 17\u201338","DOI":"10.1007\/978-3-030-20309-2_2"},{"key":"2873_CR15","unstructured":"Li S, Gupta J K, Morales P, Allen R, Kochenderfer MJ (2021) Deep implicit coordination graphs for multi-agent reinforcement learning. International Conference on Autonomous Agents and Multiagent Systems"},{"key":"2873_CR16","doi-asserted-by":"crossref","unstructured":"Liu Y, Wang W, Hu Y, Hao J, Chen X, Gao Y (2020) Multi-agent game abstraction via graph attention neural network. In: Association for the Advancement of Artificial Intelligence, pp 7211\u20137218","DOI":"10.1609\/aaai.v34i05.6211"},{"key":"2873_CR17","doi-asserted-by":"publisher","first-page":"88","DOI":"10.3389\/fnins.2020.00088","volume":"14","author":"SA Lobov","year":"2020","unstructured":"Lobov S A, Mikhaylov A N, Shamshin M, Makarov V A, Kazantsev V B (2020) Spatial properties of stdp in a self-learning spiking neural network enable controlling a mobile robot. Front Neurosci 14:88","journal-title":"Front Neurosci"},{"key":"2873_CR18","unstructured":"Lowe R, Wu YI, Tamar A, Harb J, Abbeel OP, Mordatch I (2017) Multi-agent actor-critic for mixed cooperative-competitive environments. In: Advances in Neural Information Processing Systems, pp 6379\u20136390"},{"key":"2873_CR19","doi-asserted-by":"crossref","unstructured":"Mao H, Liu W, Hao J, Luo J, Li D, Zhang Z, Wang J, Xiao Z (2019) Neighborhood cognition consistent multi-agent reinforcement learning. arXiv:191201160","DOI":"10.1609\/aaai.v34i05.6212"},{"issue":"7540","key":"2873_CR20","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu A A, Veness J, Bellemare M G, Graves A, Riedmiller M, Fidjeland A K, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"key":"2873_CR21","unstructured":"Oh J, Chockalingam V, Lee H et al (2016) Control of memory, active perception, and action in minecraft. In: International Conference on Machine Learning, pp 2790\u20132799"},{"key":"2873_CR22","doi-asserted-by":"publisher","first-page":"289","DOI":"10.1613\/jair.2447","volume":"32","author":"FA Oliehoek","year":"2008","unstructured":"Oliehoek F A, Spaan M T, Vlassis N (2008) Optimal and approximate q-value functions for decentralized pomdps. J Artif Intell Res 32:289\u2013353","journal-title":"J Artif Intell Res"},{"key":"2873_CR23","doi-asserted-by":"crossref","unstructured":"Padakandla S, Prabuchandran K, Bhatnagar S (2020) Reinforcement learning algorithm for non-stationary environments. Applied Intelligence (11):3590\u20133606","DOI":"10.1007\/s10489-020-01758-5"},{"key":"2873_CR24","unstructured":"Palmer G, Tuyls K, Bloembergen D, Savani R (2018) Lenient multi-agent deep reinforcement learning. In: International Conference on Autonomous Agents and Multiagent Systems, pp 443\u2013451"},{"key":"2873_CR25","unstructured":"Peng P, Wen Y, Yang Y, Yuan Q, Tang Z, Long H, Wang J (2017) Multiagent bidirectionally-coordinated nets: Emergence of human-level coordination in learning to play starcraft combat games. arXiv:170310069"},{"key":"2873_CR26","doi-asserted-by":"crossref","unstructured":"Prashanth L, Bhatnagar S (2011) Reinforcement learning with average cost for adaptive control of traffic lights at intersections. In: 2011 14th International IEEE Conference on Intelligent Transportation Systems, pp 1640\u20131645","DOI":"10.1109\/ITSC.2011.6082823"},{"key":"2873_CR27","unstructured":"Rashid T, Samvelyan M, Schroeder C, Farquhar G, Foerster J, Whiteson S (2018) Qmix: Monotonic value function factorisation for deep multi-agent reinforcement learning. In: International Conference on Machine Learning, pp 4295\u20134304"},{"issue":"3","key":"2873_CR28","doi-asserted-by":"publisher","first-page":"456","DOI":"10.1037\/a0012786","volume":"137","author":"JE Russo","year":"2008","unstructured":"Russo J E, Carlson K A, Meloy M G, Yong K (2008) The goal of consistency as a cause of information distortion. J Exp Psychol Gen 137(3):456\u2013470","journal-title":"J Exp Psychol Gen"},{"key":"2873_CR29","unstructured":"Samvelyan M, Rashid T, de Witt CS, Farquhar G, Nardelli N, Rudner TG, Hung CM, Torr PH, Foerster J, Whiteson S (2019) The starcraft multi-agent challenge. In: International Conference on Autonomous Agents and Multiagent Systems, pp 2186\u2013 2188"},{"issue":"6","key":"2873_CR30","doi-asserted-by":"publisher","first-page":"814","DOI":"10.1037\/0022-3514.86.6.814","volume":"86","author":"D Simon","year":"2004","unstructured":"Simon D, Snow C J, Read S J (2004) The redux of cognitive consistency theories: evidence judgments by constraint satisfaction. J Personal Social Psychol 86(6):814\u2013837","journal-title":"J Personal Social Psychol"},{"key":"2873_CR31","unstructured":"Singh A, Jain T, Sukhbaatar S (2019) Learning when to communicate at scale in multiagent cooperative and competitive tasks. In: International Conference on Learning Representations"},{"key":"2873_CR32","unstructured":"Son K, Kim D, Kang W J, Hostallero D, Yi Y (2019) Qtran: Learning to factorize with transformation for cooperative multi-agent reinforcement learning. In: International Conference on Machine Learning"},{"key":"2873_CR33","unstructured":"Sukhbaatar S, Fergus R et al (2016) Learning multiagent communication with backpropagation. In: Advances in Neural Information Processing Systems, pp 2244\u20132252"},{"key":"2873_CR34","unstructured":"Sunehag P, Lever G, Gruslys A, Czarnecki WM, Zambaldi VF, Jaderberg M, Lanctot M, Sonnerat N, Leibo JZ, Tuyls K et al (2018) Value-decomposition networks for cooperative multi-agent learning based on team reward. In: International Conference on Autonomous Agents and Multiagent Systems, pp 2085\u20132087"},{"key":"2873_CR35","doi-asserted-by":"crossref","unstructured":"Tan M (1993) Multi-agent reinforcement learning: Independent vs. cooperative agents. In: Proceedings of the Tenth International Conference on Machine Learning, pp 330\u2013337","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"2873_CR36","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. In: Advances in Neural Information Processing Systems, pp 5998\u20136008"},{"key":"2873_CR37","unstructured":"Vinyals O, Ewalds T, Bartunov S, Georgiev P, Vezhnevets AS, Yeo M, Makhzani A, K\u00fcttler H, Agapiou J, Schrittwieser J et al (2017) Starcraft ii: A new challenge for reinforcement learning. arXiv:170804782"},{"key":"2873_CR38","unstructured":"Wiering M (2000) Multi-agent reinforcement learning for traffic light control. In: Machine Learning: Proceedings of the Seventeenth International Conference, pp 1151\u20131158"},{"issue":"7","key":"2873_CR39","doi-asserted-by":"publisher","first-page":"2490","DOI":"10.1109\/TCYB.2018.2823730","volume":"49","author":"S Yang","year":"2018","unstructured":"Yang S, Wang J, Deng B, Liu C, Li H, Fietkiewicz C, Loparo K A (2018) Real-time neuromorphic system for large-scale conductance-based spiking neural networks. IEEE Trans Cybern 49(7):2490\u20132503","journal-title":"IEEE Trans Cybern"},{"issue":"1","key":"2873_CR40","doi-asserted-by":"publisher","first-page":"148","DOI":"10.1109\/TNNLS.2019.2899936","volume":"31","author":"S Yang","year":"2019","unstructured":"Yang S, Deng B, Wang J, Li H, Lu M, Che Y, Wei X, Loparo K A (2019) Scalable digital neuromorphic architecture for large-scale biophysically meaningful neural network with multi-compartment neurons. IEEE Trans Neural Networks Learn Syst 31(1):148\u2013162","journal-title":"IEEE Trans Neural Networks Learn Syst"},{"key":"2873_CR41","doi-asserted-by":"publisher","first-page":"601109","DOI":"10.3389\/fnins.2021.601109","volume":"15","author":"S Yang","year":"2021","unstructured":"Yang S, Gao T, Wang J, Deng B, Lansdell B, Linares-Barranco B (2021a) Efficient spike-driven learning with dendritic event-based processing. Front Neurosci 15:601109","journal-title":"Front Neurosci"},{"key":"2873_CR42","doi-asserted-by":"publisher","unstructured":"Yang S, Wang J, Deng B, Azghadi MR, Linares-Barranco B (2021b) Neuromorphic context-dependent learning framework with fault-tolerant spike routing. IEEE Transactions on Neural Networks and Learning Systems. https:\/\/doi.org\/10.1109\/TNNLS.2021.3084250","DOI":"10.1109\/TNNLS.2021.3084250"},{"key":"2873_CR43","doi-asserted-by":"publisher","unstructured":"Yang S, Wang J, Zhang N, Deng B, Pang Y, Azghadi MR (2021c) Cerebellumorphic: large-scale neuromorphic model and architecture for supervised motor learning. IEEE Transactions on Neural Networks and Learning Systems https:\/\/doi.org\/10.1109\/TNNLS.2021.3057070","DOI":"10.1109\/TNNLS.2021.3057070"},{"key":"2873_CR44","unstructured":"Yang Y, Hao J, Liao B, Shao K, Chen G, Liu W, Tang H (2020) Qatten: A general framework for cooperative multiagent reinforcement learning. arXiv:200203939"},{"key":"2873_CR45","unstructured":"Zhang SQ, Zhang Q, Lin J (2019) Efficient communication in multi-agent reinforcement learning via variance based control. In: Advances in Neural Information Processing Systems, pp 3235\u20133244"},{"key":"2873_CR46","unstructured":"Zhang SQ, Lin J, Zhang Q (2020) Succinct and robust multi-agent communication with temporal message control. arXiv:201014391"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-021-02873-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-021-02873-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-021-02873-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,22]],"date-time":"2023-01-22T08:31:42Z","timestamp":1674376302000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-021-02873-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,1,8]]},"references-count":46,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2022,7]]}},"alternative-id":["2873"],"URL":"https:\/\/doi.org\/10.1007\/s10489-021-02873-7","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,1,8]]},"assertion":[{"value":"24 September 2021","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 January 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}