{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T11:33:35Z","timestamp":1784806415285,"version":"3.55.0"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"31","license":[{"start":{"date-parts":[[2025,1,10]],"date-time":"2025-01-10T00:00:00Z","timestamp":1736467200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,1,10]],"date-time":"2025-01-10T00:00:00Z","timestamp":1736467200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100003130","name":"Fonds Wetenschappelijk Onderzoek","doi-asserted-by":"publisher","award":["1S94120N"],"award-info":[{"award-number":["1S94120N"]}],"id":[{"id":"10.13039\/501100003130","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003130","name":"Fonds Wetenschappelijk Onderzoek","doi-asserted-by":"publisher","award":["1S12121N"],"award-info":[{"award-number":["1S12121N"]}],"id":[{"id":"10.13039\/501100003130","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1007\/s00521-024-10598-0","type":"journal-article","created":{"date-parts":[[2025,1,10]],"date-time":"2025-01-10T03:01:30Z","timestamp":1736478090000},"page":"25645-25662","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Learning to communicate using a communication critic and counterfactual reasoning"],"prefix":"10.1007","volume":"37","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9664-9925","authenticated-orcid":false,"given":"Simon","family":"Vanneste","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Astrid","family":"Vanneste","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kevin","family":"Mets","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tom","family":"De Schepper","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ali","family":"Anwar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siegfried","family":"Mercelis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peter","family":"Hellinckx","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,1,10]]},"reference":[{"issue":"7540","key":"10598_CR1","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. nature 518(7540):529\u2013533","journal-title":"nature"},{"key":"10598_CR2","unstructured":"Mnih V, Badia AP, Mirza M, Graves A, Lillicrap T, Harley T, Silver D, Kavukcuoglu K (2016) Asynchronous methods for deep reinforcement learning. In: International conference on machine learning, PMLR, pp 1928\u20131937"},{"key":"10598_CR3","volume-title":"Introduction to reinforcement learning","author":"RS Sutton","year":"1998","unstructured":"Sutton RS, Barto AG (1998) Introduction to reinforcement learning, 1st edn. MIT Press, Cambridge, MA","edition":"1"},{"key":"10598_CR4","doi-asserted-by":"crossref","unstructured":"Van\u00a0Hasselt H, Guez A, Silver D (2016) Deep reinforcement learning with double Q-learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 30","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"10598_CR5","unstructured":"Wang Z, Schaul T, Hessel M, Hasselt H, Lanctot M, Freitas N (2016) Dueling network architectures for deep reinforcement learning. In: International conference on machine learning, PMLR, pp 1995\u20132003"},{"issue":"2","key":"10598_CR6","doi-asserted-by":"publisher","first-page":"156","DOI":"10.1109\/TSMCC.2007.913919","volume":"38","author":"L Busoniu","year":"2008","unstructured":"Busoniu L, Babuska R, De Schutter B (2008) A comprehensive survey of multiagent reinforcement learning. IEEE Trans Syst Man Cybern Part C (Appl Rev) 38(2):156\u2013172","journal-title":"IEEE Trans Syst Man Cybern Part C (Appl Rev)"},{"key":"10598_CR7","unstructured":"Foerster J, Assael IA, De\u00a0Freitas N, Whiteson S (2016) Learning to communicate with deep multi-agent reinforcement learning. Adv Neural Inform Process Syst, 29"},{"key":"10598_CR8","unstructured":"Jorge E, K\u00e5geb\u00e4ck M, Johansson FD, Gustavsson E (2016) Learning to play guess who? and inventing a grounded language as a consequence. arXiv:1611.03218"},{"key":"10598_CR9","doi-asserted-by":"crossref","unstructured":"Foerster J, Farquhar G, Afouras T, Nardelli N, Whiteson S (2018) Counterfactual multi-agent policy gradients. In: Proceedings of the AAAI conference on artificial intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"10598_CR10","unstructured":"Chang Y-H, Ho T, Kaelbling L (2003) All learning is local: Multi-agent learning in global reward games. Adv Neural Inform Process Syst, vol 16"},{"key":"10598_CR11","unstructured":"Foerster J, Nardelli N, Farquhar G, Afouras T, Torr PH, Kohli P, Whiteson S (2017) Stabilising experience replay for deep multi-agent reinforcement learning. In: International conference on machine learning, PMLR, pp 1146\u20131155"},{"key":"10598_CR12","unstructured":"Lowe R, Wu YI, Tamar A, Harb J, Pieter\u00a0Abbeel O, Mordatch I (2017) Multi-agent actor-critic for mixed cooperative-competitive environments. Adv Neural Inform Process Syst, vol 30"},{"key":"10598_CR13","unstructured":"Vanneste A, Vanneste S, Mets K, De\u00a0Schepper T, Mercelis S, Hellinckx P An in-depth analysis of discretization methods for communication learning using backpropagation with multi-agent reinforcement learning. Neural Comput. Appl"},{"key":"10598_CR14","doi-asserted-by":"crossref","unstructured":"Mordatch I, Abbeel P (2018) Emergence of grounded compositional language in multi-agent populations. In: Proceedings of the AAAI conference on artificial intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.11492"},{"key":"10598_CR15","unstructured":"Sukhbaatar S, Fergus R et al. (2016) Learning multiagent communication with backpropagation. Adv Neural Inform Process Syst, 29"},{"key":"10598_CR16","unstructured":"Peng P, Yuan Q, Wen Y, Yang Y, Tang Z, Long H, Wang J (2017) Multiagent bidirectionally-coordinated nets for learning to play starcraft combat games. arXiv:abs\/1703.10069"},{"key":"10598_CR17","unstructured":"Mao H, Gong Z, Ni Y, Xiao Z (2017) ACCNet: Actor-Coordinator-Critic Net for \u201cLearning-to-Communicate\u201d with Deep Multi-agent Reinforcement Learning"},{"key":"10598_CR18","unstructured":"Jiang J, Lu Z (2018) Learning attentional communication for multi-agent cooperation. Adv Neural Inform Process Syst 31"},{"key":"10598_CR19","unstructured":"Das A, Gervet T, Romoff J, Batra D, Parikh D, Rabbat M, Pineau J (2019) Tarmac: Targeted multi-agent communication. In: International conference on machine learning, PMLR, pp 1538\u20131546"},{"key":"10598_CR20","first-page":"22069","volume":"33","author":"Z Ding","year":"2020","unstructured":"Ding Z, Huang T, Lu Z (2020) Learning individually inferred communication for multi-agent cooperation. Adv Neural Inf Process Syst 33:22069\u201322079","journal-title":"Adv Neural Inf Process Syst"},{"key":"10598_CR21","doi-asserted-by":"publisher","unstructured":"Ossenkopf M, Jorgensen M, Geihs K (2019) Hierarchical multi-agent deep reinforcement learning to develop long-term coordination. In: Proceedings of the 34th ACM\/SIGAPP Symposium on Applied Computing. SAC \u201919. Association for Computing Machinery, New York, pp 922\u2013929. https:\/\/doi.org\/10.1145\/3297280.3297371","DOI":"10.1145\/3297280.3297371"},{"key":"10598_CR22","doi-asserted-by":"publisher","first-page":"736","DOI":"10.1007\/978-3-030-33509-0_69","volume-title":"Advances on P2P, parallel, grid, cloud and internet computing","author":"S Vanneste","year":"2020","unstructured":"Vanneste S, Vanneste A, Bosmans S, Mercelis S, Hellinckx P (2020) Learning to communicate with multi-agent reinforcement learning using value-decomposition networks. In: Barolli L, Hellinckx P, Natwichai J (eds) Advances on P2P, parallel, grid, cloud and internet computing. Springer, Cham, pp 736\u2013745"},{"key":"10598_CR23","unstructured":"Sunehag P, Lever G, Gruslys A, Czarnecki WM, Zambaldi V, Jaderberg M, Lanctot M, Sonnerat N, Leibo JZ, Tuyls K, Graepel T (2018) Value-decomposition networks for cooperative multi-agent learning based on team reward. In: Proceedings of the 17th international conference on autonomous agents and multiagent systems. AAMAS \u201918, International Foundation for Autonomous Agents and Multiagent Systems, Richland, SC, pp 2085\u20132087"},{"key":"10598_CR24","doi-asserted-by":"publisher","first-page":"40","DOI":"10.1016\/j.neucom.2020.01.079","volume":"390","author":"D Sim\u00f5es","year":"2020","unstructured":"Sim\u00f5es D, Lau N, Paulo Reis L (2020) Multi-agent actor centralized-critic with communication. Neurocomputing 390:40\u201356. https:\/\/doi.org\/10.1016\/j.neucom.2020.01.079","journal-title":"Neurocomputing"},{"key":"10598_CR25","unstructured":"Silver D, Lever G, Heess N, Degris T, Wierstra D, Riedmiller M (2014) Deterministic policy gradient algorithms. In: International conference on machine learning, PMLR, pp 387\u2013395"},{"key":"10598_CR26","unstructured":"Jaques N, Lazaridou A, Hughes E, Gulcehre C, Ortega P, Strouse D, Leibo JZ, De\u00a0Freitas N (2019) Social influence as intrinsic motivation for multi-agent deep reinforcement learning. In: International conference on machine learning, PMLR, pp 3040\u20133049"},{"key":"10598_CR27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-28929-8","volume-title":"A concise introduction to decentralized POMDPs","author":"FA Oliehoek","year":"2016","unstructured":"Oliehoek FA, Amato C (2016) A concise introduction to decentralized POMDPs. Springer, Switzerland"},{"key":"10598_CR28","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Advances in neural information processing systems, 30"},{"key":"10598_CR29","volume-title":"On-line Q-learning using connectionist systems","author":"GA Rummery","year":"1994","unstructured":"Rummery GA, Niranjan M (1994) On-line Q-learning using connectionist systems, vol 37. University of Cambridge, Department of Engineering, Cambridge"},{"key":"10598_CR30","doi-asserted-by":"crossref","unstructured":"Vanneste A, Vanneste S, Mets K, De\u00a0Schepper T, Mercelis S, Latr\u00e9 S, Hellinckx P (2022) An analysis of discretization methods for communication learning with multi-agent reinforcement learning","DOI":"10.1007\/978-3-030-89899-1_20"},{"key":"10598_CR31","unstructured":"Liang E, Liaw R, Nishihara R, Moritz P, Fox R, Goldberg K, Gonzalez JE, Jordan MI, Stoica I (2018) RLlib: Abstractions for distributed reinforcement learning. In: International conference on machine learning (ICML)"},{"key":"10598_CR32","unstructured":"Hu S, Zhong Y, Gao M, Wang W, Dong H, Li Z, Liang X, Chang X, Yang Y (2022) Marllib: Extending rllib for multi-agent reinforcement learning. arXiv:2210.13708"},{"key":"10598_CR33","unstructured":"Papoudakis G, Christianos F, Sch\u00e4fer L, Albrecht SV (2020) Benchmarking multi-agent deep reinforcement learning algorithms in cooperative tasks. arXiv:2006.07869"},{"key":"10598_CR34","unstructured":"Lowe R, Foerster J, Boureau Y-L, Pineau J, Dauphin Y (2019) On the pitfalls of measuring emergent communication. In: Proceedings of the 18th international conference on autonomous agents and multiagent systems, pp 693\u2013701"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-10598-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-024-10598-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-10598-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T05:02:35Z","timestamp":1760850155000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-024-10598-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,10]]},"references-count":34,"journal-issue":{"issue":"31","published-print":{"date-parts":[[2025,11]]}},"alternative-id":["10598"],"URL":"https:\/\/doi.org\/10.1007\/s00521-024-10598-0","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,10]]},"assertion":[{"value":"16 November 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 October 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 January 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}