{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T18:51:46Z","timestamp":1782154306819,"version":"3.54.5"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,5,13]],"date-time":"2026-05-13T00:00:00Z","timestamp":1778630400000},"content-version":"vor","delay-in-days":12,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100003130","name":"Fonds Wetenschappelijk Onderzoek","doi-asserted-by":"publisher","award":["1S12121N"],"award-info":[{"award-number":["1S12121N"]}],"id":[{"id":"10.13039\/501100003130","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003130","name":"Fonds Wetenschappelijk Onderzoek","doi-asserted-by":"publisher","award":["1S94120N"],"award-info":[{"award-number":["1S94120N"]}],"id":[{"id":"10.13039\/501100003130","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1007\/s00521-025-11782-6","type":"journal-article","created":{"date-parts":[[2026,5,13]],"date-time":"2026-05-13T06:16:35Z","timestamp":1778652995000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["An in-depth analysis of discretization methods for communication learning using backpropagation with multi-agent reinforcement learning"],"prefix":"10.1007","volume":"38","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6742-6722","authenticated-orcid":false,"given":"Astrid","family":"Vanneste","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Simon","family":"Vanneste","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tom","family":"De Schepper","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siegfried","family":"Mercelis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peter","family":"Hellinckx","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kevin","family":"Mets","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,13]]},"reference":[{"key":"11782_CR1","volume-title":"Introduction to reinforcement learning","author":"RS Sutton","year":"1998","unstructured":"Sutton RS, Barto AG (1998) Introduction to reinforcement learning, 1st edn. MIT Press, Cambridge, MA, USA","edition":"1"},{"issue":"7540","key":"11782_CR2","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"issue":"7839","key":"11782_CR3","doi-asserted-by":"publisher","first-page":"604","DOI":"10.1038\/s41586-020-03051-4","volume":"588","author":"J Schrittwieser","year":"2020","unstructured":"Schrittwieser J, Antonoglou I, Hubert T, Simonyan K, Sifre L, Schmitt S, Guez A, Lockhart E, Hassabis D, Graepel T et al (2020) Mastering atari, go, chess and shogi by planning with a learned model. Nature 588(7839):604\u2013609","journal-title":"Nature"},{"issue":"7930","key":"11782_CR4","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1038\/s41586-022-05172-4","volume":"610","author":"A Fawzi","year":"2022","unstructured":"Fawzi A, Balog M, Huang A, Hubert T, Romera-Paredes B, Barekatain M, Novikov A, Ruiz R, Schrittwieser FJ, Swirszcz G et al (2022) Discovering faster matrix multiplication algorithms with reinforcement learning. Nature 610(7930):47\u201353","journal-title":"Nature"},{"issue":"2","key":"11782_CR5","doi-asserted-by":"publisher","first-page":"156","DOI":"10.1109\/TSMCC.2007.913919","volume":"38","author":"L Busoniu","year":"2008","unstructured":"Busoniu L, Babuska R, De Schutter B (2008) A comprehensive survey of multiagent reinforcement learning. IEEE Trans Syst Man Cybern Part C (Appl Rev) 38(2):156\u2013172. https:\/\/doi.org\/10.1109\/TSMCC.2007.913919","journal-title":"IEEE Trans Syst Man Cybern Part C (Appl Rev)"},{"issue":"2","key":"11782_CR6","doi-asserted-by":"publisher","first-page":"895","DOI":"10.1007\/s10462-021-09996-w","volume":"55","author":"S Gronauer","year":"2022","unstructured":"Gronauer S, Diepold K (2022) Multi-agent deep reinforcement learning: a survey. Artif Intell Rev 55(2):895\u2013943","journal-title":"Artif Intell Rev"},{"key":"11782_CR7","unstructured":"Hausknecht M, Stone P (2015) Deep recurrent q-learning for partially observable MDPs. In: 2015 AAAi fall symposium series"},{"key":"11782_CR8","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-28929-8","volume-title":"a concise introduction to decentralized POMDPs","author":"FA Oliehoek","year":"2016","unstructured":"Oliehoek FA, Amato C (2016) a concise introduction to decentralized POMDPs. Springer, Switzerland"},{"key":"11782_CR9","doi-asserted-by":"crossref","unstructured":"Tan M (1993) Multi-agent reinforcement learning: independent vs. cooperative agents. In: Proceedings of the tenth international conference on machine learning, pp 330\u2013337","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"11782_CR10","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1007\/978-3-642-34799-3_13","volume-title":"Multi-agent systems","author":"FS Melo","year":"2012","unstructured":"Melo FS, Spaan MTJ, Witwicki SJ (2012) Querypomdp: Pomdp-based communication in multiagent systems. In: Cossentino M, Kaisers M, Tuyls K, Weiss G (eds) Multi-agent systems. Springer, Berlin, pp 189\u2013204"},{"key":"11782_CR11","unstructured":"Sukhbaatar S, Fergus R, et al (2016) Learning multiagent communication with backpropagation. In Proceedings of the 30th International Conference on Neural Information Processing Systems (pp. 2252-2260)"},{"key":"11782_CR12","unstructured":"Foerster J, Assael IA, De Freitas N, Whiteson S (2016) Learning to communicate with deep multi-agent reinforcement learning. In: Advances in neural information processing systems, pp 2137\u20132145"},{"key":"11782_CR13","unstructured":"Lowe R, Wu Y, Tamar A, Harb J, Abbeel P, Mordatch I (2020) Multi-agent actor-critic for mixed cooperative-competitive environments"},{"key":"11782_CR14","doi-asserted-by":"crossref","unstructured":"Mordatch I, Abbeel P (2018) Emergence of grounded compositional language in multi-agent populations. In: Proceedings of the AAAI conference on artificial intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.11492"},{"key":"11782_CR15","unstructured":"Lin T, Huh M, Stauffer C, Lim S-N, Isola P (2021) Learning to ground multi-agent communication with autoencoders"},{"key":"11782_CR16","doi-asserted-by":"crossref","unstructured":"Foerster JN, Farquhar G, Afouras T, Nardelli N, Whiteson S (2018) Counterfactual multi-agent policy gradients. In: Thirty-second AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"11782_CR17","doi-asserted-by":"publisher","first-page":"40","DOI":"10.1016\/j.neucom.2020.01.079","volume":"390","author":"D Sim\u00f5es","year":"2020","unstructured":"Sim\u00f5es D, Lau N, Paulo Reis L (2020) Multi-agent actor centralized-critic with communication. Neurocomputing 390:40\u201356. https:\/\/doi.org\/10.1016\/j.neucom.2020.01.079","journal-title":"Neurocomputing"},{"key":"11782_CR18","unstructured":"Jaques N, Lazaridou A, Hughes E, Gulcehre C, Ortega P, Strouse D, Leibo JZ, De Freitas N (2019) Social influence as intrinsic motivation for multi-agent deep reinforcement learning. In: International conference on machine learning, pp. 3040\u20133049"},{"key":"11782_CR19","unstructured":"Vanneste S, Vanneste A, Mets K, Anwar A, Mercelis S, Latr\u00e9 S, Hellinckx P (2021) Learning to communicate using counterfactual reasoning"},{"issue":"05","key":"11782_CR20","doi-asserted-by":"publisher","first-page":"7160","DOI":"10.1609\/aaai.v34i05.6205","volume":"34","author":"B Freed","year":"2020","unstructured":"Freed B, Sartoretti G, Hu J, Choset H (2020) Communication learning via backpropagation in discrete channels with unknown noise. Proceedings of the AAAI Conference on Artificial Intelligence 34(05):7160\u20137168. https:\/\/doi.org\/10.1609\/aaai.v34i05.6205","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"11782_CR21","unstructured":"Jang E, Gu S, Poole B (2017) Categorical Reparameterization with Gumbel-Softmax"},{"key":"11782_CR22","unstructured":"Maddison CJ, Mnih A, Teh YW (2017) The concrete distribution: a continuous relaxation of discrete random variables. In Proceedings of the international conference on learning Representations. International Conference on Learning Representations"},{"key":"11782_CR23","unstructured":"Bengio Y, L\u00e9onard N, Courville A (2013) Estimating or propagating gradients through stochastic neurons for conditional computation. arXiv preprint arXiv:1308.3432"},{"key":"11782_CR24","unstructured":"Yin P, Lyu J, Zhang S, Osher S, Qi Y, Xin J (2019) Understanding straight-through estimator in training activation quantized neural nets. In International Conference on Learning Representations"},{"key":"11782_CR25","unstructured":"Havrylov S, Titov I (2017) Emergence of language with multi-agent games: learning to communicate with sequences of symbols. In Proceedings of the 31st International Conference on Neural Information Processing Systems (pp. 2146-2156)"},{"key":"11782_CR26","unstructured":"Chrupala G, K\u00e1d\u00e1r \u00c1, Alishahi A (2015) Learning language through pictures. CoRR (2015) arXiv:1506.03694"},{"key":"11782_CR27","unstructured":"Carmeli B, Meir R, Belinkov Y (2022) Emergent quantized communication. arXiv: abs\/2211.02412"},{"key":"11782_CR28","volume-title":"Statistical theory of extreme values and some practical applications: a series of lectures","author":"EJ Gumbel","year":"1954","unstructured":"Gumbel EJ (1954) Statistical theory of extreme values and some practical applications: a series of lectures, vol 33. US Government Printing Office, Washington, DC, USA"},{"key":"11782_CR29","unstructured":"Degris T, White M, Sutton R (2012) Off-policy actor-critic. In: Langford J, Pineau J (eds) Proceedings of the 29th international conference on machine learning (ICML-12). ICML \u201912, Omnipress, New York, NY, USA, pp 457\u2013464"},{"key":"11782_CR30","unstructured":"Liang E, Liaw R, Nishihara R, Moritz P, Fox R, Goldberg K, Gonzalez JE, Jordan MI, Stoica I (2018) RLlib: Abstractions for distributed reinforcement learning. In: International conference on machine learning (ICML)"},{"key":"11782_CR31","unstructured":"Papoudakis G, Christianos F, Sch\u00e4fer L, Albrecht SV (2020) Comparative evaluation of multi-agent deep reinforcement learning algorithms. CoRR arXiv: abs\/2006.07869"},{"key":"11782_CR32","doi-asserted-by":"publisher","first-page":"736","DOI":"10.1007\/978-3-030-33509-0_69","volume-title":"Advances on P2P, parallel, grid, cloud and internet computing","author":"S Vanneste","year":"2020","unstructured":"Vanneste S, Vanneste A, Bosmans S, Mercelis S, Hellinckx P (2020) Learning to communicate with multi-agent reinforcement learning using value-decomposition networks. In: Barolli L, Hellinckx P, Natwichai J (eds) Advances on P2P, parallel, grid, cloud and internet computing. Springer, Cham, pp 736\u2013745"},{"key":"11782_CR33","doi-asserted-by":"crossref","unstructured":"Lowe R, Foerster J, Boureau Y-L, Pineau J, Dauphin Y (2019) On the pitfalls of measuring emergent communication. In: Proceedings of the 18th international conference on autonomous agents and multiagent systems. AAMAS \u201919. International Foundation for Autonomous Agents and Multiagent Systems, Richland, SC, pp 693\u2013701","DOI":"10.65109\/KNVJ7743"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-025-11782-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-025-11782-6","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-025-11782-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T18:25:51Z","timestamp":1782152751000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-025-11782-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":33,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2026,5]]}},"alternative-id":["11782"],"URL":"https:\/\/doi.org\/10.1007\/s00521-025-11782-6","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"16 November 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 October 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"404"}}