{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T15:25:10Z","timestamp":1772119510261,"version":"3.50.1"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,8,7]],"date-time":"2023-08-07T00:00:00Z","timestamp":1691366400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,8,7]],"date-time":"2023-08-07T00:00:00Z","timestamp":1691366400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001459","name":"Ministry of Education - Singapore","doi-asserted-by":"publisher","award":["Academic Research Fund Tier 1"],"award-info":[{"award-number":["Academic Research Fund Tier 1"]}],"id":[{"id":"10.13039\/501100001459","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Agent Multi-Agent Syst"],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1007\/s10458-023-09617-6","type":"journal-article","created":{"date-parts":[[2023,8,7]],"date-time":"2023-08-07T10:01:56Z","timestamp":1691402516000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Full communication memory networks for team-level cooperation learning"],"prefix":"10.1007","volume":"37","author":[{"given":"Yutong","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yizhuo","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guillaume","family":"Sartoretti","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,8,7]]},"reference":[{"key":"9617_CR1","doi-asserted-by":"crossref","unstructured":"Arulkumaran, K., Cully, A., Togelius, J. (2019). Alphastar: An evolutionary computation perspective. In: Proceedings of the Genetic and Evolutionary Computation Conference Companion, pp. 314\u2013315","DOI":"10.1145\/3319619.3321894"},{"key":"9617_CR2","unstructured":"Berner, C., Brockman, G., Chan, B., Cheung, V., Debiak, P., Dennison, C., Farhi, D., Fischer, Q., Hashme, S., Hesse, C., et al. (2019). Dota 2 with large scale deep reinforcement learning. arXiv preprint arXiv:1912.06680"},{"issue":"6","key":"9617_CR3","doi-asserted-by":"publisher","first-page":"4909","DOI":"10.1109\/TITS.2021.3054625","volume":"23","author":"BR Kiran","year":"2021","unstructured":"Kiran, B. R., Sobh, I., Talpaert, V., Mannion, P., Al Sallab, A. A., Yogamani, S., & P\u00e9rez, P. (2021). Deep reinforcement learning for autonomous driving: A survey. IEEE Transactions on Intelligent Transportation Systems., 23(6), 4909.","journal-title":"IEEE Transactions on Intelligent Transportation Systems."},{"key":"9617_CR4","first-page":"1","volume":"2021","author":"S-J Wang","year":"2021","unstructured":"Wang, S.-J., & Chang, S. (2021). Autonomous bus fleet control using multiagent reinforcement learning. Journal of Advanced Transportation, 2021, 1\u20134.","journal-title":"Journal of Advanced Transportation"},{"issue":"2","key":"9617_CR5","doi-asserted-by":"publisher","first-page":"2666","DOI":"10.1109\/LRA.2021.3062803","volume":"6","author":"M Damani","year":"2021","unstructured":"Damani, M., Luo, Z., Wenzel, E., & Sartoretti, G. (2021). Primal $$_2$$: Pathfinding via reinforcement and imitation multi-agent learning-lifelong. IEEE Robotics and Automation Letters, 6(2), 2666\u20132673.","journal-title":"IEEE Robotics and Automation Letters"},{"key":"9617_CR6","doi-asserted-by":"crossref","unstructured":"Sartoretti, G., Wu, Y., Paivine, W., Kumar, T.S., Koenig, S., Choset, H. (2019) Distributed reinforcement learning for multi-robot decentralized collective construction. In: Distributed Autonomous Robotic Systems (DARS 2018), pp. 35\u201349","DOI":"10.1007\/978-3-030-05816-6_3"},{"issue":"4","key":"9617_CR7","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1007\/s43154-022-00091-8","volume":"3","author":"Y Wang","year":"2022","unstructured":"Wang, Y., Damani, M., Wang, P., Cao, Y., & Sartoretti, G. (2022). Distributed reinforcement learning for robot teams: a review. Current Robotics Reports, 3(4), 239\u2013257.","journal-title":"Current Robotics Reports"},{"key":"9617_CR8","unstructured":"Hernandez-Leal, P., Kartal, B., Taylor, M.E. (2018). Is multiagent deep reinforcement learning the answer or the question? a brief survey. learning 21: 22"},{"key":"9617_CR9","unstructured":"Kim, D., Moon, S., Hostallero, D., Kang, W.J., Lee, T., Son, K., Yi, Y. (2019). Learning to schedule communication in multi-agent reinforcement learning. arXiv preprint arXiv:1902.01554"},{"key":"9617_CR10","doi-asserted-by":"publisher","first-page":"7211","DOI":"10.1609\/aaai.v34i05.6211","volume":"34","author":"Y Liu","year":"2020","unstructured":"Liu, Y., Wang, W., Hu, Y., Hao, J., Chen, X., & Gao, Y. (2020). Multi-agent game abstraction via graph attention neural network. Proceedings of the AAAI Conference on Artificial Intelligence, 34, 7211\u20137218.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"9617_CR11","first-page":"102","volume":"31","author":"J Jiang","year":"2018","unstructured":"Jiang, J., & Lu, Z. (2018). Learning attentional communication for multi-agent cooperation. Advances in neural information processing systems, 31, 102.","journal-title":"Advances in neural information processing systems"},{"key":"9617_CR12","unstructured":"Samvelyan, M., Rashid, T., De\u00a0Witt, C.S., Farquhar, G., Nardelli, N., Rudner, T.G., Hung, C.-M., Torr, P.H., Foerster, J., Whiteson, S. (2019). The starcraft multi-agent challenge. arXiv preprint arXiv:1902.04043"},{"key":"9617_CR13","unstructured":"Rashid, T., Samvelyan, M., Schroeder, C., Farquhar, G., Foerster, J, Whiteson, S. (2018). Qmix: Monotonic value function factorisation for deep multi-agent reinforcement learning. In: International Conference on Machine Learning, pp. 4295\u20134304. PMLR"},{"key":"9617_CR14","unstructured":"Sunehag, P., Lever, G., Gruslys, A., Czarnecki, W.M., Zambaldi, V., Jaderberg, M., Lanctot, M., Sonnerat, N., Leibo, J.Z., Tuyls, K., et al. (2017). Value-decomposition networks for cooperative multi-agent learning. arXiv preprint arXiv:1706.05296"},{"key":"9617_CR15","doi-asserted-by":"publisher","first-page":"7160","DOI":"10.1609\/aaai.v34i05.6205","volume":"34","author":"B Freed","year":"2020","unstructured":"Freed, B., Sartoretti, G., Hu, J., & Choset, H. (2020). Communication learning via backpropagation in discrete channels with unknown noise. Proceedings of the AAAI Conference on Artificial Intelligence, 34, 7160\u20137168.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"9617_CR16","first-page":"16","volume":"29","author":"J Foerster","year":"2016","unstructured":"Foerster, J., Assael, I. A., De Freitas, N., & Whiteson, S. (2016). Learning to communicate with deep multi-agent reinforcement learning. Advances in Neural Information Processing Systems, 29, 16.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9617_CR17","first-page":"2016","volume":"29","author":"S Sukhbaatar","year":"2016","unstructured":"Sukhbaatar, S., Fergus, R., et al. (2016). Learning multiagent communication with backpropagation. Advances in Neural Information Processing Systems, 29, 2016.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9617_CR18","unstructured":"Peng, P., Wen, Y., Yang, Y., Yuan, Q., Tang, Z., Long, H., Wang, J. (2017). Multiagent bidirectionally-coordinated nets: Emergence of human-level coordination in learning to play starcraft combat games. arXiv preprint arXiv:1703.10069"},{"key":"9617_CR19","unstructured":"Kong, X., Xin, B., Liu, F., Wang, Y. (2017). Revisiting the master-slave architecture in multi-agent deep reinforcement learning. arXiv preprint arXiv:1712.07305"},{"key":"9617_CR20","unstructured":"Niu, Y., Paleja, R.R., Gombolay, M.C. (2021). Multi-agent graph-attention communication and teaming. In: AAMAS, pp. 964\u2013973"},{"key":"9617_CR21","first-page":"17","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, \u0141, & Polosukhin, I. (2017). Attention is all you need. Advances in Neural Information Processing Systems, 30, 17.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9617_CR22","unstructured":"Li, W., Luo, H., Lin, Z., Zhang, C., Lu, Z., Ye, D. (2023). A survey on transformers in reinforcement learning. arXiv preprint arXiv:2301.03044"},{"key":"9617_CR23","unstructured":"Parisotto, E., Song, F., Rae, J., Pascanu, R., Gulcehre, C., Jayakumar, S., Jaderberg, M., Kaufman, R.L., Clark, A., Noury, S., et al. (2020). Stabilizing transformers for reinforcement learning. In: International Conference on Machine Learning, pp. 7487\u20137498. PMLR"},{"key":"9617_CR24","unstructured":"Cao, Y., Wang, Y., Vashisth, A., Fan, H., Sartoretti, G.A. (2022). CAtNIPP: Context-aware attention-based network for informative path planning. In: 6th Annual Conference on Robot Learning. https:\/\/openreview.net\/forum?id=cAIIbdNAeNa"},{"key":"9617_CR25","doi-asserted-by":"crossref","unstructured":"Cao, Y., Hou, T., Wang, Y., Yi, X., Sartoretti, G. (2023). Ariadne: A reinforcement learning approach using attention-based deep networks for exploration. arXiv preprint arXiv:2301.11575","DOI":"10.1109\/ICRA48891.2023.10160565"},{"key":"9617_CR26","first-page":"15084","volume":"34","author":"L Chen","year":"2021","unstructured":"Chen, L., Lu, K., Rajeswaran, A., Lee, K., Grover, A., Laskin, M., Abbeel, P., Srinivas, A., & Mordatch, I. (2021). Decision transformer: Reinforcement learning via sequence modeling. Advances in Neural Information Processing Systems, 34, 15084\u201315097.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9617_CR27","first-page":"462","volume-title":"European conference on computer vision","author":"J Shang","year":"2022","unstructured":"Shang, J., Kahatapitiya, K., Li, X., & Ryoo, M. S. (2022). Starformer: Transformer with state-action-reward representations for visual reinforcement learning. European conference on computer vision (pp. 462\u2013479). London: Springer."},{"key":"9617_CR28","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O. (2017). Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347"},{"key":"9617_CR29","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J. (2016). Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"9617_CR30","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E. (2016). Layer normalization. arXiv preprint arXiv:1607.06450."},{"issue":"13","key":"9617_CR31","doi-asserted-by":"publisher","first-page":"11352","DOI":"10.1609\/aaai.v35i13.17353","volume":"35","author":"J Su","year":"2021","unstructured":"Su, J., Adams, S., & Beling, P. (2021). Value-decomposition multi-agent actor-critics. Proceedings of the AAAI Conference on Artificial Intelligence, 35(13), 11352\u201311360. https:\/\/doi.org\/10.1609\/aaai.v35i13.17353","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"9617_CR32","unstructured":"Yu, C., Velu, A., Vinitsky, E., Wang, Y., Bayen, A., Wu, Y. (2021). The surprising effectiveness of ppo in cooperative, multi-agent games. arXiv preprint arXiv:2103.01955."},{"key":"9617_CR33","unstructured":"Hu, J., Jiang, S., Harding, S.A., Wu, H., Liao, S.-w. (2021). Rethinking the implementation tricks and monotonicity constraint in cooperative multi-agent reinforcement learning. arXiv preprint arXiv:2102.03479 ."},{"key":"9617_CR34","unstructured":"Sunehag, P., Lever, G., Gruslys, A., Czarnecki, W.M., Zambaldi, V., Jaderberg, M., Lanctot, M., Sonnerat, N., Leibo, J.Z., Tuyls, K., et al. (2018). Value-decomposition networks for cooperative multi-agent learning based on team reward. In: Proceedings of the 17th International Conference on Autonomous Agents and MultiAgent Systems, pp. 2085\u20132087."},{"key":"9617_CR35","first-page":"3123","volume":"28","author":"M Courbariaux","year":"2015","unstructured":"Courbariaux, M., Bengio, Y., & David, J.-P. (2015). Binaryconnect: Training deep neural networks with binary weights during propagations. Advances in Neural Information Processing Systems, 28, 3123\u20133131.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"3","key":"9617_CR36","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1007\/BF00992696","volume":"8","author":"RJ Williams","year":"1992","unstructured":"Williams, R. J. (1992). Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine Learning, 8(3), 229\u2013256.","journal-title":"Machine Learning"},{"key":"9617_CR37","unstructured":"Toderici, G., O\u2019Malley, S.M., Hwang, S.J., Vincent, D., Minnen, D., Baluja, S., Covell, M., Sukthankar, R. (2015). Variable rate image compression with recurrent neural networks. arXiv preprint arXiv:1511.06085."}],"container-title":["Autonomous Agents and Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-023-09617-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10458-023-09617-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-023-09617-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,17]],"date-time":"2023-12-17T23:50:46Z","timestamp":1702857046000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10458-023-09617-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,7]]},"references-count":37,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,12]]}},"alternative-id":["9617"],"URL":"https:\/\/doi.org\/10.1007\/s10458-023-09617-6","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-2563058\/v1","asserted-by":"object"}]},"ISSN":["1387-2532","1573-7454"],"issn-type":[{"value":"1387-2532","type":"print"},{"value":"1573-7454","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,8,7]]},"assertion":[{"value":"18 July 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 August 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant conflict of interest\/competing interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest\/Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"All authors approved the paper to be published","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"Code will be made available publicly upon paper acceptance.","order":6,"name":"Ethics","group":{"name":"EthicsHeading","label":"Code availability"}}],"article-number":"33"}}