{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,5]],"date-time":"2026-02-05T08:57:39Z","timestamp":1770281859221,"version":"3.49.0"},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"23","license":[{"start":{"date-parts":[[2023,9,25]],"date-time":"2023-09-25T00:00:00Z","timestamp":1695600000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,9,25]],"date-time":"2023-09-25T00:00:00Z","timestamp":1695600000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1007\/s10489-023-05007-3","type":"journal-article","created":{"date-parts":[[2023,9,25]],"date-time":"2023-09-25T11:02:22Z","timestamp":1695639742000},"page":"28207-28225","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Reinforcement learning for multi-agent formation navigation with scalability"],"prefix":"10.1007","volume":"53","author":[{"given":"Yalei","family":"Gong","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1855-2845","authenticated-orcid":false,"given":"Hongyun","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"MengMeng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haibo","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaohong","family":"Nian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,9,25]]},"reference":[{"issue":"1","key":"5007_CR1","doi-asserted-by":"publisher","first-page":"1479","DOI":"10.1109\/JSYST.2019.2917786","volume":"14","author":"J Wu","year":"2019","unstructured":"Wu J, Wang H, Li N, Su Z (2019) Formation obstacle avoidance: a fluidbased solution. IEEE Syst J 14(1):1479\u20131490. https:\/\/doi.org\/10.1109\/JSYST.2019.2917786","journal-title":"IEEE Syst J"},{"key":"5007_CR2","doi-asserted-by":"publisher","unstructured":"Li H, Zhao T, Dian S (2022) Prioritized planning algorithm for multirobot collision avoidance based on artificial untraversable vertex. Appl Intell 52(1):429\u2013451. https:\/\/doi.org\/10.1007\/s10489-021-02397-0","DOI":"10.1007\/s10489-021-02397-0"},{"issue":"12","key":"5007_CR3","doi-asserted-by":"publisher","first-page":"14413","DOI":"10.1109\/TVT.2020.3034800","volume":"69","author":"J Hu","year":"2020","unstructured":"Hu J, Niu H, Carrasco J, Lennox B, Arvin F (2020) Voronoi-based multirobot autonomous exploration in unknown environments via deep reinforcement learning. IEEE Trans Veh Technol 69(12):14413\u201314423. https:\/\/doi.org\/10.1109\/TVT.2020.3034800","journal-title":"IEEE Trans Veh Technol"},{"issue":"1","key":"5007_CR4","doi-asserted-by":"publisher","first-page":"272","DOI":"10.1109\/LRA.2022.3224667","volume":"8","author":"AH Tan","year":"2022","unstructured":"Tan AH, Bejarano FP, Zhu Y, Ren R, Nejat G (2022) Deep reinforcement learning for decentralized multi-robot exploration with macro actions. IEEE Robot Autom Lett 8(1):272\u2013279. https:\/\/doi.org\/10.1109\/LRA.2022.3224667","journal-title":"IEEE Robot Autom Lett"},{"issue":"2","key":"5007_CR5","doi-asserted-by":"publisher","first-page":"2189","DOI":"10.1007\/s10489-021-02483-3","volume":"52","author":"RJ Alitappeh","year":"2022","unstructured":"Alitappeh RJ, Jeddisaravi K (2022) Multi-robot exploration in task allocation problem. Appl Intell 52(2):2189\u20132211. https:\/\/doi.org\/10.1007\/s10489-021-02483-3","journal-title":"Appl Intell"},{"key":"5007_CR6","doi-asserted-by":"publisher","unstructured":"Okumura K, D\u00e9fago X (2023) Solving simultaneous target assignment and path planning efficiently with time-independent execution. Artif Intell 321:103946. https:\/\/doi.org\/10.1016\/j.artint.2023.103946","DOI":"10.1016\/j.artint.2023.103946"},{"issue":"2","key":"5007_CR7","doi-asserted-by":"publisher","first-page":"997","DOI":"10.1109\/TITS.2020.3019397","volume":"23","author":"F Ho","year":"2020","unstructured":"Ho F, Geraldes R, Gon\u00e7alves A, Rigault B, Sportich B, Kubo D, Cavazza M, Prendinger H (2020) Decentralized multi-agent path finding for uav traffic management. IEEE Trans Intell Transp Syst 23(2):997\u20131008. https:\/\/doi.org\/10.1109\/TITS.2020.3019397","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"5007_CR8","doi-asserted-by":"publisher","first-page":"7350","DOI":"10.1007\/s10489-020-02082-8","volume":"51","author":"W He","year":"2021","unstructured":"He W, Qi X, Liu L (2021) A novel hybrid particle swarm optimization for multi-uav cooperate path planning. Appl Intell 51:7350\u20137364. https:\/\/doi.org\/10.1007\/s10489-020-02082-8","journal-title":"Appl Intell"},{"issue":"4","key":"5007_CR9","doi-asserted-by":"publisher","first-page":"1316","DOI":"10.1007\/s10489-019-01602-5","volume":"50","author":"U Kumar","year":"2020","unstructured":"Kumar U, Banerjee A, Kala R (2020) Collision avoiding decentralized sorting of robotic swarm. Appl Intell 50(4):1316\u20131326. https:\/\/doi.org\/10.1007\/s10489-019-01602-5","journal-title":"Appl Intell"},{"issue":"5","key":"5007_CR10","doi-asserted-by":"publisher","first-page":"1109","DOI":"10.1109\/TRO.2019.2922493","volume":"35","author":"G Sartoretti","year":"2019","unstructured":"Sartoretti G, Paivine W, Shi Y, Wu Y, Choset H (2019) Distributed learning of decentralized control policies for articulated mobile robots. IEEE Trans Robot 35(5):1109\u20131122. https:\/\/doi.org\/10.1109\/TRO.2019.2922493","journal-title":"IEEE Trans Robot"},{"key":"5007_CR11","doi-asserted-by":"publisher","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533. https:\/\/doi.org\/10.1038\/nature14236","DOI":"10.1038\/nature14236"},{"key":"5007_CR12","doi-asserted-by":"publisher","unstructured":"Oroojlooy A, Hajinezhad D (2022) A review of cooperative multi-agent deep reinforcement learning. Appl Intell 1\u201346. https:\/\/doi.org\/10.1007\/s10489-022-04105-y","DOI":"10.1007\/s10489-022-04105-y"},{"issue":"6","key":"5007_CR13","doi-asserted-by":"publisher","first-page":"2358","DOI":"10.1109\/TNNLS.2020.3004893","volume":"32","author":"Z Sui","year":"2020","unstructured":"Sui Z, Pu Z, Yi J, Wu S (2020) Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration. IEEE Trans Neural Netw Learn Syst 32(6):2358\u20132372. https:\/\/doi.org\/10.1109\/TNNLS.2020.3004893","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"5007_CR14","doi-asserted-by":"publisher","unstructured":"Graves A, Graves A (2012) Long short-term memory. Supervised Sequence Labelling with Recurrent Neural Networks 37\u201345. https:\/\/doi.org\/10.1007\/978-3-642-24797-2_4","DOI":"10.1007\/978-3-642-24797-2_4"},{"issue":"8","key":"5007_CR15","doi-asserted-by":"publisher","first-page":"11811","DOI":"10.1109\/TITS.2021.3107336","volume":"23","author":"C Bai","year":"2021","unstructured":"Bai C, Yan P, Pan W, Guo J (2021) Learning-based multi-robot formation control with obstacle avoidance. IEEE Trans Intell Transp Syst 23(8):11811\u201311822. https:\/\/doi.org\/10.1109\/TITS.2021.3107336","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"13","key":"5007_CR16","doi-asserted-by":"publisher","first-page":"15600","DOI":"10.1007\/s10489-022-03191-2","volume":"52","author":"Z Zhou","year":"2022","unstructured":"Zhou Z, Zhu P, Zeng Z, Xiao J, Lu H, Zhou Z (2022) Robot navigation in a crowd by integrating deep reinforcement learning and online planning. Appl Intell 52(13):15600\u201315616. https:\/\/doi.org\/10.1007\/s10489-022-03191-2","journal-title":"Appl Intell"},{"key":"5007_CR17","doi-asserted-by":"publisher","unstructured":"Long P, Fan T, Liao X, Liu W, Zhang H, Pan J (2018) Towards optimally decentralized multi-robot collision avoidance via deep reinforcement learning. In: 2018 IEEE international conference on Robotics and automation (ICRA). IEEE, pp 6252\u20136259. https:\/\/doi.org\/10.1109\/ICRA.2018.8461113","DOI":"10.1109\/ICRA.2018.8461113"},{"issue":"7","key":"5007_CR18","doi-asserted-by":"publisher","first-page":"856","DOI":"10.1177\/0278364920916531","volume":"39","author":"T Fan","year":"2020","unstructured":"Fan T, Long P, Liu W, Pan J (2020) Distributed multi-robot collision avoidance via deep reinforcement learning for navigation in complex scenarios. Int J Robot Res 39(7):856\u2013892. https:\/\/doi.org\/10.1177\/0278364920916531","journal-title":"Int J Robot Res"},{"key":"5007_CR19","doi-asserted-by":"publisher","unstructured":"Sunehag PGA, Lever G (2018) Value-decomposition networks for cooperative multi-agent learning. Proceedings of the 17th international conference on autonomous agents and multiagent systems. Richland, USA: IFAAMAS, pp 2085\u20132087. https:\/\/doi.org\/10.48550\/arXiv.1706.05296","DOI":"10.48550\/arXiv.1706.05296"},{"issue":"1","key":"5007_CR20","doi-asserted-by":"publisher","first-page":"7234","DOI":"10.5555\/3455716.3455894","volume":"21","author":"T Rashid","year":"2020","unstructured":"Rashid T, Samvelyan M, De Witt CS, Farquhar G, Foerster J, Whiteson S (2020) Monotonic value function factorisation for deep multi-agent reinforcement learning. J Mach Learn Res 21(1):7234\u20137284. https:\/\/doi.org\/10.5555\/3455716.3455894","journal-title":"J Mach Learn Res"},{"key":"5007_CR21","doi-asserted-by":"publisher","unstructured":"Li C, He Z, Wang B, Wang Z, Li L (2022) Multi-agent reinforcement learning algorithm based on local information. In: International conference on autonomous unmanned systems. Springer, pp 3080\u20133091. https:\/\/doi.org\/10.1007\/978-981-99-0479-2_284","DOI":"10.1007\/978-981-99-0479-2_284"},{"key":"5007_CR22","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3089834","author":"Y Jin","year":"2021","unstructured":"Jin Y, Wei S, Yuan J, Zhang X (2021) Hierarchical and stable multiagent reinforcement learning for cooperative navigation control. IEEE Trans Neural Netw Learn Syst. https:\/\/doi.org\/10.1109\/TNNLS.2021.3089834","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"5007_CR23","doi-asserted-by":"publisher","unstructured":"Khan A, Zhang C, Lee DD, Kumar V, Ribeiro A (2018) Scalable centralized deep multi-agent reinforcement learning via policy gradients. https:\/\/doi.org\/10.48550\/arXiv.1805.08776, arXiv:1805.08776","DOI":"10.48550\/arXiv.1805.08776"},{"key":"5007_CR24","doi-asserted-by":"publisher","unstructured":"Khan A, Tolstaya E, Ribeiro A, Kumar V (2020) Graph policy gradients for large scale robot control. In: Conference on robot learning. PMLR, pp 823\u2013834. https:\/\/doi.org\/10.48550\/arXiv.1907.03822","DOI":"10.48550\/arXiv.1907.03822"},{"key":"5007_CR25","doi-asserted-by":"publisher","first-page":"4195","DOI":"10.1007\/s10489-020-01755-8","volume":"50","author":"H Chen","year":"2020","unstructured":"Chen H, Liu Y, Zhou Z, Hu D, Zhang M (2020) Gama: graph attention multi-agent reinforcement learning algorithm for cooperation. Appl Intell 50:4195\u20134205. https:\/\/doi.org\/10.1007\/s10489-020-01755-8","journal-title":"Appl Intell"},{"key":"5007_CR26","doi-asserted-by":"publisher","unstructured":"Lillicrap TP, Hunt JJ, Pritzel A, Heess N, Erez T, Tassa Y, Silver D, Wierstra D (2015) Continuous control with deep reinforcement learning. https:\/\/doi.org\/10.48550\/arXiv.1509.02971, arXiv:1509.02971","DOI":"10.48550\/arXiv.1509.02971"},{"key":"5007_CR27","doi-asserted-by":"publisher","unstructured":"Lowe R, Wu YI, Tamar A, Harb J, Pieter Abbeel O, Mordatch I (2017) Multi-agent actor-critic for mixed cooperative-competitive environments. Adv Neural Inf Process Syst 30. https:\/\/doi.org\/10.48550\/arXiv.1706.02275","DOI":"10.48550\/arXiv.1706.02275"},{"issue":"7","key":"5007_CR28","doi-asserted-by":"publisher","first-page":"5847","DOI":"10.1109\/TIE.2017.2782229","volume":"65","author":"G Wen","year":"2017","unstructured":"Wen G, Chen CP, Liu YJ (2017) Formation control with obstacle avoidance for a class of stochastic multiagent systems. IEEE Trans Ind Electron 65(7):5847\u20135855. https:\/\/doi.org\/10.1109\/TIE.2017.2782229","journal-title":"IEEE Trans Ind Electron"},{"key":"5007_CR29","doi-asserted-by":"publisher","unstructured":"Zhu Y, Li S, Zhang J, Xu X (2021) Combined reinforcement learning via artificial potential field: A case study in pommerman. In: 2021 IEEE international symposium on circuits and systems (ISCAS). IEEE, pp 1\u20135. https:\/\/doi.org\/10.1109\/ISCAS51556.2021.9401286","DOI":"10.1109\/ISCAS51556.2021.9401286"},{"issue":"3","key":"5007_CR30","doi-asserted-by":"publisher","first-page":"5896","DOI":"10.1109\/LRA.2022.3161699","volume":"7","author":"R Han","year":"2022","unstructured":"Han R, Chen S, Wang S, Zhang Z, Gao R, Hao Q, Pan J (2022) Reinforcement learned distributed multi-robot navigation with reciprocal velocity obstacle shaped rewards. IEEE Robot Autom Lett 7(3):5896\u20135903. https:\/\/doi.org\/10.1109\/LRA.2022.3161699","journal-title":"IEEE Robot Autom Lett"},{"key":"5007_CR31","doi-asserted-by":"publisher","unstructured":"Iqbal S, Sha F (2019) Actor-attention-critic for multi-agent reinforcement learning. In: International conference on machine learning. PMLR, pp 2961\u20132970. https:\/\/doi.org\/10.48550\/arXiv.1810.02912","DOI":"10.48550\/arXiv.1810.02912"},{"key":"5007_CR32","doi-asserted-by":"publisher","unstructured":"Yu C, Velu A, Vinitsky E, Gao J, Wang Y, Bayen A, Wu Y (2022) The surprising effectiveness of ppo in cooperative multi-agent games. Adv Neural Inf Process Syst 35:24611\u201324624. https:\/\/doi.org\/10.48550\/arXiv.2103.01955","DOI":"10.48550\/arXiv.2103.01955"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-023-05007-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-023-05007-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-023-05007-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,29]],"date-time":"2023-11-29T14:12:40Z","timestamp":1701267160000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-023-05007-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,25]]},"references-count":32,"journal-issue":{"issue":"23","published-print":{"date-parts":[[2023,12]]}},"alternative-id":["5007"],"URL":"https:\/\/doi.org\/10.1007\/s10489-023-05007-3","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,9,25]]},"assertion":[{"value":"5 September 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 September 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}]}}