{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T02:16:26Z","timestamp":1783044986878,"version":"3.54.6"},"reference-count":45,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62263017"],"award-info":[{"award-number":["62263017"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Advanced Engineering Informatics"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.aei.2026.104844","type":"journal-article","created":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T16:26:15Z","timestamp":1780676775000},"page":"104844","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["Scalable multi-agent path finding through dynamic-aware deep reinforcement learning"],"prefix":"10.1016","volume":"76","author":[{"given":"Lin","family":"Huo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1407-4863","authenticated-orcid":false,"given":"Jianlin","family":"Mao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongjun","family":"San","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiwei","family":"Xuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhigang","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.aei.2026.104844_b1","doi-asserted-by":"crossref","DOI":"10.1109\/LRA.2025.3579647","article-title":"Reinforcement learning for multi-agent path finding in large-scale warehouses via distributed policy evolution","author":"Shi","year":"2025","journal-title":"IEEE Robot. Autom. Lett."},{"issue":"1","key":"10.1016\/j.aei.2026.104844_b2","doi-asserted-by":"crossref","first-page":"7331","DOI":"10.1038\/s41598-025-88305-9","article-title":"Real time task planning for order picking in intelligent logistics warehousing","volume":"15","author":"Zhang","year":"2025","journal-title":"Sci. Rep."},{"key":"10.1016\/j.aei.2026.104844_b3","doi-asserted-by":"crossref","unstructured":"A. Andreychuk, K. Yakovlev, A. Panov, A. Skrynnik, Mapf-gpt: Imitation learning for multi-agent pathfinding at scale, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 39, 2025, pp. 23126\u201323134.","DOI":"10.1609\/aaai.v39i22.34477"},{"issue":"4","key":"10.1016\/j.aei.2026.104844_b4","doi-asserted-by":"crossref","first-page":"6932","DOI":"10.1109\/LRA.2020.3026638","article-title":"Mobile robot path planning in dynamic environments through globally guided reinforcement learning","volume":"5","author":"Wang","year":"2020","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.aei.2026.104844_b5","doi-asserted-by":"crossref","first-page":"57390","DOI":"10.1109\/ACCESS.2024.3392305","article-title":"A comprehensive review on leveraging machine learning for multi-agent path finding","volume":"12","author":"Alkazzi","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.aei.2026.104844_b6","doi-asserted-by":"crossref","unstructured":"H. Ma, T.S. Kumar, S. Koenig, Multi-agent path finding with delay probabilities, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 31, 2017.","DOI":"10.1609\/aaai.v31i1.11035"},{"issue":"1","key":"10.1016\/j.aei.2026.104844_b7","doi-asserted-by":"crossref","first-page":"145","DOI":"10.3934\/mbe.2023008","article-title":"Path planning and collision avoidance methods for distributed multi-robot systems in complex dynamic environments","volume":"20","author":"Yang","year":"2023","journal-title":"Math. Biosci. Eng."},{"key":"10.1016\/j.aei.2026.104844_b8","doi-asserted-by":"crossref","DOI":"10.1016\/j.swevo.2024.101833","article-title":"Evolution of path costs for efficient decentralized multi-agent pathfinding","volume":"93","author":"Farhadi","year":"2025","journal-title":"Swarm Evol. Comput."},{"key":"10.1016\/j.aei.2026.104844_b9","series-title":"Where paths collide: A comprehensive survey of classic and learning-based multi-agent pathfinding","author":"Wang","year":"2025"},{"issue":"2","key":"10.1016\/j.aei.2026.104844_b10","doi-asserted-by":"crossref","first-page":"41","DOI":"10.1007\/s10462-023-10670-6","article-title":"Learning team-based navigation: A review of deep reinforcement learning techniques for multi-agent pathfinding","volume":"57","author":"Chung","year":"2024","journal-title":"Artif. Intell. Rev."},{"key":"10.1016\/j.aei.2026.104844_b11","doi-asserted-by":"crossref","first-page":"269","DOI":"10.1007\/BF01386390","article-title":"A note on two problems in connexion with graphs","volume":"1","author":"Dijkstra","year":"1959","journal-title":"Numer. Math."},{"issue":"3","key":"10.1016\/j.aei.2026.104844_b12","doi-asserted-by":"crossref","first-page":"665","DOI":"10.1007\/s10514-018-9735-4","article-title":"An integer linear programming model for fair multitarget tracking in cooperative multirobot systems","volume":"43","author":"Banfi","year":"2019","journal-title":"Auton. Robots"},{"issue":"5","key":"10.1016\/j.aei.2026.104844_b13","doi-asserted-by":"crossref","first-page":"674","DOI":"10.26599\/TST.2021.9010012","article-title":"Deep reinforcement learning based mobile robot navigation: A review","volume":"26","author":"Zhu","year":"2021","journal-title":"Tsinghua Sci. Technol."},{"key":"10.1016\/j.aei.2026.104844_b14","series-title":"LayeredMAPF: A decomposition of MAPF instance to reduce solving costs","author":"Yao","year":"2024"},{"key":"10.1016\/j.aei.2026.104844_b15","doi-asserted-by":"crossref","first-page":"40","DOI":"10.1016\/j.artint.2014.11.006","article-title":"Conflict-based search for optimal multi-agent pathfinding","volume":"219","author":"Sharon","year":"2015","journal-title":"Artificial Intelligence"},{"key":"10.1016\/j.aei.2026.104844_b16","doi-asserted-by":"crossref","unstructured":"K. Okumura, LACAM: Search-based algorithm for quick multi-agent pathfinding, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 37, 2023, pp. 11655\u201311662.","DOI":"10.1609\/aaai.v37i10.26377"},{"key":"10.1016\/j.aei.2026.104844_b17","series-title":"A comprehensive survey on multi-agent cooperative decision-making: Scenarios, approaches, challenges and perspectives","author":"Jin","year":"2025"},{"key":"10.1016\/j.aei.2026.104844_b18","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.110978","article-title":"Deep reinforcement learning of group consciousness for multi-robot pathfinding","volume":"155","author":"Huo","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.aei.2026.104844_b19","series-title":"Stochastic prediction of multi-agent interactions from partial observations","author":"Sun","year":"2019"},{"key":"10.1016\/j.aei.2026.104844_b20","doi-asserted-by":"crossref","unstructured":"R. Veerapaneni, M.S. Saleem, J. Li, M. Likhachev, Windowed MAPF with Completeness Guarantees, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 39, 2025, pp. 23323\u201323332.","DOI":"10.1609\/aaai.v39i22.34499"},{"issue":"2","key":"10.1016\/j.aei.2026.104844_b21","doi-asserted-by":"crossref","DOI":"10.1007\/s11432-022-3696-5","article-title":"A survey on model-based reinforcement learning","volume":"67","author":"Luo","year":"2024","journal-title":"Sci. China Inf. Sci."},{"key":"10.1016\/j.aei.2026.104844_b22","series-title":"2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems","first-page":"9301","article-title":"Scrimp: Scalable communication for reinforcement-and imitation-learning-based multi-agent pathfinding","author":"Wang","year":"2023"},{"key":"10.1016\/j.aei.2026.104844_b23","doi-asserted-by":"crossref","first-page":"15849","DOI":"10.52202\/068431-1153","article-title":"Plan to predict: Learning an uncertainty-foreseeing model for model-based reinforcement learning","volume":"35","author":"Wu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"2","key":"10.1016\/j.aei.2026.104844_b24","doi-asserted-by":"crossref","first-page":"2666","DOI":"10.1109\/LRA.2021.3062803","article-title":"PRIMAL _2 : Pathfinding via reinforcement and imitation multi-agent learning-lifelong","volume":"6","author":"Damani","year":"2021","journal-title":"IEEE Robot. Autom. Lett."},{"issue":"2","key":"10.1016\/j.aei.2026.104844_b25","doi-asserted-by":"crossref","first-page":"1455","DOI":"10.1109\/LRA.2021.3139145","article-title":"Learning selective communication for multi-agent path finding","volume":"7","author":"Ma","year":"2021","journal-title":"IEEE Robot. Autom. Lett."},{"issue":"10","key":"10.1016\/j.aei.2026.104844_b26","doi-asserted-by":"crossref","first-page":"10233","DOI":"10.1109\/TII.2023.3240585","article-title":"Transformer-based imitative reinforcement learning for multirobot path planning","volume":"19","author":"Chen","year":"2023","journal-title":"IEEE Trans. Ind. Informatics"},{"issue":"3","key":"10.1016\/j.aei.2026.104844_b27","doi-asserted-by":"crossref","first-page":"2378","DOI":"10.1109\/LRA.2019.2903261","article-title":"Primal: Pathfinding via reinforcement and imitation multi-agent learning","volume":"4","author":"Sartoretti","year":"2019","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.aei.2026.104844_b28","series-title":"2020 IEEE\/RSJ International Conference on Intelligent Robots and Systems","first-page":"11748","article-title":"Mapper: Multi-agent path planning with evolutionary reinforcement learning in mixed dynamic environments","author":"Liu","year":"2020"},{"key":"10.1016\/j.aei.2026.104844_b29","series-title":"Value-decomposition networks for cooperative multi-agent learning","author":"Sunehag","year":"2017"},{"key":"10.1016\/j.aei.2026.104844_b30","doi-asserted-by":"crossref","unstructured":"M. Tan, Multi-agent reinforcement learning: Independent vs. cooperative agents, in: Proceedings of the International Conference on Machine Learning, 1993, pp. 330\u2013337.","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"10.1016\/j.aei.2026.104844_b31","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume":"30","author":"Lowe","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"178","key":"10.1016\/j.aei.2026.104844_b32","first-page":"1","article-title":"Monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"21","author":"Rashid","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.aei.2026.104844_b33","series-title":"Robotics Research: The 14th International Symposium ISRR","first-page":"3","article-title":"Reciprocal n-body collision avoidance","author":"Van Den Berg","year":"2011"},{"key":"10.1016\/j.aei.2026.104844_b34","doi-asserted-by":"crossref","unstructured":"P. Hernandez-Leal, B. Kartal, M.E. Taylor, Agent modeling as auxiliary task for deep reinforcement learning, in: Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment, Vol. 15, 2019, pp. 31\u201337.","DOI":"10.1609\/aiide.v15i1.5221"},{"issue":"2","key":"10.1016\/j.aei.2026.104844_b35","first-page":"73","article-title":"A survey on multi-agent reinforcement learning and its application","volume":"3","author":"Ning","year":"2024","journal-title":"J. Autom. Intell."},{"key":"10.1016\/j.aei.2026.104844_b36","series-title":"Trust region policy optimisation in multi-agent reinforcement learning","author":"Kuba","year":"2021"},{"key":"10.1016\/j.aei.2026.104844_b37","doi-asserted-by":"crossref","first-page":"24611","DOI":"10.52202\/068431-1787","article-title":"The surprising effectiveness of ppo in cooperative multi-agent games","volume":"35","author":"Yu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.aei.2026.104844_b38","series-title":"Dealing with non-stationarity in multi-agent deep reinforcement learning","author":"Papoudakis","year":"2019"},{"key":"10.1016\/j.aei.2026.104844_b39","series-title":"Benchmarking multi-agent deep reinforcement learning algorithms in cooperative tasks","author":"Papoudakis","year":"2020"},{"key":"10.1016\/j.aei.2026.104844_b40","series-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"1998"},{"key":"10.1016\/j.aei.2026.104844_b41","series-title":"SGDR: Stochastic Gradient Descent with Warm Restarts","author":"Loshchilov","year":"2016"},{"key":"10.1016\/j.aei.2026.104844_b42","series-title":"Social behavior as a key to learning-based multi-agent pathfinding dilemmas","author":"He","year":"2024"},{"issue":"2","key":"10.1016\/j.aei.2026.104844_b43","doi-asserted-by":"crossref","first-page":"100","DOI":"10.1109\/TSSC.1968.300136","article-title":"A formal basis for the heuristic determination of minimum cost paths","volume":"4","author":"Hart","year":"1968","journal-title":"IEEE Trans. Syst. Sci. Cybern."},{"key":"10.1016\/j.aei.2026.104844_b44","unstructured":"D. Hafner, T. Lillicrap, J. Ba, M. Norouzi, Dream to Control: Learning Behaviors by Latent Imagination, in: Proceedings of the International Conference on Learning Representations, 2020."},{"key":"10.1016\/j.aei.2026.104844_b45","series-title":"Proceedings of the Conference on International Symposium on Combinatorial Search","first-page":"151","article-title":"Multi-agent pathfinding: Definitions, variants, and benchmarks","author":"Stern","year":"2019"}],"container-title":["Advanced Engineering Informatics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1474034626005367?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1474034626005367?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T01:23:20Z","timestamp":1783041800000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1474034626005367"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":45,"alternative-id":["S1474034626005367"],"URL":"https:\/\/doi.org\/10.1016\/j.aei.2026.104844","relation":{},"ISSN":["1474-0346"],"issn-type":[{"value":"1474-0346","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Scalable multi-agent path finding through dynamic-aware deep reinforcement learning","name":"articletitle","label":"Article Title"},{"value":"Advanced Engineering Informatics","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.aei.2026.104844","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104844"}}