{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T15:18:19Z","timestamp":1784906299775,"version":"3.55.0"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2024,12,9]],"date-time":"2024-12-09T00:00:00Z","timestamp":1733702400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,9]],"date-time":"2024-12-09T00:00:00Z","timestamp":1733702400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Science Foundation of China","doi-asserted-by":"crossref","award":["62272426"],"award-info":[{"award-number":["62272426"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100018568","name":"Special Fund Project for Science and Technology Innovation Strategy of Guangdong Province","doi-asserted-by":"publisher","award":["202201150401021"],"award-info":[{"award-number":["202201150401021"]}],"id":[{"id":"10.13039\/501100018568","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1007\/s10489-024-05933-w","type":"journal-article","created":{"date-parts":[[2024,12,9]],"date-time":"2024-12-09T08:30:11Z","timestamp":1733733011000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Multi-agent dual actor-critic framework for reinforcement learning navigation"],"prefix":"10.1007","volume":"55","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3596-6457","authenticated-orcid":false,"given":"Fengguang","family":"Xiong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yaodan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinhe","family":"Kuang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ligang","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xie","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,9]]},"reference":[{"key":"5933_CR1","doi-asserted-by":"publisher","first-page":"1662","DOI":"10.1109\/LRA.2021.3059628","volume":"6","author":"C Jung","year":"2021","unstructured":"Jung C, Shim DH (2021) Incorporating Multi-Context Into the Traversability Map for Urban Autonomous Driving Using Deep Inverse Reinforcement Learning. IEEE Robotics Automation Lett 6:1662\u20131669","journal-title":"IEEE Robotics Automation Lett"},{"key":"5933_CR2","doi-asserted-by":"publisher","first-page":"120495","DOI":"10.1016\/j.eswa.2023.120495","volume":"231","author":"AK Shakya","year":"2023","unstructured":"Shakya AK, Pillai G, Chakrabarty S (2023) Reinforcement learning algorithms: A brief survey. Expert Syst Appl 231:120495","journal-title":"Expert Syst Appl"},{"key":"5933_CR3","doi-asserted-by":"publisher","first-page":"945","DOI":"10.1007\/s10462-021-09997-9","volume":"55","author":"B Singh","year":"2022","unstructured":"Singh B, Kumar R, Singh VP (2022) Reinforcement learning in robotic applications: a comprehensive survey. Artif Intell Rev 55:945\u2013990","journal-title":"Artif Intell Rev"},{"key":"5933_CR4","doi-asserted-by":"publisher","first-page":"1015","DOI":"10.1109\/TCYB.2019.2932203","volume":"51","author":"Z Zhang","year":"2021","unstructured":"Zhang Z, Ong Y-S, Wang D, Xue B (2021) A Collaborative Multiagent Reinforcement Learning Method Based on Policy Gradient Potential. IEEE Trans Cybernetics 51:1015\u20131027","journal-title":"IEEE Trans Cybernetics"},{"key":"5933_CR5","doi-asserted-by":"publisher","first-page":"311","DOI":"10.1109\/TCCN.2021.3130993","volume":"8","author":"M Krouka","year":"2022","unstructured":"Krouka M, Elgabli A, Ben Issaid C, Bennis M (2022) Communication-Efficient and Federated Multi-Agent Reinforcement Learning. IEEE Trans Cognitive Commun Networking 8:311\u2013320","journal-title":"IEEE Trans Cognitive Commun Networking"},{"key":"5933_CR6","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1109\/TNNLS.2021.3089834","volume":"34","author":"Y Jin","year":"2023","unstructured":"Jin Y, Wei S, Yuan J, Zhang X (2023) Hierarchical and Stable Multiagent Reinforcement Learning for Cooperative Navigation Control. IEEE Trans Neural Networks Learning Syst 34:90\u2013103","journal-title":"IEEE Trans Neural Networks Learning Syst"},{"key":"5933_CR7","doi-asserted-by":"publisher","first-page":"174","DOI":"10.1109\/TCYB.2020.3015811","volume":"51","author":"X Wang","year":"2021","unstructured":"Wang X, Ke L, Qiao Z, Chai X (2021) Large-Scale Traffic Signal Control Using a Novel Multiagent Reinforcement Learning. IEEE Trans Cybernetics 51:174\u2013187","journal-title":"IEEE Trans Cybernetics"},{"key":"5933_CR8","doi-asserted-by":"publisher","first-page":"6991","DOI":"10.1109\/TITS.2021.3066366","volume":"23","author":"M Liu","year":"2022","unstructured":"Liu M, Zhao F, Yin J, Niu J, Liu Y (2022) Reinforcement-Tracking: An Effective Trajectory Tracking and Navigation Method for Autonomous Urban Driving. IEEE Trans Intell Transp Syst 23:6991\u20137007","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"5933_CR9","doi-asserted-by":"publisher","first-page":"103955","DOI":"10.1016\/j.trc.2022.103955","volume":"146","author":"H Su","year":"2023","unstructured":"Su H, Zhong YD, Chow JYJ, Dey B, Jin L (2023) EMVLight: A multi-agent reinforcement learning framework for an emergency vehicle decentralized routing and traffic signal control system. Transportation Res Part C-Emerging Technol 146:103955","journal-title":"Transportation Res Part C-Emerging Technol"},{"key":"5933_CR10","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1109\/MSP.2017.2743240","volume":"34","author":"K Arulkumaran","year":"2017","unstructured":"Arulkumaran K, Deisenroth MP, Brundage M, Bharath AA (2017) Deep Reinforcement Learning: A brief survey. IEEE Signal Process Mag 34:26\u201338","journal-title":"IEEE Signal Process Mag"},{"key":"5933_CR11","doi-asserted-by":"publisher","first-page":"1180","DOI":"10.1016\/j.neucom.2007.11.026","volume":"71","author":"J Peters","year":"2008","unstructured":"Peters J, Schaal S (2008) Natural Actor-Critic. Neurocomputing 71:1180\u20131190","journal-title":"Neurocomputing"},{"key":"5933_CR12","doi-asserted-by":"publisher","first-page":"4933","DOI":"10.1109\/TNNLS.2019.2959129","volume":"31","author":"D Wu","year":"2020","unstructured":"Wu D, Dong X, Shen J, Hoi SC (2020) Reducing estimation bias via triplet-average deep deterministic policy gradient. IEEE Trans Neural Networks Learning Syst 31:4933\u20134945","journal-title":"IEEE Trans Neural Networks Learning Syst"},{"key":"5933_CR13","unstructured":"Mnih V (2013) Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602"},{"key":"5933_CR14","doi-asserted-by":"publisher","unstructured":"Tan H (2021) Reinforcement learning with deep deterministic policy gradient, 2021 International Conference on Artificial Intelligence, Big Data and Algorithms (CAIBDA), pp 82\u201385. https:\/\/doi.org\/10.1109\/CAIBDA53561.2021.00025","DOI":"10.1109\/CAIBDA53561.2021.00025"},{"key":"5933_CR15","first-page":"6379","volume":"30","author":"R Lowe","year":"2017","unstructured":"Lowe R, Wu YI, Tamar A, Harb J, Abbeel P, Mordatch I (2017) Multi-agent actor-critic for mixed cooperative-competitive environments. Adv Neural Inf Process Syst 30:6379\u20136390","journal-title":"Adv Neural Inf Process Syst"},{"key":"5933_CR16","doi-asserted-by":"publisher","unstructured":"Iqbal S, Sha F (2019)\u00a0Actor-attention-critic for multi-agent reinforcement learning. In: International Conference on Machine Learning, PMLR, pp 2961\u20132970.\u00a0\u00a0https:\/\/doi.org\/10.48550\/arXiv.1810.02912","DOI":"10.48550\/arXiv.1810.02912"},{"issue":"1","key":"5933_CR17","doi-asserted-by":"publisher","first-page":"660","DOI":"10.1109\/TMC.2022.3233879","volume":"23","author":"B Liu","year":"2023","unstructured":"Liu B, Han W, Wang E, Xiong S, Wu L, Wang Q, Wang J, Qiao C (2023) Multi-agent attention double actor-critic framework for intelligent traffic light control in urban scenarios with hybrid traffic. IEEE Trans Mob Comput 23(1):660\u2013672","journal-title":"IEEE Trans Mob Comput"},{"key":"5933_CR18","doi-asserted-by":"publisher","first-page":"5625","DOI":"10.1007\/s00500-023-09365-5","volume":"28","author":"L Ding","year":"2024","unstructured":"Ding L, Du W, Zhang J, Guo L, Zhang C, Jin D, Ding S (2024) Better value estimation in Q-learning-based multi-agent reinforcement learning. Soft Comput 28:5625\u20135638","journal-title":"Soft Comput"},{"key":"5933_CR19","unstructured":"Barth-Maron G, Hoffman MW, Budden D, Dabney W, Horgan D, Tb D, Muldal A, Heess N, Lillicrap T (2018) Distributed distributional deterministic policy gradients. arXiv preprint arXiv:1804.08617"},{"key":"5933_CR20","doi-asserted-by":"publisher","unstructured":"Bellemare MG, Dabney W, Munos R (2017)\u00a0A distributional perspective on reinforcement learning. In: International Conference on Machine Learning, PMLR, pp 449\u2013458.\u00a0https:\/\/doi.org\/10.48550\/arXiv.1707.06887","DOI":"10.48550\/arXiv.1707.06887"},{"key":"5933_CR21","doi-asserted-by":"publisher","unstructured":"Gu S, Lillicrap T, Sutskever I, Levine S (2016)\u00a0Continuous deep q-learning with model-based acceleration. In: International Conference on Machine Learning, PMLR, pp 2829\u20132838. https:\/\/doi.org\/10.48550\/arXiv.1603.00748","DOI":"10.48550\/arXiv.1603.00748"},{"key":"5933_CR22","unstructured":"Feinberg V, Wan A, Stoica I, Jordan MI, Gonzalez JE, Levine S (2018) Model-based value estimation for efficient model-free reinforcement learning. arXiv preprint arXiv:1803.00101"},{"key":"5933_CR23","doi-asserted-by":"publisher","unstructured":"Horgan D, Quan J, Budden D, Barth-Maron G, Hessel M, Van Hasselt, Silver D (2018) Distributed prioritized experience replay. https:\/\/doi.org\/10.48550\/arXiv.1803.00933","DOI":"10.48550\/arXiv.1803.00933"},{"issue":"16","key":"5933_CR24","first-page":"17453","volume":"38","author":"C Li","year":"2024","unstructured":"Li C, Zhang Y, Wang J, Hu Y, Dong S, Li W, Lv T, Fan C, Gao Y (2024) Optimistic Value Instructors for Cooperative Multi-Agent Reinforcement Learning. Proc AAAI Conf Artificial Intell 38(16):17453\u201317460","journal-title":"Proc AAAI Conf Artificial Intell"},{"key":"5933_CR25","unstructured":"Thrun S, Schwartz A (2014)Issues in using function approximation for reinforcement learning. In: Proceedings of the 1993 connectionist models summer school, Psychology Press, pp 255\u2013263"},{"issue":"6","key":"5933_CR26","first-page":"6971","volume":"37","author":"E Cetin","year":"2023","unstructured":"Cetin E, Celiktutan O (2023) Learning pessimism for reinforcement learning. Proc AAAI Conf Artificial Intell 37(6):6971\u20136979","journal-title":"Proc AAAI Conf Artificial Intell"},{"key":"5933_CR27","doi-asserted-by":"publisher","unstructured":"Fujimoto S, Hoof H, Meger D (2018)\u00a0Addressing function approximation error in actor-critic methods. In: International Conference on Machine Learning, PMLR, pp 1587\u20131596. https:\/\/doi.org\/10.48550\/arXiv.1802.09477","DOI":"10.48550\/arXiv.1802.09477"},{"issue":"8","key":"5933_CR28","doi-asserted-by":"publisher","first-page":"8621","DOI":"10.1609\/aaai.v36i8.20840","volume":"36","author":"W Wei","year":"2022","unstructured":"Wei W, Zhang Y, Liang J, Li L, Li Y (2022) Controlling underestimation bias in reinforcement learning via quasi-median operation. Proc AAAI Conference Artificial Intell 36(8):8621\u20138628","journal-title":"Proc AAAI Conference Artificial Intell"},{"key":"5933_CR29","doi-asserted-by":"publisher","unstructured":"Wagenmaker AJ, Chen Y, Simchowitz M, Du S, Jamieson K (2022)\u00a0First-order regret in reinforcement learning with linear function approximation: A robust estimation approach. In: International Conference on Machine Learning, PMLR, pp 22384\u201322429. https:\/\/doi.org\/10.48550\/arXiv.2112.03432","DOI":"10.48550\/arXiv.2112.03432"},{"key":"5933_CR30","doi-asserted-by":"publisher","unstructured":"Sherman U, Koren T, Mansour Y (2023)\u00a0Improved regret for efficient online reinforcement learning with linear function approximation. In: International Conference on Machine Learning, PMLR, pp 31117\u201331150. https:\/\/doi.org\/10.48550\/arXiv.2301.13087","DOI":"10.48550\/arXiv.2301.13087"},{"key":"5933_CR31","doi-asserted-by":"publisher","first-page":"108012","DOI":"10.1016\/j.engappai.2024.108012","volume":"133","author":"H Ma","year":"2024","unstructured":"Ma H, Zhang H, Tian D, Yue D, Hancke GP (2024) Optimal demand response based dynamic pricing strategy via Multi-Agent Federated Twin Delayed Deep Deterministic policy gradient algorithm. Eng Appl Artif Intell 133:108012","journal-title":"Eng Appl Artif Intell"},{"key":"5933_CR32","doi-asserted-by":"publisher","unstructured":"Sheikh HU, B\u00f6l\u00f6ni L (2020) Multi-Agent Reinforcement Learning for Problems with Combined Individual and Team Reward, 2020 International Joint Conference on Neural Networks (IJCNN), Glasgow, UK, pp 1-8. https:\/\/doi.org\/10.1109\/IJCNN48605.2020.9206879","DOI":"10.1109\/IJCNN48605.2020.9206879"},{"key":"5933_CR33","doi-asserted-by":"publisher","first-page":"1155","DOI":"10.1109\/TNNLS.2019.2919338","volume":"31","author":"S Al-Dabooni","year":"2019","unstructured":"Al-Dabooni S, Wunsch DC (2019) An improved n-step value gradient learning adaptive dynamic programming algorithm for online learning. IEEE Trans Neural Networks Learning Syst 31:1155\u20131169","journal-title":"IEEE Trans Neural Networks Learning Syst"},{"key":"5933_CR34","unstructured":"Ackermann J, Gabler V, Osa T, Sugiyama M (2019) Reducing overestimation bias in multi-agent domains using double centralized critics. arXiv preprint arXiv:1910.01465"},{"key":"5933_CR35","doi-asserted-by":"publisher","first-page":"859","DOI":"10.1007\/s10994-022-06187-8","volume":"112","author":"Q Yang","year":"2023","unstructured":"Yang Q, Sim\u00e3o TD, Tindemans SH, Spaan MT (2023) Safety-constrained reinforcement learning with a distributional safety critic. Mach Learn 112:859\u2013887","journal-title":"Mach Learn"},{"issue":"3","key":"5933_CR36","doi-asserted-by":"publisher","first-page":"4599","DOI":"10.1109\/TASE.2023.3299275","volume":"21","author":"K Wang","year":"2023","unstructured":"Wang K, Mu C, Ni Z, Liu D (2023) Safe reinforcement learning and adaptive optimal control with applications to obstacle avoidance problem. IEEE Trans Autom Sci Eng 21(3):4599\u20134612","journal-title":"IEEE Trans Autom Sci Eng"},{"key":"5933_CR37","doi-asserted-by":"publisher","first-page":"102927","DOI":"10.1016\/j.sysarc.2023.102927","volume":"142","author":"Z Wang","year":"2023","unstructured":"Wang Z, Wen M, Xu Y, Zhou Y, Wang JH, Zhang L (2023) Communication compression techniques in distributed deep learning: A survey. J Syst Architect 142:102927","journal-title":"J Syst Architect"},{"key":"5933_CR38","doi-asserted-by":"publisher","unstructured":"Gerstgrasser M, Danino T, Keren S (2024) Selectively sharing experiences improves multi-agent reinforcement learning. Advances in Neural Information Processing Systems. p 36. https:\/\/doi.org\/10.48550\/arXiv.2311.00865","DOI":"10.48550\/arXiv.2311.00865"},{"key":"5933_CR39","doi-asserted-by":"crossref","unstructured":"Uehara Y, Matumae S (2023) Dimensionality Reduction Methods Using VAE for Deep Reinforcement Learning of Autonomous Driving. In: 2023 Eleventh International Symposium on Computing and Networking Workshops (CANDARW), IEEE, pp 338\u2013342","DOI":"10.1109\/CANDARW60564.2023.00064"},{"issue":"10","key":"5933_CR40","first-page":"11735","volume":"37","author":"Z Xu","year":"2023","unstructured":"Xu Z, Bai Y, Zhang B, Li D, Fan G (2023) Haven: Hierarchical cooperative multi-agent reinforcement learning with dual coordination mechanism. Proc AAAI Conf Artificial Intell 37(10):11735\u201311743","journal-title":"Proc AAAI Conf Artificial Intell"},{"key":"5933_CR41","doi-asserted-by":"publisher","unstructured":"Chen D, Zhang K, Wang Y, Yin X, Li Z, Filev D (2024) Communication-efficient decentralized multi-agent reinforcement learning for cooperative adaptive cruise control. IEEE Trans Intell Vehicles. https:\/\/doi.org\/10.1109\/TIV.2024.3368025","DOI":"10.1109\/TIV.2024.3368025"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05933-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-05933-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05933-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T15:05:22Z","timestamp":1737385522000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-05933-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,9]]},"references-count":41,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2025,1]]}},"alternative-id":["5933"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-05933-w","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,9]]},"assertion":[{"value":"14 October 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 December 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"105"}}