{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T12:04:27Z","timestamp":1784289867702,"version":"3.55.0"},"reference-count":151,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,12,28]],"date-time":"2023-12-28T00:00:00Z","timestamp":1703721600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,12,28]],"date-time":"2023-12-28T00:00:00Z","timestamp":1703721600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["12102077"],"award-info":[{"award-number":["12102077"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["12161076"],"award-info":[{"award-number":["12161076"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2241263"],"award-info":[{"award-number":["U2241263"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["DUT22RC(3)010"],"award-info":[{"award-number":["DUT22RC(3)010"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["DUT22LAB305"],"award-info":[{"award-number":["DUT22LAB305"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["DUT22ZD211"],"award-info":[{"award-number":["DUT22ZD211"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s10462-023-10620-2","type":"journal-article","created":{"date-parts":[[2023,12,28]],"date-time":"2023-12-28T05:03:58Z","timestamp":1703739838000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":152,"title":["Deep reinforcement learning-based air combat maneuver decision-making: literature review, implementation tutorial and future direction"],"prefix":"10.1007","volume":"57","author":[{"given":"Xinwei","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yihui","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xichao","family":"Su","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chen","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haijun","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,12,28]]},"reference":[{"key":"10620_CR1","unstructured":"Air Combat Evolution Project Overview. (Air Combat Evolution Project Overview. https:\/\/www.darpa.mil\/program\/air-combat-evolution. 2023\u2013May\u201321"},{"key":"10620_CR2","unstructured":"Air combat reinforcement learning. https:\/\/github.com\/y8107928\/air-combat-Reinforcement-Learning. 2023\u2013May\u201321"},{"key":"10620_CR3","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1007\/3-540-31182-3_15","volume":"10","author":"S Akabari","year":"2005","unstructured":"Akabari S, Menhaj MB, Nikravesh SK (2005) Fuzzy modeling of offensive maneuvers in an air-to-air combat. computational intelligence. Theory Appl 10:171\u2013184. https:\/\/doi.org\/10.1007\/3-540-31182-3_15","journal-title":"Theory Appl"},{"key":"10620_CR4","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2208.12328","author":"F AlMahamid","year":"2022","unstructured":"AlMahamid F, Grolinger K (2022) Autonomous unmanned aerial vehicle navigation using reinforcement learning: a systematic review. Eng Appl Artificial Intell. https:\/\/doi.org\/10.48550\/arXiv.2208.12328","journal-title":"Eng Appl Artificial Intell"},{"key":"10620_CR5","doi-asserted-by":"publisher","first-page":"5649","DOI":"10.1007\/s00521-021-06702-3","volume":"34","author":"MN Alpdemir","year":"2022","unstructured":"Alpdemir MN (2022) Tactical UAV path optimization under radar threat using deep reinforcement learning. Neural Comput Appl 34:5649\u20135664. https:\/\/doi.org\/10.1007\/s00521-021-06702-3","journal-title":"Neural Comput Appl"},{"key":"10620_CR6","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1109\/MSP.2017.2743240","volume":"34","author":"K Arulkumaran","year":"2017","unstructured":"Arulkumaran K, Deisenroth MP, Brundage M, Bharath AA (2017) Deep reinforcement learning: a brief survey. IEEE Signal Process Mag 34:26\u201338. https:\/\/doi.org\/10.1109\/MSP.2017.2743240","journal-title":"IEEE Signal Process Mag"},{"key":"10620_CR7","doi-asserted-by":"publisher","DOI":"10.2514\/6.1987-2393","author":"F Austin","year":"1987","unstructured":"Austin F, Carbone G, Falco M, Hinz H, Lewis M (1987) Automated maneuvering decisions for air-to-air combat. American Institute Aeronaut Astronautics. https:\/\/doi.org\/10.2514\/6.1987-2393","journal-title":"American Institute Aeronaut Astronautics"},{"key":"10620_CR8","doi-asserted-by":"publisher","DOI":"10.2514\/3.20590","author":"F Austin","year":"1991","unstructured":"Austin F, Carbone G, Hinz H, Lewis M, Falco M (1991) Game theory for automated maneuvering during air-to-air combat. J Guid Control Dyn. https:\/\/doi.org\/10.2514\/3.20590","journal-title":"J Guid Control Dyn"},{"key":"10620_CR9","doi-asserted-by":"publisher","first-page":"999","DOI":"10.3390\/electronics10090999","volume":"10","author":"AT Azar","year":"2021","unstructured":"Azar AT, Koubaa A, Ali Mohamed N, Ibrahim HA, Ibrahim ZF, Kazim M, Ammar A, Benjdira B, Khamis AM, Hameed IA, Casalino G (2021) Drone deep reinforcement learning: a review. Electronics 10:999. https:\/\/doi.org\/10.3390\/electronics10090999","journal-title":"Electronics"},{"key":"10620_CR10","doi-asserted-by":"publisher","first-page":"26427","DOI":"10.1109\/ACCESS.2023.3257849","volume":"11","author":"J Bae","year":"2023","unstructured":"Bae J, Jung H, Kim S, Kim S, Kim Y-D (2023) Deep reinforcement learning-based air-to-air combat maneuver generation in a realistic environment. IEEE Access 11:26427\u201326440. https:\/\/doi.org\/10.1109\/ACCESS.2023.3257849","journal-title":"IEEE Access"},{"key":"10620_CR11","doi-asserted-by":"publisher","first-page":"1171","DOI":"10.1109\/OJCOMS.2021.3081996","volume":"2","author":"H Bayerlein","year":"2021","unstructured":"Bayerlein H, Theile M, Caccamo M, Gesbert D (2021) Multi-UAV path planning for wireless data harvesting with deep reinforcement learning. IEEE Open J Commun Soc 2:1171\u20131187. https:\/\/doi.org\/10.1109\/OJCOMS.2021.3081996","journal-title":"IEEE Open J Commun Soc"},{"key":"10620_CR12","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2103.15819","author":"J Bergdahl","year":"2021","unstructured":"Bergdahl J, Gordillo C, Tollmar K, Gissl\u00e9n L (2021) Augmenting automated game testing with deep reinforcement learning. ArXiv. https:\/\/doi.org\/10.48550\/arXiv.2103.15819","journal-title":"ArXiv"},{"key":"10620_CR13","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1912.06680","author":"C Berner","year":"2019","unstructured":"Berner C, Brockman G, Chan B, Cheung V, D\u0119biak P, Dennison C, Farhi D, Fischer Q, Hashme S, Hesse C, J\u00f3zefowicz R, Gray S, Olsson C, Pachocki J, Petrov M, Pinto H, Raiman J, Salimans T, Schlatter J, Zhang S (2019) Dota 2 with large scale deep reinforcement learning. ArXiv. https:\/\/doi.org\/10.48550\/arXiv.1912.06680","journal-title":"ArXiv"},{"key":"10620_CR14","doi-asserted-by":"publisher","first-page":"1510","DOI":"10.1109\/ICTAI.2019.00215","volume":"2019","author":"X Cao","year":"2019","unstructured":"Cao X, Wan H, Lin Y, Han S (2019) High-value prioritized experience replay for off-policy reinforcement learning. IEEE Int Conference Tools with Artificial Intell 2019:1510\u20131514. https:\/\/doi.org\/10.1109\/ICTAI.2019.00215","journal-title":"IEEE Int Conference Tools with Artificial Intell"},{"key":"10620_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2023\/3657814","volume":"2023","author":"Y Cao","year":"2023","unstructured":"Cao Y, Kou Y, Li Z, Xu A (2023) Autonomous maneuver decision of UCAV air combat based on double deep Q network algorithm and stochastic game theory. Int J Aerospace Eng 2023:1\u201320. https:\/\/doi.org\/10.1155\/2023\/3657814","journal-title":"Int J Aerospace Eng"},{"key":"10620_CR16","doi-asserted-by":"publisher","first-page":"1400","DOI":"10.1109\/TNNLS.2020.3042120","volume":"33","author":"R Chai","year":"2020","unstructured":"Chai R, Tsourdos A, Savvaris A, Chai S, Xia Y (2020a) Design and implementation of deep neural network-based control for automatic parking maneuver process. IEEE Trans Neural Net Learn Syst 33:1400\u20131413. https:\/\/doi.org\/10.1109\/TNNLS.2020.3042120","journal-title":"IEEE Trans Neural Net Learn Syst"},{"key":"10620_CR17","doi-asserted-by":"publisher","first-page":"5005","DOI":"10.1109\/TNNLS.2019.2955400","volume":"31","author":"R Chai","year":"2020","unstructured":"Chai R, Tsourdos A, Savvaris A, Chai S, Xia Y, Chen CLP (2020b) Six-DOF spacecraft optimal trajectory planning and real-time attitude control: a deep neural network-based approach. IEEE Trans Neural Net Learn Syst 31:5005\u20135013. https:\/\/doi.org\/10.1109\/TNNLS.2019.2955400","journal-title":"IEEE Trans Neural Net Learn Syst"},{"key":"10620_CR18","doi-asserted-by":"publisher","first-page":"6904","DOI":"10.1109\/TIE.2019.2939934","volume":"67","author":"R Chai","year":"2020","unstructured":"Chai R, Tsourdos A, Savvaris A, Xia Y, Chai S (2020c) Real-time reentry trajectory planning of hypersonic vehicles: a two-step strategy incorporating fuzzy multiobjective transcription and deep neural network. IEEE Trans Industr Electron 67:6904\u20136915. https:\/\/doi.org\/10.1109\/TIE.2019.2939934","journal-title":"IEEE Trans Industr Electron"},{"key":"10620_CR19","doi-asserted-by":"publisher","DOI":"10.1016\/j.paerosci.2021.100696","author":"R Chai","year":"2021","unstructured":"Chai R, Tsourdos A, Savvaris A, Chai S (2021a) Review of advanced guidance and control algorithms for space\/aerospace vehicles. Prog Aerosp Sci. https:\/\/doi.org\/10.1016\/j.paerosci.2021.100696","journal-title":"Prog Aerosp Sci"},{"key":"10620_CR20","doi-asserted-by":"publisher","first-page":"1685","DOI":"10.1109\/TAES.2021.3050645","volume":"57","author":"R Chai","year":"2021","unstructured":"Chai R, Tsourdos A, Savvaris A, Chai S, Xia Y (2021b) Solving constrained trajectory planning problems using biased particle swarm optimization. IEEE Trans Aerosp Electron Syst 57:1685\u20131701. https:\/\/doi.org\/10.1109\/TAES.2021.3050645","journal-title":"IEEE Trans Aerosp Electron Syst"},{"key":"10620_CR21","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2022.110561","author":"R Chai","year":"2022","unstructured":"Chai R, Tsourdos A, Gao H, Chai S, Xia Y (2022a) Attitude tracking control for reentry vehicles using centralised robust model predictive control. Automatica. https:\/\/doi.org\/10.1016\/j.automatica.2022.110561","journal-title":"Automatica"},{"key":"10620_CR22","doi-asserted-by":"publisher","first-page":"4022","DOI":"10.1109\/TIE.2021.3076729","volume":"69","author":"R Chai","year":"2022","unstructured":"Chai R, Tsourdos A, Gao H, Xia Y, Chai S (2022b) Dual-loop tube-based robust model predictive attitude tracking control for spacecraft with system constraints and additive disturbances. IEEE Trans Industr Electron 69:4022\u20134033. https:\/\/doi.org\/10.1109\/TIE.2021.3076729","journal-title":"IEEE Trans Industr Electron"},{"key":"10620_CR23","doi-asserted-by":"publisher","first-page":"4035","DOI":"10.1109\/TII.2022.3168434","volume":"51","author":"R Chai","year":"2022","unstructured":"Chai R, Tsourdos A, Chai S, Xia Y, Savvaris A (2022c) Multi-phase overtaking maneuver planning for autonomous ground vehicles via a desensitized trajectory optimization approach. IEEE Trans Industr Inf 51:4035\u20134049. https:\/\/doi.org\/10.1109\/TII.2022.3168434","journal-title":"IEEE Trans Industr Inf"},{"key":"10620_CR24","doi-asserted-by":"publisher","first-page":"1633","DOI":"10.1109\/TASE.2022.3183610","volume":"20","author":"R Chai","year":"2023","unstructured":"Chai R, Liu D, Liu T, Tsourdos A, Xia Y, Chai S (2023) Deep learning-based trajectory planning and control for autonomous ground vehicle parking maneuver. IEEE Trans Autom Sci Eng 20:1633\u20131647. https:\/\/doi.org\/10.1109\/TASE.2022.3183610","journal-title":"IEEE Trans Autom Sci Eng"},{"key":"10620_CR78","doi-asserted-by":"publisher","first-page":"1712","DOI":"10.1016\/j.asoc.2007.10.011","volume":"8","author":"C Chen","year":"2008","unstructured":"Chen M, Wu Q, Jiang C (2008) A modified ant optimization algorithm for path planning of UCAV. Appl Soft Comput 8:1712\u20131718. https:\/\/doi.org\/10.1016\/j.asoc.2007.10.011","journal-title":"Appl Soft Comput"},{"key":"10620_CR25","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.117448","volume":"203","author":"JB Crumpacker","year":"2022","unstructured":"Crumpacker JB, Robbins MJ, Jenkins PR (2022) An approximate dynamic programming approach for solving an air combat maneuvering problem. Expert Syst Appl 203:117448. https:\/\/doi.org\/10.1016\/j.eswa.2022.117448","journal-title":"Expert Syst Appl"},{"key":"10620_CR26","doi-asserted-by":"publisher","first-page":"1393","DOI":"10.1109\/7.976974","volume":"37","author":"J Cruz","year":"2001","unstructured":"Cruz J, Simaan M, Gacic A, Jiang H, Letelliier B, Li M, Liu Y (2001) Game-theoretic modeling and control of a military air operation. IEEE Trans Aerosp Electron Syst 37:1393\u20131405. https:\/\/doi.org\/10.1109\/7.976974","journal-title":"IEEE Trans Aerosp Electron Syst"},{"key":"10620_CR27","doi-asserted-by":"publisher","first-page":"8613498","DOI":"10.1155\/2021\/8613498","volume":"2021","author":"K Cui","year":"2021","unstructured":"Cui K, Han W, Liu Y, Wang X, Su X, Liu J, Shao X (2021) Model predictive control for automatic carrier landing with time delay. Int J Aerospace Eng 2021:8613498. https:\/\/doi.org\/10.1155\/2021\/8613498","journal-title":"Int J Aerospace Eng"},{"key":"10620_CR28","unstructured":"DARPA AlphaDogfight program overview. (DARPA AlphaDogfight program overview. https:\/\/en.wikipedia.org\/wiki\/DARPA_AlphaDogfight. 2023\u2013May\u201321"},{"key":"10620_CR29","unstructured":"DARPA's Gremlins Program. (DARPA's Gremlins Program. https:\/\/www.darpa.mil\/program\/gremlins. 2023\u2013May\u201321"},{"key":"10620_CR30","unstructured":"Dassault nEUROn. https:\/\/zh.wikipedia.org\/zh-cn. 2023\u2013Aug\u201308"},{"key":"10620_CR31","doi-asserted-by":"publisher","first-page":"4005","DOI":"10.1007\/s12652-022-04467-8","volume":"14","author":"A Din","year":"2022","unstructured":"Din A, Mir I, Faiza SA (2022) Development of reinforced learning based non-linear controller for unmanned aerial vehicle. J Ambient Intell Humaniz Comput 14:4005\u20134022. https:\/\/doi.org\/10.1007\/s12652-022-04467-8","journal-title":"J Ambient Intell Humaniz Comput"},{"key":"10620_CR32","doi-asserted-by":"publisher","DOI":"10.2514\/6.2023-1071","author":"A Din","year":"2023","unstructured":"Din A, Mir I, Gul F, Mir S (2023) Non-linear intelligent control design for unconventional unmanned aerial vehicle. American Institute Aeronautics Astronautics. https:\/\/doi.org\/10.2514\/6.2023-1071","journal-title":"American Institute Aeronautics Astronautics"},{"key":"10620_CR33","doi-asserted-by":"publisher","first-page":"3048","DOI":"10.1007\/s10489-022-03510-7","volume":"53","author":"A Din","year":"2023","unstructured":"Din A, Akhtar S, Maqsood A, Habib M, Mir I (2023b) Modified model free dynamic programming: an augmented approach for unmanned aerial vehicle. Appl Intell 53:3048\u20133068. https:\/\/doi.org\/10.1007\/s10489-022-03510-7","journal-title":"Appl Intell"},{"key":"10620_CR34","doi-asserted-by":"publisher","first-page":"5943","DOI":"10.1177\/0954410019889447","volume":"233","author":"Y Dong","year":"2019","unstructured":"Dong Y, Ai J, Liu J (2019) Guidance and control for own aircraft in the autonomous air combat: a historical review and future prospects. J Aerosp Eng 233:5943\u20135991. https:\/\/doi.org\/10.1177\/0954410019889447","journal-title":"J Aerosp Eng"},{"key":"10620_CR35","unstructured":"European Horizons Program. (European Horizons Program. https:\/\/irp.fas.org\/program\/collect\/uav_roadmap2005.pdf. 2023\u2013May\u201321"},{"key":"10620_CR36","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1007\/s10479-012-1261-8","volume":"222","author":"L Evers","year":"2014","unstructured":"Evers L, Dollevoet T, Barros AI, Monsuur H (2014) Robust UAV mission planning. Ann Oper Res 222:293\u2013315. https:\/\/doi.org\/10.1007\/s10479-012-1261-8","journal-title":"Ann Oper Res"},{"key":"10620_CR37","doi-asserted-by":"publisher","first-page":"1033","DOI":"10.3390\/machines10111033","volume":"10","author":"Z Fan","year":"2022","unstructured":"Fan Z, Xu Y, Kang Y, Luo D (2022) Air combat maneuver decision method based on A3C deep reinforcement learning. MACHINES 10:1033. https:\/\/doi.org\/10.3390\/machines10111033","journal-title":"MACHINES"},{"key":"10620_CR71","doi-asserted-by":"publisher","unstructured":"Fu L, Wang Q, Xu J, Zhou Y, Zhu K (2012) Target assignment and sorting for multi-target attack in multi-aircraft coordinated based on RBF. 2012 Chinese control and decision conference. https:\/\/doi.org\/10.1109\/CCDC.2012.6244311","DOI":"10.1109\/CCDC.2012.6244311"},{"key":"10620_CR38","doi-asserted-by":"publisher","first-page":"3380","DOI":"10.1109\/CCDC.2014.6852760","volume":"2014","author":"L Fu","year":"2014","unstructured":"Fu L, Xie F, Wang D, Meng G (2014) The overview for UAV air-combat decision method. Chinese Control and Decision Conference 2014:3380\u20133384. https:\/\/doi.org\/10.1109\/CCDC.2014.6852760","journal-title":"Chinese Control and Decision Conference"},{"key":"10620_CR39","unstructured":"Future combat air system project overview. https:\/\/en.wikipedia.org\/wiki\/Future_Combat_Air_System#Contractors. 2023\u2013May\u201321"},{"key":"10620_CR40","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2023.106404","author":"X Gao","year":"2023","unstructured":"Gao X, Wang L, Yu X, Su X, Ding Y, Lu C, Peng H, Wang X (2023) Conditional probability based multi-objective cooperative task assignment for heterogeneous UAVs. Eng Appl Artificial Intell. https:\/\/doi.org\/10.1016\/j.engappai.2023.106404","journal-title":"Eng Appl Artificial Intell"},{"key":"10620_CR41","doi-asserted-by":"publisher","first-page":"1291","DOI":"10.1109\/TSMCC.2012.2218595","volume":"42","author":"I Grondman","year":"2012","unstructured":"Grondman I, Busoniu L, Lopes G, Babuska R (2012) A survey of actor-critic reinforcement learning: standard and natural policy gradients. IEEE Trans Syst 42:1291\u20131307. https:\/\/doi.org\/10.1109\/TSMCC.2012.2218595","journal-title":"IEEE Trans Syst"},{"key":"10620_CR42","doi-asserted-by":"publisher","first-page":"160","DOI":"10.3969\/j.issn.1000-1093.2017.01.021","volume":"38","author":"H Guo","year":"2017","unstructured":"Guo H, Hou M, Zhang Q, Tang C (2017) UCAV robust maneuver decision based on statistics principle. Binggong Xuebao\/acta Armamentarii 38:160\u2013167. https:\/\/doi.org\/10.3969\/j.issn.1000-1093.2017.01.021","journal-title":"Binggong Xuebao\/acta Armamentarii"},{"key":"10620_CR43","doi-asserted-by":"publisher","first-page":"479","DOI":"10.1016\/j.cja.2020.05.011","volume":"34","author":"T Guo","year":"2021","unstructured":"Guo T, Jiang N, Li B, Zhu X, Wang Y, Du W (2021) UAV navigation in high dynamic environments: A deep reinforcement learning approach. Chin J Aeronaut 34:479\u2013489. https:\/\/doi.org\/10.1016\/j.cja.2020.05.011","journal-title":"Chin J Aeronaut"},{"key":"10620_CR44","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/IJCNN55064.2022.9892208","volume":"2022","author":"Y Han","year":"2022","unstructured":"Han Y, Piao H, Hou Y, Sun Y, Sun Z, Zhou D, Yang S, Peng X, Fan S (2022) Deep relationship graph reinforcement learning for multi-aircraft air combat. International Joint Conference on Neural Net 2022:1\u20138. https:\/\/doi.org\/10.1109\/IJCNN55064.2022.9892208","journal-title":"International Joint Conference on Neural Net"},{"key":"10620_CR45","doi-asserted-by":"publisher","first-page":"11565","DOI":"10.1109\/TIE.2020.3038072","volume":"11","author":"Z Hou","year":"2021","unstructured":"Hou Z, Fei J, Deng Y, Xu J (2021) Data-Efficient hierarchical reinforcement learning for robotic assembly control applications. IEEE Trans Industr Electron 11:11565\u201311575. https:\/\/doi.org\/10.1109\/TIE.2020.3038072","journal-title":"IEEE Trans Industr Electron"},{"key":"10620_CR46","doi-asserted-by":"publisher","DOI":"10.1155\/2018\/6481635","author":"X Hu","year":"2018","unstructured":"Hu X, Luo P, Zhang X, Wang J (2018) Improved ant colony optimization for weapon-target assignment. Math Prob Eng. https:\/\/doi.org\/10.1155\/2018\/6481635","journal-title":"Math Prob Eng"},{"key":"10620_CR47","doi-asserted-by":"publisher","first-page":"32282","DOI":"10.1109\/ACCESS.2021.3060426","volume":"9","author":"D Hu","year":"2021","unstructured":"Hu D, Yang R, Zuo J, Zhang Z, Wu J, Wang Y (2021) Application of deep reinforcement learning in maneuver planning of beyond-visual-range air combat. IEEE Access 9:32282\u201332297. https:\/\/doi.org\/10.1109\/ACCESS.2021.3060426","journal-title":"IEEE Access"},{"key":"10620_CR48","doi-asserted-by":"publisher","first-page":"467","DOI":"10.3390\/electronics11030467","volume":"11","author":"J Hu","year":"2022","unstructured":"Hu J, Wang L, Hu T, Guo C, Wang Y (2022) Autonomous maneuver decision making of dual-uav cooperative air combat based on deep reinforcement learning. Electronics 11:467. https:\/\/doi.org\/10.3390\/electronics11030467","journal-title":"Electronics"},{"key":"10620_CR49","unstructured":"Hu Z (2020) Research on tactical decision-making of ucav based on deep reinforcement learning. Master of engineering, Harbin Institute of Technology, Shenzhen"},{"key":"10620_CR50","doi-asserted-by":"publisher","first-page":"86","DOI":"10.21629\/JSEE.2018.01.09","volume":"29","author":"C Huang","year":"2018","unstructured":"Huang C, Dong K, Huang H, Tang S (2018) Autonomous air combat maneuver decision using Bayesian inference and moving horizon optimization. J Syst Eng Electron 29:86\u201397. https:\/\/doi.org\/10.21629\/JSEE.2018.01.09","journal-title":"J Syst Eng Electron"},{"key":"10620_CR51","doi-asserted-by":"publisher","unstructured":"Huang C, Wei Z, Yang Y, Ku S, Zhang H (2019) Knowledge acquisition for the air combat based on GWO. In: 2019 International conference on artificial intelligence technologies and applications vol 1325, pp 12\u201378. https:\/\/doi.org\/10.1088\/1742-6596\/1325\/1\/012078","DOI":"10.1088\/1742-6596\/1325\/1\/012078"},{"key":"10620_CR52","doi-asserted-by":"publisher","first-page":"133653","DOI":"10.1109\/ACCESS.2019.2941229","volume":"7","author":"B Jang","year":"2019","unstructured":"Jang B, Kim M, Harerimana G, Kim JW (2019) Q-learning algorithms: a comprehensive classification and applications. IEEE Access 7:133653\u2013133667. https:\/\/doi.org\/10.1109\/ACCESS.2019.2941229","journal-title":"IEEE Access"},{"key":"10620_CR53","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1016\/j.neucom.2019.06.024","volume":"360","author":"N Jiang","year":"2019","unstructured":"Jiang N, Jin S, Zhang C (2019) Hierarchical automatic curriculum learning: Converting a sparse reward navigation task into dense reward. Neurocomputing 360:265\u2013278. https:\/\/doi.org\/10.1016\/j.neucom.2019.06.024","journal-title":"Neurocomputing"},{"key":"10620_CR54","doi-asserted-by":"publisher","first-page":"516","DOI":"10.1109\/YAC57282.2022.10023870","volume":"2022","author":"Y Jiang","year":"2022","unstructured":"Jiang Y, Yu J, Li Q (2022) A novel decision-making algorithm for beyond visual range air combat based on deep reinforcement learning. Youth Academic Annual Conference of Chinese Association of Automation 2022:516\u2013521. https:\/\/doi.org\/10.1109\/YAC57282.2022.10023870","journal-title":"Youth Academic Annual Conference of Chinese Association of Automation"},{"key":"10620_CR55","doi-asserted-by":"publisher","first-page":"92426","DOI":"10.1109\/ACCESS.2022.3202918","volume":"10","author":"X Jing","year":"2022","unstructured":"Jing X, Hou M, Wu G, Ma Z, Tao Z (2022) Research on maneuvering decision algorithm based on improved deep deterministic policy gradient. IEEE Access 10:92426\u201392445. https:\/\/doi.org\/10.1109\/ACCESS.2022.3202918","journal-title":"IEEE Access"},{"key":"10620_CR56","doi-asserted-by":"publisher","DOI":"10.1117\/12718892","author":"J Kaneshige","year":"2007","unstructured":"Kaneshige J, Krishnakumar K (2007) Artificial immune system approach for air combat maneuvering. Intell Comput. https:\/\/doi.org\/10.1117\/12718892","journal-title":"Intell Comput"},{"key":"10620_CR57","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1177\/1687814020936790","volume":"12","author":"C Kim","year":"2020","unstructured":"Kim C, Ji C, Kim BS (2020) Development of a control law to improve the handling qualities for short-range air-to-air combat maneuvers. Adv Mech Eng 12:207\u2013226. https:\/\/doi.org\/10.1177\/1687814020936790","journal-title":"Adv Mech Eng"},{"key":"10620_CR58","doi-asserted-by":"publisher","first-page":"1238","DOI":"10.1177\/0278364913495721","volume":"32","author":"J Kober","year":"2013","unstructured":"Kober J, Bagnell J, Peters J (2013) Reinforcement learning in robotics: a survey. Int J Robot Res 32:1238\u20131274. https:\/\/doi.org\/10.1177\/0278364913495721","journal-title":"Int J Robot Res"},{"key":"10620_CR59","doi-asserted-by":"publisher","first-page":"506","DOI":"10.1109\/ICCA51439.2020.9264567","volume":"2020","author":"W Kong","year":"2020","unstructured":"Kong W, Zhou D, Zhang K, Yang Z (2020) Air combat autonomous maneuver decision for one-on- one within visual range engagement base on robust multi-agent reinforcement learning. IEEE Int Conference Control Automation 2020:506\u2013512. https:\/\/doi.org\/10.1109\/ICCA51439.2020.9264567","journal-title":"IEEE Int Conference Control Automation"},{"key":"10620_CR60","doi-asserted-by":"publisher","DOI":"10.1109\/JSEN.2022.3220324","author":"W Kong","year":"2022","unstructured":"Kong W, Zhou D, Du Y, Zhou Y, Zhao Y (2022a) Reinforcement Learning for Multi-aircraft autonomous air combat in multi-sensor UCAV platform. IEEE Sens J. https:\/\/doi.org\/10.1109\/JSEN.2022.3220324","journal-title":"IEEE Sens J"},{"key":"10620_CR61","doi-asserted-by":"publisher","DOI":"10.1049\/cth2.12413","author":"W Kong","year":"2022","unstructured":"Kong W, Zhou D, Du Y, Zhou Y, Zhao YY (2022b) Hierarchical multi-agent reinforcement learning for multi-aircraft close-range air combat. IET Control Theory Appl. https:\/\/doi.org\/10.1049\/cth2.12413","journal-title":"IET Control Theory Appl"},{"key":"10620_CR62","doi-asserted-by":"publisher","unstructured":"Kumar M, Agrawal K, Dutt V (2019) Modeling Decisions in Collective Risk Social Dilemma Games for Climate Change Using Reinforcement Learning. 2019 IEEE conference on cognitive and computational aspects of situation management. https:\/\/doi.org\/10.1109\/COGSIMA.2019.8724273.","DOI":"10.1109\/COGSIMA.2019.8724273"},{"key":"10620_CR63","doi-asserted-by":"publisher","unstructured":"Lange S, Riedmiller M (2010) Deep auto-encoder neural networks in reinforcement learning. 2010 International Joint Conference on Neural Networks. https:\/\/doi.org\/10.1109\/IJCNN.2010.5596468","DOI":"10.1109\/IJCNN.2010.5596468"},{"key":"10620_CR64","doi-asserted-by":"publisher","first-page":"29064","DOI":"10.1109\/ACCESS.2020.2971780","volume":"8","author":"B Li","year":"2020","unstructured":"Li B, Wu Y (2020) Path planning for uav ground target tracking via deep reinforcement learning. IEEE Access 8:29064\u201329074. https:\/\/doi.org\/10.1109\/ACCESS.2020.2971780","journal-title":"IEEE Access"},{"key":"10620_CR65","doi-asserted-by":"publisher","first-page":"3789","DOI":"10.3390\/rs12223789","volume":"12","author":"B Li","year":"2020","unstructured":"Li B, Gan Z, Chen D, Sergey D (2020a) UAV maneuvering target tracking in uncertain environments based on deep reinforcement learning and meta-learning. Remote Sensing 12:3789. https:\/\/doi.org\/10.3390\/rs12223789","journal-title":"Remote Sensing"},{"key":"10620_CR66","doi-asserted-by":"publisher","first-page":"67887","DOI":"10.1109\/ACCESS.2020.2985576","volume":"8","author":"Y Li","year":"2020","unstructured":"Li Y, Han W, Wang Y (2020b) Deep reinforcement learning with application to air confrontation intelligent decision-making of manned\/unmanned aerial vehicle cooperative system. IEEE Access 8:67887\u201367898. https:\/\/doi.org\/10.1109\/ACCESS.2020.2985576","journal-title":"IEEE Access"},{"key":"10620_CR67","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1049\/cit2.12109","volume":"8","author":"B Li","year":"2022","unstructured":"Li B, Bai S, Gan Z, Liang S, Evgeny N, Yao S (2022a) Autonomous air combat decision-making of UAV based on parallel self-play reinforcement learning. CAAI Trans Intell Technol 8:64\u201381. https:\/\/doi.org\/10.1049\/cit2.12109","journal-title":"CAAI Trans Intell Technol"},{"key":"10620_CR68","doi-asserted-by":"publisher","first-page":"1697","DOI":"10.1016\/j.dt.2021.09.014","volume":"18","author":"Y Li","year":"2022","unstructured":"Li Y, Shi J, Jiang W, Zhang W, Lyu Y (2022b) Autonomous maneuver decision-making for a UCAV in short-range aerial combat based on an MS-DDQN algorithm. Def Technol 18:1697\u20131714. https:\/\/doi.org\/10.1016\/j.dt.2021.09.014","journal-title":"Def Technol"},{"key":"10620_CR69","doi-asserted-by":"publisher","DOI":"10.1049\/cit2.12195","author":"B Li","year":"2023","unstructured":"Li B, Bai S, Liang S, Ma R, Neretin E, Huang J (2023) Manoeuvre decision-making of unmanned aerial vehicles in air combat based on an expert actor-based soft actor critic algorithm. CAAI Trans Intell Technol. https:\/\/doi.org\/10.1049\/cit2.12195","journal-title":"CAAI Trans Intell Technol"},{"key":"10620_CR70","doi-asserted-by":"publisher","first-page":"157","DOI":"10.3390\/drones7030157","volume":"7","author":"S Li","year":"2023","unstructured":"Li S, Wu Q, Du B, Wang Y, Chen M (2023b) Autonomous maneuver decision-making of ucav with incomplete information in human-computer gaming. Drones 7:157. https:\/\/doi.org\/10.3390\/drones7030157","journal-title":"Drones"},{"key":"10620_CR74","doi-asserted-by":"publisher","first-page":"563","DOI":"10.3390\/aerospace9100563","volume":"9","author":"X Liu","year":"2022","unstructured":"Liu X, Yin Y, Su Y, Ming R (2022) A Multi-UCAV cooperative decision-making method based on an MAPPO algorithm for beyond-visual-range air combat. Aerospace 9:563. https:\/\/doi.org\/10.3390\/aerospace9100563","journal-title":"Aerospace"},{"key":"10620_CR75","doi-asserted-by":"publisher","first-page":"3133","DOI":"10.1109\/COMST.2019.2916583","volume":"21","author":"NC Luong","year":"2019","unstructured":"Luong NC, Hoang DT, Gong S, Niyato D, Wang P, Liang Y, Kim DI (2019) Applications of deep reinforcement learning in communications and networking: a survey. IEEE Commun Surveys Tutorials 21:3133\u20133174. https:\/\/doi.org\/10.1109\/COMST.2019.2916583","journal-title":"IEEE Commun Surveys Tutorials"},{"key":"10620_CR76","doi-asserted-by":"publisher","unstructured":"Lyu L, Shen Y, Zhang S (2022) The advance of reinforcement learning and deep reinforcement learning. 2022 IEEE International conference on electrical engineering p 644\u2013648. https:\/\/doi.org\/10.1109\/EEBDA53927.2022.9744760","DOI":"10.1109\/EEBDA53927.2022.9744760"},{"key":"10620_CR77","doi-asserted-by":"publisher","first-page":"773","DOI":"10.1007\/s11370-021-00398-z","volume":"14","author":"EF Morales","year":"2021","unstructured":"Morales EF, Murrieta-Cid R, Becerra I, Esquivel-Basaldua MA (2021) A survey on deep learning and deep reinforcement learning in robotics with a tutorial on deep reinforcement learning. Intel Serv Robot 14:773\u2013805. https:\/\/doi.org\/10.1007\/s11370-021-00398-z","journal-title":"Intel Serv Robot"},{"key":"10620_CR79","unstructured":"MQ-9. https:\/\/zh.wikipedia.org\/zh-cn\/MQ-9. 2023\u2013Aug\u201308"},{"key":"10620_CR80","doi-asserted-by":"publisher","first-page":"3826","DOI":"10.1109\/TCYB.2020.2977374","volume":"50","author":"TT Nguyen","year":"2020","unstructured":"Nguyen TT, Nguyen ND, Nahavandi S (2020) Deep reinforcement learning for multiagent systems: a review of challenges, solutions, and applications. IEEE Trans Cybernet 50:3826\u20133839. https:\/\/doi.org\/10.1109\/TCYB.2020.2977374","journal-title":"IEEE Trans Cybernet"},{"key":"10620_CR81","unstructured":"OFFensive Swarm-Enabled Tactics (OFFSET) program. https:\/\/apps.dtic.mil\/sti\/pdfs\/AD1125864.pdf. 2023\u2013May\u201321"},{"key":"10620_CR82","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2210.07282","author":"M \u00d6zbek","year":"2022","unstructured":"\u00d6zbek M, Y\u0131ld\u0131r\u0131m S, Aksoy M, Kernin E, Koyuncu E (2022) Harfang3D dog-fight sandbox: a reinforcement learning research platform for the customized control tasks of fighter aircrafts. ArXiv. https:\/\/doi.org\/10.48550\/arXiv.2210.07282","journal-title":"ArXiv"},{"key":"10620_CR83","doi-asserted-by":"publisher","first-page":"81","DOI":"10.3390\/a15030081","volume":"15","author":"S Parisi","year":"2022","unstructured":"Parisi S, Tateo D, Hensel M, Eramo CD, Peters J, Pajarinen J (2022) Long-term visitation value for deep exploration in sparse-reward reinforcement learning. Algorithms 15:81. https:\/\/doi.org\/10.3390\/a15030081","journal-title":"Algorithms"},{"key":"10620_CR84","doi-asserted-by":"publisher","first-page":"204","DOI":"10.5139\/IJASS.2016.17.2.204","volume":"17","author":"H Park","year":"2016","unstructured":"Park H, Lee B, Tahk M, Yoo D (2016) Differential game based air combat maneuver generation using scoring function matrix. Int J Aeronautical Space Sci 17:204\u2013213. https:\/\/doi.org\/10.5139\/IJASS.2016.17.2.204","journal-title":"Int J Aeronautical Space Sci"},{"key":"10620_CR85","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/IJCNN48605.2020.9207088","volume":"2020","author":"H Piao","year":"2020","unstructured":"Piao H, Sun Z, Meng G, Chen H, Qu B, Lang K, Sun Y, Yang S, Peng X (2020) Beyond-visual-range air combat tactics auto-generation by reinforcement learning. Int Joint Conference on Neural Net 2020:1\u20138. https:\/\/doi.org\/10.1109\/IJCNN48605.2020.9207088","journal-title":"Int Joint Conference on Neural Net"},{"key":"10620_CR86","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.119285","volume":"215","author":"H Piao","year":"2023","unstructured":"Piao H, Han Y, Chen H, Peng X, Fan S, Sun Y, Liang C, Liu Z, Sun Z, Zhou D (2023) Complex relationship graph abstraction for autonomous air combat collaboration: A learning and expert knowledge hybrid approach. Expert Syst Appl 215:119285. https:\/\/doi.org\/10.1016\/j.eswa.2022.119285","journal-title":"Expert Syst Appl"},{"key":"10620_CR87","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2105.00990","author":"AP Pope","year":"2021","unstructured":"Pope AP, Ide JS, Micovic D, Diaz H, Rosenbluth D, Ritholtz L, Twedt JC, Walker TT, Alcedo K, Javorsek D (2021) Hierarchical reinforcement learning for air-to-air combat. International Conference Unmanned Aircraft Syst. https:\/\/doi.org\/10.48550\/arXiv.2105.00990","journal-title":"International Conference Unmanned Aircraft Syst"},{"key":"10620_CR88","doi-asserted-by":"publisher","first-page":"1057","DOI":"10.1109\/TSMCA.2010.2044997","volume":"40","author":"J Poropudas","year":"2010","unstructured":"Poropudas J, Virtanen K (2010) Game-theoretic validation and analysis of air combat simulation models. IEEE Trans Syst, Man, Cybernet - Part a: Syst Humans 40:1057\u20131070. https:\/\/doi.org\/10.1109\/TSMCA.2010.2044997","journal-title":"IEEE Trans Syst, Man, Cybernet - Part a: Syst Humans"},{"key":"10620_CR89","unstructured":"Russia National Weapons Program. https:\/\/www.foi.se\/rest-api\/report\/FOI-R--4239--SE. 2023\u2013May\u201321"},{"key":"10620_CR90","doi-asserted-by":"publisher","first-page":"146264","DOI":"10.1109\/ACCESS.2019.2943253","volume":"7","author":"H Qie","year":"2019","unstructured":"Qie H, Shi D, Shen T, Xu X, Li Y, Wang L (2019) Joint optimization of multi-UAV target assignment and path planning based on multi-agent reinforcement learning. IEEE Access 7:146264\u2013146272. https:\/\/doi.org\/10.1109\/ACCESS.2019.2943253","journal-title":"IEEE Access"},{"key":"10620_CR91","doi-asserted-by":"publisher","first-page":"5719","DOI":"10.1109\/CAC51589.2020.9327310","volume":"2020","author":"X Qiu","year":"2020","unstructured":"Qiu X, Yao Z, Tan F, Zhu Z, Lu J (2020) One-to-one air-combat maneuver strategy based on improved TD3 algorithm. Chinese Automation Congress 2020:5719\u20135725. https:\/\/doi.org\/10.1109\/CAC51589.2020.9327310","journal-title":"Chinese Automation Congress"},{"key":"10620_CR92","doi-asserted-by":"publisher","first-page":"261","DOI":"10.1023\/A:1011319115230","volume":"7","author":"R Rardin","year":"2001","unstructured":"Rardin R, Uzsoy R (2001) Experimental evaluation of heuristic optimization algorithms: a tutorial. J Heurist 7:261\u2013304. https:\/\/doi.org\/10.1023\/A:1011319115230","journal-title":"J Heurist"},{"key":"10620_CR93","unstructured":"RL air combat. https:\/\/github.com\/Linaom1214\/RL_air-combat. 2023\u2013May\u201321"},{"key":"10620_CR94","doi-asserted-by":"publisher","first-page":"351","DOI":"10.1007\/s10846-018-0891-8","volume":"93","author":"A Rodriguez-Ramos","year":"2019","unstructured":"Rodriguez-Ramos A, Sampedro C, Bavle H, de la Puente P, Campoy P (2019) A deep reinforcement learning strategy for UAV autonomous landing on a moving platform. J Intell Rob Syst 93:351\u2013366. https:\/\/doi.org\/10.1007\/s10846-018-0891-8","journal-title":"J Intell Rob Syst"},{"key":"10620_CR95","doi-asserted-by":"publisher","first-page":"1639","DOI":"10.1109\/JAS.2022.105803","volume":"9","author":"W Ruan","year":"2022","unstructured":"Ruan W, Duan H, Deng Y (2022) Autonomous maneuver decisions via transfer learning pigeon-inspired optimization for UCAVs in dogfight engagements. IEEE\/CAA J Automatica Sinica 9:1639\u20131657. https:\/\/doi.org\/10.1109\/JAS.2022.105803","journal-title":"IEEE\/CAA J Automatica Sinica"},{"key":"10620_CR96","unstructured":"Russia is testing its own 'loyal wingman' drone for its Su-57 stealth fighter. https:\/\/tass.com\/defense\/1012351. 2023\u2013May\u201321"},{"key":"10620_CR97","doi-asserted-by":"publisher","first-page":"322","DOI":"10.3390\/drones7050322","volume":"7","author":"N Sarkar","year":"2023","unstructured":"Sarkar N, Gul S (2023) Artificial intelligence-based autonomous UAV networks: a survey. Drones 7:322. https:\/\/doi.org\/10.3390\/drones7050322","journal-title":"Drones"},{"key":"10620_CR98","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver D, Huang A, Maddison C, Guez A, Sifre L, Driessche G, Schrittwieser J, Antonoglou I, Panneershelvam V, Lanctot M, Dieleman S, Grewe D, Nham J, Kalchbrenner N, Sutskever I, Lillicrap T, Leach M, Kavukcuoglu K, Graepel T, Hassabis D (2016) Mastering the game of go with deep neural networks and tree search. Nature 529:484\u2013489. https:\/\/doi.org\/10.1038\/nature16961","journal-title":"Nature"},{"key":"10620_CR99","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver D, Schrittwieser J, Simonyan K, Antonoglou I, Huang A, Guez A, Hubert T, Baker L, Bolton A, Chen Y, Lillicrap T, Hui F, Sifre L, Driessche G, Graepel T, Hassabis D (2017) Mastering the game of go without human knowledge. Nature 550:354\u2013359. https:\/\/doi.org\/10.1038\/nature24270","journal-title":"Nature"},{"key":"10620_CR100","first-page":"247","volume":"8","author":"R Smith","year":"1995","unstructured":"Smith R, Dike B (1995) Learning novel fighter combat maneuver rules via genetic algorithms. Int J Expert Syst 8:247\u2013276","journal-title":"Int J Expert Syst"},{"key":"10620_CR101","doi-asserted-by":"publisher","DOI":"10.1145\/176567.176571","author":"VS Subrahmanian","year":"1994","unstructured":"Subrahmanian VS (1994) Amalgamating knowledge bases. Association for Comput Machinery. https:\/\/doi.org\/10.1145\/176567.176571","journal-title":"Association for Comput Machinery"},{"key":"10620_CR102","doi-asserted-by":"publisher","first-page":"5596","DOI":"10.1109\/CAC51589.2020.9327613","volume":"2020","author":"Y Sun","year":"2020","unstructured":"Sun Y, Wang X, Wang T, Gao P (2020) Modeling of air-to-air missile dynamic attack zone based on bayesian networks. Chinese Automation Congress 2020:5596\u20135601. https:\/\/doi.org\/10.1109\/CAC51589.2020.9327613","journal-title":"Chinese Automation Congress"},{"key":"10620_CR103","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/ICEET53442.2021.9659753","volume":"2021","author":"S Tasbas","year":"2021","unstructured":"Tasbas S, Aydinli S (2021) 2-D air combat maneuver decision using reinforcement learning. Int Conference Eng Emerg Technol 2021:1\u20136. https:\/\/doi.org\/10.1109\/ICEET53442.2021.9659753","journal-title":"Int Conference Eng Emerg Technol"},{"key":"10620_CR104","doi-asserted-by":"publisher","first-page":"1072","DOI":"10.1016\/j.apenergy.2018.11.002","volume":"235","author":"JR V\u00e1zquez-Canteli","year":"2019","unstructured":"V\u00e1zquez-Canteli JR, Nagy Z (2019) Reinforcement learning for demand response: a review of algorithms and modeling techniques. Appl Energy 235:1072\u20131089. https:\/\/doi.org\/10.1016\/j.apenergy.2018.11.002","journal-title":"Appl Energy"},{"key":"10620_CR105","doi-asserted-by":"publisher","first-page":"1671","DOI":"10.1016\/j.ins.2011.01.001","volume":"181","author":"NA Vien","year":"2011","unstructured":"Vien NA, Yu H, Chung T (2011) Hessian matrix distribution for Bayesian policy gradient reinforcement learning. Inf Sci 181:1671\u20131685. https:\/\/doi.org\/10.1016\/j.ins.2011.01.001","journal-title":"Inf Sci"},{"key":"10620_CR106","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals O, Babuschkin I, Czarnecki WM, Mathieu M, Dudzik A, Chung J, Choi DH, Powell R, Ewalds T, Georgiev P, Oh J, Horgan D, Kroiss M, Danihelka I, Huang A, Sifre L, Cai T, Agapiou JP, Jaderberg M, Vezhnevets AS, Leblond R, Pohlen T, Dalibard V, Budden D, Sulsky Y, Molloy J, Paine TL, Gulcehre C, Wang Z, Pfaff T, Wu Y, Ring R, Yogatama D, W\u00fcnsch D, McKinney K, Smith O, Schaul T, Lillicrap T, Kavukcuoglu K, Hassabis D, Apps C, Silver D (2019) Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature 575:350\u2013354. https:\/\/doi.org\/10.1038\/s41586-019-1724-z","journal-title":"Nature"},{"key":"10620_CR107","doi-asserted-by":"publisher","first-page":"122","DOI":"10.1109\/ICTC55111.2022.9778652","volume":"2022","author":"L Wang","year":"2022","unstructured":"Wang L, Wei H (2022) Research on autonomous decision-making of UCAV based on deep reinforcement learning. Inform Commun Technol Conference 2022:122\u2013126. https:\/\/doi.org\/10.1109\/ICTC55111.2022.9778652","journal-title":"Inform Commun Technol Conference"},{"key":"10620_CR108","doi-asserted-by":"publisher","first-page":"256","DOI":"10.1109\/ICAwST.2011.6163151","volume":"2011","author":"J Wang","year":"2011","unstructured":"Wang J, Zhao X, Zhang Y, Wang B (2011) Cooperative air-defense system of system model based on immune multi-agent for surface warship formation. Int Conference Awareness Sci Technol 2011:256\u2013260. https:\/\/doi.org\/10.1109\/ICAwST.2011.6163151","journal-title":"Int Conference Awareness Sci Technol"},{"key":"10620_CR109","doi-asserted-by":"publisher","first-page":"2184","DOI":"10.1016\/j.engappai.2013.06.016","volume":"26","author":"Y Wang","year":"2013","unstructured":"Wang Y, Li TS, Lin C (2013) Backward Q-learning: the combination of Sarsa algorithm and Q-learning. Eng Appl Artif Intell 26:2184\u20132193. https:\/\/doi.org\/10.1016\/j.engappai.2013.06.016","journal-title":"Eng Appl Artif Intell"},{"key":"10620_CR110","doi-asserted-by":"publisher","DOI":"10.1177\/1687814016674384","author":"Y Wang","year":"2016","unstructured":"Wang Y, Huang C, Tang C (2016) Research on unmanned combat aerial vehicle robust maneuvering decision under incomplete target information. Adv Mech Eng. https:\/\/doi.org\/10.1177\/1687814016674384","journal-title":"Adv Mech Eng"},{"key":"10620_CR111","doi-asserted-by":"publisher","first-page":"6180","DOI":"10.1109\/JIOT.2020.2973193","volume":"7","author":"C Wang","year":"2020","unstructured":"Wang C, Wang J, Wang J, Zhang X (2020a) Deep reinforcement-learning-based autonomous UAV navigation with sparse rewards. IEEE Internet Things J 7:6180\u20136190. https:\/\/doi.org\/10.1109\/JIOT.2020.2973193","journal-title":"IEEE Internet Things J"},{"key":"10620_CR112","doi-asserted-by":"publisher","DOI":"10.1016\/j.ast.2019.105534","volume":"96","author":"M Wang","year":"2020","unstructured":"Wang M, Wang L, Yue T, Liu H (2020b) Influence of unmanned combat aerial vehicle agility on short-range aerial combat effectiveness. Aerosp Sci Technol 96:105534. https:\/\/doi.org\/10.1016\/j.ast.2019.105534","journal-title":"Aerosp Sci Technol"},{"key":"10620_CR113","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2020\/7180639","volume":"2020","author":"Z Wang","year":"2020","unstructured":"Wang Z, Li H, Wu H, Wu Z (2020c) Improving maneuver strategy in air combat by alternate freeze games with a deep reinforcement learning algorithm. Math Probl Eng 2020:1\u201317. https:\/\/doi.org\/10.1155\/2020\/7180639","journal-title":"Math Probl Eng"},{"key":"10620_CR114","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1109\/TCCN.2020.3027695","volume":"7","author":"L Wang","year":"2021","unstructured":"Wang L, Wang K, Pan C, Xu W, Aslam N, Hanzo L (2021a) Multi-agent deep reinforcement learning-based trajectory planning for multi-uav assisted mobile edge computing. IEEE Trans Commun 7:73\u201384. https:\/\/doi.org\/10.1109\/TCCN.2020.3027695","journal-title":"IEEE Trans Commun"},{"key":"10620_CR115","doi-asserted-by":"publisher","first-page":"4555","DOI":"10.1109\/TPAMI.2021.3069908","volume":"44","author":"X Wang","year":"2021","unstructured":"Wang X, Chen Y, Zhu W (2021b) A survey on curriculum learning. IEEE Trans Pattern Anal Mach Intell 44:4555\u20134576. https:\/\/doi.org\/10.1109\/TPAMI.2021.3069908","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"10620_CR116","doi-asserted-by":"publisher","first-page":"238","DOI":"10.1016\/j.dt.2020.11.013","volume":"18","author":"X Wang","year":"2022","unstructured":"Wang X, Peng H, Liu J, Dong X, Zhao X, Lu C (2022) Optimal control based coordinated taxiing path planning and tracking for multiple carrier aircraft on flight deck. Def Technol 18:238\u2013248. https:\/\/doi.org\/10.1016\/j.dt.2020.11.013","journal-title":"Def Technol"},{"key":"10620_CR117","doi-asserted-by":"publisher","first-page":"4857","DOI":"10.1109\/CCDC55256.2022.10033863","volume":"2022","author":"Y Wang","year":"2022","unstructured":"Wang Y, Ren T, Fan Z (2022b) Autonomous maneuver decision of uav based on deep reinforcement learning: comparison of DQN and DDPG. Chinese Control and Decision Conference 2022:4857\u20134860. https:\/\/doi.org\/10.1109\/CCDC55256.2022.10033863","journal-title":"Chinese Control and Decision Conference"},{"key":"10620_CR118","doi-asserted-by":"publisher","first-page":"105792","DOI":"10.1016\/j.engappai.2022.105792","volume":"119","author":"X Wang","year":"2023","unstructured":"Wang X, Li B, Su X, Peng H, Wang L, Lu C, Wang C (2023) Autonomous dispatch trajectory planning on flight deck: a search-resampling-optimization framework. Eng Appl Artificial Intell 119:105792. https:\/\/doi.org\/10.1016\/j.engappai.2022.105792","journal-title":"Eng Appl Artificial Intell"},{"key":"10620_CR119","doi-asserted-by":"publisher","unstructured":"Wang Y, Jiang T, Li Y, Zhang Z (2021) A hierarchical reinforcement learning method on multi UCAV air combat. Society of photo-optical instrumentation engineers 119330K\u2013119337K. https:\/\/doi.org\/10.1117\/12.2615268","DOI":"10.1117\/12.2615268"},{"key":"10620_CR120","doi-asserted-by":"publisher","first-page":"799","DOI":"10.1016\/j.apenergy.2018.03.104","volume":"222","author":"J Wu","year":"2018","unstructured":"Wu J, He H, Peng J, Li Y, Li Z (2018) Continuous reinforcement learning of energy management with deep Q network for a power split hybrid electric bus. Appl Energy 222:799\u2013811. https:\/\/doi.org\/10.1016\/j.apenergy.2018.03.104","journal-title":"Appl Energy"},{"key":"10620_CR121","doi-asserted-by":"publisher","first-page":"238","DOI":"10.3390\/drones6090238","volume":"6","author":"L Wu","year":"2022","unstructured":"Wu L, Wang C, Zhang P, Wei C (2022) Deep reinforcement learning with corrective feedback for autonomous uav landing on a mobile platform. Drones 6:238. https:\/\/doi.org\/10.3390\/drones6090238","journal-title":"Drones"},{"key":"10620_CR122","doi-asserted-by":"publisher","unstructured":"Wu Y, Lei Y, Z Z, Wang Y (2022) Decision modeling and simulation of fighter air-to-ground combat based on reinforcement learning: association for computing machinery 8:102\u2013109. https:\/\/doi.org\/10.1145\/3529446.3529463","DOI":"10.1145\/3529446.3529463"},{"key":"10620_CR123","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2020\/8325498","volume":"2020","author":"Z Xi","year":"2020","unstructured":"Xi Z, Xu A, Kou Y, Li Z, Yang A (2020) Air combat maneuver trajectory prediction model of target based on chaotic theory and IGA-VNN. Math Probl Eng 2020:1\u201323. https:\/\/doi.org\/10.1155\/2020\/8325498","journal-title":"Math Probl Eng"},{"key":"10620_CR124","doi-asserted-by":"publisher","first-page":"498","DOI":"10.23919\/JSEE.2021.000042","volume":"32","author":"Z Xi","year":"2021","unstructured":"Xi Z, An X, Kou Y, Li Z, Yang A (2021) Target maneuver trajectory prediction based on RBF neural network optimized by hybrid algorithm. J Syst Eng Electron 32:498\u2013516. https:\/\/doi.org\/10.23919\/JSEE.2021.000042","journal-title":"J Syst Eng Electron"},{"key":"10620_CR125","doi-asserted-by":"publisher","first-page":"340","DOI":"10.1016\/j.cja.2023.04.020","volume":"36","author":"Z Xi","year":"2023","unstructured":"Xi Z, Yu Y, Kou Y, Li Z, Li Y (2023) An online ensemble semi-supervised classification framework for air combat target maneuver recognition. Chinese J Aeronaut 36:340\u2013360. https:\/\/doi.org\/10.1016\/j.cja.2023.04.020","journal-title":"Chinese J Aeronaut"},{"key":"10620_CR126","doi-asserted-by":"publisher","first-page":"5630","DOI":"10.3390\/s20195630","volume":"20","author":"J Xie","year":"2020","unstructured":"Xie J, Peng X, Wang H, Niu W, Zheng X (2020) UAV autonomous tracking and landing based on deep reinforcement learning strategy. Sensors 20:5630. https:\/\/doi.org\/10.3390\/s20195630","journal-title":"Sensors"},{"key":"10620_CR127","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2017EDP7278","author":"Z Xu","year":"2018","unstructured":"Xu Z, Cao L, Chen X, Li C, Zhang Y, Lai J (2018) Deep reinforcement learning with sarsa and q-learning: a hybrid approach. IEICE Trans Inform Syst. https:\/\/doi.org\/10.1587\/transinf.2017EDP7278","journal-title":"IEICE Trans Inform Syst"},{"key":"10620_CR128","doi-asserted-by":"publisher","first-page":"28","DOI":"10.3390\/drones7010028","volume":"7","author":"D Xu","year":"2023","unstructured":"Xu D, Guo Y, Yu Z, Wang Z, Lan R, Zhao R, Xie X, Long H (2023) PPO-Exp: keeping fixed-wing UAV formation with deep reinforcement learning. Drones 7:28. https:\/\/doi.org\/10.3390\/drones7010028","journal-title":"Drones"},{"key":"10620_CR129","doi-asserted-by":"publisher","first-page":"114","DOI":"10.4028\/www.scientific.net\/AMM.69.114","volume":"69","author":"Y Xuan","year":"2011","unstructured":"Xuan Y, Huang C, Li W (2011) Air combat situation assessment by gray fuzzy bayesian network. Appl Mech Mater 69:114\u2013119. https:\/\/doi.org\/10.4028\/www.scientific.net\/AMM.69.114","journal-title":"Appl Mech Mater"},{"key":"10620_CR130","doi-asserted-by":"publisher","first-page":"43013","DOI":"10.1109\/ACCESS.2022.3168359","volume":"10","author":"J Yan","year":"2022","unstructured":"Yan J, Daobo W, Tingting B, Zongyuan Y (2022) Multi-UAV objective assignment using hungarian fusion genetic algorithm. IEEE Access 10:43013\u201343021. https:\/\/doi.org\/10.1109\/ACCESS.2022.3168359","journal-title":"IEEE Access"},{"key":"10620_CR131","doi-asserted-by":"publisher","first-page":"363","DOI":"10.1109\/ACCESS.2019.2961426","volume":"8","author":"Q Yang","year":"2020","unstructured":"Yang Q, Zhang J, Shi G, Hu J, Wu Y (2020) Maneuver decision of uav in short-range air combat based on deep reinforcement learning. IEEE Access 8:363\u2013378. https:\/\/doi.org\/10.1109\/ACCESS.2019.2961426","journal-title":"IEEE Access"},{"key":"10620_CR132","doi-asserted-by":"publisher","first-page":"2602","DOI":"10.3390\/electronics11162602","volume":"11","author":"K Yang","year":"2022","unstructured":"Yang K, Dong W, Cai M, Jia S, Liu R (2022) UCAV air combat maneuver decisions based on a proximal policy optimization algorithm with situation reward shaping. Electronics 11:2602. https:\/\/doi.org\/10.3390\/electronics11162602","journal-title":"Electronics"},{"key":"10620_CR133","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/DASC55683.2022.9925811","volume":"2022","author":"J Yoo","year":"2022","unstructured":"Yoo J, Seong H, Shim D, Bae J, Kim Y (2022) Deep reinforcement learning-based intelligent agent for autonomous air combat. IEEE\/AIAA Digital Avionics Syst Conference 2022:1\u20139. https:\/\/doi.org\/10.1109\/DASC55683.2022.9925811","journal-title":"IEEE\/AIAA Digital Avionics Syst Conference"},{"key":"10620_CR134","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2020.106490","author":"S You","year":"2020","unstructured":"You S, Diao M, Gao L, Zhang F, Wang H (2020) Target tracking strategy using deep deterministic policy gradient. Appl Soft Comput. https:\/\/doi.org\/10.1016\/j.asoc.2020.106490","journal-title":"Appl Soft Comput"},{"key":"10620_CR135","doi-asserted-by":"publisher","DOI":"10.3390\/drones6030077","author":"X Yu","year":"2022","unstructured":"Yu X, Gao X, Wang L, Wang X, Ding Y, Lu C, Zhang S (2022) Cooperative multi-UAV task assignment in cross-regional joint operations considering ammunition inventory. Drones. https:\/\/doi.org\/10.3390\/drones6030077","journal-title":"Drones"},{"key":"10620_CR136","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2022\/3551508","volume":"2022","author":"L Yue","year":"2022","unstructured":"Yue L, Yang R, Zhang Y, Yu L, Wang Z (2022) Deep reinforcement learning for uav intelligent mission planning. Complexity 2022:1\u201313. https:\/\/doi.org\/10.1155\/2022\/3551508","journal-title":"Complexity"},{"key":"10620_CR72","doi-asserted-by":"publisher","first-page":"1554","DOI":"10.1016\/j.proeng.2012.01.172","volume":"29","author":"Z Zhang","year":"2012","unstructured":"Zhang L, Yuan Z, Liu W (2012) The design of target assignment model based on the reverse mutation ant colony algorithm. Procedia Eng 29:1554\u20131558. https:\/\/doi.org\/10.1016\/j.proeng.2012.01.172","journal-title":"Procedia Eng"},{"key":"10620_CR137","doi-asserted-by":"publisher","first-page":"1421","DOI":"10.23919\/JSEE.2021.000121","volume":"32","author":"J Zhang","year":"2021","unstructured":"Zhang J, Yang Q, Shi G, Lu Y, Wu Y (2021) UAV cooperative air combat maneuver decision based on multi-agent reinforcement learning. J syst Eng Electron 32:1421\u20131438. https:\/\/doi.org\/10.23919\/JSEE.2021.000121","journal-title":"J syst Eng Electron"},{"key":"10620_CR138","doi-asserted-by":"publisher","DOI":"10.3389\/fnbot.2022.996412","author":"H Zhang","year":"2022","unstructured":"Zhang H, Zhou H, Wei Y, Huang C (2022) Autonomous maneuver decision-making method based on reinforcement learning and monte carlo tree search. Front Neurorobot. https:\/\/doi.org\/10.3389\/fnbot.2022.996412","journal-title":"Front Neurorobot"},{"key":"10620_CR139","doi-asserted-by":"publisher","first-page":"10230","DOI":"10.3390\/app122010230","volume":"12","author":"H Zhang","year":"2022","unstructured":"Zhang H, Wei Y, Zhou H, Huang C (2022b) Maneuver decision-making for autonomous air combat based on FRE-PPO. Appl Sci 12:10230. https:\/\/doi.org\/10.3390\/app122010230","journal-title":"Appl Sci"},{"key":"10620_CR140","doi-asserted-by":"publisher","first-page":"1772","DOI":"10.1109\/CCDC.2018.8407414","volume":"2018","author":"K Zhao","year":"2018","unstructured":"Zhao K, Huang C (2018) Air combat situation assessment for UAV based on improved decision tree. Chinese Control and Decision Conference 2018:1772\u20131776. https:\/\/doi.org\/10.1109\/CCDC.2018.8407414","journal-title":"Chinese Control and Decision Conference"},{"key":"10620_CR141","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1016\/j.neunet.2011.09.005","volume":"26","author":"T Zhao","year":"2012","unstructured":"Zhao T, Hachiya H, Niu G, Sugiyama M (2012) Analysis and improvement of policy gradient estimation. Neural Netw 26:118\u2013129. https:\/\/doi.org\/10.1016\/j.neunet.2011.09.005","journal-title":"Neural Netw"},{"key":"10620_CR142","doi-asserted-by":"publisher","first-page":"4546","DOI":"10.3390\/s20164546","volume":"20","author":"W Zhao","year":"2020","unstructured":"Zhao W, Chu H, Miao X, Guo L, Shen H, Zhu C, Zhang F, Liang D (2020a) Research on the multiagent joint proximal policy optimization algorithm controlling cooperative fixed-wing UAV obstacle avoidance. Sensors 20:4546. https:\/\/doi.org\/10.3390\/s20164546","journal-title":"Sensors"},{"key":"10620_CR143","doi-asserted-by":"publisher","DOI":"10.1177\/1729881420905922","author":"Y Zhao","year":"2020","unstructured":"Zhao Y, Chen Y, Zhen Z, Jiang J (2020b) Multi-weapon multi-target assignment based on hybrid genetic algorithm in uncertain environment. Int J Adv Rob Syst. https:\/\/doi.org\/10.1177\/1729881420905922","journal-title":"Int J Adv Rob Syst"},{"key":"10620_CR144","doi-asserted-by":"publisher","first-page":"10595","DOI":"10.3390\/app112210595","volume":"11","author":"W Zhao","year":"2021","unstructured":"Zhao W, Meng Z, Wang K, Zhang J, Lu S (2021) Hierarchical active tracking control for UAVs via deep reinforcement learning. Appl Sci 11:10595. https:\/\/doi.org\/10.3390\/app112210595","journal-title":"Appl Sci"},{"key":"10620_CR145","doi-asserted-by":"publisher","first-page":"2031","DOI":"10.3390\/electronics11132031","volume":"11","author":"X Zhao","year":"2022","unstructured":"Zhao X, Yang R, Zhang Y, Yan M, Yue L (2022) Deep reinforcement learning for intelligent dual-uav reconnaissance mission planning. Electronics 11:2031. https:\/\/doi.org\/10.3390\/electronics11132031","journal-title":"Electronics"},{"key":"10620_CR146","doi-asserted-by":"publisher","first-page":"76","DOI":"10.20517\/ir.2023.04","volume":"3","author":"Z Zheng","year":"2023","unstructured":"Zheng Z, Duan H (2023) UAV maneuver decision-making via deep reinforcement learning for short-range air combat. Intell Robot 3:76\u201394. https:\/\/doi.org\/10.20517\/ir.2023.04","journal-title":"Intell Robot"},{"key":"10620_CR73","doi-asserted-by":"publisher","first-page":"551","DOI":"10.1016\/S1004-4132(07)60128-5","volume":"18","author":"Z Zhong","year":"2007","unstructured":"Zhong L, Tong M, Zhong W, Zhagn S (2007) Sequential maneuvering decisions based on multi-stage influence diagram in air combat. J Syst Eng Electron 18:551\u2013555. https:\/\/doi.org\/10.1016\/S1004-4132(07)60128-5","journal-title":"J Syst Eng Electron"},{"key":"10620_CR147","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2016\/6051046","volume":"2016","author":"Y Zhong","year":"2016","unstructured":"Zhong Y, Yao P, Sun Y, Yang J (2016) Cooperative task allocation method of MCAV\/UCAV formation. Math Probl Eng 2016:1\u20139. https:\/\/doi.org\/10.1155\/2016\/6051046","journal-title":"Math Probl Eng"},{"key":"10620_CR148","doi-asserted-by":"publisher","DOI":"10.1117\/122631651","author":"H Zhou","year":"2022","unstructured":"Zhou H, Zhang X, Zhang Z, Wu F, Liu J, Chen Y (2022) Reinforcement learning technology for air combat confrontation of unmanned aerial vehicle. Soc Photo-Optical Instrument Eng. https:\/\/doi.org\/10.1117\/122631651","journal-title":"Soc Photo-Optical Instrument Eng"},{"key":"10620_CR149","doi-asserted-by":"publisher","unstructured":"Zhou K, Wei R, Xu Z, Zhang Q (2018) A brain like air combat learning system inspired by human learning mechanism. In: 2018 IEEE CSAA guidance navigation and control conference. https:\/\/doi.org\/10.1109\/GNCC42960.2018.9018975","DOI":"10.1109\/GNCC42960.2018.9018975"},{"key":"10620_CR150","doi-asserted-by":"publisher","first-page":"2375","DOI":"10.1109\/JIOT.2017.2759728","volume":"5","author":"J Zhu","year":"2018","unstructured":"Zhu J, Song Y, Jiang D, Song H (2018) A new deep-Q-learning-based transmission scheduling mechanism for the cognitive internet of things. IEEE Int Things 5:2375\u20132385. https:\/\/doi.org\/10.1109\/JIOT.2017.2759728","journal-title":"IEEE Int Things"},{"key":"10620_CR151","doi-asserted-by":"publisher","first-page":"9540","DOI":"10.1109\/TVT.2021.3102161","volume":"70","author":"B Zhu","year":"2021","unstructured":"Zhu B, Bedeer E, Nguyen HH, Barton R, Henry J (2021) UAV trajectory planning in wireless sensor networks for energy consumption minimization by deep reinforcement learning. IEEE Trans Veh Technol 70:9540\u20139554. https:\/\/doi.org\/10.1109\/TVT.2021.3102161","journal-title":"IEEE Trans Veh Technol"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-023-10620-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-023-10620-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-023-10620-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T06:07:36Z","timestamp":1705903656000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-023-10620-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,28]]},"references-count":151,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["10620"],"URL":"https:\/\/doi.org\/10.1007\/s10462-023-10620-2","relation":{},"ISSN":["0269-2821","1573-7462"],"issn-type":[{"value":"0269-2821","type":"print"},{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,12,28]]},"assertion":[{"value":"1 October 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 December 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}},{"value":"The code is available at the URL gitee.com\/wangyyhhh.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Code availability"}}],"article-number":"1"}}