{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T02:53:34Z","timestamp":1773888814584,"version":"3.50.1"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2021,10,1]],"date-time":"2021-10-01T00:00:00Z","timestamp":1633046400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,10,1]],"date-time":"2021-10-01T00:00:00Z","timestamp":1633046400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100008812","name":"Defence Science and Technology Group","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100008812","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2022,2]]},"DOI":"10.1007\/s00521-021-06270-6","type":"journal-article","created":{"date-parts":[[2021,10,1]],"date-time":"2021-10-01T11:28:29Z","timestamp":1633087709000},"page":"1713-1733","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Discrete-to-deep reinforcement learning methods"],"prefix":"10.1007","volume":"34","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9016-8912","authenticated-orcid":false,"given":"Budi","family":"Kurniawan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8687-4424","authenticated-orcid":false,"given":"Peter","family":"Vamplew","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1184-8376","authenticated-orcid":false,"given":"Michael","family":"Papasimeon","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6199-9685","authenticated-orcid":false,"given":"Richard","family":"Dazeley","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2537-0326","authenticated-orcid":false,"given":"Cameron","family":"Foale","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,10,1]]},"reference":[{"key":"6270_CR1","unstructured":"Achiam J, Knight E, Abbeel P (2019) Towards characterizing divergence in deep q-learning. arXiv:1903.08894"},{"key":"6270_CR2","first-page":"833","volume":"5","author":"A Barto","year":"1983","unstructured":"Barto A, Sutton R, Anderson C (1983) Neuronlike adaptive elements that can solve difficult learning control problems. IEEE Trans Syst Man Cybern 5:833\u2013836","journal-title":"IEEE Trans Syst Man Cybern"},{"key":"6270_CR3","unstructured":"Boyan J, Moore A (1995) Generalization in reinforcement learning: Safely approximating the value function. NIPS-7. San Mateo, CA: Morgan Kaufmann"},{"key":"6270_CR4","doi-asserted-by":"crossref","unstructured":"Brady T, Paschall S (2010) The challenge of safe lunar landing. IEEE Aerospace Conference. IEEE","DOI":"10.1109\/AERO.2010.5447029"},{"key":"6270_CR5","unstructured":"Brockman G, Cheung V, Pettersson L, Schneider J, Schulman J, Tang J, Zaremba W (2016) Openai gym"},{"key":"6270_CR6","doi-asserted-by":"crossref","unstructured":"Chebotar Y, Hausman K, Kroemer O, Sukhatme G, Schaal S (2016) Generalizing regrasping with supervised policy learning. International Symposium on Experimental Robotics","DOI":"10.1007\/978-3-319-50115-4_54"},{"issue":"1","key":"6270_CR7","doi-asserted-by":"publisher","first-page":"67","DOI":"10.5139\/IJASS.2009.10.1.067","volume":"10","author":"D Cho","year":"2009","unstructured":"Cho D, Jeong D, Bang H (2009) Optimal perilune altitude of lunar landing trajectory. Int J Aeronaut Space Sci 10(1):67\u201374","journal-title":"Int J Aeronaut Space Sci"},{"key":"6270_CR8","doi-asserted-by":"crossref","unstructured":"Gadgil S, Xin Y, Xu C (2020) Solving the lunar lander problem under uncertainty using reinforcement learning. arXiv:2011.11850","DOI":"10.1109\/SoutheastCon44009.2020.9368267"},{"key":"6270_CR9","volume-title":"Deep Learning","author":"I Goodfellow","year":"2016","unstructured":"Goodfellow I, Bengio Y, Courville A (2016) Deep Learning. MIT Press, Cambridge"},{"key":"6270_CR10","unstructured":"van Hasselt H, Doron Y, Strub F, Hessel M, Sonnerat N, Modayil J (2018) Deep reinforcement learning and the deadly triad. arXiv:1812.02648"},{"key":"6270_CR11","doi-asserted-by":"crossref","unstructured":"van Hasselt H, Guez A, Siver D (2016) Deep reinforcement learning with double q-learning. Proceedings of the Thirtieth AAAI Conference on Artificial Intelligence","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"6270_CR12","doi-asserted-by":"crossref","unstructured":"Hessel M, Modayil J, Van\u00a0Hasselt H, Schaul T, Ostrovski G, Dabney W, Horgan D, Piot B, Azar M, Silver D (2018) Rainbow: Combining improvements in deep reinforcement learning. AAAI Conference on Artificial Intelligence","DOI":"10.1609\/aaai.v33i01.33013796"},{"key":"6270_CR13","unstructured":"Horgan D, Quan J, Budden D, Barth-Maron G, Hessel M, van Hasselt H, Silver D (2018) Distributed prioritized experience replay. ICLR"},{"key":"6270_CR14","doi-asserted-by":"crossref","unstructured":"Kurniawan B, Vamplew P, Papasimeon M, Dazeley R, Foale C (2019) An empirical study of reward structures for actor-critic reinforcement learning in air combat manoeuvring simulation. 32nd Australasian Joint Conference on Artificial Intelligence","DOI":"10.1007\/978-3-030-35288-2_5"},{"key":"6270_CR15","unstructured":"Kurniawan B, Vamplew P, Papasimeon M, Dazeley R, Foale C (2020) Discrete-to-deep supervised policy learning. arXiv:2005.02057"},{"key":"6270_CR16","unstructured":"Lagoudakis MG, Parr R (2003) Reinforcement learning as classification: leveraging modern classifiers. Proc. 20th Int. Conf. Mach. Learn. p. 424\u2013431"},{"key":"6270_CR17","doi-asserted-by":"crossref","unstructured":"Lam C, Masek M, Kelly L, Papasimeon M, Benke L (2019) A simheuristic approach for evolving agent behaviour in the exploration for novel combat tactics. Operations Research Perspectives 6","DOI":"10.1016\/j.orp.2019.100123"},{"key":"6270_CR18","doi-asserted-by":"crossref","unstructured":"Levine S, Wagener N, Abbeel P (2015) Learning contact-rich manipulation skills with guided policy search. IEEE International Conference on Robotics and Automation pp. 156\u2013163","DOI":"10.1109\/ICRA.2015.7138994"},{"key":"6270_CR19","unstructured":"Lillicrap T, Hunt J, Pritzel A, Heess N, Erez T, Tassa Y, Silver D, Wierstra D (2015) Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971"},{"key":"6270_CR20","unstructured":"Lin L (1993) Reinforcement learning for robots using neural networks. Phd thesis, Carnegie Mellon University"},{"issue":"4","key":"6270_CR21","doi-asserted-by":"publisher","first-page":"1097","DOI":"10.1016\/j.automatica.2007.08.021","volume":"44","author":"X Liu","year":"2008","unstructured":"Liu X, Duan G, Teo K (2008) Optimal soft landing control for moon lander. Automatica 44(4):1097","journal-title":"Automatica"},{"key":"6270_CR22","doi-asserted-by":"crossref","unstructured":"Masek M, Lam C, Benke L, Kelly L, Papasimeon M (2018) Discovering emergent agent behaviour with evolutionary finite state machines. Conf. on Principles and Practice of Multi-Agent Systems, Int","DOI":"10.1007\/978-3-030-03098-8_2"},{"issue":"5","key":"6270_CR23","doi-asserted-by":"publisher","first-page":"1641","DOI":"10.2514\/1.46815","volume":"33","author":"J McGrew","year":"2010","unstructured":"McGrew J, How J, Williams B, Roy N (2010) Air-combat strategy using approximate dynamic programming. J Guid Control Dyn 33(5):1641","journal-title":"J Guid Control Dyn"},{"issue":"2","key":"6270_CR24","first-page":"137","volume":"2","author":"D Michie","year":"1968","unstructured":"Michie D, Chambers R (1968) Boxes: an experiment in adaptive control. Mach Intell 2(2):137\u2013152","journal-title":"Mach Intell"},{"key":"6270_CR25","unstructured":"Mnih V, Badia A, Mirza M, Graves A, Harley T, Lillicrap T, Silver D, Kavukcuoglu K (2016) Asynchronous methods for deep reinforcement learning. Proc. 33rd Int. Conf. Mach. Learn. 48"},{"key":"6270_CR26","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Graves A, Antonoglou I, Wierstra D, Riedmiller M (2013) Playing atari with deep reinforcement learning. NIPS Deep Learning Workshop"},{"key":"6270_CR27","doi-asserted-by":"crossref","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu A, Veness J, Bellemare M, Graves A, Riedmiller M, Fidjeland A, Ostrovski G, Petersen S, Beattie C, Sadik A, Antonoglou I, King H, Kumaran D, Wierstra D, Legg S, Hassabis D (2015) Human-level control through deep reinforcement learning. Nature pp. 29\u201333","DOI":"10.1038\/nature14236"},{"key":"6270_CR28","unstructured":"Nair A, Srinivasan P, Blackwell S, Alcicek C, Fearon R, DeMaria A, Panneershelvam V, Suleyman M, Beattie C, Petersen S, Legg S, Mnih V, Kavukcuoglu K, Silver D (2015) Massively parallel methods for deep reinforcement learning. ICML Deep Learning Workshop"},{"key":"6270_CR29","unstructured":"Pollack J, Blair A (1997) Why did td-gammon work? Advances in Neural Information Processing Systems p. 10\u201316"},{"key":"6270_CR30","unstructured":"Ramirez M, Papasimeon M, Lipovetzky N, Benke L, Miller T, Pearce A, Scala E, Zamani M (2018) Integrated hybrid planning and programmed control for real time uav maneuvering. Proc. 17th International Conference on Autonomous Agents and MultiAgent Systems pp. 1318\u20131326"},{"key":"6270_CR31","doi-asserted-by":"crossref","unstructured":"Riedmiller M (2005) Neural fitted q iteration - first experiences with a data efficient neural reinforcement learning method. Machine Learning: ECML 2005","DOI":"10.1007\/11564096_32"},{"key":"6270_CR32","doi-asserted-by":"crossref","unstructured":"Roijers D, Vamplew P, Whiteson S, Dazeley R (2013) A survey of multi-objective sequential decision-making. Journal of Artificial Intelligence Research 48","DOI":"10.1613\/jair.3987"},{"key":"6270_CR33","unstructured":"Rosenstein M, Barto AG (2002) Supervised learning combined with an actor critic architecture. Technical report, Amherst, MA, USA"},{"key":"6270_CR34","unstructured":"Schaul T, Quan J, Antonoglou I, Silver D (2015) Prioritized experience replay. arXiv preprint arXiv:1511.05952"},{"key":"6270_CR35","unstructured":"Shaw R (1985) Fighter Combat: Tactics and Maneuvering. Naval Institute Press"},{"key":"6270_CR36","unstructured":"Sobh I, Darwish N (2018) End-to-end framework for fast learning asynchronous agents. Proceedings of the 32nd Conference on Neural Information Processing Systems, Imitation Learning and its Challenges in Robotics workshop"},{"key":"6270_CR37","volume-title":"Reinforcement learning: an introduction","author":"R Sutton","year":"2018","unstructured":"Sutton R, Barto A (2018) Reinforcement learning: an introduction, 2nd edn. MIT Press, Cambridge","edition":"2"},{"key":"6270_CR38","doi-asserted-by":"crossref","unstructured":"Tesauro G (1995) TD-Gammon: A self-teaching backgammon program. Applications of Neural Networks","DOI":"10.1007\/978-1-4757-2379-3_11"},{"key":"6270_CR39","unstructured":"Uther W, Veloso M (1998) Tree-based discretization for continuous state space reinforcement learning. AAAI-98 Proceedings"},{"issue":"51","key":"6270_CR40","doi-asserted-by":"publisher","first-page":"51","DOI":"10.1007\/s10994-010-5232-5","volume":"84","author":"P Vamplew","year":"2011","unstructured":"Vamplew P, Dazeley R, Berry A, Issabekov R, Dekker E (2011) Empirical evaluation methods for multiobjective reinforcement learning algorithms. Mach Learn 84(51):51","journal-title":"Mach Learn"},{"key":"6270_CR41","doi-asserted-by":"crossref","unstructured":"Wang L, Zhang W, X, H, Zha H (2018) Supervised reinforcement learning with recurrent neural network for dynamic treatment recommendation. International Conference on Knowledge Discovery and Data Mining pp. 2447\u20132456","DOI":"10.1145\/3219819.3219961"},{"key":"6270_CR42","unstructured":"Wang Z, Bapst V, Heess N, Mnih V, Munos R, Kavukcuoglu K, de\u00a0Freitas N (2016) Sample efficient actor-critic with experience replay. arXiv preprint arXiv:1611.01224"},{"key":"6270_CR43","unstructured":"Wang Z, de\u00a0Freitas N, Lanctot M (2015) Dueling network architectures for deep reinforcement learning. ArXiv e-prints"},{"key":"6270_CR44","unstructured":"Zhang S, Sutton RS (2017) A deeper look at experience replay. CoRR, abs\/1712.01275"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-021-06270-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-021-06270-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-021-06270-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,10]],"date-time":"2023-01-10T23:08:41Z","timestamp":1673392121000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-021-06270-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,1]]},"references-count":44,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2022,2]]}},"alternative-id":["6270"],"URL":"https:\/\/doi.org\/10.1007\/s00521-021-06270-6","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,10,1]]},"assertion":[{"value":"28 October 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 June 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 October 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The source code for D2D-SPL is available at:.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Source Code:"}},{"value":"The authors declare that they have no conflict of interests.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interests:"}}]}}