{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,20]],"date-time":"2026-08-20T15:20:46Z","timestamp":1787239246414,"version":"build-2736575974"},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,10,31]],"date-time":"2023-10-31T00:00:00Z","timestamp":1698710400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,10,31]],"date-time":"2023-10-31T00:00:00Z","timestamp":1698710400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100016073","name":"Key Technologies Research and Development Program of Anhui Province","doi-asserted-by":"publisher","award":["2018AAA0101400"],"award-info":[{"award-number":["2018AAA0101400"]}],"id":[{"id":"10.13039\/100016073","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100014718","name":"Innovative Research Group Project of the National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61921004"],"award-info":[{"award-number":["61921004"]}],"id":[{"id":"10.13039\/100014718","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010023","name":"Natural Science Research of Jiangsu Higher Education Institutions of China","doi-asserted-by":"publisher","award":["BK20202006"],"award-info":[{"award-number":["BK20202006"]}],"id":[{"id":"10.13039\/501100010023","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62173251"],"award-info":[{"award-number":["62173251"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s00521-023-08991-2","type":"journal-article","created":{"date-parts":[[2023,10,31]],"date-time":"2023-10-31T15:02:38Z","timestamp":1698764558000},"page":"323-336","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Hierarchical reinforcement learning for kinematic control tasks with parameterized action spaces"],"prefix":"10.1007","volume":"36","author":[{"given":"Jingyu","family":"Cao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lu","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9269-334X","authenticated-orcid":false,"given":"Changyin","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,10,31]]},"reference":[{"key":"8991_CR1","volume-title":"Surface path tracking method of autonomous surface underwater vehicle based on deep reinforcement learning","author":"D Song","year":"2022","unstructured":"Song D, Gan W, Yao P, Zang W, Qu X (2022) Surface path tracking method of autonomous surface underwater vehicle based on deep reinforcement learning. In press, Neural Computing and Applications"},{"issue":"17","key":"8991_CR2","doi-asserted-by":"publisher","first-page":"14599","DOI":"10.1007\/s00521-022-07244-y","volume":"34","author":"C Fu","year":"2022","unstructured":"Fu C, Xu X, Zhang Y, Lyu Y, Xia Y, Zhou Z, Wu W (2022) Memory-enhanced deep reinforcement learning for uav navigation in 3d environment. Neural Comput Appl 34(17):14599\u201314607","journal-title":"Neural Comput Appl"},{"issue":"5","key":"8991_CR3","doi-asserted-by":"publisher","first-page":"2054","DOI":"10.1109\/TNNLS.2020.2996209","volume":"32","author":"C Sun","year":"2020","unstructured":"Sun C, Liu W, Dong L (2020) Reinforcement learning with task decomposition for cooperative multiagent systems. IEEE Transact Neural Netw Lear Syst 32(5):2054\u20132065","journal-title":"IEEE Transact Neural Netw Lear Syst"},{"issue":"4","key":"8991_CR4","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1109\/TG.2018.2849942","volume":"10","author":"Y Wang","year":"2018","unstructured":"Wang Y, He H, Sun C (2018) Learning to navigate through complex dynamic environment with modular deep reinforcement learning. IEEE Transact Games 10(4):400\u2013412","journal-title":"IEEE Transact Games"},{"key":"8991_CR5","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Graves A, Antonoglou I, Wierstra D, Riedmiller M (2013) Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602"},{"key":"8991_CR6","unstructured":"Lillicrap T.P, Hunt J.J, Pritzel A, Heess N, Erez T, Tassa Y, Silver D, Wierstra D (2015) Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971"},{"key":"8991_CR7","doi-asserted-by":"crossref","unstructured":"Masson W, Ranchod P, Konidaris G (2016) Reinforcement learning with parameterized actions. In: Proceedings of the AAAI conference on artificial intelligence, vol. 30, pp 1934\u20131940","DOI":"10.1609\/aaai.v30i1.10226"},{"key":"8991_CR8","unstructured":"Hausknecht M, Stone P (2016) Deep reinforcement learning in parameterized action space. In: Proceedings of the international conference on learning representations (ICLR)"},{"key":"8991_CR9","unstructured":"Xiong J, Wang Q, Yang Z, Sun P, Han L, Zheng Y, Fu H, Zhang T, Liu J, Liu H (2018) Parametrized deep q-networks learning: reinforcement learning with discrete-continuous hybrid action space. arXiv preprint arXiv:1810.06394"},{"key":"8991_CR10","unstructured":"Bester CJ, James SD, Konidaris GD (2019) Multi-pass q-networks for deep reinforcement learning with parameterised action spaces. arXiv preprint arXiv:1905.04388"},{"key":"8991_CR11","doi-asserted-by":"crossref","unstructured":"Fu H, Tang H, Hao J, Lei Z, Chen Y, Fan C (2019) Deep multi-agent reinforcement learning with discrete-continuous hybrid action spaces. In: Twenty-Eighth international joint conference on artificial intelligence IJCAI-19","DOI":"10.24963\/ijcai.2019\/323"},{"key":"8991_CR12","doi-asserted-by":"crossref","unstructured":"Zhang X, Jin S, Wang C, Zhu X, Tomizuka M (2022) Learning insertion primitives with discrete-continuous hybrid action space for robotic assembly tasks. In: 2022 International conference on robotics and automation (ICRA), pp 9881\u20139887 . IEEE","DOI":"10.1109\/ICRA46639.2022.9811973"},{"issue":"4","key":"8991_CR13","doi-asserted-by":"publisher","first-page":"892","DOI":"10.1177\/01423312211037847","volume":"44","author":"Q Zheng","year":"2022","unstructured":"Zheng Q, Wang D, Chen Z, Sun Y, Liang B (2022) Continuous reinforcement learning based ramp jump control for single-track two-wheeled robots. Transact Instit Meas Control 44(4):892\u2013904","journal-title":"Transact Instit Meas Control"},{"issue":"6","key":"8991_CR14","doi-asserted-by":"publisher","first-page":"2067","DOI":"10.1109\/TRO.2021.3073771","volume":"37","author":"M Lombardi","year":"2021","unstructured":"Lombardi M, Liuzza D, Bernardo M (2021) Using learning to control artificial avatars in human motor coordination tasks. IEEE Transact Robot 37(6):2067\u20132082","journal-title":"IEEE Transact Robot"},{"key":"8991_CR15","doi-asserted-by":"publisher","DOI":"10.1007\/s00521-021-06476-8","volume-title":"Control of an auv with completely unknown dynamics and multi-asymmetric input constraints via off-policy reinforcement learning","author":"M Mohammadi","year":"2022","unstructured":"Mohammadi M, Arefi MM, Vafamand N, Kaynak O (2022) Control of an auv with completely unknown dynamics and multi-asymmetric input constraints via off-policy reinforcement learning. In press, Neural Computing and Applications"},{"issue":"7","key":"8991_CR16","doi-asserted-by":"publisher","first-page":"5649","DOI":"10.1007\/s00521-021-06702-3","volume":"34","author":"MN Alpdemir","year":"2022","unstructured":"Alpdemir MN (2022) Tactical uav path optimization under radar threat using deep reinforcement learning. Neural Comput Appl 34(7):5649\u20135664","journal-title":"Neural Comput Appl"},{"key":"8991_CR17","unstructured":"Ma J, Wu F (2020) Feudal multi-agent deep reinforcement learning for traffic signal control. In: Proceeding of the 19th international conference on autonomous agents and multiagent systems(AAMAS), pp 816\u2013824"},{"issue":"12","key":"8991_CR18","doi-asserted-by":"publisher","first-page":"7778","DOI":"10.1109\/TNNLS.2021.3087733","volume":"33","author":"S Pateria","year":"2022","unstructured":"Pateria S, Subagdja B, Tan AH, Chai Q (2022) End-to-end hierarchical reinforcement learning with integrated subgoal discovery. IEEE Transact Neural Netw Learn Syst 33(12):7778\u20137790","journal-title":"IEEE Transact Neural Netw Learn Syst"},{"issue":"11","key":"8991_CR19","doi-asserted-by":"publisher","first-page":"3409","DOI":"10.1109\/TNNLS.2019.2891792","volume":"30","author":"N Dilokthanakul","year":"2019","unstructured":"Dilokthanakul N, Kaplanis C, Pawlowski N, Shanahan M (2019) Feature control as intrinsic motivation for hierarchical reinforcement learning. IEEE Transact Neural Netw Learn Syst 30(11):3409\u20133418","journal-title":"IEEE Transact Neural Netw Learn Syst"},{"issue":"2","key":"8991_CR20","doi-asserted-by":"publisher","first-page":"1086","DOI":"10.1007\/s10489-020-01849-3","volume":"51","author":"N Bougie","year":"2021","unstructured":"Bougie N, Ichise R (2021) Fast and slow curiosity for high-level exploration in reinforcement learning. Appl Intell 51(2):1086\u20131107","journal-title":"Appl Intell"},{"issue":"6","key":"8991_CR21","doi-asserted-by":"publisher","first-page":"4367","DOI":"10.1109\/TII.2020.3004857","volume":"17","author":"T Ren","year":"2020","unstructured":"Ren T, Niu J, Liu X, Wu J, Zhang Z (2020) An efficient model-free approach for controlling large-scale canals via hierarchical reinforcement learning. IEEE Transact Indus Inform 17(6):4367\u20134378","journal-title":"IEEE Transact Indus Inform"},{"issue":"11","key":"8991_CR22","doi-asserted-by":"publisher","first-page":"5174","DOI":"10.1109\/TNNLS.2018.2805379","volume":"29","author":"Z Yang","year":"2018","unstructured":"Yang Z, Merrick K, Jin L, Abbass HA (2018) Hierarchical deep reinforcement learning for continuous action control. IEEE Transact Neural Netw Learn Syst 29(11):5174\u20135184","journal-title":"IEEE Transact Neural Netw Learn Syst"},{"key":"8991_CR23","unstructured":"Nachum O, Gu S, Lee H, Levine S (2018) Data-efficient hierarchical reinforcement learning. arXiv preprint arXiv:1805.08296"},{"issue":"5","key":"8991_CR24","doi-asserted-by":"publisher","first-page":"1546","DOI":"10.1109\/TRO.2020.2994002","volume":"36","author":"A Devo","year":"2020","unstructured":"Devo A, Mezzetti G, Costante G, Fravolini ML, Valigi P (2020) Towards generalization in target-driven visual navigation by using deep reinforcement learning. IEEE Transact Robot 36(5):1546\u20131561","journal-title":"IEEE Transact Robot"},{"key":"8991_CR25","doi-asserted-by":"crossref","unstructured":"Whlke J, Schmitt F, Hoof H.V (2021) Hierarchies of planning and reinforcement learning for robot navigation. In: 2021 IEEE international conference on robotics and automation (ICRA), pp 10682\u201310688","DOI":"10.1109\/ICRA48506.2021.9561151"},{"issue":"2","key":"8991_CR26","doi-asserted-by":"publisher","first-page":"3623","DOI":"10.1109\/LRA.2021.3060403","volume":"6","author":"S Christen","year":"2021","unstructured":"Christen S, Jendele L, Aksan E, Hilliges O (2021) Learning functionally decomposed hierarchies for continuous control tasks with path planning. IEEE Robot Autom Lett 6(2):3623\u20133630","journal-title":"IEEE Robot Autom Lett"},{"issue":"2","key":"8991_CR27","doi-asserted-by":"publisher","first-page":"2985","DOI":"10.1109\/LRA.2022.3145971","volume":"7","author":"R Bigazzi","year":"2022","unstructured":"Bigazzi R, Landi F, Cascianelli S, Baraldi L, Cornia M, Cucchiara R (2022) Focus on impact: indoor exploration with intrinsic motivation. IEEE Robot Autom Lett 7(2):2985\u20132992","journal-title":"IEEE Robot Autom Lett"},{"key":"8991_CR28","doi-asserted-by":"crossref","unstructured":"Xia F, Li C, Mart\u00edn-Mart\u00edn R, Litany O, Toshev A, Savarese S (2021) Relmogen: Leveraging motion generation in reinforcement learning for mobile manipulation. In: 2021 international conference on robotics and automation (ICRA)","DOI":"10.1109\/ICRA48506.2021.9561315"},{"issue":"10","key":"8991_CR29","doi-asserted-by":"publisher","first-page":"1686","DOI":"10.1109\/JAS.2021.1004141","volume":"8","author":"C Liu","year":"2021","unstructured":"Liu C, Zhu F, Liu Q, Fu Y (2021) Hierarchical reinforcement learning with automatic sub-goal identification. IEEE\/CAA J Autom Sin 8(10):1686\u20131696","journal-title":"IEEE\/CAA J Autom Sin"},{"issue":"9","key":"8991_CR30","doi-asserted-by":"publisher","first-page":"4727","DOI":"10.1109\/TNNLS.2021.3059912","volume":"33","author":"X Yang","year":"2022","unstructured":"Yang X, Ji Z, Wu J, Lai YK, Setchi R (2022) Hierarchical reinforcement learning with universal policies for multistep robotic manipulation. IEEE Transact Neural Netw Learn Syst 33(9):4727\u20134741","journal-title":"IEEE Transact Neural Netw Learn Syst"},{"key":"8991_CR31","unstructured":"Peng X.B, Chang M, Zhang G, Abbeel P, Levine S (2019) Mcp: Learning composable hierarchical control with multiplicative compositional policies. In: Proc. NIPS, pp 3681\u20133692"},{"issue":"358","key":"8991_CR32","first-page":"120","volume":"3","author":"RA Howard","year":"1960","unstructured":"Howard RA (1960) Dynamic programming and markov processes. Math Gazette 3(358):120","journal-title":"Math Gazette"},{"key":"8991_CR33","unstructured":"Haarnoja T, Zhou A, Abbeel P, Levine S (2018) Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: International conference on machine learning, pp 1861\u20131870. PMLR"},{"key":"8991_CR34","unstructured":"Haarnoja T, Zhou A, Abbeel P, Levine S (2019) Soft actor-critic algorithm and applications. arXiv preprint arXiv:1812.05905"},{"key":"8991_CR35","unstructured":"Christodoulou P (2019) Soft actor-critic for discrete action settings. arXiv preprint arXiv:1910.07207"},{"issue":"3731","key":"8991_CR36","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1126\/science.153.3731.34","volume":"153","author":"R Bellman","year":"1966","unstructured":"Bellman R (1966) Dynamic programming. Science 153(3731):34\u201337","journal-title":"Science"},{"key":"8991_CR37","unstructured":"Paszke A, Gross S, Chintala S, Chanan G, Yang E, Devito Z, Lin Z, Desmaison A, Antiga L, Lerer A (2017) Automatic differentiation in pytorch. In: cNIPS 2017 autodiff workshop: the future of gradient-based machine learning software and techniques"},{"key":"8991_CR38","unstructured":"Brockman G, Cheung V, Pettersson L, Schneider J, Schulman J, Tang J, Zaremba W (2016) Openai gym. arXiv preprint arXiv:1606.01540"},{"key":"8991_CR39","unstructured":"Kitano H, M, A, Y, K, I, N (1997) Robocup : a challenge ai problem. Ai Magazine, 18\u20137385"}],"updated-by":[{"DOI":"10.1007\/s00521-023-09305-2","type":"correction","label":"Correction","source":"publisher","updated":{"date-parts":[[2023,12,13]],"date-time":"2023-12-13T00:00:00Z","timestamp":1702425600000}}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-023-08991-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-023-08991-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-023-08991-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,4]],"date-time":"2024-01-04T12:10:56Z","timestamp":1704370256000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-023-08991-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,31]]},"references-count":39,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["8991"],"URL":"https:\/\/doi.org\/10.1007\/s00521-023-08991-2","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10,31]]},"assertion":[{"value":"15 February 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 August 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 October 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 December 2023","order":4,"name":"change_date","label":"Change Date","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Correction","order":5,"name":"change_type","label":"Change Type","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"A Correction to this paper has been published:","order":6,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"https:\/\/doi.org\/10.1007\/s00521-023-09305-2","URL":"https:\/\/doi.org\/10.1007\/s00521-023-09305-2","order":7,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"No potential conflict of interest was reported by the authors.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}