{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T15:25:07Z","timestamp":1786980307699,"version":"build-2736575974"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,10,10]],"date-time":"2023-10-10T00:00:00Z","timestamp":1696896000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,10,10]],"date-time":"2023-10-10T00:00:00Z","timestamp":1696896000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61921004"],"award-info":[{"award-number":["61921004"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62173251"],"award-info":[{"award-number":["62173251"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U1713209"],"award-info":[{"award-number":["U1713209"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62103104"],"award-info":[{"award-number":["62103104"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62136008"],"award-info":[{"award-number":["62136008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004608","name":"Natural Science Foundation of Jiangsu Province","doi-asserted-by":"publisher","award":["BK20210215"],"award-info":[{"award-number":["BK20210215"]}],"id":[{"id":"10.13039\/501100004608","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s00521-023-08882-6","type":"journal-article","created":{"date-parts":[[2023,10,10]],"date-time":"2023-10-10T08:02:02Z","timestamp":1696924922000},"page":"273-287","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Hierarchical multi-agent reinforcement learning for cooperative tasks with sparse rewards in continuous domain"],"prefix":"10.1007","volume":"36","author":[{"given":"Jingyu","family":"Cao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lu","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xin","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuanda","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9269-334X","authenticated-orcid":false,"given":"Changyin","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,10,10]]},"reference":[{"key":"8882_CR1","doi-asserted-by":"publisher","first-page":"101","DOI":"10.1016\/j.neucom.2020.06.031","volume":"412","author":"Y Wang","year":"2020","unstructured":"Wang Y, Dong L, Sun C (2020) Cooperative control for multi-player pursuit-evasion games with reinforcement learning. Neurocomputing 412:101\u2013114","journal-title":"Neurocomputing"},{"issue":"5","key":"8882_CR2","doi-asserted-by":"publisher","first-page":"2054","DOI":"10.1109\/TNNLS.2020.2996209","volume":"32","author":"C Sun","year":"2020","unstructured":"Sun C, Liu W, Dong L (2020) Reinforcement learning with task decomposition for cooperative multiagent systems. IEEE Trans Neural Netw Learn Syst 32(5):2054\u20132065","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"10","key":"8882_CR3","doi-asserted-by":"publisher","first-page":"4639","DOI":"10.1109\/TNNLS.2020.3025711","volume":"32","author":"Z Zhang","year":"2021","unstructured":"Zhang Z, Wang D, Gao J (2021) Learning automata-based multiagent reinforcement learning for optimization of cooperative tasks. IEEE Trans Neural Netw Learn Syst 32(10):4639\u20134652","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"4","key":"8882_CR4","doi-asserted-by":"publisher","first-page":"3143","DOI":"10.1007\/s00521-022-07880-4","volume":"35","author":"Y Shike","year":"2023","unstructured":"Shike Y, Jingchen L, Haobin S (2023) Mix-attention approximation for homogeneous large-scale multi-agent reinforcement learning. Neural Comput Appl 35(4):3143\u20133154","journal-title":"Neural Comput Appl"},{"key":"8882_CR5","doi-asserted-by":"crossref","unstructured":"Tan M (1993) Multi-agent reinforcement learning-independent vs. cooperative agent. In: Proceedings of the 10th International Conference on Machine Learning, pp 330\u2013337","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"issue":"3","key":"8882_CR6","doi-asserted-by":"publisher","first-page":"1086","DOI":"10.1109\/TITS.2019.2901791","volume":"21","author":"T Chu","year":"2020","unstructured":"Chu T, Wang J, Codec\u00e0 L, Li Z (2020) Multi-agent deep reinforcement learning for large-scale traffic signal control. IEEE Trans Intell Transp Syst 21(3):1086\u20131095","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"8882_CR7","unstructured":"Lowe R, Wu Y, Tamar A, Harb J (2017) Multi-agent actor-critic for mixed cooperative-competitive environments. arXiv preprint arXiv:1706.02275"},{"key":"8882_CR8","doi-asserted-by":"crossref","unstructured":"Wen C, Yao X, Wang Y, Tan X (2020) Smix ($$\\lambda $$): enhancing centralized value functions for cooperative multi-agent reinforcement learning. In: Proceedings of the AAAI Conference on Artificial Intelligence (vol 34, pp 7301\u20137308)","DOI":"10.1609\/aaai.v34i05.6223"},{"key":"8882_CR9","doi-asserted-by":"crossref","unstructured":"Sun Q, Yao Y, Yi P, Hu Y, Yang Z, Yang G, Zhou X (2022) Learning controlled and targeted communication with the centralized critic for the multi-agent system. Appl Intell (in Press)","DOI":"10.1007\/s10489-022-04225-5"},{"issue":"17","key":"8882_CR10","doi-asserted-by":"publisher","first-page":"14599","DOI":"10.1007\/s00521-022-07244-y","volume":"34","author":"C Fu","year":"2022","unstructured":"Fu C, Xu X, Zhang Y, Lyu Y, Xia Y, Zhou Z, Wu W (2022) Memory-enhanced deep reinforcement learning for UAV navigation in 3D environment. Neural Comput Appl 34(17):14599\u201314607","journal-title":"Neural Comput Appl"},{"issue":"11","key":"8882_CR11","doi-asserted-by":"publisher","first-page":"5174","DOI":"10.1109\/TNNLS.2018.2805379","volume":"29","author":"Z Yang","year":"2018","unstructured":"Yang Z, Merrick K, Jin L, Abbass HA (2018) Hierarchical deep reinforcement learning for continuous action control. IEEE Trans Neural Netw Learn Syst 29(11):5174\u20135184","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"9","key":"8882_CR12","doi-asserted-by":"publisher","first-page":"4227","DOI":"10.1007\/s00521-019-04330-6","volume":"32","author":"N Passalis","year":"2020","unstructured":"Passalis N, Tefas A (2020) Continuous drone control using deep reinforcement learning for frontal view person shooting. Neural Comput Appl 32(9):4227\u20134238","journal-title":"Neural Comput Appl"},{"key":"8882_CR13","unstructured":"Lee SY, Sungik C, Chung S-Y (2019) Sample-efficient deep reinforcement learning via episodic backward update. In: Proceedings of the NeurIPS, pp 2110\u20132119"},{"key":"8882_CR14","unstructured":"Schaul T, Quan J, Antonoglou I, Silver D (2015) Prioritized experience replay. arXiv preprint arXiv:1511.05952"},{"key":"8882_CR15","doi-asserted-by":"crossref","unstructured":"Lee S, Lee J, Hasuo I (2021) Predictive per: Balancing priority and diversity towards stable deep reinforcement learning. In: 2021 International Joint Conference on Neural Networks (IJCNN), pp 1\u201310","DOI":"10.1109\/IJCNN52387.2021.9534243"},{"issue":"7","key":"8882_CR16","doi-asserted-by":"publisher","first-page":"5649","DOI":"10.1007\/s00521-021-06702-3","volume":"34","author":"MN Alpdemir","year":"2022","unstructured":"Alpdemir MN (2022) Tactical UAV path optimization under radar threat using deep reinforcement learning. Neural Comput Appl 34(7):5649\u20135664","journal-title":"Neural Comput Appl"},{"issue":"12","key":"8882_CR17","doi-asserted-by":"publisher","first-page":"11547","DOI":"10.1109\/JIOT.2020.3022611","volume":"7","author":"X Tao","year":"2020","unstructured":"Tao X, Hafid AS (2020) Deepsensing: a novel mobile crowdsensing framework with double deep q-network and prioritized experience replay. IEEE Internet Things J 7(12):11547\u201311558","journal-title":"IEEE Internet Things J"},{"key":"8882_CR18","unstructured":"Andrychowicz M, Wolski F, Ray A, Schneider J, Fong R, Welinder P, Mcgrew B, Tobin J, Abbeel P, Zaremba W (2017) Hindsight experience replay. arXiv preprint arXiv:1707.01495"},{"key":"8882_CR19","doi-asserted-by":"crossref","unstructured":"Andres A, Villar-Rodriguez E, Ser JD (2022) Collaborative training of heterogeneous reinforcement learning agents in environments with sparse rewards: what and when to share?. Neural Comput Appl (in Press)","DOI":"10.1007\/s00521-022-07774-5"},{"key":"8882_CR20","first-page":"1471","volume":"29","author":"MG Bellemare","year":"2016","unstructured":"Bellemare MG, Srinivasan S, Ostrovski G, Schaul T, Saxton D, Munos R (2016) Unifying count-based exploration and intrinsic motivation. Adv Neural Inf Process Syst 29:1471\u20131479","journal-title":"Adv Neural Inf Process Syst"},{"key":"8882_CR21","unstructured":"Ostrovski G, Bellemare MG, Oord AVD, Munos R (2017) Count-based exploration with neural density models. In: International conference on machine learning, pp 2721\u20132730. PMLR"},{"key":"8882_CR22","unstructured":"Tang H, Houthooft R, Foote D, Stooke A, Chen X, Duan Y, Schulman J, Turck FD, Abbeel P (2017) #Exploration: a study of count-based exploration for deep reinforcement learning. In: 31st Conference on Neural Information Processing Systems(NIPS), vol 30, pp 1\u201318"},{"key":"8882_CR23","unstructured":"Stadie BC, Levine S, Abbeel P (2015) Incentivizing exploration in reinforcement learning with deep predictive models. arXiv preprint arXiv:1507.00814"},{"key":"8882_CR24","doi-asserted-by":"crossref","unstructured":"Pathak D, Agrawal P, Efros AA, Darrell T (2017) Curiosity-driven exploration by self-supervised prediction. In: Proceedings of the 34th international conference on machine learning, pp 2778\u20132787. PMLR","DOI":"10.1109\/CVPRW.2017.70"},{"issue":"9","key":"8882_CR25","first-page":"4555","volume":"44","author":"X Wang","year":"2021","unstructured":"Wang X, Chen Y, Zhu W (2021) A survey on curriculum learning. IEEE Trans Pattern Anal Mach Intell 44(9):4555\u20134576","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"8882_CR26","doi-asserted-by":"crossref","unstructured":"Rafati J, Noelle DC (2019) Learning representations in model-free hierarchical reinforcement learning. In: Proceeding of the AAAI conference on artificial intelligence, vol 33, pp 10009\u201310010","DOI":"10.1609\/aaai.v33i01.330110009"},{"key":"8882_CR27","doi-asserted-by":"crossref","unstructured":"Bacon PL, Harb J, Precup D (2017) The option-critic architecture. In: Proceeding of the AAAI conference on artificial intelligence, vol 31","DOI":"10.1609\/aaai.v31i1.10916"},{"issue":"9","key":"8882_CR28","doi-asserted-by":"publisher","first-page":"4727","DOI":"10.1109\/TNNLS.2021.3059912","volume":"33","author":"X Yang","year":"2021","unstructured":"Yang X, Ji Z, Wu J, Lai Y-K, Wei C, Liu G, Setchi R (2021) Hierarchical reinforcement learning with universal policies for multistep robotic manipulation. IEEE Trans Neural Netw Learn Syst 33(9):4727\u20134741","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"11","key":"8882_CR29","doi-asserted-by":"publisher","first-page":"3409","DOI":"10.1109\/TNNLS.2019.2891792","volume":"30","author":"N Dilokthanakul","year":"2019","unstructured":"Dilokthanakul N, Kaplanis C, Pawlowski N, Shanahan M (2019) Feature control as intrinsic motivation for hierarchical reinforcement learning. IEEE Trans Neural Netw Learn Syst 30(11):3409\u20133418","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"12","key":"8882_CR30","doi-asserted-by":"publisher","first-page":"7778","DOI":"10.1109\/TNNLS.2021.3087733","volume":"33","author":"S Pateria","year":"2021","unstructured":"Pateria S, Subagdja B, Tan A-H, Quek C (2021) End-to-end hierarchical reinforcement learning with integrated subgoal discovery. IEEE Trans Neural Netw Learn Syst 33(12):7778\u20137790","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"8882_CR31","unstructured":"Vezhnevets AS, Osindero S (2017) Feudal networks for hierarchical reinforcement learning. In: International conference on machine learning, pp 3540\u20133549. PMLR"},{"key":"8882_CR32","doi-asserted-by":"crossref","unstructured":"Whlke J, Schmitt F, Hoof HV (2021) Hierarchies of planning and reinforcement learning for robot navigation. In: 2021 IEEE international conference on robotics and automation (ICRA), pp 10682\u201310688","DOI":"10.1109\/ICRA48506.2021.9561151"},{"key":"8882_CR33","unstructured":"Nachum O, Gu S, Lee H, Levine S (2018) Data-efficient hierarchical reinforcement learning. arXiv preprint arXiv:1805.08296"},{"key":"8882_CR34","unstructured":"Ma J, Wu F (2020) Feudal multi-agent deep reinforcement learning for traffic signal control. In: Proceeding of the 19th international conference on autonomous agents and multiagent systems (AAMAS), pp 816\u2013824"},{"issue":"6","key":"8882_CR35","doi-asserted-by":"publisher","first-page":"4367","DOI":"10.1109\/TII.2020.3004857","volume":"17","author":"T Ren","year":"2020","unstructured":"Ren T, Niu J, Liu X, Wu J, Zhang Z (2020) An efficient model-free approach for controlling large-scale canals via hierarchical reinforcement learning. IEEE Trans Ind Inform 17(6):4367\u20134378","journal-title":"IEEE Trans Ind Inform"},{"key":"8882_CR36","unstructured":"Jin Y, Wei S, Yuan J, Zhang X (2021) Hierarchical and stable multiagent reinforcement learning for cooperative navigation control. IEEE Trans Neural Netw Learn Syst (in Press)"},{"key":"8882_CR37","doi-asserted-by":"crossref","unstructured":"Zhou J, Chen J, Tong Y, Zhang J (2022) Screening goals and selecting policies in hierarchical reinforcement learning. Appl Intell (in Press)","DOI":"10.1007\/s10489-021-03093-9"},{"issue":"358","key":"8882_CR38","first-page":"120","volume":"3","author":"RA Howard","year":"1960","unstructured":"Howard RA (1960) Dynamic programming and Markov processes. Math Gaz 3(358):120","journal-title":"Math Gaz"},{"issue":"8","key":"8882_CR39","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"issue":"6088","key":"8882_CR40","doi-asserted-by":"publisher","first-page":"533","DOI":"10.1038\/323533a0","volume":"323","author":"DE Rumelhart","year":"1986","unstructured":"Rumelhart DE, Hinton GE, Williams RJ (1986) Learning representations by back-propagating errors. Nature 323(6088):533\u2013536","journal-title":"Nature"},{"key":"8882_CR41","doi-asserted-by":"crossref","unstructured":"Zhang T, Guo S, Tan T, Hu X, Chen F (2022) Adjacency constraint for efficient hierarchical reinforcement learning. IEEE Trans Pattern Anal Mach Intell (in Press)","DOI":"10.1109\/TPAMI.2022.3192418"},{"issue":"4","key":"8882_CR42","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1109\/TG.2018.2849942","volume":"10","author":"Y Wang","year":"2018","unstructured":"Wang Y, He H, Sun C (2018) Learning to navigate through complex dynamic environment with modular deep reinforcement learning. IEEE Trans Games 10(4):400\u2013412","journal-title":"IEEE Trans Games"},{"key":"8882_CR43","unstructured":"Simonyan K, Zisserman A (2014) Two-stream convolutional networks for action recognition in videos. arXiv preprint arXiv:1406.2199"},{"key":"8882_CR44","unstructured":"Kingma D, Ba J (2014) Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"key":"8882_CR45","doi-asserted-by":"crossref","unstructured":"Gupta JK, Egorov M, Kochenderfer M (2017) Cooperative multi-agent control using deep reinforcement learning. In: International conference on autonomous agents and multiagent systems, pp 66\u201383","DOI":"10.1007\/978-3-319-71682-4_5"},{"key":"8882_CR46","unstructured":"Martin A, Barham P, Chen J, Chen Z, Zhang X (2016) Tensorflow: a system for large-scale machine learning. In: 12th USENIX symposium on operating systems design and implementation, pp 265\u2013283"},{"key":"8882_CR47","unstructured":"Senadeera M, Karimpanal TG, Gupta S, Rana S (2022) Sympathy-based reinforcement learning agents. In: Proceedings of the 21st international conference on autonomous agents and multiagent systems, pp 1164\u20131172"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-023-08882-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-023-08882-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-023-08882-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,4]],"date-time":"2024-01-04T12:08:57Z","timestamp":1704370137000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-023-08882-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,10]]},"references-count":47,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["8882"],"URL":"https:\/\/doi.org\/10.1007\/s00521-023-08882-6","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10,10]]},"assertion":[{"value":"26 January 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 July 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 October 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"No potential conflict of interest was reported by the authors.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}