{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T19:43:42Z","timestamp":1784058222286,"version":"3.55.0"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,5,5]],"date-time":"2025-05-05T00:00:00Z","timestamp":1746403200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,5]],"date-time":"2025-05-05T00:00:00Z","timestamp":1746403200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1007\/s10489-025-06584-1","type":"journal-article","created":{"date-parts":[[2025,5,5]],"date-time":"2025-05-05T18:38:22Z","timestamp":1746470302000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Quadruped robot locomotion via soft actor-critic with muti-head critic and dynamic policy gradient"],"prefix":"10.1007","volume":"55","author":[{"given":"Yanan","family":"Fan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongcai","family":"Pei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongbing","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Meng","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianyuan","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiyong","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,5]]},"reference":[{"issue":"4","key":"6584_CR1","doi-asserted-by":"publisher","first-page":"4062","DOI":"10.1109\/LRA.2018.2862431","volume":"3","author":"Y Shen","year":"2018","unstructured":"Shen Y, Zhang G, Tian Y, Ma S (2018) Development of a wheel-paddle integrated quadruped robot for rough terrain and its verification on hybrid mode. IEEE Robot Autom Lett 3(4):4062\u20134067","journal-title":"IEEE Robot Autom Lett"},{"issue":"4","key":"6584_CR2","doi-asserted-by":"publisher","first-page":"1866","DOI":"10.1007\/s12555-016-0798-8","volume":"16","author":"N Hu","year":"2018","unstructured":"Hu N, Li S, Gao F (2018) Multi-objective hierarchical optimal control for quadruped rescue robot. Int J Control Autom Syst 16(4):1866\u20131877","journal-title":"Int J Control Autom Syst"},{"issue":"1","key":"6584_CR3","doi-asserted-by":"publisher","first-page":"382","DOI":"10.1109\/JAS.2017.7510790","volume":"5","author":"D Gong","year":"2017","unstructured":"Gong D, Wang P, Zhao S, Du L, Duan Y (2017) Bionic quadruped robot dynamic gait control strategy based on twenty degrees of freedom. IEEE CAA J Autom Sin 5(1):382\u2013388","journal-title":"IEEE CAA J Autom Sin"},{"issue":"8","key":"6584_CR4","doi-asserted-by":"publisher","first-page":"3177","DOI":"10.1007\/s40815-023-01563-5","volume":"25","author":"X Song","year":"2023","unstructured":"Song X, Song Y, Stojanovic V, Song S (2023) Improved dynamic event-triggered security control for t-s fuzzy lpv-pde systems via pointwise measurements and point control. Int J Fuzzy Syst 25(8):3177\u20133192","journal-title":"Int J Fuzzy Syst"},{"key":"6584_CR5","doi-asserted-by":"publisher","unstructured":"Yu L, Wang Y, Tao W (2010) Gait analysis and implementation of a simple quadruped robot. In: 2010 The 2nd international conference on industrial mechatronics and automation, vol. 2. IEEE, pp 431\u2013434. https:\/\/doi.org\/10.1109\/ICINDMA.2010.5538277","DOI":"10.1109\/ICINDMA.2010.5538277"},{"issue":"2","key":"6584_CR6","doi-asserted-by":"publisher","first-page":"691","DOI":"10.1109\/TRO.2020.2976304","volume":"37","author":"H Yu","year":"2020","unstructured":"Yu H, Gao H, Deng Z (2020) Toward a unified approximate analytical representation for spatially running spring-loaded inverted pendulum model. IEEE Trans Rob 37(2):691\u2013698","journal-title":"IEEE Trans Rob"},{"issue":"4","key":"6584_CR7","doi-asserted-by":"publisher","first-page":"2201","DOI":"10.1109\/LRA.2017.2723931","volume":"2","author":"AW Winkler","year":"2017","unstructured":"Winkler AW, Farshidian F, Pardo D, Neunert M, Buchli J (2017) Fast trajectory optimization for legged robots using vertex-based zmp constraints. IEEE Robot Autom Lett 2(4):2201\u20132208","journal-title":"IEEE Robot Autom Lett"},{"key":"6584_CR8","doi-asserted-by":"publisher","unstructured":"Ding Y, Pandala A, Park H-W (2019) Real-time model predictive control for versatile dynamic motions in quadrupedal robots. In: 2019 International Conference on Robotics and Automation (ICRA). IEEE, pp 8484\u20138490. https:\/\/doi.org\/10.1109\/ICRA.2019.8793669","DOI":"10.1109\/ICRA.2019.8793669"},{"issue":"3","key":"6584_CR9","doi-asserted-by":"publisher","first-page":"2553","DOI":"10.1109\/LRA.2019.2908502","volume":"4","author":"S Fahmi","year":"2019","unstructured":"Fahmi S, Mastalli C, Focchi M, Semini C (2019) Passive whole-body control for quadruped robots: experimental validation over challenging terrain. IEEE Robot Autom Lett 4(3):2553\u20132560","journal-title":"IEEE Robot Autom Lett"},{"issue":"3","key":"6584_CR10","doi-asserted-by":"publisher","first-page":"867","DOI":"10.1109\/TSMCB.2010.2097589","volume":"41","author":"C Liu","year":"2011","unstructured":"Liu C, Chen Q, Wang D (2011) Cpg-inspired workspace trajectory generation and adaptive locomotion control for quadruped robots. IEEE Trans Syst Man Cybern B Cybern 41(3):867\u2013880","journal-title":"IEEE Trans Syst Man Cybern B Cybern"},{"issue":"11","key":"6584_CR11","doi-asserted-by":"publisher","first-page":"2015","DOI":"10.1177\/01423312221142564","volume":"45","author":"S Guan","year":"2023","unstructured":"Guan S, Zhuang Z, Tao H, Chen Y, Stojanovic V, Paszke W (2023) Feedback-aided pd-type iterative learning control for time-varying systems with non-uniform trial lengths. Trans Inst Meas Control 45(11):2015\u20132026","journal-title":"Trans Inst Meas Control"},{"key":"6584_CR12","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.jprocont.2023.103112","volume":"132","author":"H Tao","year":"2023","unstructured":"Tao H, Zheng J, Wei J, Paszke W, Rogers E, Stojanovic V (2023) Repetitive process based indirect-type iterative learning control for batch processes with model uncertainty and input delay. J Process Control 132:1\u201312","journal-title":"J Process Control"},{"issue":"12","key":"6584_CR13","doi-asserted-by":"publisher","first-page":"1198","DOI":"10.1038\/s42256-022-00576-3","volume":"4","author":"Y Jin","year":"2022","unstructured":"Jin Y, Liu X, Shao Y, Wang H, Yang W (2022) High-speed quadrupedal locomotion by imitation-relaxation reinforcement learning. Nat Mach Intell 4(12):1198\u20131208","journal-title":"Nat Mach Intell"},{"issue":"4","key":"6584_CR14","doi-asserted-by":"publisher","first-page":"7193","DOI":"10.1109\/LRA.2021.3092647","volume":"6","author":"J Wang","year":"2021","unstructured":"Wang J, Hu C, Zhu Y (2021) Cpg-based hierarchical locomotion control for modular quadrupedal robots using deep reinforcement learning. IEEE Robot Autom Lett 6(4):7193\u20137200","journal-title":"IEEE Robot Autom Lett"},{"key":"6584_CR15","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton RS, Barto AG (1998) Reinforcement learning: an introduction. MIT Press, Cambridge"},{"key":"6584_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.compchemeng.2022.107760","volume":"161","author":"O Dogru","year":"2022","unstructured":"Dogru O, Velswamy K, Ibrahim F, Wu Y, Sundaramoorthy AS, Huang B, Xu S, Nixon M, Bell N (2022) Reinforcement learning approach to autonomous pid tuning. Comput Chem Eng 161:1\u201317","journal-title":"Comput Chem Eng"},{"issue":"6","key":"6584_CR17","doi-asserted-by":"publisher","first-page":"1280","DOI":"10.1007\/s42235-021-00104-w","volume":"18","author":"YK Kim","year":"2021","unstructured":"Kim YK, Seol W, Park J (2021) Biomimetic quadruped robot with a spinal joint and optimal spinal motion via reinforcement learning. J Bionic Eng 18(6):1280\u20131290","journal-title":"J Bionic Eng"},{"key":"6584_CR18","doi-asserted-by":"publisher","first-page":"123182","DOI":"10.1016\/j.energy.2022.123182","volume":"245","author":"Z Chen","year":"2022","unstructured":"Chen Z, Gu H, Shen S, Shen J (2022) Energy management strategy for power-split plug-in hybrid electric vehicle based on mpc and double q-learning. Energy 245:123182","journal-title":"Energy"},{"key":"6584_CR19","doi-asserted-by":"crossref","unstructured":"Ogum BN, Schomaker LR, Carloni R (2024) Learning to walk with deep reinforcement learning: forward dynamic simulation of a physics-based musculoskeletal model of an osseointegrated transfemoral amputee. IEEE Trans Neural Syst Rehabil Eng 32:431\u2013441","DOI":"10.1109\/TNSRE.2024.3352416"},{"issue":"26","key":"6584_CR20","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1126\/scirobotics.aau5872","volume":"4","author":"J Hwangbo","year":"2019","unstructured":"Hwangbo J, Lee J, Dosovitskiy A, Bellicoso D, Tsounis V, Koltun V, Hutter M (2019) Learning agile and dynamic motor skills for legged robots. Sci Robot 4(26):1\u201314","journal-title":"Sci Robot"},{"issue":"47","key":"6584_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1126\/scirobotics.abc5986","volume":"5","author":"J Lee","year":"2020","unstructured":"Lee J, Hwangbo J, Wellhausen L, Koltun V, Hutter M (2020) Learning quadrupedal locomotion over challenging terrain. Sci Robot 5(47):1\u201313","journal-title":"Sci Robot"},{"issue":"7540","key":"6584_CR22","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"key":"6584_CR23","doi-asserted-by":"publisher","first-page":"591","DOI":"10.1007\/s10846-019-01004-2","volume":"96","author":"Y Sun","year":"2019","unstructured":"Sun Y, Cheng J, Zhang G, Xu H (2019) Mapless motion planning system for an autonomous underwater vehicle using policy gradient-based deep reinforcement learning. J Intell Robot Syst 96:591\u2013601","journal-title":"J Intell Robot Syst"},{"key":"6584_CR24","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.knosys.2020.106736","volume":"214","author":"S Han","year":"2021","unstructured":"Han S, Zhou W, L\u00fc S, Yu J (2021) Regularly updated deterministic policy gradient algorithm. Knowl-Based Syst 214:1\u201312","journal-title":"Knowl-Based Syst"},{"key":"6584_CR25","doi-asserted-by":"publisher","first-page":"40","DOI":"10.1016\/j.jprocont.2018.11.004","volume":"75","author":"Y Ma","year":"2019","unstructured":"Ma Y, Zhu W, Benton MG, Romagnoli J (2019) Continuous control of a polymerization system with deep reinforcement learning. J Process Control 75:40\u201347","journal-title":"J Process Control"},{"issue":"2","key":"6584_CR26","doi-asserted-by":"publisher","first-page":"2122","DOI":"10.1109\/JSYST.2022.3222262","volume":"17","author":"OE Egbomwan","year":"2022","unstructured":"Egbomwan OE, Liu S, Chaoui H (2022) Twin delayed deep deterministic policy gradient (td3) based virtual inertia control for inverter-interfacing dgs in microgrids. IEEE Syst J 17(2):2122\u20132132","journal-title":"IEEE Syst J"},{"issue":"5","key":"6584_CR27","doi-asserted-by":"publisher","first-page":"2908","DOI":"10.1109\/TRO.2022.3172469","volume":"38","author":"S Gangapurwala","year":"2022","unstructured":"Gangapurwala S, Geisert M, Orsolino R, Fallon M, Havoutis I (2022) Rloc: Terrain-aware legged locomotion using reinforcement learning and optimal control. IEEE Trans Rob 38(5):2908\u20132927","journal-title":"IEEE Trans Rob"},{"key":"6584_CR28","unstructured":"Fujimoto S, Hoof H, Meger D (2018) Addressing function approximation error in actor-critic methods. In: International conference on machine learning, vol. 80. PMLR, pp 1587\u20131596"},{"key":"6584_CR29","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.apenergy.2022.119353","volume":"321","author":"R Huang","year":"2022","unstructured":"Huang R, He H, Zhao X, Wang Y, Li M (2022) Battery health-aware and naturalistic data-driven energy management for hybrid electric bus based on td3 deep reinforcement learning algorithm. Appl Energy 321:1\u201315","journal-title":"Appl Energy"},{"key":"6584_CR30","doi-asserted-by":"publisher","unstructured":"Wang Y, Jia W, Sun Y (2022) A hierarchical reinforcement learning framework based on soft actor-critic for quadruped gait generation. In: 2022 IEEE international conference on Robotics and Biomimetics (ROBIO). IEEE, pp 1970\u20131975. https:\/\/doi.org\/10.1109\/ROBIO55434.2022.10011919","DOI":"10.1109\/ROBIO55434.2022.10011919"},{"issue":"12","key":"6584_CR31","doi-asserted-by":"publisher","first-page":"1198","DOI":"10.1038\/s42256-022-00576-3","volume":"4","author":"Y Jin","year":"2022","unstructured":"Jin Y, Liu X, Shao Y, Wang H, Yang W (2022) High-speed quadrupedal locomotion by imitation-relaxation reinforcement learning. Nat Mach Intell 4(12):1198\u20131208","journal-title":"Nat Mach Intell"},{"key":"6584_CR32","doi-asserted-by":"publisher","unstructured":"Xie Z, Berseth G, Clary P, Hurst J, Panne M (2018) Feedback control for cassie with deep reinforcement learning. In: 2018 IEEE\/RSJ international conference on Intelligent Robots and Systems (IROS). IEEE, pp 1241\u20131246. https:\/\/doi.org\/10.1109\/IROS.2018.8593722","DOI":"10.1109\/IROS.2018.8593722"},{"issue":"6","key":"6584_CR33","doi-asserted-by":"publisher","first-page":"3812","DOI":"10.1109\/LRA.2023.3271445","volume":"8","author":"G Cheng","year":"2023","unstructured":"Cheng G, Dong L, Cai W, Sun C (2023) Multi-task reinforcement learning with attention-based mixture of experts. IEEE Robot Autom Lett 8(6):3812\u20133819","journal-title":"IEEE Robot Autom Lett"},{"issue":"6","key":"6584_CR34","doi-asserted-by":"publisher","first-page":"7686","DOI":"10.1109\/TPAMI.2022.3223407","volume":"45","author":"C Huang","year":"2022","unstructured":"Huang C, Wang G, Zhou Z, Zhang R, Lin L (2022) Reward-adaptive reinforcement learning: dynamic policy gradient optimization for bipedal locomotion. IEEE Trans Pattern Anal Mach Intell 45(6):7686\u20137695","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"3","key":"6584_CR35","doi-asserted-by":"publisher","first-page":"3121","DOI":"10.1109\/TNNLS.2022.3174051","volume":"35","author":"C Banerjee","year":"2022","unstructured":"Banerjee C, Chen Z, Noman N (2022) Improved soft actor-critic: mixing prioritized off-policy samples with on-policy experiences. IEEE Trans Neural Netw Learn Syst 35(3):3121\u20133129","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"4","key":"6584_CR36","doi-asserted-by":"publisher","first-page":"1608","DOI":"10.1049\/cit2.12195","volume":"8","author":"B Li","year":"2023","unstructured":"Li B, Bai S, Liang S, Ma R, Neretin E, Huang J (2023) Manoeuvre decision-making of unmanned aerial vehicles in air combat based on an expert actor-based soft actor critic algorithm. CAAI Trans Intell Technol 8(4):1608\u20131619","journal-title":"CAAI Trans Intell Technol"},{"key":"6584_CR37","unstructured":"Haarnoja T, Zhou A, Abbeel P, Levine S (2018) Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: International conference on machine learning, vol. 80. PMLR, pp 1861\u20131870"},{"key":"6584_CR38","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.jpowsour.2022.231099","volume":"524","author":"D Xu","year":"2022","unstructured":"Xu D, Cui Y, Ye J, Cha SW, Li A, Zheng C (2022) A soft actor-critic-based energy management strategy for electric vehicles with hybrid energy storage systems. J Power Sourc 524:1\u201313","journal-title":"J Power Sourc"},{"issue":"1","key":"6584_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2021\/9920591","volume":"2021","author":"H Ali","year":"2021","unstructured":"Ali H, Majeed H, Usman I, Almejalli KA (2021) Reducing entropy overestimation in soft actor critic using dual policy network. Wirel Commun Mob Comput 2021(1):1\u201313","journal-title":"Wirel Commun Mob Comput"},{"key":"6584_CR40","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.jpowsour.2023.232648","volume":"559","author":"R Huang","year":"2023","unstructured":"Huang R, He H (2023) Naturalistic data-driven and emission reduction-conscious energy management for hybrid electric vehicle based on improved soft actor-critic algorithm. J Power Sourc 559:1\u201313","journal-title":"J Power Sourc"},{"key":"6584_CR41","doi-asserted-by":"publisher","first-page":"113740","DOI":"10.1109\/ACCESS.2023.3324193","volume":"11","author":"S Li","year":"2023","unstructured":"Li S, He X, Xu X, Zhao T, Song C, Li J (2023) Weapon-target assignment strategy in joint combat decision-making based on multi-head deep reinforcement learning. IEEE Access 11:113740\u2013113751","journal-title":"IEEE Access"},{"key":"6584_CR42","first-page":"1","volume":"30","author":"H Van Seijen","year":"2017","unstructured":"Van Seijen H, Fatemi M, Romoff J, Laroche R, Barnes T, Tsang J (2017) Hybrid reward architecture for reinforcement learning. Adv Neural Inf Process Syst 30:1\u201311","journal-title":"Adv Neural Inf Process Syst"},{"key":"6584_CR43","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Graves A, Antonoglou I, Wierstra D, Riedmiller M (2013) Playing atari with deep reinforcement learning. arXiv:1312.5602"},{"issue":"4","key":"6584_CR44","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1109\/TAI.2021.3125918","volume":"3","author":"A Tittaferrante","year":"2021","unstructured":"Tittaferrante A, Yassine A (2021) Multiadvisor reinforcement learning for multiagent multiobjective smart home energy control. IEEE Trans Artif Intell 3(4):581\u2013594","journal-title":"IEEE Trans Artif Intell"},{"key":"6584_CR45","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.ast.2023.108689","volume":"142","author":"J Qi","year":"2023","unstructured":"Qi J, Gao H, Su H, Han L, Su B, Huo M, Yu H, Deng Z (2023) Reinforcement learning-based stable jump control method for asteroid-exploration quadruped robots. Aerosp Sci Technol 142:1\u201312","journal-title":"Aerosp Sci Technol"},{"key":"6584_CR46","doi-asserted-by":"crossref","unstructured":"Grondman I, Busoniu L, Lopes GA, Babuska R (2012) A survey of actor-critic reinforcement learning: standard and natural policy gradients. IEEE Trans Syst Man Cybern Pt C Appl Rev 42(6):1291\u20131307","DOI":"10.1109\/TSMCC.2012.2218595"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06584-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-06584-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06584-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T13:57:06Z","timestamp":1758290226000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-06584-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,5]]},"references-count":46,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,7]]}},"alternative-id":["6584"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-06584-1","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,5]]},"assertion":[{"value":"17 April 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 May 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}},{"value":"Ethical approval was not required for this study as it involved no human or animal subjects.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Consent"}}],"article-number":"720"}}