{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T18:49:40Z","timestamp":1772736580449,"version":"3.50.1"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2024,8,9]],"date-time":"2024-08-09T00:00:00Z","timestamp":1723161600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,8,9]],"date-time":"2024-08-09T00:00:00Z","timestamp":1723161600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61773402"],"award-info":[{"award-number":["61773402"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Major Program\/Open Project of Xiangjiang Laboratory","award":["23XJ01010"],"award-info":[{"award-number":["23XJ01010"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s13042-024-02283-4","type":"journal-article","created":{"date-parts":[[2024,8,9]],"date-time":"2024-08-09T03:30:47Z","timestamp":1723174247000},"page":"5839-5861","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["A deep reinforcement learning control method guided by RBF-ARX pseudo LQR"],"prefix":"10.1007","volume":"15","author":[{"given":"Tianbo","family":"Peng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hui","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,8,9]]},"reference":[{"issue":"1","key":"2283_CR1","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1002\/oca.2240","volume":"38","author":"X Zhang","year":"2017","unstructured":"Zhang X, Cheng L, Hao S, Gao W, Lai Y (2017) Optimization design of RBF-ARX model and application research on flatness control system. Optimal Control Appl Methods 38(1):19\u201335","journal-title":"Optimal Control Appl Methods"},{"issue":"1","key":"2283_CR2","doi-asserted-by":"publisher","first-page":"191","DOI":"10.1109\/TCST.2008.922507","volume":"17","author":"V Haggan-Ozaki","year":"2009","unstructured":"Haggan-Ozaki V, Ozaki T, Toyoda Y (2009) An Akaike state-space controller for RBF-ARX models. IEEE Trans Control Syst Technol 17(1):191\u2013198","journal-title":"IEEE Trans Control Syst Technol"},{"issue":"3","key":"2283_CR3","doi-asserted-by":"publisher","first-page":"2530","DOI":"10.1109\/TAES.2022.3215946","volume":"59","author":"Y Zhou","year":"2023","unstructured":"Zhou Y, Ling K, Ding F, Hu Y (2023) Online network-based identification and its application in satellite attitude control systems. IEEE Trans Aerosp Electron Syst 59(3):2530\u20132543","journal-title":"IEEE Trans Aerosp Electron Syst"},{"key":"2283_CR4","doi-asserted-by":"publisher","first-page":"106","DOI":"10.1016\/j.enconman.2012.04.013","volume":"64","author":"P Casanova-Pelaez","year":"2012","unstructured":"Casanova-Pelaez P, Cruz-Peragon F, Palomar-Carnicero J, Dorado R, Lopez-Garcia R (2012) RBF-ARX model of an industrial furnace for drying olive pomace. Energy Convers Manage 64:106\u2013112","journal-title":"Energy Convers Manage"},{"issue":"5","key":"2283_CR5","first-page":"2774","volume":"71","author":"C Li","year":"2024","unstructured":"Li C, You C, Gu Y, Zhu Y (2024) Parameter identification of the RBF-ARX model based on the hybrid whale optimization algorithm. IEEE Trans Circuits and Syst II-Express Briefs 71(5):2774\u20132778","journal-title":"IEEE Trans Circuits and Syst II-Express Briefs"},{"issue":"2","key":"2283_CR6","doi-asserted-by":"publisher","first-page":"351","DOI":"10.1080\/00207179.2019.1594386","volume":"94","author":"X Tian","year":"2021","unstructured":"Tian X, Peng H, Zeng X, Zhou F, Xu W, Peng X (2021) A modelling and predictive control approach to linear two-stage inverted pendulum based on RBF-ARX model. Int J Control 94(2):351\u2013369","journal-title":"Int J Control"},{"issue":"7540","key":"2283_CR7","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"key":"2283_CR8","doi-asserted-by":"publisher","unstructured":"Hasselt H, Guez A, Silver D (2016) Deep reinforcement learning with double Q-Learning. in: Proceedings of the thirtieth AAAI Conference on Artificial Intelligence. pp. 2094\u20132100. https:\/\/doi.org\/10.1609\/aaai.v30i1.10295","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"2283_CR9","doi-asserted-by":"publisher","first-page":"2579","DOI":"10.1109\/TSP.2023.3268475","volume":"71","author":"H Shen","year":"2023","unstructured":"Shen H, Zhang K, Hong M, Chen T (2023) Towards understanding asynchronous advantage actor-critic: convergence and linear speedup. IEEE Trans Signal Process 71:2579\u20132594","journal-title":"IEEE Trans Signal Process"},{"key":"2283_CR10","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3265358","author":"H Li","year":"2023","unstructured":"Li H, He H (2023) Multiagent trust region policy optimization. IEEE Trans Neural Netw Learn Syst. https:\/\/doi.org\/10.1109\/TNNLS.2023.3265358","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"2283_CR11","doi-asserted-by":"publisher","first-page":"750","DOI":"10.1016\/j.ins.2022.07.111","volume":"609","author":"J Zhang","year":"2022","unstructured":"Zhang J, Zhang Z, Han S, Lue S (2022) Proximal policy optimization via enhanced exploration efficiency. Inf Sci 609:750\u2013765","journal-title":"Inf Sci"},{"key":"2283_CR12","doi-asserted-by":"publisher","DOI":"10.1111\/exsy.13205","author":"Q Wang","year":"2023","unstructured":"Wang Q, Sanchez F, McCarthy R, Bulens D, McGuinness K, O\u2019Connor N, W\u00fcthrich M, Widmaier F, Bauer S, Redmond S (2023) Dexterous robotic manipulation using deep reinforcement learning and knowledge transfer for complex sparse reward-based tasks. Expert Syst. https:\/\/doi.org\/10.1111\/exsy.13205","journal-title":"Expert Syst"},{"key":"2283_CR13","doi-asserted-by":"publisher","DOI":"10.1016\/j.jobe.2023.106774","volume":"73","author":"Y Xu","year":"2023","unstructured":"Xu Y, Gao W, Li Y, Xiao F (2023) Operational optimization for the grid-connected residential photovoltaic-battery system using model-based reinforcement learning. J Build Eng 73:106774","journal-title":"J Build Eng"},{"key":"2283_CR14","doi-asserted-by":"publisher","first-page":"281","DOI":"10.1007\/s10844-022-00738-0","volume":"60","author":"M Ghanem","year":"2023","unstructured":"Ghanem M, Chen T, Nepomuceno E (2023) Hierarchical reinforcement learning for efficient and effective automated penetration testing of large networks. J Intell Inf Syst 60:281\u2013303","journal-title":"J Intell Inf Syst"},{"key":"2283_CR15","doi-asserted-by":"publisher","first-page":"16893","DOI":"10.1007\/s10489-022-04354-x","volume":"53","author":"G Wu","year":"2023","unstructured":"Wu G, Fang W, Wang J, Ge P, Cao J, Ping Y, Gou P (2023) Dyna-PPO reinforcement learning with Gaussian process for the continuous action decision-making in autonomous driving. Appl Intell 53:16893\u201316907","journal-title":"Appl Intell"},{"key":"2283_CR16","doi-asserted-by":"publisher","first-page":"2669","DOI":"10.1007\/s12555-020-0788-8","volume":"20","author":"Y Liu","year":"2022","unstructured":"Liu Y, Chen Z, Li Y, Lu M, Chen C, Zhang X (2022) Robot search path planning method based on prioritized deep reinforcement learning. Int J Control Autom Syst 20:2669\u20132680","journal-title":"Int J Control Autom Syst"},{"key":"2283_CR17","doi-asserted-by":"publisher","unstructured":"Nair A, McGrew B, Andrychowicz M, Zaremba W, Abbeel P (2018) Overcoming exploration in reinforcement learning with demonstrations. In: 2018 IEEE International Conference on Robotics and Automation (ICRA). pp. 6292\u20136299. https:\/\/doi.org\/10.1109\/ICRA.2018.8463162","DOI":"10.1109\/ICRA.2018.8463162"},{"key":"2283_CR18","doi-asserted-by":"publisher","first-page":"3296","DOI":"10.1007\/s12555-021-0511-4","volume":"20","author":"CH Min","year":"2022","unstructured":"Min CH, Song JB (2022) Hierarchical end-to-end control policy for multi-degree-of-freedom manipulators. Int J Control Autom Syst 20:3296\u20133311","journal-title":"Int J Control Autom Syst"},{"key":"2283_CR19","unstructured":"Ross S, Gordon G, Bagnell D (2011) A Reduction of imitation learning and structured prediction to no-regret online learning. In: Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics. pp. 627\u2013635."},{"key":"2283_CR20","doi-asserted-by":"publisher","first-page":"563","DOI":"10.1007\/s12555-021-0642-7","volume":"21","author":"W Li","year":"2023","unstructured":"Li W, Yue M, Shangguan J, Jin Y (2023) Navigation of mobile robots based on deep reinforcement learning: reward function optimization and knowledge transfer. Int J Control Autom Syst 21:563\u2013574","journal-title":"Int J Control Autom Syst"},{"key":"2283_CR21","doi-asserted-by":"publisher","unstructured":"Peng T, Peng H, Liu F (2022) Guided deep reinforcement learning based on RBF-ARX pseudo LQR in single stage inverted pendulum. In: 2022 International Conference on Intelligent Systems and Computational Intelligence (ICISCI). pp. 62\u201367. https:\/\/doi.org\/10.1109\/ICISCI53188.2022.9941450","DOI":"10.1109\/ICISCI53188.2022.9941450"},{"key":"2283_CR22","doi-asserted-by":"publisher","unstructured":"Zeng X, Peng H, Li A (2023) Effective and stable role-based multi-agent collaboration by structural information principles. In: Proceedings of the AAAI Conference on Artificial Intelligence. pp. 11772\u201311780.\u00a0https:\/\/doi.org\/10.1609\/aaai.v37i10.26390","DOI":"10.1609\/aaai.v37i10.26390"},{"key":"2283_CR23","doi-asserted-by":"crossref","unstructured":"Zeng X, Peng H, Li A, Liu C, He L, Yu PS (2023) Hierarchical state abstraction based on structural information principles. In: Proceedings of the Thirty-Second International Joint Conference on Artificial Intelligence (IJCAI-23). pp. 4549\u20134557.","DOI":"10.24963\/ijcai.2023\/506"},{"key":"2283_CR24","doi-asserted-by":"publisher","first-page":"287","DOI":"10.1016\/S0005-1098(99)00140-5","volume":"36","author":"KJ Astr\u00f6m","year":"2000","unstructured":"Astr\u00f6m KJ, Furuta K (2000) Swinging up a pendulum by energy control. Automatica 36:287\u2013295","journal-title":"Automatica"},{"key":"2283_CR25","first-page":"2897","volume":"73","author":"A Shahzad","year":"2022","unstructured":"Shahzad A, Munshi S, Azam S, Khan MN (2022) Design and implementation of a state-feedback controller using LQR technique. Comput Mater Cont 73:2897\u20132911","journal-title":"Comput Mater Cont"},{"key":"2283_CR26","doi-asserted-by":"publisher","first-page":"17397","DOI":"10.1007\/s00521-023-08599-6","volume":"35","author":"Z Ben Hazem","year":"2023","unstructured":"Ben Hazem Z, Binguel Z (2023) A comparative study of anti-swing radial basis neural-fuzzy LQR controller for multi-degree-of-freedom rotary pendulum systems. Neural Comput Appl 35:17397\u201317413","journal-title":"Neural Comput Appl"},{"key":"2283_CR27","doi-asserted-by":"publisher","first-page":"2564","DOI":"10.1002\/asjc.2978","volume":"25","author":"FM Escalante","year":"2023","unstructured":"Escalante FM, Jutinico AL, Terra MH, Siqueira AAG (2023) Robust linear quadratic regulator applied to an inverted pendulum. Asian J Control 25:2564\u20132576","journal-title":"Asian J Control"},{"key":"2283_CR28","doi-asserted-by":"publisher","first-page":"5115","DOI":"10.1007\/s00500-022-07008-9","volume":"26","author":"S Alimoradpour","year":"2022","unstructured":"Alimoradpour S, Rafie M, Ahmadzadeh B (2022) Providing a genetic algorithm-based method to optimize the fuzzy logic controller for the inverted pendulum. Soft Comput 26:5115\u20135130","journal-title":"Soft Comput"},{"key":"2283_CR29","doi-asserted-by":"publisher","DOI":"10.1111\/exsy.13140","author":"V Srivastava","year":"2022","unstructured":"Srivastava V, Srivastava S, Chaudhary G, Valencia XPB (2022) Performance improvement and Lyapunov stability analysis of nonlinear systems using hybrid optimization techniques. Expert Syst. https:\/\/doi.org\/10.1111\/exsy.13140","journal-title":"Expert Syst"},{"key":"2283_CR30","doi-asserted-by":"publisher","DOI":"10.1080\/00207179.2023.2177135","author":"X Cai","year":"2023","unstructured":"Cai X, Lou XY (2023) Reset control and Script capital L2-gain analysis of piecewise-affine systems. Int J Control. https:\/\/doi.org\/10.1080\/00207179.2023.2177135","journal-title":"Int J Control"},{"issue":"18","key":"2283_CR31","doi-asserted-by":"publisher","first-page":"9150","DOI":"10.1016\/j.jfranklin.2017.01.035","volume":"355","author":"H Gritli","year":"2018","unstructured":"Gritli H, Belghith S (2018) Robust feedback control of the underactuated inertia wheel inverted pendulum under parametric uncertainties and subject to external disturbances: LMI formulation. J Franklin Inst 355(18):9150\u20139191","journal-title":"J Franklin Inst"},{"key":"2283_CR32","unstructured":"Fujimoto S, Meger D, Precup D (2019) Off-policy deep reinforcement learning without exploration. In: Proceedings of the 36th International Conference on Machine Learning. pp. 2052\u20132062. https:\/\/proceedings.mlr.press\/v97\/fujimoto19a.html"},{"key":"2283_CR33","doi-asserted-by":"publisher","first-page":"1274","DOI":"10.1109\/TMC.2019.2908171","volume":"19","author":"CH Liu","year":"2020","unstructured":"Liu CH, Ma XX, Gao XD, Tang J (2020) Distributed energy-efficient multi-UAV navigation for long-term communication coverage by deep reinforcement learning. IEEE Trans Mob Comput 19:1274\u20131285","journal-title":"IEEE Trans Mob Comput"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-024-02283-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-024-02283-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-024-02283-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T10:28:44Z","timestamp":1730197724000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-024-02283-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,9]]},"references-count":33,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["2283"],"URL":"https:\/\/doi.org\/10.1007\/s13042-024-02283-4","relation":{},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"value":"1868-8071","type":"print"},{"value":"1868-808X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8,9]]},"assertion":[{"value":"1 November 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 July 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 August 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflicts of interest to declare that are relevant to the content of this paper.The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Manuscript is approved by all authors for publication.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}]}}