{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T18:37:21Z","timestamp":1781635041888,"version":"3.54.5"},"reference-count":85,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2022,8,20]],"date-time":"2022-08-20T00:00:00Z","timestamp":1660953600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,8,20]],"date-time":"2022-08-20T00:00:00Z","timestamp":1660953600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Intell Robot Syst"],"published-print":{"date-parts":[[2022,9]]},"DOI":"10.1007\/s10846-022-01656-7","type":"journal-article","created":{"date-parts":[[2022,8,20]],"date-time":"2022-08-20T05:02:50Z","timestamp":1660971770000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Learning Push Recovery Behaviors for Humanoid Walking Using Deep Reinforcement Learning"],"prefix":"10.1007","volume":"106","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1263-5110","authenticated-orcid":false,"given":"Dicksiano C.","family":"Melo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2944-4476","authenticated-orcid":false,"given":"Marcos R. O. A.","family":"Maximo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2399-5066","authenticated-orcid":false,"given":"Adilson Marques","family":"da Cunha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,8,20]]},"reference":[{"key":"1656_CR1","doi-asserted-by":"publisher","unstructured":"Abreu, M., Lau, N., Sousa, A., Reis, L. P.: Learning Low Level Skills from Scratch for Humanoid Robot Soccer Using Deep Reinforcement Learning. In: 2019 IEEE International Conference on Autonomous Robot Systems and Competitions (ICARSC), pp. 1\u20138 (2019), https:\/\/doi.org\/10.1109\/ICARSC.2019.8733632","DOI":"10.1109\/ICARSC.2019.8733632"},{"key":"1656_CR2","doi-asserted-by":"crossref","unstructured":"Abreu, M., Reis, L. P., Lau, N.: Learning to Run Faster in a Humanoid Robot Soccer Environment through Reinforcement Learning. In: Chalup, S., Niemueller, T., Suthakorn, J., Williams, M. A. (eds.) Robocup 2019: Robot World Cup XXIII, pp 3\u201315. Springer International Publishing, Cham (2019)","DOI":"10.1007\/978-3-030-35699-6_1"},{"key":"1656_CR3","doi-asserted-by":"crossref","unstructured":"Abreu, M., Simes, D., Lau, N., Reis, L.P.: Fast, human-like running and sprinting. https:\/\/archive.robocup.info\/Soccer\/Simulation\/3D\/FCPs\/RoboCup\/2019\/FCPortugal_SS3D_RC2019_FCP.pdf (2019)","DOI":"10.1007\/978-3-030-35699-6_1"},{"key":"1656_CR4","unstructured":"de Albuquerque Maximo, M. R. O.: Automatic Walking Step Duration through Model Predictive Control. Ph.D. thesis, Aeronautics Institute of Technology (2017)"},{"key":"1656_CR5","unstructured":"Bain, M., Sammut, C.: A Framework for Behavioural Cloning. In: Machine Intelligence 15 (1995)"},{"key":"1656_CR6","doi-asserted-by":"crossref","unstructured":"Carvalho Melo, D., Quartucci Forster, C.H., Omena de Albuquerque Maximo\u0301, M.R.: Learning When to Kick through Deep Neural Networks. In: 2019 Latin American Robotics Symposium (LARS), 2019 Brazilian Symposium on Robotics (SBR) and 2019 Workshop on Robotics in Education (WRE), pp. 43\u201348 (2019)","DOI":"10.1109\/LARS-SBR-WRE48964.2019.00016"},{"key":"1656_CR7","doi-asserted-by":"crossref","unstructured":"Carvalho Melo, L., Omena Albuquerque Maximo\u0301, M.R.: Learning Humanoid Robot Running Skills through Proximal Policy Optimization. In: 2019 Latin American Robotics Symposium (LARS), 2019 Brazilian Symposium on Robotics (SBR) and 2019 Workshop on Robotics in Education (WRE), pp. 37\u201342 (2019)","DOI":"10.1109\/LARS-SBR-WRE48964.2019.00015"},{"key":"1656_CR8","doi-asserted-by":"crossref","unstructured":"Chaffre, T., Moras, J., Chan-Hon-Tong, A., Marzat, J.: Sim-to-real transfer with incremental environment complexity for reinforcement learning of depth-based robot navigation (2020)","DOI":"10.5220\/0009821603140323"},{"key":"1656_CR9","unstructured":"Colas, C., Sigaud, O., Oudeyer, P.: How many random seeds? statistical power analysis in deep reinforcement learning experiments. arXiv:1806.08295 (2018)"},{"key":"1656_CR10","doi-asserted-by":"crossref","unstructured":"Depinet, M., MacAlpine, P., Stone, P.: Keyframe Sampling, Optimization, and Behavior Integration: Towards Long-Distance Kicking in the Robocup 3D Simulation League. In: Bianchi, R. A. C., Akin, H. L., Ramamoorthy, S., Sugiura, K. (eds.) RoboCup-2014: Robot Soccer World Cup XVIII, Lecture Notes in Artificial Intelligence. Springer Verlag, Berlin (2015)","DOI":"10.1007\/978-3-319-18615-3_47"},{"key":"1656_CR11","unstructured":"Dhariwal, P., Hesse, C., Klimov, O., Nichol, A., Plappert, M., Radford, A., Schulman, J., Sidor, S., Wu, Y., Zhokhov, P.: Openai baselines https:\/\/github.com\/openai\/baselines (2017)"},{"key":"1656_CR12","doi-asserted-by":"crossref","unstructured":"Dorer, K.: Learning to Use Toes in a Humanoid Robot. In: Akiyama, H., Obst, O., Sammut, C., Tonidandel, F. (eds.) Robocup 2017: Robot World Cup XXI, pp 168\u2013179. Springer International Publishing, Cham (2018)","DOI":"10.1007\/978-3-030-00308-1_14"},{"key":"1656_CR13","unstructured":"Duan, Y., Andrychowicz, M., Stadie, B.C., Ho, J., Schneider, J., Sutskever, I., Abbeel, P., Zaremba, W.: One-shot imitation learning. arXiv:1703.07326 (2017)"},{"issue":"1","key":"1656_CR14","doi-asserted-by":"publisher","first-page":"93","DOI":"10.1002\/ajpa.1330690111","volume":"69","author":"DC Dunbar","year":"1986","unstructured":"Dunbar, D. C., Horak, F. B., Macpherson, J., Rushmer, D. S.: Neural control of quadrupedal and bipedal stance: implications for the evolution of erect posture. American journal of physical anthropology 69 (1), 93\u2013105 (1986)","journal-title":"American journal of physical anthropology"},{"issue":"1","key":"1656_CR15","doi-asserted-by":"publisher","first-page":"54","DOI":"10.1214\/ss\/1177013815","volume":"1","author":"B Efron","year":"1986","unstructured":"Efron, B., Tibshirani, R.: Bootstrap methods for standard errors, confidence intervals, and other measures of statistical accuracy. Statist. Sci. 1(1), 54\u201375 (1986). https:\/\/doi.org\/10.1214\/ss\/1177013815","journal-title":"Statist. Sci."},{"key":"1656_CR16","unstructured":"Farchy, A., Barrett, S., MacAlpine, P., Stone, P.: Humanoid Robots Learning to Walk Faster: from the Real World to Simulation and Back. In: Proceedings of 12Th International Conference on Autonomous Agents and Multiagent Systems (AAMAS) (2013)"},{"key":"1656_CR17","unstructured":"Fischer, J., Dorer, K.: Learning a walk behavior utilizing toes from scratch. https:\/\/archive.robocup.info\/Soccer\/Simulation\/3D\/FCPs\/RoboCup\/2019\/magmaOffenburg_SS3D_RC2019_FCP.pdf (2019)"},{"key":"1656_CR18","unstructured":"Goodfellow, I., Bengio, Y., Courville, A.: Deep learning. MIT press (2016)"},{"key":"1656_CR19","unstructured":"Goodfellow, I. J., Mirza, M., Xiao, D., Courville, A., Bengio, Y.: an empirical investigation of catastrophic forgetting in gradient-based neural networks (2015)"},{"key":"1656_CR20","unstructured":"Haarnoja, T., Zhou, A., Abbeel, P., Levine, S.: Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. arXiv:1801.01290 (2018)"},{"key":"1656_CR21","unstructured":"Hofmann, A.: Robust execution of bipedal walking tasks from biomechanical principles (2006)"},{"key":"1656_CR22","doi-asserted-by":"publisher","first-page":"517","DOI":"10.1093\/ptj\/77.5.517","volume":"77","author":"F Horak","year":"1997","unstructured":"Horak, F., Henry, S., Shumway-Cook, A.: Postural perturbations: New insights for treatment of balance disorders. Physical therapy 77, 517\u201333 (1997). https:\/\/doi.org\/10.1093\/ptj\/77.5.517","journal-title":"Physical therapy"},{"key":"1656_CR23","doi-asserted-by":"crossref","unstructured":"Horak, F., Macpherson, J.: Postural Orientation and Equilibrium. In: Handbook of Physiology. Exercise: Regulation and Integration of Multiple Systems. MD1 am Physiol Soc pp. 255\u2013292 (1996)","DOI":"10.1002\/cphy.cp120107"},{"key":"1656_CR24","doi-asserted-by":"crossref","unstructured":"James, S., Wohlhart, P., Kalakrishnan, M., Kalashnikov, D., Irpan, A., Ibarz, J., Levine, S., Hadsell, R., Bousmalis, K.: Sim-to-real via sim-to-sim: Data-efficient robotic grasping via randomized-to-canonical adaptation networks (2019)","DOI":"10.1109\/CVPR.2019.01291"},{"key":"1656_CR25","unstructured":"Kajita, S., Kanehiro, F., Kaneko, K., Yokoi, K., Hirukawa, H.: The 3D Linear Inverted Pendulum mode: A simple modeling for a biped walking pattern generation. In: Proceedings of the 2001IEEE\/RSJ International Conference on Intelligent Robots and Systems. IEEE, Hawaii, USA (2001)"},{"key":"1656_CR26","doi-asserted-by":"publisher","unstructured":"Kim, H., Seo, D., Kim, D.: Push Recovery Control for Humanoid Robot Using Reinforcement Learning. In: 2019 Third IEEE International Conference on Robotic Computing (IRC), pp. 488\u2013492 (2019), https:\/\/doi.org\/10.1109\/IRC.2019.00102","DOI":"10.1109\/IRC.2019.00102"},{"key":"1656_CR27","unstructured":"Leike, J., Martic, M., Krakovna, V., Ortega, P. A., Everitt, T., Lefrancq, A., Orseau, L., Legg, S.: Ai safety gridworlds (2017)"},{"key":"1656_CR28","unstructured":"Lillicrap, T.P., Hunt, J.J., Pritzel, A., Heess, N., Erez, T., Tassa, Y., Silver, D., Wierstra, D.: Continuous control with deep reinforcement learning. arXiv:1509.02971 (2015)"},{"key":"1656_CR29","unstructured":"MacAlpine, P., Barrett, S., Urieli, D., Vu, V., Stone, P.: Design and optimization of an omnidirectional humanoid walk: A winning approach at the roboCup 2011 3D simulation competition. In: Proceedings of the Twenty-Sixth AAAI Conference on Artificial Intelligence (AAAI) (2012)"},{"key":"1656_CR30","doi-asserted-by":"crossref","unstructured":"MacAlpine, P., Collins, N., Lopez-Mobilia, A., Stone, P.: UT Austin Villa: RoboCup 2012 3D Simulation League Champion. In: Chen, X., Stone, P., Sucar, L. E., der Zant, T. V. (eds.) RoboCup-2012: Robot Soccer World Cup XVI, Lecture Notes in Artificial Intelligence. Springer Verlag, Berlin (2013)","DOI":"10.1007\/978-3-642-39250-4_8"},{"key":"1656_CR31","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1016\/j.artint.2017.09.001","volume":"254","author":"P MacAlpine","year":"2018","unstructured":"MacAlpine, P., Stone, P.: Overlapping layered learning. Artificial Intelligence 254, 21\u201343 (2018). https:\/\/doi.org\/10.1016\/j.artint.2017.09.001 . https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0004370217301066","journal-title":"Artificial Intelligence"},{"key":"1656_CR32","doi-asserted-by":"crossref","unstructured":"MacAlpine, P., Stone, P.: UT Austin Villa: RoboCup 2017 3D Simulation League Competition and Technical Challenges Champions. In: Sammut, C., Obst, O., Tonidandel, F., Akyama, H. (eds.) RoboCup 2017: Robot Soccer World Cup XXI, Lecture Notes in Artificial Intelligence. Springer (2018)","DOI":"10.1007\/978-3-030-00308-1_39"},{"issue":"3","key":"1656_CR33","doi-asserted-by":"publisher","first-page":"172988141667513","DOI":"10.1177\/1729881416675135","volume":"14","author":"MR Maximo","year":"2017","unstructured":"Maximo, M.R., Colombini, E.L., Ribeiro, C.H.: Stable and fast model-free walk with arms movement for humanoid robots. International Journal of Advanced Robotic Systems 14(3), 1729881416675135 (2017). https:\/\/doi.org\/10.1177\/1729881416675135","journal-title":"International Journal of Advanced Robotic Systems"},{"key":"1656_CR34","unstructured":"Maximo, M. R. O. A.: Omnidirectional ZMP-based walking for a humanoid robot. Master\u2019s Thesis, Instituto tecnol\u00f3gico de aeron\u00e1utica, s\u00e3o jos\u00e9 dos Campos, SP Brazil (2015)"},{"key":"1656_CR35","unstructured":"Maximo, M. R. O. A., Ribeiro, C. H. C.: ZMP-Based Humanoid Walking Engine with Arms Movement and Stabilization. In: Proceedings of the 2016 Congresso Brasileiro de Autom\u00e1tica (CBA). SBA, Vit\u00f3ria, ES, Brazil (2016)"},{"key":"1656_CR36","unstructured":"Maximo, M. R. O. A., Ribeiro, C. H. C., Afonso, R. J. M.: Modeling of a position servo used in robotics applications. In: Proceedings of the 2017 Simp\u00f3sio Brasileiro de Automa\u00e7\u00e3o Inteligente (SBAI). SBA, Porto Alegre, SC, Brazil (2017)"},{"key":"1656_CR37","unstructured":"Melo, D. C.: Learning Push Recovery Strategies for Bipedal Walking. Master\u2019s Thesis, Instituto tecnol\u00f3gico de aeron\u00e1utica, s\u00e3o jos\u00e9 dos Campos, SP Brazil (2021)"},{"key":"1656_CR38","doi-asserted-by":"publisher","unstructured":"Melo, D. C., M\u00e1ximo, M.R.O.A., da Cunha, A.M.: Push recovery strategies through deep reinforcement learning. In: 2020 Latin American Robotics Symposium (LARS), 2020 Brazilian Symposium on Robotics (SBR) and 2020 Workshop on Robotics in Education (WRE), pp. 1\u20136 (2020), https:\/\/doi.org\/10.1109\/LARS\/SBR\/WRE51543.2020.9306967","DOI":"10.1109\/LARS\/SBR\/WRE51543.2020.9306967"},{"key":"1656_CR39","unstructured":"Melo, L. C., Maximo, M.R.O.A.: Learning humanoid robot running skills through proximal policy optimization (2019)"},{"key":"1656_CR40","unstructured":"Melo, L. C., Maximo, M. R. O. A., da Cunha, A. M.: Bottom-up meta-policy search. In: Proceedings of the Deep Reinforcement Learning Workshop of NeurIPS 2019 (2019)"},{"key":"1656_CR41","unstructured":"Mitchell, E., Rafailov, R., Peng, X. B., Levine, S., Finn, C.: Offline meta-reinforcement learning with advantage weighting (2020)"},{"key":"1656_CR42","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Graves, A., Antonoglou, I., Wierstra, D., Riedmiller, M.: Playing Atari with Deep Reinforcement Learning. In: NIPS Deep Learning Workshop (2013)"},{"key":"1656_CR43","doi-asserted-by":"publisher","unstructured":"Muniz, F., Maximo, M. R. O. A., Ribeiro, C. H. C.: Keyframe Movement Optimization for Simulated Humanoid Robot Using a Parallel Optimization Framework. In: 2016 XIII Latin American Robotics Symposium and IV Brazilian Robotics Symposium (LARS\/SBR), pp. 79\u201384 (2016), https:\/\/doi.org\/10.1109\/LARS-SBR.2016.20https:\/\/doi.org\/10.1109\/LARS-SBR.2016.20","DOI":"10.1109\/LARS-SBR.2016.20 10.1109\/LARS-SBR.2016.20"},{"key":"1656_CR44","doi-asserted-by":"publisher","unstructured":"Muzio, A., Aguiar, L., Maximo, M. R. O. A., Pinto, S. C.: Monte Carlo Localization with Field Lines Observations for Simulated Humanoid Robotic Soccer. In: 2016 XIII Latin American Robotics Symposium and IV Brazilian Robotics Symposium (LARS\/SBR), pp 334\u2013339. IEEE, Recife, PE, Brazil (2016), https:\/\/doi.org\/10.1109\/LARS-SBR.2016.63","DOI":"10.1109\/LARS-SBR.2016.63"},{"key":"1656_CR45","unstructured":"Muzio, A.F.V.: Deep reinforcement learning applied to humanoid robots (2017)"},{"key":"1656_CR46","doi-asserted-by":"publisher","unstructured":"Muzio, A. F. V., Maximo, M. R. O. A., Yoneyama, T.: Deep Reinforcement Learning for Humanoid Robot Dribbling. In: 2020 Latin American Robotics Symposium (LARS), 2020 Brazilian Symposium on Robotics (SBR) and 2020 Workshop on Robotics in Education (WRE), pp. 1\u20136 (2020), https:\/\/doi.org\/10.1109\/LARS\/SBR\/WRE51543.2020.9307084","DOI":"10.1109\/LARS\/SBR\/WRE51543.2020.9307084"},{"key":"1656_CR47","doi-asserted-by":"crossref","unstructured":"Nashner, L.: Analysis of stance posture in humans (1981)","DOI":"10.1007\/978-1-4684-3884-0_10"},{"issue":"1","key":"1656_CR48","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1017\/S0140525X00020008","volume":"8","author":"LM Nashner","year":"1985","unstructured":"Nashner, L. M., McCollum, G.: The organization of human postural movements: a formal basis and experimental synthesis. Behavioral and Brain Sciences 8(1), 135\u2013150 (1985). https:\/\/doi.org\/10.1017\/S0140525X00020008","journal-title":"Behavioral and Brain Sciences"},{"key":"1656_CR49","unstructured":"Oh, J., Singh, S.P., Lee, H., Kohli, P.: Zero-shot task generalization with multi-task deep reinforcement learning. arXiv:1706.05064 (2017)"},{"key":"1656_CR50","doi-asserted-by":"crossref","unstructured":"OpenAI, Andrychowicz, M., Baker, B., Chociej, M., J\u00f3zefowicz, R., McGrew, B., Pachocki, J., Petron, A., Plappert, M., Powell, G., Ray, A., Schneider, J., Sidor, S., Tobin, J., Welinder, P., Weng, L., Zaremba, W.: Learning dexterous in-hand manipulation. arXiv:1808.00177(2018)","DOI":"10.1177\/0278364919887447"},{"key":"1656_CR51","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1007\/s10514-013-9341-4","volume":"35","author":"DE Orin","year":"2013","unstructured":"Orin, D. E., Goswani, A., Lee, S. H.: Centroidal dynamics of a humanoid robot. Auton. Robot. 35, 161\u2013176 (2013)","journal-title":"Auton. Robot."},{"key":"1656_CR52","doi-asserted-by":"publisher","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. https:\/\/doi.org\/10.3115\/1073083.1073135https:\/\/doi.org\/10.3115\/1073083.1073135 (2002)","DOI":"10.3115\/1073083.1073135 10.3115\/1073083.1073135"},{"key":"1656_CR53","doi-asserted-by":"crossref","unstructured":"Peng, X.B., Abbeel, P., Levine, S., van de Panne, M.: Deepmimic: Example-guided deep reinforcement learning of physics-based character skills. ACM Trans. Graph. 37, 4 (2018)","DOI":"10.1145\/3197517.3201311"},{"key":"1656_CR54","doi-asserted-by":"crossref","unstructured":"Peng, X. B., Berseth, G., Yin, K., van de Panne, M.: Deeploco: Dynamic locomotion skills using hierarchical deep reinforcement learning. ACM Transactions on Graphics (Proc SIGGRAPH 2017) 36(4) (2017)","DOI":"10.1145\/3072959.3073602"},{"key":"1656_CR55","doi-asserted-by":"publisher","unstructured":"Rebula, J., Canas, F., Pratt, J., Goswami, A.: Learning capture points for bipedal push recovery. pp. 1774\u20131774. https:\/\/doi.org\/10.1109\/ROBOT.2008.4543460 (2008)","DOI":"10.1109\/ROBOT.2008.4543460"},{"issue":"11","key":"1656_CR56","doi-asserted-by":"publisher","first-page":"1149","DOI":"10.1016\/S0021-9290(99)00116-5","volume":"32","author":"S Rietdyk","year":"1999","unstructured":"Rietdyk, S., Patla, A., Winter, D., Ishac, M., Little, C.: Balance recovery from medio-lateral perturbations of the upper body during standing. Journal of Biomechanics 32(11), 1149\u20131158 (1999). https:\/\/doi.org\/10.1016\/S0021-9290(99)00116-5. http:\/\/www.sciencedirect.com\/science\/article\/pii\/S0021929099001165","journal-title":"Journal of Biomechanics"},{"issue":"2","key":"1656_CR57","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1016\/S0966-6362(99)00032-6","volume":"10","author":"C Runge","year":"1999","unstructured":"Runge, C., Shupert, C., Horak, F., Zajac, F.: Ankle and hip postural strategies defined by joint torques. Gait and Posture 10(2), 161\u2013170 (1999). https:\/\/doi.org\/10.1016\/S0966-6362(99)00032-6","journal-title":"Gait and Posture"},{"key":"1656_CR58","doi-asserted-by":"publisher","first-page":"233","DOI":"10.1016\/S1364-6613(99)01327-3","volume":"3","author":"S Schaal","year":"1999","unstructured":"Schaal, S.: Is imitation learning the route to humanoid robots? Trends Cogn. Sci. 3, 233\u2013242 (1999)","journal-title":"Trends Cogn. Sci."},{"key":"1656_CR59","doi-asserted-by":"publisher","unstructured":"Schroff, F., Philbin, J.: Facenet: A unified embedding for face recognition and clustering. 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). https:\/\/doi.org\/10.1109\/cvpr.2015.7298682https:\/\/doi.org\/10.1109\/cvpr.2015.7298682 (2015)","DOI":"10.1109\/cvpr.2015.7298682 10.1109\/cvpr.2015.7298682"},{"key":"1656_CR60","unstructured":"Schulman, J., Levine, S., Moritz, P., Jordan, M.I., Abbeel, P.: Trust region policy optimization. arXiv:1502.05477 (2015)"},{"key":"1656_CR61","unstructured":"Schulman, J., Moritz, P., Levine, S., Jordan, M.I., Abbeel, P.: High-dimensional continuous control using generalized advantage estimation. In: Bengio, Y., LeCun, Y. (eds.) 4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings. arXiv:1506.02438 (2016)"},{"key":"1656_CR62","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv:1707.06347 (2017)"},{"key":"1656_CR63","doi-asserted-by":"publisher","unstructured":"Yi, S.-J., Zhang, B.-T., Hong, D., Lee, D.D.: Learning Full Body Push Recovery Control for Small Humanoid Robots. In: 2011 IEEE International Conference on Robotics and Automation, pp. 2047\u20132052 (2011), https:\/\/doi.org\/10.1109\/ICRA.2011.5980531","DOI":"10.1109\/ICRA.2011.5980531"},{"key":"1656_CR64","doi-asserted-by":"crossref","unstructured":"Shafiee-Ashtiani, M., Yousefi-Koma, A., Mirjalili, R., Maleki, H., Karimi, M.: Push recovery of a position-controlled humanoid robot based on capture point feedback control (2017)","DOI":"10.1109\/ICRoM.2017.8466226"},{"key":"1656_CR65","doi-asserted-by":"crossref","unstructured":"Shafii, N., Aslani, S., Nezami, O. M., Shiry, S.: Evolution of Biped Walking Using Truncated Fourier Series and Particle Swarm Optimization. In: Robocup 2009: Robot Soccer World Cup XIII, pp 344\u2013354. Springer, Singapore (2010)","DOI":"10.1007\/978-3-642-11876-0_30"},{"key":"1656_CR66","volume-title":"Introduction to autonomous mobile robots","author":"R Siegwart","year":"2011","unstructured":"Siegwart, R., Nourbakhsh, I. R., Scaramuzza, D.: Introduction to autonomous mobile robots. The MIT press, Cambridge (2011)"},{"key":"1656_CR67","doi-asserted-by":"crossref","unstructured":"Singh, A., Jang, E., Irpan, A., Kappler, D., Dalal, M., Levine, S., Khansari, M., Finn, C.: Scalable multi-task imitation learning with autonomous improvement (2020)","DOI":"10.1109\/ICRA40945.2020.9197020"},{"key":"1656_CR68","doi-asserted-by":"publisher","unstructured":"Stephens, B.: Humanoid Push Recovery. In: 2007 7Th IEEE-RAS International Conference on Humanoid Robots, pp. 589\u2013595 (2007), https:\/\/doi.org\/10.1109\/ICHR.2007.4813931","DOI":"10.1109\/ICHR.2007.4813931"},{"key":"1656_CR69","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction, 2nd edn. The MIT Press. http:\/\/incompleteideas.net\/book\/the-book-2nd.html (2018)"},{"key":"1656_CR70","doi-asserted-by":"crossref","unstructured":"Tan, C., Sun, F., Kong, T., Zhang, W., Yang, C., Liu, C.: A survey on deep transfer learning. arXiv:1808.01974 (2018)","DOI":"10.1007\/978-3-030-01424-7_27"},{"key":"1656_CR71","unstructured":"Tanwani, A.K.: Domain-invariant representation learning for sim-to-real transfer (2020)"},{"key":"1656_CR72","unstructured":"Tedrake, R. L.: Applied Optimal Control for Dynamically Stable Legged Locomotion. Ph.D. thesis Massachusetts Institute of Technology (2004)"},{"key":"1656_CR73","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-29678-2\u2216_4716 10.1007\/978-3-540-29678-2\u2216_4716","volume-title":"Postural Synergies, pp. 3228\u20133233","author":"LH Ting","year":"2009","unstructured":"Ting, L.H.: Postural Synergies, pp. 3228\u20133233. Springer, Berlin Heidelberg (2009). https:\/\/doi.org\/10.1007\/978-3-540-29678-2\u2216_4716https:\/\/doi.org\/10.1007\/978-3-540-29678-2\u2216_4716"},{"key":"1656_CR74","doi-asserted-by":"crossref","unstructured":"Todorov, E., Erez, T., Tassa, Y.: Mujoco: a Physics Engine for Model-Based Control. In: IROS, pp. 5026\u20135033. IEEE (2012)","DOI":"10.1109\/IROS.2012.6386109"},{"key":"1656_CR75","doi-asserted-by":"crossref","unstructured":"Torabi, F., Warnell, G., Stone, P.: Behavioral cloning from observation. arXiv:1805.01954(2018)","DOI":"10.24963\/ijcai.2018\/687"},{"key":"1656_CR76","unstructured":"Vatankhah, H., Lau, N., MacAlpine, P., van Dijk, S., Glaser, S.: Simspark https:\/\/gitlab.com\/robocup-sim\/SimSpark (2018)"},{"issue":"1","key":"1656_CR77","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1142\/S0219843604000083","volume":"1","author":"M Vukobratovi\u0107","year":"2004","unstructured":"Vukobratovi\u0107, M., Borovac, B.: Zero-Moment Point \u2013 thirty five years of its life. International Journal of Humanoid Robots 1(1), 157\u2013173 (2004)","journal-title":"International Journal of Humanoid Robots"},{"key":"1656_CR78","unstructured":"Wang, Z., Bapst, V., Heess, N., Mnih, V., Munos, R., Kavukcuoglu, K., de Freitas, N.: Sample efficient actor-critic with experience replay. arXiv:1611.01224 (2016)"},{"key":"1656_CR79","unstructured":"Xie, Z., Clary, P., Dao, J., Morais, P., Hurst, J., van de Panne, M.: Iterative reinforcement learning based design of dynamic locomotion skills for cassie (2019)"},{"key":"1656_CR80","doi-asserted-by":"crossref","unstructured":"Xu, Y., Vatankhah, H.: Simspark: an Open Source Robot Simulator Developed by the Robocup Community. In: Behnke, S., Veloso, M., Visser, A., Xiong, R. (eds.) Robocup 2013: Robot World Cup XVII, pp 632\u2013639. Springer, Berlin, Heidelberg (2014)","DOI":"10.1007\/978-3-662-44468-9_59"},{"key":"1656_CR81","doi-asserted-by":"publisher","unstructured":"Yang, C., Komura, T., Li, Z.: Emergence of Human-Comparable Balancing Behaviours by Deep Reinforcement Learning. In: 2017 IEEE-RAS 17Th International Conference on Humanoid Robotics (Humanoids), pp. 372\u2013377 (2017), https:\/\/doi.org\/10.1109\/HUMANOIDS.2017.8246900","DOI":"10.1109\/HUMANOIDS.2017.8246900"},{"key":"1656_CR82","doi-asserted-by":"publisher","unstructured":"Yang, C., Yuan, K., Merkt, W., Komura, T., Vijayakumar, S., Li, Z.: Learning Whole-Body Motor Skills for Humanoids. In: 2018 IEEE-RAS 18Th International Conference on Humanoid Robots (Humanoids), pp. 270\u2013276 (2018), https:\/\/doi.org\/10.1109\/HUMANOIDS.2018.8625045","DOI":"10.1109\/HUMANOIDS.2018.8625045"},{"key":"1656_CR83","doi-asserted-by":"publisher","unstructured":"Yi, S., Zhang, B., Hong, D., Lee, D. D.: Online Learning of Low Dimensional Strategies for High-Level Push Recovery in Bipedal Humanoid Robots. In: 2013 IEEE International Conference on Robotics and Automation, pp. 1649\u20131655 (2013), https:\/\/doi.org\/10.1109\/ICRA.2013.6630791","DOI":"10.1109\/ICRA.2013.6630791"},{"key":"1656_CR84","doi-asserted-by":"publisher","unstructured":"Yi, S. J., Zhang, B. T., Hong, D., Lee, D.: Online learning of a full body push recovery controller for omnidirectional walking. pp. 1\u20136. https:\/\/doi.org\/10.1109\/Humanoids.2011.6100896 (2011)","DOI":"10.1109\/Humanoids.2011.6100896"},{"key":"1656_CR85","doi-asserted-by":"publisher","unstructured":"Yi, S. J., Zhang, B. T., Hong, D., Lee, D.: Practical bipedal walking control on uneven terrain using surface learning and push recovery. pp. 3963\u20133968. https:\/\/doi.org\/10.1109\/IROS.2011.6095131 (2011)","DOI":"10.1109\/IROS.2011.6095131"}],"container-title":["Journal of Intelligent &amp; Robotic Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10846-022-01656-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10846-022-01656-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10846-022-01656-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,22]],"date-time":"2022-09-22T12:32:31Z","timestamp":1663849951000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10846-022-01656-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,20]]},"references-count":85,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2022,9]]}},"alternative-id":["1656"],"URL":"https:\/\/doi.org\/10.1007\/s10846-022-01656-7","relation":{},"ISSN":["0921-0296","1573-0409"],"issn-type":[{"value":"0921-0296","type":"print"},{"value":"1573-0409","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,8,20]]},"assertion":[{"value":"30 July 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 November 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 August 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration"}},{"value":"The authors declare that they have no conflicts of interest\/competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflicts of interest\/Competing interests"}}],"article-number":"8"}}