{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,30]],"date-time":"2026-01-30T04:42:35Z","timestamp":1769748155645,"version":"3.49.0"},"reference-count":69,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2022,4,27]],"date-time":"2022-04-27T00:00:00Z","timestamp":1651017600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,4,27]],"date-time":"2022-04-27T00:00:00Z","timestamp":1651017600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100002322","name":"Coordena\u00e7\u00e3o de Aperfei\u00e7oamento de Pessoal de N\u00edvel Superior","doi-asserted-by":"publisher","award":["88882.161989\/2017-01"],"award-info":[{"award-number":["88882.161989\/2017-01"]}],"id":[{"id":"10.13039\/501100002322","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003593","name":"Conselho Nacional de Desenvolvimento Cient\u00edfico e Tecnol\u00f3gico","doi-asserted-by":"publisher","award":["304134\/2-18-0"],"award-info":[{"award-number":["304134\/2-18-0"]}],"id":[{"id":"10.13039\/501100003593","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Intell Robot Syst"],"published-print":{"date-parts":[[2022,5]]},"DOI":"10.1007\/s10846-022-01619-y","type":"journal-article","created":{"date-parts":[[2022,4,27]],"date-time":"2022-04-27T15:07:18Z","timestamp":1651072038000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":26,"title":["Deep Reinforcement Learning for Humanoid Robot Behaviors"],"prefix":"10.1007","volume":"105","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4665-0039","authenticated-orcid":false,"given":"Alexandre F. V.","family":"Muzio","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2944-4476","authenticated-orcid":false,"given":"Marcos R. O. A.","family":"Maximo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5375-1076","authenticated-orcid":false,"given":"Takashi","family":"Yoneyama","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,4,27]]},"reference":[{"key":"1619_CR1","unstructured":"Simspark. http:\/\/simspark.sourceforge.net\/wiki\/index.php\/Main_Page (2004)"},{"key":"1619_CR2","doi-asserted-by":"crossref","unstructured":"Abdolmaleki, A., Sim\u00f5es, D., Lau, N., Reis, L.P., Neumann, G., Sar\u0131el, S., Lee, D.D. Behnke, S., Sheh, R. (eds.): Learning a humanoid kick with controlled distance. Springer International Publishing, Cham (2017)","DOI":"10.1007\/978-3-319-68792-6_4"},{"key":"1619_CR3","doi-asserted-by":"crossref","unstructured":"Abrel, M., Reis, L.P., Lau, N.: Learning to run faster in a humanoid robot soccer environment through reinforcement learning. In: Proceedings of the 2019 RoboCup Symposium. RoboCup, Australia (2019)","DOI":"10.1007\/978-3-030-35699-6_1"},{"key":"1619_CR4","unstructured":"Al-Shedivat, M., Bansal, T., Burda, Y., Sutskever, I., Mordatch, I., Abbeel, P.: Continuous adaptation via meta-learning in nonstationary and competitive environments. arXiv:1710.03641(2017)"},{"key":"1619_CR5","unstructured":"Alcaraz-Jim\u00e9nez, J., Herrero-Perez, D., Barber\u00e1, H: A closed-loop dribbling gait for the standard platform league (2014)"},{"key":"1619_CR6","unstructured":"Bansal, T., Pachocki, J., Sidor, S., Sutskever, I., Mordatch, I.: Emergent complexity via multi-agent competition. arXiv:1710.03748 (2017)"},{"key":"1619_CR7","doi-asserted-by":"crossref","unstructured":"Barto, A.G., Sutton, R.S., Anderson, C.W.: Neuronlike adaptive elements that can solve difficult learning control problems (1983)","DOI":"10.1109\/TSMC.1983.6313077"},{"key":"1619_CR8","unstructured":"Bengio, Y., Courville, A.C., Vincent, P.: Unsupervised feature learning and deep learning: A review and new perspectives. arXiv:1206.5538 (2012)"},{"key":"1619_CR9","doi-asserted-by":"publisher","unstructured":"Bengio, Y., Louradour, J., Collobert, R., Weston, J.: Curriculum learning. In: Proceedings of the 26th Annual International Conference on Machine Learning, ICML \u201909, pp. 41\u201348. ACM, New York (2009), https:\/\/doi.org\/10.1145\/1553374.1553380","DOI":"10.1145\/1553374.1553380"},{"key":"1619_CR10","first-page":"281","volume":"13","author":"J Bergstra","year":"2012","unstructured":"Bergstra, J., Bengio, Y.: Random search for hyper-parameter optimization. J. Mach. Learn. Res. 13, 281\u2013305 (2012). http:\/\/dl.acm.org\/citation.cfm?id=2188385.2188395","journal-title":"J. Mach. Learn. Res."},{"key":"1619_CR11","doi-asserted-by":"crossref","unstructured":"Carvalho Melo, D., Quartucci Forster, C.H., Omena de Albuquerque M\u00e1ximo, M.R.: Learning when to kick through deep neural networks. In: 2019 Latin American Robotics Symposium (LARS), 2019 Brazilian Symposium on Robotics (SBR) and 2019 Workshop on Robotics in Education (WRE), pp. 43\u201348 (2019)","DOI":"10.1109\/LARS-SBR-WRE48964.2019.00016"},{"key":"1619_CR12","doi-asserted-by":"crossref","unstructured":"Carvalho Melo, L., Omena Albuquerque M\u00e1ximo, M.R.: Learning humanoid robot running skills through proximal policy optimization. In: 2019 Latin American Robotics Symposium (LARS), 2019 Brazilian Symposium on Robotics (SBR) and 2019 Workshop on Robotics in Education (WRE), pp. 37\u201342 (2019)","DOI":"10.1109\/LARS-SBR-WRE48964.2019.00015"},{"key":"1619_CR13","doi-asserted-by":"crossref","unstructured":"Depinet, M., MacAlpine, P., Stone, P. Bianchi, R.A.C., Akin, H.L., Ramamoorthy, S., Sugiura, K. (eds.): Keyframe sampling, optimization, and behavior integration: Towards long-distance kicking in the Robocup 3D simulation league. Springer, Berlin (2015)","DOI":"10.1007\/978-3-319-18615-3_47"},{"key":"1619_CR14","unstructured":"Dhariwal, P., Hesse, C., Plappert, M., Radford, A., Schulman, J., Sidor, S., Wu, Y., Openai baselines. https:\/\/github.com\/openai\/baselines (2017)"},{"key":"1619_CR15","unstructured":"Duan, Y., Chen, X., Houthooft, R., Schulman, J., Abbeel, P.: Benchmarking deep reinforcement learning for continuous control. arXiv:1604.06778 (2016)"},{"key":"1619_CR16","unstructured":"Farchy, A., Barrett, S., MacAlpine, P., Stone, P.: Humanoid robots learning to walk faster: From the real world to simulation and back. In: Proc. of 12Th Int. Conf. on Autonomous Agents and Multiagent Systems (AAMAS). AAMAS, Saint Paul (2013)"},{"key":"1619_CR17","unstructured":"Farchy, A., Barrett, S., MacAlpine, P., Stone, P.: Humanoid Robots Learning to Walk Faster: From the Real World to Simulation and Back. In: Proc. of 12Th Int. Conf. on Autonomous Agents and Multiagent Systems (AAMAS) (2013)"},{"key":"1619_CR18","unstructured":"Florensa, C., Held, D., Wulfmeier, M., Abbeel, P.: Reverse curriculum generation for reinforcement learning. arXiv:1707.05300 (2017)"},{"key":"1619_CR19","unstructured":"Frans, K., Ho, J., Chen, X., Abbeel, P., Schulman, J.: Meta learning shared hierarchies. arXiv:1710.09767 (2017)"},{"key":"1619_CR20","first-page":"61","volume-title":"A Case Study on Improving Defense Behavior in Soccer Simulation 2D: The NeuroHassle Approach","author":"T Gabel","year":"2009","unstructured":"Gabel, T., Riedmiller, M., Trost, F.: A Case Study on Improving Defense Behavior in Soccer Simulation 2D: The NeuroHassle Approach, pp. 61\u201372. Springer, Berlin (2009)"},{"key":"1619_CR21","unstructured":"Google: Protocol buffers. https:\/\/developers.google.com\/protocol-buffers\/ (2017)"},{"key":"1619_CR22","unstructured":"Haarnoja, T., Zhou, A., Abbeel, P., Levine, S.: Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor (2018)"},{"key":"1619_CR23","unstructured":"Hausknecht, M., Stone, P.: Deep reinforcement learning in parameterized action space. In: Proceedings of the International Conference on Learning Representations (ICLR). ICLR, San Juan (2016)"},{"key":"1619_CR24","unstructured":"Heess, N., TB, D., Sriram, S., Lemmon, J., Merel, J., Wayne, G., Tassa, Y., Erez, T., Wang, Z., Eslami, S.M.A., Riedmiller, M., Silver, D.: Emergence of locomotion behaviours in rich environments (2017)"},{"key":"1619_CR25","unstructured":"Kajita, S., Kanehiro, F., Kaneko, K., Yokoi, K., Hirukawa, H.: The 3D linear inverted pendulum mode: A simple modeling for a biped walking pattern generation. In: Proceedings of the 2001 IEEE\/RSJ International Conference on Intelligent Robots and Systems. IEEE, Hawaii (2001)"},{"key":"1619_CR26","doi-asserted-by":"crossref","unstructured":"Kim, J., Kim, B., Yoon, J., Lee, M., Jung, S.Y., Choi, J.: Robot soccer using deep q network. In: 2018 International Conference on Platform Technology and Service (Platcon), pp. 1\u20136 (2018)","DOI":"10.1109\/PlatCon.2018.8472776"},{"issue":"1","key":"1619_CR27","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1609\/aimag.v18i1.1276","volume":"18","author":"H Kitano","year":"1997","unstructured":"Kitano, H., Asada, M., Kuniyoshi, Y., Noda, I., Osawa, E., Matsubara, H.: Robocup: A challenge problem for ai. AI Magazine 18(1), 73 (1997). https:\/\/doi.org\/10.1609\/aimag.v18i1.1276. https:\/\/aaai.org\/ojs\/index.php\/aimagazine\/article\/view\/1276","journal-title":"AI Magazine"},{"key":"1619_CR28","doi-asserted-by":"crossref","unstructured":"Leottau, D.L., del Solar, J.R., MacAlpine, P., Stone, P.: A study of layered learning strategies applied to individual behaviors in robot soccer. In: Almeida, L., Ji, J., Steinbauer, G., Luke, S. (eds.) RoboCup-2015: Robot Soccer World Cup XIX, Lecture Notes in Artificial Intelligence. Springer, Berlin (2016)","DOI":"10.1007\/978-3-319-29339-4_24"},{"key":"1619_CR29","doi-asserted-by":"crossref","unstructured":"Leottau, L., Celemin, C., del solar, J.R.: Ball dribbling for humanoid biped robots: A reinforcement learning and fuzzy control approach (2014)","DOI":"10.1007\/978-3-319-18615-3_45"},{"key":"1619_CR30","unstructured":"Lillicrap, T.P., Hunt, J.J., Pritzel, A., Heess, N., Erez, T., Tassa, Y., Silver, D., Wierstra, D.: Continuous control with deep reinforcement learning. arXiv:1509.02971 (2015)"},{"key":"1619_CR31","unstructured":"MacAlpine, P., Barrett, S., Urieli, D., Vu, V., Stone, P.: Design and optimization of an omnidirectional humanoid walk: A winning approach at the roboCup 2011 3D simulation competition. In: Proceedings of the Twenty-Sixth AAAI Conference on Artificial Intelligence (AAAI). AAAI, Toronto (2012)"},{"key":"1619_CR32","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1016\/j.artint.2017.09.001","volume":"254","author":"P MacAlpine","year":"2018","unstructured":"MacAlpine, P., Stone, P.: Overlapping layered learning. Artificial Intelligence 254, 21\u201343 (2018). https:\/\/doi.org\/10.1016\/j.artint.2017.09.001. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0004370217301066","journal-title":"Artificial Intelligence"},{"key":"1619_CR33","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1016\/j.artint.2017.09.001","volume":"254","author":"P MacAlpine","year":"2018","unstructured":"MacAlpine, P., Stone, P.: Overlapping layered learning. Artificial Intelligence 254, 21\u201343 (2018). https:\/\/doi.org\/10.1016\/j.artint.2017.09.001. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0004370217301066","journal-title":"Artificial Intelligence"},{"key":"1619_CR34","doi-asserted-by":"crossref","unstructured":"MacAlpine, P., Stone, P.: UT Austin Villa: RoboCup 2017 3D simulation league competition and technical challenges champions. In: Sammut, C., Obst, O., Tonidandel, F., Akyama, H. (eds.) RoboCup 2017: Robot Soccer World Cup XXI, Lecture Notes in Artificial Intelligence. Springer, Berlin (2018)","DOI":"10.1007\/978-3-030-00308-1_39"},{"key":"1619_CR35","unstructured":"Matiisen, T., Oliver, A., Cohen, T., Schulman, J.: Teacher-student curriculum learning. arXiv:1707.00183 (2017)"},{"issue":"3","key":"1619_CR36","doi-asserted-by":"publisher","first-page":"172988141667513","DOI":"10.1177\/1729881416675135","volume":"14","author":"MR Maximo","year":"2017","unstructured":"Maximo, M.R., Colombini, E.L., Ribeiro, C.H.: Stable and fast model-free walk with arms movement for humanoid robots. Int. J. Adv. Robot. Syst 14(3), 1729881416675135 (2017). https:\/\/doi.org\/10.1177\/1729881416675135","journal-title":"Int. J. Adv. Robot. Syst"},{"key":"1619_CR37","unstructured":"Maximo, M.R.O.A.: Omnidirectional Zmp-based walking for a humanoid robot. Master\u2019s thesis, Instituto Tecnol\u00f3gico de Aeron\u00e1utica (2015)"},{"key":"1619_CR38","unstructured":"Maximo, M.R.O.A., Ribeiro, C.H.C.: ZMP-based humanoid walking engine with arms movement and stabilization. In: Proceedings of the 2016 Congresso Brasileiro de Autom\u00e1tica (CBA), SBA. Vit\u00f3ria, Brazil (2016)"},{"key":"1619_CR39","doi-asserted-by":"publisher","unstructured":"de Medeiros, T.F., de M\u00e1ximo, A., M.R.O., Yoneyama, T.: Deep reinforcement learning applied to ieee very small size soccer strategy. In: 2020 Latin American Robotics Symposium (LARS), 2020 Brazilian Symposium on Robotics (SBR) and 2020 Workshop on Robotics in Education (WRE), pp. 1\u20136 (2020), https:\/\/doi.org\/10.1109\/LARS\/SBR\/WRE51543.2020.9306954","DOI":"10.1109\/LARS\/SBR\/WRE51543.2020.9306954"},{"key":"1619_CR40","unstructured":"Melo, D., Soares, E.E., Moreira, E., Muniz, F., Marra, G., Nahum, G., Lopes, H., Saraiva, J.L., Jos\u00e9 Ot\u00e1vio Vidal, J.F., Melo, L., Maximo, M.: Itandroids soccer3d team description paper 2017. https:\/\/www.robocup2017.org\/file\/symposium\/soccer_sim_3D\/ITAndroids3D_TDP.pdf (2017)"},{"key":"1619_CR41","doi-asserted-by":"publisher","unstructured":"Melo, D.C., M\u00e1ximo, M. R. O. A., da Cunha, A.M.: Push recovery strategies through deep reinforcement learning. In: 2020 Latin American Robotics Symposium (LARS), 2020 Brazilian Symposium on Robotics (SBR) and 2020 Workshop on Robotics in Education (WRE), pp. 1\u20136 (2020), https:\/\/doi.org\/10.1109\/LARS\/SBR\/WRE51543.2020.9306967","DOI":"10.1109\/LARS\/SBR\/WRE51543.2020.9306967"},{"key":"1619_CR42","unstructured":"Melo, L.C.: Imitation Learning and Meta-Learning for Optimizing Humanoid Robot Motions. Master\u2019s Thesis, Instituto tecnol\u00f3gico de aeron\u00e1utica, s\u00e3o jos\u00e9 dos Campos, SP Brazil (2019)"},{"key":"1619_CR43","unstructured":"Melo, L.C., Maximo, M.R.O.A., da Cunha, A.M.: Bottom-up meta-policy search. In: Proceedings of the Deep Reinforcement Learning Workshop of NeurIPS 2019 (2019)"},{"issue":"3","key":"1619_CR44","doi-asserted-by":"publisher","first-page":"54","DOI":"10.1007\/s10846-021-01355-9","volume":"102","author":"LC Melo","year":"2021","unstructured":"Melo, L.C., Melo, D.C., Maximo, M.R.O.A.: Learning humanoid robot running motions with symmetry incentive through proximal policy optimization. Journal of Intelligent &, Robotic Systems 102(3), 54 (2021). https:\/\/doi.org\/10.1007\/s10846-021-01355-9","journal-title":"Journal of Intelligent &, Robotic Systems"},{"key":"1619_CR45","unstructured":"Mnih, V., Badia, A.P., Mirza, M., Graves, A., Lillicrap, T.P., Harley, T., Silver, D., Kavukcuoglu, K.: Asynchronous methods for deep reinforcement learning. arXiv:1602.01783 (2016)"},{"issue":"7540","key":"1619_CR46","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Rusu, A.A., Veness, J., Bellemare, M.G., Graves, A., Riedmiller, M., Fidjeland, A.K., Ostrovski, G., Petersen, S., Beattie, C., Sadik, A., Antonoglou, I., King, H., Kumaran, D., Wierstra, D., Legg, S., Hassabis, D.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015). https:\/\/doi.org\/10.1038\/nature14236. Letter","journal-title":"Nature"},{"key":"1619_CR47","doi-asserted-by":"publisher","unstructured":"Muniz, F., Maximo, M.R., Ribeiro, C.H.: Keyframe movement optimization for simulated humanoid robot using a parallel optimization framework. In: 2016 XIII Latin American Robotics Symposium and IV Brazilian Robotics Symposium (LARS\/SBR), pp. 79\u201384 (2016), https:\/\/doi.org\/10.1109\/LARS-SBR.2016.20","DOI":"10.1109\/LARS-SBR.2016.20"},{"key":"1619_CR48","unstructured":"Muzio, A., Melo, D., Henrique, E., Muniz, F., Marzzo, I., Saraiva, J.L., Melo, L., Aguiar, L.G., Maximo, M., Bertolino, M.: Itandroids soccer3d team description paper 2016. http:\/\/www.robocup2016.org\/media\/symposium\/Team-Description-Papers\/Simulation3D\/RoboCup_2016_Sim3D_TDP_ITAndroids3D.pdf\/ (2016)"},{"key":"1619_CR49","unstructured":"Muzio, A.F.V.: Curriculum-based Deep Reinforcement Learning Applied to Humanoid Robots. Master\u2019s Thesis, Instituto tecnol\u00f3gico de aeron\u00e1utica, s\u00e3o jos\u00e9 dos Campos, SP Brazil (2018)"},{"key":"1619_CR50","doi-asserted-by":"publisher","unstructured":"Muzio, A.F.V., Maximo, M.R.A., Yoneyama, T.: Deep reinforcement learning for humanoid robot dribbling. In: 2020 Latin American Robotics Symposium (LARS), 2020 Brazilian Symposium on Robotics (SBR) and 2020 Workshop on Robotics in Education (WRE), pp. 1\u20136 (2020), https:\/\/doi.org\/10.1109\/LARS\/SBR\/WRE51543.2020.9307084","DOI":"10.1109\/LARS\/SBR\/WRE51543.2020.9307084"},{"key":"1619_CR51","unstructured":"Obst, O., Murray, J., Boedecker, J., Rollmann, M., Ebrahimi, M., Vatankhah, H., van Dijk, S., Yuan, X.: Simspark effectors. https:\/\/gitlab.com\/robocup-sim\/SimSpark\/wikis\/Effectors (2004)"},{"key":"1619_CR52","unstructured":"ODE: Open dynamics engine (ode). http:\/\/www.ode.org\/ (2004)"},{"issue":"4","key":"1619_CR53","first-page":"2017","volume":"36","author":"XB Peng","year":"2017","unstructured":"Peng, X.B., Berseth, G., Yin, K., van de Panne, M.: Deeploco: Dynamic locomotion skills using hierarchical deep reinforcement learning. ACM Transactions on Graphics Proc SIGGRAPH 36(4), 2017 (2017)","journal-title":"ACM Transactions on Graphics Proc SIGGRAPH"},{"key":"1619_CR54","unstructured":"Peng, X.B., Chang, M., Zhang, G., Abbeel, P., Levine, S.: Mcp: Learning composable hierarchical control with multiplicative compositional policies. In: Wallach, H., Larochelle, H., Beygelzimer, A., D\u2019Alch\u00e9-Buc, F., Fox, E., Garnett, R. (eds.) Advances in Neural Information Processing Systems, vol. 32, pp. 3681\u20133692. Curran Associates Inc (2019). http:\/\/papers.nips.cc\/paper\/8626-mcp-learning-composable-hierarchical-control-with-multiplicative-compositional-policies.pdf"},{"key":"1619_CR55","unstructured":"Plappert, M., Houthooft, R., Dhariwal, P., Sidor, S., Chen, R.Y., Chen, X., Asfour, T., Abbeel, P., Andrychowicz, M.: Parameter space noise for exploration. arXiv:1706.01905 (2017)"},{"key":"1619_CR56","unstructured":"Robotics, S.: Nao robot. https:\/\/www.ald.softbankrobotics.com\/en\/robots\/nao (2018)"},{"key":"1619_CR57","unstructured":"Schulman, J., Levine, S., Moritz, P., Jordan, M.I., Abbeel, P.: Trust region policy optimization. arXiv:1502.05477 (2015)"},{"key":"1619_CR58","unstructured":"Schulman, J., Moritz, P., Levine, S., Jordan, M.I., Abbeel, P.: High-dimensional continuous control using generalized advantage estimation. In: Bengio, Y. (ed.) 4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings. arXiv:1506.02438 (2016)"},{"key":"1619_CR59","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv:1707.06347 (2017)"},{"key":"1619_CR60","unstructured":"Schwab, D.: Robot deep reinforcement learning: Tensor state-action spaces and auxiliary task learning with multiple state representations. Ph.D. thesis, Carnegie Mellon University (2020)"},{"key":"1619_CR61","unstructured":"Silver, D., Lever, G., Heess, N., Degris, T., Wierstra, D., Riedmiller, M.: Deterministic policy gradient algorithms. In: Proceedings of the 31st International Conference on International Conference on Machine Learning - vol 32, ICML\u201914, pp. I\u2013387\u2013I\u2013395. JMLR.org. http:\/\/dl.acm.org\/citation.cfm?id=3044805.3044850 (2014)"},{"key":"1619_CR62","doi-asserted-by":"publisher","unstructured":"Spitznagel, M., Weiler, D., Dorer, K.: Deep reinforcement multi-directional kick-learning of a simulated robot with toes. In: 2021 IEEE International Conference on Autonomous Robot Systems and Competitions (ICARSC), pp. 104\u2013110 (2021), https:\/\/doi.org\/10.1109\/ICARSC52212.2021.9429811","DOI":"10.1109\/ICARSC52212.2021.9429811"},{"key":"1619_CR63","unstructured":"Stoecker, J.: Roboviz. https:\/\/github.com\/magmaOffenburg\/RoboViz (2011)"},{"key":"1619_CR64","volume-title":"Introduction to Reinforcement Learning","author":"RS Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Introduction to Reinforcement Learning, 1st edn. MIT Press, Cambridge (1998)","edition":"1st edn."},{"key":"1619_CR65","unstructured":"Urieli, D., MacAlpine, P., Kalyanakrishnan, S., Bentor, Y., Stone, P.: On optimizing interdependent skills: A case study in simulated 3d humanoid robot soccer. In: Tumer, K., Yolum, P., Sonenberg, L., Stone, P. (eds.) Proc. of 10th Int. Conf. on Autonomous Agents and Multiagent Systems (AAMAS), vol. 2, pp. 769\u2013776. IFAAMAS (2011)"},{"key":"1619_CR66","unstructured":"Wang, Z., Bapst, V., Heess, N., Mnih, V., Munos, R., Kavukcuoglu, K., de Freitas, N.: Sample efficient actor-critic with experience replay. arXiv:1611.01224 (2016)"},{"key":"1619_CR67","unstructured":"Watkins, C.J.C.H.: Learning from delayed rewards. Ph.D. thesis, King\u2019s College (1989)"},{"key":"1619_CR68","doi-asserted-by":"crossref","unstructured":"Wiliams, R.J.: Simple statistical gradient-following algorithms for connectionist reinforcement learning (1992)","DOI":"10.1007\/978-1-4615-3618-5_2"},{"key":"1619_CR69","unstructured":"Zaremba, W., Sutskever, I.: Learning to execute. arXiv:1410.4615 (2014)"}],"container-title":["Journal of Intelligent &amp; Robotic Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10846-022-01619-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10846-022-01619-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10846-022-01619-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,23]],"date-time":"2024-09-23T02:43:07Z","timestamp":1727059387000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10846-022-01619-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,4,27]]},"references-count":69,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2022,5]]}},"alternative-id":["1619"],"URL":"https:\/\/doi.org\/10.1007\/s10846-022-01619-y","relation":{},"ISSN":["0921-0296","1573-0409"],"issn-type":[{"value":"0921-0296","type":"print"},{"value":"1573-0409","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,4,27]]},"assertion":[{"value":"21 February 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 February 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 April 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflicts of interest\/competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of Interests"}}],"article-number":"12"}}