{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,11]],"date-time":"2025-11-11T13:02:29Z","timestamp":1762866149645},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2013,3,29]],"date-time":"2013-03-29T00:00:00Z","timestamp":1364515200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Auton Robot"],"published-print":{"date-parts":[[2013,5]]},"DOI":"10.1007\/s10514-013-9328-1","type":"journal-article","created":{"date-parts":[[2013,3,28]],"date-time":"2013-03-28T08:38:57Z","timestamp":1364459937000},"page":"327-346","source":"Crossref","is-referenced-by-count":12,"title":["DCOB: Action space for reinforcement learning of high DoF robots"],"prefix":"10.1007","volume":"34","author":[{"given":"Akihiko","family":"Yamaguchi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Takamatsu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tsukasa","family":"Ogasawara","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,3,29]]},"reference":[{"key":"9328_CR1","doi-asserted-by":"crossref","unstructured":"Asada, M., Noda, S., & Hosoda, K. (1996). Action-based sensor space categorization for robot learning. In The IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS \u201996) (pp. 1502\u20131509).","DOI":"10.1109\/IROS.1996.569012"},{"key":"9328_CR2","doi-asserted-by":"crossref","unstructured":"Baird, L.C., & Klopf, A.H. (1993). Reinforcement learning with high-dimensional, continuous actions. Technical Report WL-TR-93-1147, Wright Laboratory, Wright-Patterson Air Force Base.","DOI":"10.21236\/ADA280844"},{"issue":"3","key":"9328_CR3","doi-asserted-by":"crossref","first-page":"930","DOI":"10.1109\/18.256500","volume":"39","author":"A Barron","year":"1993","unstructured":"Barron, A. (1993). Universal approximation bounds for superpositions of a sigmoidal function. IEEE Transactions on Information Theory, 39(3), 930\u2013945. doi: 10.1109\/18.256500 .","journal-title":"IEEE Transactions on Information Theory"},{"issue":"6","key":"9328_CR4","doi-asserted-by":"crossref","first-page":"1347","DOI":"10.1162\/089976602753712972","volume":"14","author":"K Doya","year":"2002","unstructured":"Doya, K., Samejima, K., Katagiri, K., & Kawato, M. (2002). Multiple model-based reinforcement learning. Neural Computation, 14(6), 1347\u20131369. doi: 10.1162\/089976602753712972 .","journal-title":"Neural Computation"},{"key":"9328_CR5","doi-asserted-by":"crossref","unstructured":"Gaskett, C., Fletcher, L., & Zelinsky, A. (2000). Reinforcement learning for a vision based mobile robot. In The IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS\u201900).","DOI":"10.1109\/IROS.2000.894638"},{"key":"9328_CR6","first-page":"1547","volume-title":"Advances in neural information processing systems","author":"A Ijspeert","year":"2002","unstructured":"Ijspeert, A., & Schaal, S. (2002). Learning attractor landscapes for learning motor primitives. In S. Becker, S. Thrun, & K. Obermayer (Eds.), Advances in neural information processing systems (pp. 1547\u20131554). Cambridge: MIT Press."},{"key":"9328_CR7","doi-asserted-by":"crossref","unstructured":"Kimura, H., Yamashita, T., & Kobayashi, S. (2001). Reinforcement learning of walking behavior for a four-legged robot. In Proceedings of the 40th IEEE Conference on Decision and Control. Portugal.","DOI":"10.1109\/CDC.2001.980135"},{"issue":"3\u20134","key":"9328_CR8","doi-asserted-by":"crossref","first-page":"253","DOI":"10.1016\/S0921-8890(98)00054-2","volume":"25","author":"F Kirchner","year":"1998","unstructured":"Kirchner, F. (1998). Q-learning of complex behaviours on a six-legged walking machine. Robotics and Autonomous Systems, 25(3\u20134), 253\u2013262. doi: 10.1016\/S0921-8890(98)00054-2 .","journal-title":"Robotics and Autonomous Systems"},{"key":"9328_CR9","doi-asserted-by":"crossref","unstructured":"Kober, J., & Peters, J. (2009). Learning motor primitives for robotics. In The IEEE International Conference on Robotics and Automation (ICRA\u201909) (pp. 2509\u20132515).","DOI":"10.1109\/ROBOT.2009.5152577"},{"issue":"2","key":"9328_CR10","doi-asserted-by":"crossref","first-page":"111","DOI":"10.1016\/j.robot.2003.11.006","volume":"46","author":"T Kondo","year":"2004","unstructured":"Kondo, T., & Ito, K. (2004). A reinforcement learning with evolutionary state recruitment strategy for autonomous mobile robots control. Robotics and Autonomous Systems, 46(2), 111\u2013124.","journal-title":"Robotics and Autonomous Systems"},{"key":"9328_CR11","unstructured":"Loch, J., & Singh, S. (1998). Using eligibility traces to find the best memoryless policy in partially observable markov decision processes. In Proceedings of the Fifteenth International Conference on Machine Learning. (pp. 323\u2013331)."},{"key":"9328_CR12","doi-asserted-by":"crossref","unstructured":"Matsubara, T., Morimoto, J., Nakanishi, J., Hyon, S., Hale, J.G., & Cheng, G. (2007). Learning to acquire whole-body humanoid CoM movements to achieve dynamic tasks. In The IEEE International Conference on Robotics and Automation (ICRA\u201907). (pp. 2688\u20132693). doi: 10.1109\/ROBOT.2007.363871 .","DOI":"10.1109\/ROBOT.2007.363871"},{"key":"9328_CR13","unstructured":"Mcgovern, A., & Barto, A.G. (2001). Automatic discovery of subgoals in reinforcement learning using diverse density. In The Eighteenth International Conference on Machine Learning. (pp. 361\u2013368). San Mateo, CA: Morgan Kaufmann."},{"key":"9328_CR14","unstructured":"Menache, I., Mannor, S., & Shimkin, N. (2002). Q-cut - dynamic discovery of sub-goals in reinforcement learning. In ECML \u201902: Proceedings of the 13th European Conference on Machine Learning (pp. 295\u2013306). London: Springer."},{"issue":"3","key":"9328_CR15","doi-asserted-by":"crossref","first-page":"299","DOI":"10.1016\/j.neunet.2003.11.004","volume":"17","author":"H Miyamoto","year":"2004","unstructured":"Miyamoto, H., Morimoto, J., Doya, K., & Kawato, M. (2004). Reinforcement learning with via-point representation. Neural Networks, 17(3), 299\u2013305. doi: 10.1016\/j.neunet.2003.11.004 .","journal-title":"Neural Networks"},{"issue":"3","key":"9328_CR16","first-page":"199","volume":"21","author":"AW Moore","year":"1995","unstructured":"Moore, A. W., & Atkeson, C. G. (1995). The parti-game algorithm for variable resolution reinforcement learning in multidimensional state-spaces. Machine Learning, 21(3), 199\u2013233. doi: 10.1023\/A:1022656217772 .","journal-title":"Machine Learning"},{"key":"9328_CR17","doi-asserted-by":"crossref","unstructured":"Morimoto, J., & Doya, K. (1998). Reinforcement learning of dynamic motor sequence: Learning to stand up. In The IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS\u201998). (pp 1721\u20131726).","DOI":"10.1109\/IROS.1998.724846"},{"issue":"1","key":"9328_CR18","doi-asserted-by":"crossref","first-page":"37","DOI":"10.1016\/S0921-8890(01)00113-0","volume":"36","author":"J Morimoto","year":"2001","unstructured":"Morimoto, J., & Doya, K. (2001). Acquisition of stand-up behavior by a real robot using hierarchical reinforcement learning. Robotics and Autonomous Systems, 36(1), 37\u201351. doi: 10.1016\/S0921-8890(01)00113-0 .","journal-title":"Robotics and Autonomous Systems"},{"issue":"6","key":"9328_CR19","doi-asserted-by":"crossref","first-page":"723","DOI":"10.1016\/j.neunet.2007.01.002","volume":"20","author":"Y Nakamura","year":"2007","unstructured":"Nakamura, Y., Mori, T., Sato, M., & Ishii, S. (2007). Reinforcement learning for a biped robot based on a CPG-actor-critic method. Neural Networks, 20(6), 723\u2013735. doi: 10.1016\/j.neunet.2007.01.002 .","journal-title":"Neural Networks"},{"key":"9328_CR20","doi-asserted-by":"crossref","unstructured":"Peng, J., & Williams, R. J. (1994). Incremental multi-step Q-learning. In International Conference on Machine Learning. (pp. 226\u2013232).","DOI":"10.1016\/B978-1-55860-335-6.50035-0"},{"key":"9328_CR21","unstructured":"Peters, J., Vijayakumar, S., & Schaal, S. (2003). Reinforcement learning for humanoid robotics. In IEEE-RAS International Conference on Humanoid Robots. Karlsruhe, Germany."},{"issue":"2","key":"9328_CR22","doi-asserted-by":"crossref","first-page":"407","DOI":"10.1162\/089976600300015853","volume":"12","author":"M Sato","year":"2000","unstructured":"Sato, M., & Ishii, S. (2000). On-line EM algorithm for the normalized Gaussian network. Neural Computation, 12(2), 407\u2013432.","journal-title":"Neural Computation"},{"key":"9328_CR23","volume-title":"Algorithms","author":"R Sedgewick","year":"2011","unstructured":"Sedgewick, R., & Wayne, K. (2011). Algorithms. Boston: Addison-Wesley."},{"key":"9328_CR24","unstructured":"Stolle, M. (2004). Automated discovery of options in reinforcement learning (Master\u2019s thesis, McGill University)."},{"key":"9328_CR25","unstructured":"Sutton, R., & Barto, A. (1998). Reinforcement Learning: An Introduction. Cambridge: MIT Press. Retrieved from http:\/\/citeseer.ist.psu.edu\/sutton98reinforcement.html ."},{"key":"9328_CR26","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"RS Sutton","year":"1999","unstructured":"Sutton, R. S., Precup, D., & Singh, S. (1999). Between mdps and semi-mdps: A framework for temporal abstraction in reinforcement learning. Artificial Intelligence, 112, 181\u2013211.","journal-title":"Artificial Intelligence"},{"key":"9328_CR27","unstructured":"Takahashi, Y., & Asada, M. (2003). Multi-layered learning systems for vision-based behavior acquisition of a real mobile robot. In Proceedings of SICE Annual Conference 2003 (pp. 2937\u20132942)."},{"key":"9328_CR28","doi-asserted-by":"crossref","unstructured":"Tham, C. K., & Prager, R. W. (1994). A modular Q-learning architecture for manipulator task decomposition. In The Eleventh International Conference on Machine Learning (pp. 309\u2013317).","DOI":"10.1016\/B978-1-55860-335-6.50045-3"},{"key":"9328_CR29","doi-asserted-by":"crossref","unstructured":"Theodorou, E., Buchli, J., & Schaal, S. (2010). Reinforcement learning of motor skills in high dimensions: A path integral approach. In The IEEE International Conference on Robotics and Automation (ICRA\u201910) (pp. 2397\u20132403). doi: 10.1109\/ROBOT.2010.5509336 .","DOI":"10.1109\/ROBOT.2010.5509336"},{"key":"9328_CR30","first-page":"59","volume":"22","author":"JN Tsitsiklis","year":"1996","unstructured":"Tsitsiklis, J. N., & Roy, B. V. (1996). Feature-based methods for large scale dynamic programming. Machine Learning, 22, 59\u201394.","journal-title":"Machine Learning"},{"key":"9328_CR31","doi-asserted-by":"crossref","unstructured":"Tsitsiklis, J. N., & Roy, B. V. (1997). An analysis of temporal-difference learning with function approximation. IEEE Transactions on Automatic Control, 42(5), 674\u2013690.","DOI":"10.1109\/9.580874"},{"key":"9328_CR32","unstructured":"Uchibe, E., Doya, K. (2004). Competitive-cooperative-concurrent reinforcement learning with importance sampling. In The International Conference on Simulation of Adaptive Behavior: From Animals and Animats (pp. 287\u2013296)."},{"issue":"7","key":"9328_CR33","doi-asserted-by":"crossref","first-page":"1317","DOI":"10.1016\/S0893-6080(98)00066-5","volume":"11","author":"DM Wolpert","year":"1998","unstructured":"Wolpert, D. M., & Kawato, M. (1998). Multiple paired forward and inverse models for motor control. Neural Networks, 11(7), 1317\u20131329.","journal-title":"Neural Networks"},{"key":"9328_CR34","unstructured":"Yamaguchi, A. (2011). Highly modularized learning system for behavior acquisition of functional robots. Ph.D. Thesis, Nara Institute of Science and Technology, Japan."},{"issue":"2","key":"9328_CR35","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1016\/j.robot.2004.03.006","volume":"47","author":"J Zhang","year":"2004","unstructured":"Zhang, J., & R\u00f6ssler, B. (2004). Self-valuing learning and generalization with application in visually guided grasping of complex objects. Robotics and Autonomous Systems, 47(2), 117\u2013127.","journal-title":"Robotics and Autonomous Systems"}],"container-title":["Autonomous Robots"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10514-013-9328-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10514-013-9328-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10514-013-9328-1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,7,11]],"date-time":"2019-07-11T13:12:28Z","timestamp":1562850748000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10514-013-9328-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,3,29]]},"references-count":35,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2013,5]]}},"alternative-id":["9328"],"URL":"https:\/\/doi.org\/10.1007\/s10514-013-9328-1","relation":{},"ISSN":["0929-5593","1573-7527"],"issn-type":[{"value":"0929-5593","type":"print"},{"value":"1573-7527","type":"electronic"}],"subject":[],"published":{"date-parts":[[2013,3,29]]}}}