{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,13]],"date-time":"2025-10-13T09:01:04Z","timestamp":1760346064682,"version":"3.37.0"},"reference-count":60,"publisher":"Springer Science and Business Media LLC","issue":"3","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Biol Cybern"],"published-print":{"date-parts":[[2009,3]]},"DOI":"10.1007\/s00422-009-0295-8","type":"journal-article","created":{"date-parts":[[2009,2,19]],"date-time":"2009-02-19T16:57:36Z","timestamp":1235062656000},"page":"249-260","source":"Crossref","is-referenced-by-count":12,"title":["Learning to reach by reinforcement learning using a receptive field based function approximation approach with continuous actions"],"prefix":"10.1007","volume":"100","author":[{"given":"Minija","family":"Tamosiunaite","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tamim","family":"Asfour","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Florentin","family":"W\u00f6rg\u00f6tter","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2009,2,20]]},"reference":[{"issue":"3","key":"295_CR1","doi-asserted-by":"crossref","first-page":"287","DOI":"10.1007\/s004220000171","volume":"83","author":"A Arleo","year":"2000","unstructured":"Arleo A, Gerstner W (2000) Spatial cognition and neuro-mimetic navigation: a model of hippocampal place cell activity. Biol Cyber 83(3): 287\u2013299","journal-title":"Biol Cyber"},{"key":"295_CR2","doi-asserted-by":"crossref","unstructured":"Asfour T, Dillmann R (2003) Human-like motion of a humanoid robot arm based on a closed-form solution of the inverse kinematics problem. In: IEEE\/RSJ international conference on intelligent robots and systems","DOI":"10.1109\/IROS.2003.1248841"},{"key":"295_CR3","doi-asserted-by":"crossref","unstructured":"Asfour T, Regenstein K, Azad P, Schr\u00f6der J, Vahrenkamp N, Dillmann R (2006) ARMAR-III: an integrated humanoid platform for sensory-motor control. In: IEEE\/RAS International conference on humanoid robots","DOI":"10.1109\/ICHR.2006.321380"},{"key":"295_CR4","doi-asserted-by":"crossref","unstructured":"Baxter J, Bartlett PL (2000) Direct gradient-based reinforcement learning. In: Proceedings of the ISCAS, Geneva, vol 3, pp 271\u201374","DOI":"10.1109\/ISCAS.2000.856049"},{"issue":"11","key":"295_CR5","first-page":"481","volume":"6","author":"C Breazeal","year":"2008","unstructured":"Breazeal C, Scassellati B (2008) Robots that imitate humans. TICS 6(11): 481\u2013487","journal-title":"TICS"},{"issue":"4","key":"295_CR6","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MRA.2006.250573","volume":"13","author":"F Chaumette","year":"2007","unstructured":"Chaumette F, Hutchinson S (2007a) Visual servo control, part i: basic approaches. IEEE Robot Autom Mag 13(4): 82\u201390","journal-title":"IEEE Robot Autom Mag"},{"issue":"1","key":"295_CR7","doi-asserted-by":"crossref","first-page":"109","DOI":"10.1109\/MRA.2007.339609","volume":"14","author":"F Chaumette","year":"2007","unstructured":"Chaumette F, Hutchinson S (2007b) Visual servo control, part ii: advanced approaches. IEEE Robot Autom Mag 14(1): 109\u2013118","journal-title":"IEEE Robot Autom Mag"},{"key":"295_CR8","doi-asserted-by":"crossref","first-page":"109","DOI":"10.1016\/j.robot.2004.03.005","volume":"47","author":"R Dillmann","year":"2004","unstructured":"Dillmann R (2004) Teaching and learning of robot tasks via observation of human performance. Robot Autonom Sys 47: 109\u2013116","journal-title":"Robot Autonom Sys"},{"key":"295_CR9","doi-asserted-by":"crossref","unstructured":"Enokida S, Ohashi T, Yoshida T, Ejima T (1999) Stochastic field model for autonomous robot learning. In: IEEE international conference on systems, Man, and Cybernetics, vol 2, pp 752\u2013757. doi: 10.1109\/ICSMC.1999.825356","DOI":"10.1109\/ICSMC.1999.825356"},{"issue":"3","key":"295_CR10","doi-asserted-by":"crossref","first-page":"313","DOI":"10.1109\/70.143350","volume":"8","author":"B Espiau","year":"1992","unstructured":"Espiau B, Cahumette F, Rives P (1992) A new approach to visual servoing in robotics. IEEE Trans Robot Autom 8(3): 313\u2013326","journal-title":"IEEE Trans Robot Autom"},{"issue":"1","key":"295_CR11","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1002\/(SICI)1098-1063(2000)10:1<1::AID-HIPO1>3.0.CO;2-1","volume":"10","author":"DJ Foster","year":"2000","unstructured":"Foster DJ, Morris RG, Dayan P (2000) A model of hippocampally dependent navigation, using the temporal difference learning rule. Hippocampus 10(1): 1\u201316","journal-title":"Hippocampus"},{"key":"295_CR12","doi-asserted-by":"crossref","unstructured":"Fukao T, Sumitomo T, Ineyama N, Adachi N (1998) Q-learning based on regularization theory to treat the continuous states and actions. In: IEEE international joint conference on neural networks, pp 1057\u2013062","DOI":"10.1109\/IJCNN.1998.685918"},{"key":"295_CR13","doi-asserted-by":"crossref","unstructured":"Gaskett C, Fletcher L, Zelinsky A (2000) Reinforcement learning for a vision based mobile robot. In: IEEE\/RSJ international conference on intelligent robots and systems, pp 403\u201309","DOI":"10.1109\/IROS.2000.894638"},{"issue":"6","key":"295_CR14","first-page":"1040","volume":"13","author":"GJ Gordon","year":"2001","unstructured":"Gordon GJ (2001) Reinforcement learning with function approximation converges to a region. Adv Neural Inform Process Syst 13(6): 1040\u20131046","journal-title":"Adv Neural Inform Process Syst"},{"key":"295_CR15","doi-asserted-by":"crossref","unstructured":"Gross H, Stephan V, Krabbes M (1998) A neural field approach to topological reinforcement learning in continuous action spaces. In: IEEE world congress on computational intelligence and international joint conference on neural networks, Anchorage, Alaska, pp 1992\u2013997. http:\/\/www.citeseer.ist.psu.edu\/article\/gross98neural.html","DOI":"10.1109\/IJCNN.1998.687165"},{"issue":"4","key":"295_CR16","doi-asserted-by":"crossref","first-page":"525","DOI":"10.1109\/70.704214","volume":"14","author":"R Horaud","year":"1998","unstructured":"Horaud R, Dornaika F, Espiau B (1998) Visually guided object grasping. IEEE Trans Robot Autom 14(4): 525\u2013532","journal-title":"IEEE Trans Robot Autom"},{"key":"295_CR17","doi-asserted-by":"crossref","unstructured":"Horiuchi T, Fujino A, Katai O, Sawaragi T (1997) Fuzzy interpolation-based Q-learning with profit sharing planscheme. In: Proceedings of the sixth IEEE international conference on fuzzy systems, vol 3, pp 1707\u2013712","DOI":"10.1109\/FUZZY.1997.619797"},{"key":"295_CR18","doi-asserted-by":"crossref","unstructured":"Hosoda K, Asada M (1994) Versatile visual servoing without knowledge of true jacobian. In: IEEE\/RSJ international conference on intelligent robots and systems","DOI":"10.1109\/IROS.1994.407392"},{"key":"295_CR19","doi-asserted-by":"crossref","unstructured":"Hutchinson SA, Hager GD, Corke PI (1996) A tutorial on visual servo control. IEEE Trans Robot Autom 12(5): 651\u201370. http:\/\/www.citeseer.ist.psu.edu\/hutchinson96tutorial.html","DOI":"10.1109\/70.538972"},{"key":"295_CR20","doi-asserted-by":"crossref","unstructured":"Kabudian J, Meybodi MR, Homayounpour MM (2004) Applying continuous action reinforcement learning automata(carla) to global training of hidden markov models. In: Proceedings of the international conference on information technology: coding and computing, IEEE Computer Society, Washington, DC, vol 4, pp 638\u201342","DOI":"10.1109\/ITCC.2004.1286725"},{"key":"295_CR21","doi-asserted-by":"crossref","unstructured":"Kobayashi Y, Fujii H, Hosoe S (2005) Reinforcement learning for manipulation using constraint between object and robot. In: IEEE international conference on systems, man and cybernetics, vol 1, pp 871\u201376","DOI":"10.1109\/ICSMC.2005.1571256"},{"key":"295_CR22","doi-asserted-by":"crossref","unstructured":"Kolodziejski C, Porr B, W\u00f6rg\u00f6tter F (2008) On the equivalence between differential hebbian and temporal difference learning. Neural Comp","DOI":"10.1162\/neco.2008.04-08-750"},{"key":"295_CR23","doi-asserted-by":"crossref","unstructured":"Leonard S, Jagersand M (2004) Learning based visual servoing. In: International conference on intelligent robots and systems, Sendai, Japan, pp 680\u201385","DOI":"10.1109\/IROS.2004.1389431"},{"key":"295_CR24","unstructured":"Li J, Lilienthal AJ, Martinez-Marin T, Duckett T (2006) Q-ran: a constructive reinforcement learning approach for robot behavior learning. In: Proceedings of the IEEE\/RSJ international conference on intelligent robots and systems, pp 2656\u2013662"},{"key":"295_CR25","unstructured":"Martinez-Marin T, Duckett T (2004) Robot docking by reinforcement learning in a visual servoing framework. In: IEEE conference on robotics, automation and mechatronics, vol 1, pp 159\u201364"},{"key":"295_CR26","doi-asserted-by":"crossref","unstructured":"Martinetz TM, Ritter HJ, Schulten KJ (1990) Three-dimensional neural net for learning visuomotor coordination of a robot arm. IEEE Trans Neural Netw 1(1): 131\u201336. http:\/\/www.citeseer.ist.psu.edu\/martinetz90threedimensional.html","DOI":"10.1109\/72.80212"},{"issue":"3","key":"295_CR27","doi-asserted-by":"crossref","first-page":"629","DOI":"10.1109\/TNN.2004.824412","volume":"15","author":"M Moussa","year":"2004","unstructured":"Moussa M (2004) Combining expert neural networks using reinforcement feedback for learning primitive grasping behavior. IEEE Trans Neural Netw 15(3): 629\u2013638","journal-title":"IEEE Trans Neural Netw"},{"key":"295_CR28","doi-asserted-by":"crossref","first-page":"239","DOI":"10.1109\/5326.669561","volume":"28","author":"M Moussa","year":"1998","unstructured":"Moussa M, Kamel M (1998) An experimental approach to robotic grasping using a connectionist architecture and generic grasping functions. IEEE Trans Systems Man Cybernetics Part C: Appl Rev 28: 239\u2013253","journal-title":"IEEE Trans Systems Man Cybernetics Part C: Appl Rev"},{"key":"295_CR29","unstructured":"Perez MA, Cook PA (2004) Actor-critic architecture to increase the performance of a 6-dof visual servoing task. In: IEEE 4th international conference on intelligent systems design and application, Budapest, pp 669\u201374"},{"key":"295_CR30","doi-asserted-by":"crossref","unstructured":"Peters J, Schaal S (2006a) Reinforcement learning for parameterized motor primitives. In: International joint conference on neural networks, pp 73\u20130. http:\/\/www-clmc.usc.edu\/publications\/P\/peters-IJCNN2006.pdf","DOI":"10.1109\/IJCNN.2006.246662"},{"key":"295_CR31","doi-asserted-by":"crossref","unstructured":"Peters J, Schaal S (2006b) Policy gradient methods in robotics. In: IEEE\/RSJ international conference on intelligent robots and systems, IROS2006, pp 2219\u2013225","DOI":"10.1109\/IROS.2006.282564"},{"key":"295_CR32","unstructured":"Peters J, Schaal S (2007) Reinforcement learning for operation space control. In: IEEE international conference robotics and automatation, pp 2111\u2013116"},{"key":"295_CR33","doi-asserted-by":"crossref","first-page":"682","DOI":"10.1016\/j.neunet.2008.02.003","volume":"21","author":"J Peters","year":"2008","unstructured":"Peters J, Schaal S (2008) Reinforcement learning of motor skills with policy gradients. Neural Netw 21: 682\u2013697","journal-title":"Neural Netw"},{"key":"295_CR34","doi-asserted-by":"crossref","unstructured":"Qiang L, Hai ZH, Ming LL, Zheng YG (2000) Reinforcement learning with continuous vector output. In: IEEE international conference on systems, man, and cybernetics, vol 1, pp 188\u2013193. doi: 10.1109\/ICSMC.2000.884987","DOI":"10.1109\/ICSMC.2000.884987"},{"issue":"6","key":"295_CR35","doi-asserted-by":"crossref","first-page":"1504","DOI":"10.1109\/TNN.2005.852970","volume":"16","author":"V Ruis de Angulo","year":"2005","unstructured":"Ruis de Angulo V, Torras C (2005a) Speeding up the learning of robot kinematics through function decomposition. IEEE Trans Neural Netw 16(6): 1504\u20131512","journal-title":"IEEE Trans Neural Netw"},{"key":"295_CR36","doi-asserted-by":"crossref","unstructured":"Ruis de Angulo V, Torras C (2005b) Using psoms to learn inverse kinematics through virtual decomposition of the robot. In: International work-conference on artificial and natural neural networks (IWANN), pp 701\u201308","DOI":"10.1007\/11494669_86"},{"key":"295_CR37","unstructured":"Reynolds SI (2002) The stability of general discounted reinforcement learning with linear function approximation. In: UK workshop on computational intelligence (UKCI-02), pp 139\u201346"},{"key":"295_CR38","doi-asserted-by":"crossref","unstructured":"Rezzoug N, Gorce P, Abellard A, Khelifa MB, Abellard P (2006) Learning to grasp in unknown environment by reinforcement learning and shaping. In: IEEE international conference on systems, man and cybernetics, vol 6, pp 4487\u20134492. doi: 10.1109\/ICSMC.2006.384851","DOI":"10.1109\/ICSMC.2006.384851"},{"key":"295_CR39","unstructured":"Schaal S, Ijspeert A, Billard A (2003) Decoding, imitating and influencing the actions of others: the mechanisms of social interaction. In: Computational approaches to motor learning by imitation, vol 358, pp 537\u201347"},{"key":"295_CR40","doi-asserted-by":"crossref","unstructured":"Shibata K, Ito K (1999) Hand\u2013eye coordination in robot arm reaching task by reinforcement learning using a neural network. In: IEEE international conference on systems, man, and cybernetics, vol 5, pp 458\u201363","DOI":"10.1109\/ICSMC.1999.815594"},{"issue":"2","key":"295_CR41","doi-asserted-by":"crossref","first-page":"595","DOI":"10.1152\/jn.1989.62.2.595","volume":"62","author":"J Soechting","year":"1989","unstructured":"Soechting J, Flanders M (1989a) Errors in pointing are due to approximations in targets in sensorimotor transformations. J Neurophysiol 62(2): 595\u2013608","journal-title":"J Neurophysiol"},{"issue":"2","key":"295_CR42","doi-asserted-by":"crossref","first-page":"582","DOI":"10.1152\/jn.1989.62.2.582","volume":"62","author":"J Soechting","year":"1989","unstructured":"Soechting J, Flanders M (1989b) Sensorimotor representations for pointing to targets in three-dimensional space. J Neurophysiol 62(2): 582\u2013594","journal-title":"J Neurophysiol"},{"issue":"9","key":"295_CR43","doi-asserted-by":"crossref","first-page":"1125","DOI":"10.1016\/j.neunet.2005.08.012","volume":"18","author":"T Str\u00f6sslin","year":"2005","unstructured":"Str\u00f6sslin T, Sheynikhovich D, Chavarriaga R, Gerstner W (2005) Robust self-localisation and navigation based on hippocampal place cells. Neural Netw 18(9): 1125\u20131140","journal-title":"Neural Netw"},{"key":"295_CR44","doi-asserted-by":"crossref","unstructured":"Sugiyama M, Hachiya H, Towell, Vijayakumar S (2007) Value function approximation on non-linear manifolds for robot motor control. In: IEEE international conference on robotics and automation, pp 1733\u20131740. doi: 10.1109\/ROBOT.2007.363573","DOI":"10.1109\/ROBOT.2007.363573"},{"key":"295_CR45","first-page":"9","volume":"3","author":"RS Sutton","year":"1988","unstructured":"Sutton RS (1988) Learning to predict by the methods of temporal differences. Mach Learn 3: 9\u201344","journal-title":"Mach Learn"},{"key":"295_CR46","volume-title":"Reinforcement learning: an introduction","author":"R Sutton","year":"1998","unstructured":"Sutton R, Barto A (1998) Reinforcement learning: an introduction. MIT Press, Cambridge"},{"key":"295_CR47","first-page":"1057","volume":"12","author":"RS Sutton","year":"2000","unstructured":"Sutton RS, McAllester D, Singh S, Mansour Y (2000) Policy gradient methods for reinforcement learning with function approximation. Adv Neural Inform Process Syst 12: 1057\u20131063","journal-title":"Adv Neural Inform Process Syst"},{"key":"295_CR48","doi-asserted-by":"crossref","unstructured":"Szepesvari C, Smart WD (2004) Interpolation-based Q-learning. In: Twenty-first international conference on machine learning (ICML04), vol 21, pp 791\u201398","DOI":"10.1145\/1015330.1015445"},{"key":"295_CR49","doi-asserted-by":"crossref","unstructured":"Takahashi Y, Takeda M, Asada M (1999) Continuous valued Q-learning for vision-guided behavior. In: Proceedings of the IEEE\/SICE\/RSJ international conference on multisensor fusion and integration for intelligent systems, pp 255\u201360","DOI":"10.1109\/MFI.1999.815999"},{"key":"295_CR50","doi-asserted-by":"crossref","unstructured":"Takeda M, Nakamura T, Ogasawara T (2001) Continuous valued Q-learning method able to incrementally refine state space. In: IEEE\/RSJ International conference on intelligent robots and systems, vol 1, pp 265\u2013271. doi: 10.1109\/IROS.2001.973369","DOI":"10.1109\/IROS.2001.973369"},{"key":"295_CR51","doi-asserted-by":"crossref","first-page":"562","DOI":"10.1007\/s10827-008-0094-6","volume":"25","author":"M Tamosiunaite","year":"2008","unstructured":"Tamosiunaite M, Ainge J, Kulvicius T, Porr B, Dudchenko P, W\u00f6rg\u00f6tter F (2008) Path-finding in real and simulated rats: assessing the influence of path characteristics on navigation learning. J Comput Neurosci 25: 562\u2013582","journal-title":"J Comput Neurosci"},{"issue":"3","key":"295_CR52","doi-asserted-by":"crossref","first-page":"58","DOI":"10.1145\/203330.203343","volume":"38","author":"G Tesauro","year":"1995","unstructured":"Tesauro G (1995) Temporal difference learning and TD-gammon. Comm ACM 38(3): 58\u201367","journal-title":"Comm ACM"},{"key":"295_CR53","doi-asserted-by":"crossref","unstructured":"Tham C, Prager R (1993) Reinforcement learning methods for multi-linked manipulator obstacle avoidance and control. In: Proceedings of the IEEE Asia-Pacific workshop on advances in motion control, Singapore, pp 140\u201345","DOI":"10.1109\/APWAM.1993.316204"},{"key":"295_CR54","doi-asserted-by":"crossref","unstructured":"van Hasselt H, Wiering M (2007) Reinforcement learning in continuous action spaces. In: IEEE international symposium on approximate dynamic programming and reinforcement learning, pp 272\u2013279. doi: 10.1109\/ADPRL.2007.368199","DOI":"10.1109\/ADPRL.2007.368199"},{"key":"295_CR55","doi-asserted-by":"crossref","unstructured":"Wang B, Li J, Liu H (2006) A heuristic reinforcement learning for robot approaching objects. In: IEEE conference on robotics, automation and mechatronics, pp 1\u20135. doi: 10.1109\/RAMECH.2006.252749","DOI":"10.1109\/RAMECH.2006.252749"},{"key":"295_CR56","unstructured":"Watkins CJ (1989) Learning from delayed rewards. Ph.D. thesis, Cambridge University, Cambridge"},{"key":"295_CR57","first-page":"279","volume":"8","author":"CJ Watkins","year":"1992","unstructured":"Watkins CJ, Dayan P (1992) Q-learning. Mach Learn 8: 279\u2013292","journal-title":"Mach Learn"},{"key":"295_CR58","doi-asserted-by":"crossref","unstructured":"Wiering M (2004) Convergence and divergence in standard and averaging reinforcement learning. In: Boulicaut J, Esposito F, Giannotti F, Pedreschi D (eds) Proceedings of the 15th European conference on machine learning ECML\u201904, pp 477\u201388","DOI":"10.1007\/978-3-540-30115-8_44"},{"key":"295_CR59","first-page":"229","volume":"8","author":"RJ Williams","year":"1992","unstructured":"Williams RJ (1992) Simple statistical gradient-following algorithms for connectionists reinforcement learning. Mach Learn 8: 229\u2013256","journal-title":"Mach Learn"},{"issue":"2","key":"295_CR60","doi-asserted-by":"crossref","first-page":"245","DOI":"10.1162\/0899766053011555","volume":"17","author":"F W\u00f6rg\u00f6tter","year":"2005","unstructured":"W\u00f6rg\u00f6tter F, Porr B (2005) Temporal sequence learning, prediction, and control: a review of different models and their relation to biological mechanisms. Neural Comput 17(2): 245\u2013319","journal-title":"Neural Comput"}],"container-title":["Biological Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00422-009-0295-8.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,7]],"date-time":"2025-02-07T20:37:21Z","timestamp":1738960641000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s00422-009-0295-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2009,2,20]]},"references-count":60,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2009,3]]}},"alternative-id":["295"],"URL":"https:\/\/doi.org\/10.1007\/s00422-009-0295-8","relation":{},"ISSN":["0340-1200","1432-0770"],"issn-type":[{"type":"print","value":"0340-1200"},{"type":"electronic","value":"1432-0770"}],"subject":[],"published":{"date-parts":[[2009,2,20]]}}}