{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,1,28]],"date-time":"2023-01-28T12:32:24Z","timestamp":1674909144985},"reference-count":32,"publisher":"Elsevier BV","issue":"4","license":[{"start":{"date-parts":[[2003,12,1]],"date-time":"2003-12-01T00:00:00Z","timestamp":1070236800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Cognitive Systems Research"],"published-print":{"date-parts":[[2003,12]]},"DOI":"10.1016\/s1389-0417(03)00014-7","type":"journal-article","created":{"date-parts":[[2003,5,27]],"date-time":"2003-05-27T18:47:42Z","timestamp":1054061262000},"page":"319-337","source":"Crossref","is-referenced-by-count":4,"title":["Event-learning and robust policy heuristics"],"prefix":"10.1016","volume":"4","author":[{"given":"Andr\u00e1s","family":"L\u00f6rincz","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Imre","family":"P\u00f3lik","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Istv\u00e1n","family":"Szita","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/S1389-0417(03)00014-7_BIB22","unstructured":"Aamodt, T. M. (1997). Intelligent control via reinforcement learning, Bachelor\u2019s thesis, University of Toronto, http:\/\/www.eecg.utoronto.ca\/~aamodt\/."},{"key":"10.1016\/S1389-0417(03)00014-7_BIB1","series-title":"Encyclopedia of information, linguistics and control","first-page":"261","article-title":"Learning machines\u2014a unified view","author":"Andreae","year":"1969"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB8","doi-asserted-by":"crossref","first-page":"163","DOI":"10.1080\/03081077808960681","article-title":"Discrete and continuous models","volume":"4","author":"Barto","year":"1978","journal-title":"International Journal of General Systems"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB13","series-title":"Neuro-dynamic programming","author":"Bertsekas","year":"1996"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB26","doi-asserted-by":"crossref","first-page":"931","DOI":"10.1002\/rob.4620100704","article-title":"On the application of harmonic function to robotics","volume":"10","author":"Connolly","year":"1993","journal-title":"Journal of Robotic Systems"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB29","article-title":"Temporal difference learning in continuous time and space","volume":"vol. 8","author":"Doya","year":"1996"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB31","doi-asserted-by":"crossref","first-page":"243","DOI":"10.1162\/089976600300015961","article-title":"Reinforcement learning in continuous time and space","volume":"12","author":"Doya","year":"2000","journal-title":"Neural Computation"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB16","doi-asserted-by":"crossref","first-page":"757","DOI":"10.1142\/S0129065796000713","article-title":"Self-organizing multi-resolution grid for motion planning and control","volume":"7","author":"Fomin","year":"1997","journal-title":"International Journal of Neural Systems"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB27","doi-asserted-by":"crossref","first-page":"125","DOI":"10.1016\/0893-6080(94)E0045-M","article-title":"Neural network dynamics for path planning and obstacle avoidance","volume":"8","author":"Glausius","year":"1995","journal-title":"Neural Networks"},{"issue":"3","key":"10.1016\/S1389-0417(03)00014-7_BIB23","doi-asserted-by":"crossref","first-page":"219","DOI":"10.1145\/136035.136037","article-title":"Gross motion planning \u2013 a survey","volume":"24","author":"Hwang","year":"1992","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB21","series-title":"Nonlinear control systems: an introduction","author":"Isidori","year":"1989"},{"issue":"6","key":"10.1016\/S1389-0417(03)00014-7_BIB4","doi-asserted-by":"crossref","first-page":"1185","DOI":"10.1162\/neco.1994.6.6.1185","article-title":"On the convergence of stochastic iterative dynamic programming algorithms","volume":"6","author":"Jaakkola","year":"1994","journal-title":"Neural Computation"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB25","series-title":"Toward a practice of autonomous systems","article-title":"On the self-organizing properties of topological maps","author":"Keymeulen","year":"1992"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB24","doi-asserted-by":"crossref","first-page":"61","DOI":"10.1007\/BF00203631","article-title":"A neural model with fluid properties for solving the labyrinthian puzzle","volume":"64","author":"Lei","year":"1990","journal-title":"Biological Cybernetics"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB9","series-title":"Proceedings of the 12th international conference on machine learning","article-title":"Learning policies for partially observable environments: scaling up","author":"Littman","year":"1995"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB11","unstructured":"Littman, M. L. (1996). Algorithms for sequential decision making, PhD thesis, Department of Computer Science, Brown University, February."},{"key":"10.1016\/S1389-0417(03)00014-7_BIB20","unstructured":"L\u00f6rincz, A., P\u00f3lik, I., & Szita, I. (2001). Event-learning and robust policy heuristics, Technical report NIPG-ELU-15-05-2001, ELTE, http:\/\/people.inf.elte.hu\/lorincz\/Files\/NIPG-ELU-14-05-2001.pdf."},{"key":"10.1016\/S1389-0417(03)00014-7_BIB19","first-page":"125","article-title":"Ockham\u2019s razor modeling of the matrisome channels of the basal ganglia thalamocortical loops","volume":"11","author":"L\u00f6rincz","year":"2001","journal-title":"International Journal of Neural Computation"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB14","series-title":"On-line Q-learning using connectionist systems","author":"Rummery","year":"1996"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB10","series-title":"Proceedings of the 11th machine learning conference","first-page":"284","article-title":"Learning without state-estimation in partially observable markovian decision processes","author":"Singh","year":"1995"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB15","doi-asserted-by":"crossref","first-page":"287","DOI":"10.1023\/A:1007678930559","article-title":"Convergence results for single-step on-policy reinforcement-learning algorithms","volume":"38","author":"Singh","year":"2000","journal-title":"Machine Learning"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB12","series-title":"Reinforcement learning: an introduction","author":"Sutton","year":"1998"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB6","first-page":"1038","article-title":"Generalization in reinforcement learning: Successful examples using sparse coarse coding","volume":"8","author":"Sutton","year":"1996","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB18","doi-asserted-by":"crossref","first-page":"267","DOI":"10.1016\/0925-2312(95)00116-6","article-title":"Approximate geometry representation and sensory fusion","volume":"12","author":"Szepesv\u00e1ri","year":"1996","journal-title":"Neurocomputing"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB28","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1002\/(SICI)1097-4563(199812)15:1<1::AID-ROB1>3.0.CO;2-V","article-title":"An integrated architecture for motion-control and path-planning","volume":"15","author":"Szepesv\u00e1ri","year":"1998","journal-title":"Journal of Robotic Systems"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB7","doi-asserted-by":"crossref","first-page":"2017","DOI":"10.1162\/089976699300016070","article-title":"A unified analysis of value-function-based reinforcement-learning algorithms","volume":"11","author":"Szepesv\u00e1ri","year":"1999","journal-title":"Neural Computation"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB17","doi-asserted-by":"crossref","first-page":"1691","DOI":"10.1016\/S0893-6080(97)00043-9","article-title":"Dynamic state feedback neurocontroller for compensatory control","volume":"10","author":"Szepesv\u00e1ri","year":"1997","journal-title":"Neural Networks"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB30","series-title":"Proceedings of the 8th Belgian\u2013Dutch conference on machine learning","article-title":"Linear quadratic regulation using reinforcement learning","author":"ten Haagen","year":"1998"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB32","unstructured":"ten Haagen, S. (2001). Continuous state space Q-learning for control of nonlinear systems, PhD thesis, University of Amsterdam, Amsterdam."},{"key":"10.1016\/S1389-0417(03)00014-7_BIB5","doi-asserted-by":"crossref","first-page":"59","DOI":"10.1007\/BF00114724","article-title":"Feature-based methods for large scale dynamic programming","volume":"22","author":"Tsitsiklis","year":"1996","journal-title":"Machine Learning"},{"key":"10.1016\/S1389-0417(03)00014-7_BIB3","unstructured":"Watkins, C. (1989). Learning from delayed rewards, PhD thesis, King\u2019s College, Cambridge, UK."},{"key":"10.1016\/S1389-0417(03)00014-7_BIB2","doi-asserted-by":"crossref","first-page":"286","DOI":"10.1016\/S0019-9958(77)90354-0","article-title":"An adaptive optimal controller for discrete-time Markov environments","volume":"34","author":"Witten","year":"1977","journal-title":"Information and Control"}],"container-title":["Cognitive Systems Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1389041703000147?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1389041703000147?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2019,3,21]],"date-time":"2019-03-21T03:54:02Z","timestamp":1553140442000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1389041703000147"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2003,12]]},"references-count":32,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2003,12]]}},"alternative-id":["S1389041703000147"],"URL":"https:\/\/doi.org\/10.1016\/s1389-0417(03)00014-7","relation":{},"ISSN":["1389-0417"],"issn-type":[{"value":"1389-0417","type":"print"}],"subject":[],"published":{"date-parts":[[2003,12]]}}}