{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,4]],"date-time":"2024-09-04T21:05:32Z","timestamp":1725483932297},"publisher-location":"Berlin, Heidelberg","reference-count":9,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540404552"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.1007\/3-540-45034-3_47","type":"book-chapter","created":{"date-parts":[[2007,5,20]],"date-time":"2007-05-20T08:38:50Z","timestamp":1179650330000},"page":"471-480","source":"Crossref","is-referenced-by-count":0,"title":["Hybrid Least-Squares Methods for Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Hailin","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cihan H.","family":"Dagli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"47_CR1","first-page":"9","volume":"3","author":"R.S. Sutton","year":"1988","unstructured":"R.S. Sutton.: Learning to Predict by the Methods of Temporal Difference. Machine Learning, Vol.3, No.1 (1988) 9\u201344","journal-title":"Machine Learning"},{"key":"47_CR2","volume-title":"Learning From Delayed Rewards","author":"C.J.C.H. Watkins","year":"1989","unstructured":"C.J.C.H. Watkins.: Learning From Delayed Rewards. PhD thesis, Cambridge University, Cambridge, UK (1989)"},{"key":"47_CR3","volume-title":"Reinforcement Learning: An Introduction","author":"R.S. Sutton","year":"1998","unstructured":"R.S. Sutton, A. Barto.: Reinforcement Learning: An Introduction. MIT Press, Cambridge, MA (1998)"},{"issue":"1\/2\/3","key":"47_CR4","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1023\/A:1018056104778","volume":"22","author":"S. J. Bradtke","year":"1996","unstructured":"Steven J. Bradtke, A. Barto.: Linear Least-Squares Algorithms for Temporal Difference Learning. Machine Learning. 22(1\/2\/3) (1996) 33\u201357","journal-title":"Machine Learning"},{"key":"47_CR5","unstructured":"Daphne Koller, Ronald Parr.: Policy Iteration for factored MDPs. Proceedings of the 16th Conference on Uncertainty in Artificial Intelligence (UAI-00), Morgan Kaufmann. (2000) 326\u2013334"},{"key":"47_CR6","unstructured":"Michail Lagoudakis, Ronald Parr.: Model Free Least Squares Policy Iteration. Proceedings of the 14th Neural Information Processing Systems (NIPS-14), Vancouver, Canada. December (2001)"},{"key":"47_CR7","first-page":"2513","volume":"21","author":"S. Chen","year":"1990","unstructured":"S. Chen, C.F. Cowan, P.M. Grant.: Orthogonal Least Squares Algorithm for Radial Basis Function Networks, IEEE Transactions on Neural Networks, vol.21. (1990) 2513\u201339","journal-title":"IEEE Transactions on Neural Networks"},{"key":"47_CR8","unstructured":"Michail Lagoudakis, Michael L. Littman.: Algorithm Selection Using Reinforcement Learning. Proceedings of the 7th International Conference on Machine Learning. San Francisco, CA (2000) 511\u2013518"},{"key":"47_CR9","unstructured":"R.S. Sutton.: Temporal Aspects of Credit Assignment in Reinforcement Learning. PhD thesis, University of Massachusetts (1984)"}],"container-title":["Lecture Notes in Computer Science","Developments in Applied Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/3-540-45034-3_47.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,17]],"date-time":"2020-11-17T21:08:05Z","timestamp":1605647285000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/3-540-45034-3_47"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[null]]},"ISBN":["9783540404552"],"references-count":9,"URL":"https:\/\/doi.org\/10.1007\/3-540-45034-3_47","relation":{},"subject":[]}}