{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,16]],"date-time":"2026-05-16T16:06:57Z","timestamp":1778947617156,"version":"3.51.4"},"reference-count":21,"publisher":"Springer Science and Business Media LLC","issue":"1-2","license":[{"start":{"date-parts":[[2003,1,1]],"date-time":"2003-01-01T00:00:00Z","timestamp":1041379200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2003,1,1]],"date-time":"2003-01-01T00:00:00Z","timestamp":1041379200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Discrete Event Dynamic Systems"],"published-print":{"date-parts":[[2003,1]]},"DOI":"10.1023\/a:1022192903948","type":"journal-article","created":{"date-parts":[[2003,3,21]],"date-time":"2003-03-21T19:29:05Z","timestamp":1048274945000},"page":"79-110","source":"Crossref","is-referenced-by-count":101,"title":["Least Squares Policy Evaluation Algorithms with Linear Function Approximation"],"prefix":"10.1007","volume":"13","author":[{"given":"A.","family":"Nedi\u0106","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"D. P.","family":"Bertsekas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"5110822_CR1","volume-title":"Real Analysis and Probability","author":"R. B. Ash","year":"1972","unstructured":"Ash, R. B. 1972. Real Analysis and Probability. New York: Academic Press Inc."},{"key":"5110822_CR2","doi-asserted-by":"crossref","first-page":"270","DOI":"10.1162\/neco.1995.7.2.270","volume":"7","author":"D. P. Bertsekas","year":"1995","unstructured":"Bertsekas, D. P. 1995. A counterexample to temporal differences learning. Neural Computation 7: 270\u2013279.","journal-title":"Neural Computation"},{"key":"5110822_CR3","volume-title":"Lab. for Info. and Decision Systems Report","author":"D. P. Bertsekas","year":"1996","unstructured":"Bertsekas, D. P., and Ioffe, S. 1996. Temporal differences-based policy iteration and application in neuro-dynamic programming. Lab. for Info. and Decision Systems Report LIDS-P-2349. Cambridge, MA: MIT."},{"key":"5110822_CR4","volume-title":"Nonlinear Programming","author":"D. P. Bertsekas","year":"1999","unstructured":"Bertsekas, D. P. 1999. Nonlinear Programming, 2nd edition. Belmont, MA: Athena Scientific.","edition":"2nd"},{"key":"5110822_CR5","volume-title":"Dynamic Programming and Optimal Control","author":"D. P. Bertsekas","year":"2001","unstructured":"Bertsekas, D. P. 2001. Dynamic Programming and Optimal Control, 2nd edition. Belmont, MA: Athena Scientific.","edition":"2nd"},{"key":"5110822_CR6","volume-title":"Neuro-Dynamic Programming","author":"D. P. Bertsekas","year":"1996","unstructured":"Bertsekas, D. P., and Tsitsiklis, J. N. 1996. Neuro-Dynamic Programming. Belmont, MA: Athena Scientific."},{"key":"5110822_CR7","doi-asserted-by":"crossref","first-page":"627","DOI":"10.1137\/S1052623497331063","volume":"10","author":"D. P. Bertsekas","year":"2000","unstructured":"Bertsekas, D. P., and Tsitsiklis, J. N. 2000. Gradient convergence in gradient methods with errors. SIAM J. Optim. 10: 627\u2013642.","journal-title":"SIAM J. Optim."},{"key":"5110822_CR8","doi-asserted-by":"crossref","unstructured":"Boyan, J. A. 2002. Technical update: least-squares temporal difference learning. To appear in Machine Learning, 49.","DOI":"10.1023\/A:1017936530646"},{"key":"5110822_CR9","doi-asserted-by":"crossref","first-page":"33","DOI":"10.1023\/A:1018056104778","volume":"22","author":"S. J. Bradtke","year":"1996","unstructured":"Bradtke, S. J., and Barto, A. G. 1996. Linear least-squares algorithms for temporal difference learning. Machine Learning 22: 33\u201357.","journal-title":"Machine Learning"},{"key":"5110822_CR10","doi-asserted-by":"crossref","first-page":"295","DOI":"10.1023\/A:1022657612745","volume":"14","author":"P. Dayan","year":"1994","unstructured":"Dayan, P., and Sejnowski, T. J. 1994. TD(l) converges with probability 1. Machine Learning 14: 295\u2013301.","journal-title":"Machine Learning"},{"key":"5110822_CR11","volume-title":"Discrete Stochastic Processes","author":"R. G. Gallager","year":"1995","unstructured":"Gallager, R. G. 1995. Discrete Stochastic Processes. Boston, MA: Kluwer Academic Publishers."},{"key":"5110822_CR12","volume-title":"Incremental Learning of Evaluation Functions for Absorbing Markov Chains: New Methods and Theorems","author":"L. Gurvits","year":"1994","unstructured":"Gurvits, L., Lin, L., and Hanson, S. J. 1994. Incremental Learning of Evaluation Functions for Absorbing Markov Chains: New Methods and Theorems. Working paper. Princeton, NJ: Siemens Corporate Research."},{"key":"5110822_CR13","volume-title":"Matrix Computations","author":"G. H. Golub","year":"1996","unstructured":"Golub, G. H., and Van Loan, C. F. 1996. Matrix Computations, 3rd edition. Baltimore, MD: Johns Hopkins University Press","edition":"3rd"},{"key":"5110822_CR14","doi-asserted-by":"crossref","first-page":"1185","DOI":"10.1162\/neco.1994.6.6.1185","volume":"6","author":"T. Jaakkola","year":"1994","unstructured":"Jaakkola, T., Jordan, M. I., and Singh S. P. 1994. On the convergence of stochastic iterative dynamic programming algorithms. Neural Computation 6: 1185\u20131201.","journal-title":"Neural Computation"},{"key":"5110822_CR15","volume-title":"Finite Markov Chains","author":"J. G. Kemeny","year":"1967","unstructured":"Kemeny, J. G., and Snell, J. L. 1967. Finite Markov Chains. New York: Van Nostrand Company."},{"key":"5110822_CR16","volume-title":"Discrete Parameter Martingales","author":"J. Neveu","year":"1975","unstructured":"Neveu, J. 1975. Discrete Parameter Martingales. Amsterdam: North-Holland."},{"key":"5110822_CR17","volume-title":"Modern Probability Theory and Its Applications","author":"E. Parzen","year":"1962","unstructured":"Parzen, E. 1962. Modern Probability Theory and Its Applications. New York: John Wiley Inc."},{"key":"5110822_CR18","doi-asserted-by":"crossref","DOI":"10.1002\/9780470316887","volume-title":"Markovian Decision Problems","author":"M. L. Puterman","year":"1994","unstructured":"Puterman, M. L. 1994. Markovian Decision Problems. New York: John Wiley Inc."},{"key":"5110822_CR19","doi-asserted-by":"crossref","first-page":"9","DOI":"10.1023\/A:1022633531479","volume":"3","author":"R. S. Sutton","year":"1988","unstructured":"Sutton, R. S. 1988. Learning to predict by the methods of temporal differences. Machine Learning 3: 9\u201344.","journal-title":"Machine Learning"},{"key":"5110822_CR20","doi-asserted-by":"crossref","first-page":"241","DOI":"10.1023\/A:1007609817671","volume":"42","author":"\u00c2. V. Tadic","year":"2001","unstructured":"Tadic \u00c2, V. 2001. On the convergence of temporal-difference learning with linear function approximation. Machine Learning 42: 241\u2013267.","journal-title":"Machine Learning"},{"key":"5110822_CR21","doi-asserted-by":"crossref","first-page":"674","DOI":"10.1109\/9.580874","volume":"42","author":"J. N. Tsitsiklis","year":"1997","unstructured":"Tsitsiklis, J. N., and Van Roy, B. 1997. An analysis of temporal-difference learning with function approximation. IEEE Transactions on Automatic Control 42: 674\u2013690.","journal-title":"IEEE Transactions on Automatic Control"}],"container-title":["Discrete Event Dynamic Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1023\/A:1022192903948.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1023\/A:1022192903948\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1023\/A:1022192903948.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,29]],"date-time":"2025-07-29T04:14:54Z","timestamp":1753762494000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1023\/A:1022192903948"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2003,1]]},"references-count":21,"journal-issue":{"issue":"1-2","published-print":{"date-parts":[[2003,1]]}},"alternative-id":["5110822"],"URL":"https:\/\/doi.org\/10.1023\/a:1022192903948","relation":{},"ISSN":["0924-6703","1573-7594"],"issn-type":[{"value":"0924-6703","type":"print"},{"value":"1573-7594","type":"electronic"}],"subject":[],"published":{"date-parts":[[2003,1]]}}}