{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,9]],"date-time":"2025-11-09T07:53:38Z","timestamp":1762674818832},"reference-count":17,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[1991,11,1]],"date-time":"1991-11-01T00:00:00Z","timestamp":688953600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["ZOR - Methods and Models of Operations Research"],"published-print":{"date-parts":[[1991,11]]},"DOI":"10.1007\/bf01415991","type":"journal-article","created":{"date-parts":[[2005,4,3]],"date-time":"2005-04-03T10:51:26Z","timestamp":1112525486000},"page":"491-503","source":"Crossref","is-referenced-by-count":0,"title":["Adaptive policy-iteration and policy-value-iteration for discounted Markov decision processes"],"prefix":"10.1007","volume":"35","author":[{"given":"G.","family":"H\ufffdbner","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"M.","family":"Sch\ufffdl","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"CR1","doi-asserted-by":"crossref","first-page":"815","DOI":"10.1007\/BF01069233","volume":"17","author":"VV Baranov","year":"1981","unstructured":"Baranov VV (1981) Recursive algorithms of adaptive control in stochastic systems. Cybernetics 17:815?824","journal-title":"Cybernetics"},{"key":"CR2","doi-asserted-by":"crossref","first-page":"207","DOI":"10.1007\/BF00935474","volume":"34","author":"A Federgruen","year":"1981","unstructured":"Federgruen A, Schweitzer PJ (1981) Nonstationary Markov decision problems with converging parameters. J Optim Theory Appl 34:207?241","journal-title":"J Optim Theory Appl"},{"key":"CR3","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4419-8714-3","volume-title":"Adaptive Control Processes","author":"O Hernandez-Lerma","year":"1989","unstructured":"Hernandez-Lerma O (1989) Adaptive Control Processes. Springer, Berlin Heidelberg New York"},{"key":"CR4","volume-title":"Lecture Notes in Operations Research and Mathematical Systems 33","author":"K Hinderer","year":"1970","unstructured":"Hinderer K (1970) Foundations of non-stationary dynamic programming with discrete time parameter. Lecture Notes in Operations Research and Mathematical Systems 33. Springer, Berlin Heidelberg New York"},{"key":"CR5","doi-asserted-by":"crossref","first-page":"289","DOI":"10.1016\/B978-0-12-568150-6.50022-2","volume-title":"Dynamic Programming and its aplications","author":"K Hinderer","year":"1978","unstructured":"Hinderer K (1978) On approximate solutions of finite-stage dynamic programs. In: Puterman ML (ed) Dynamic Programming and its aplications. Academic Press, New York, pp 289?317"},{"key":"CR6","doi-asserted-by":"crossref","first-page":"161","DOI":"10.1007\/BF01740510","volume":"10","author":"G H\u00fcbner","year":"1988","unstructured":"H\u00fcbner G (1988) A unified approach to adaptive control of average reward Markov decision processes. OR Spektrum 10:161?166","journal-title":"OR Spektrum"},{"key":"CR7","unstructured":"H\u00fcbner G (1989) Estimation and adaptive control of span-contracting Markov decision processes. Universit\u00e4t Hamburg, Institut f\u00fcr Mathematische Stochastik, Preprint 89-7. To appear in Kybernetika"},{"key":"CR8","first-page":"67","volume":"15","author":"M Kurano","year":"1972","unstructured":"Kurano M (1972) Discrete-time Markovian decision processes with an unknown parameter ? average return criterion. J Operat Res Soc Japan 15:67?76","journal-title":"J Operat Res Soc Japan"},{"key":"CR9","first-page":"21","volume":"4","author":"M Kurano","year":"1983","unstructured":"Kurano M (1983) Adaptive policies in Markov decision processes with uncertain matrices. J Inf Optim 4:21?40","journal-title":"J Inf Optim"},{"key":"CR10","doi-asserted-by":"crossref","first-page":"270","DOI":"10.2307\/3214080","volume":"24","author":"M Kurano","year":"1987","unstructured":"Kurano M (1987) Learming algorithms for Markov decision processes. J Appl Probab 24:270?276","journal-title":"J Appl Probab"},{"key":"CR11","doi-asserted-by":"crossref","first-page":"40","DOI":"10.2307\/1426206","volume":"6","author":"P Mandl","year":"1974","unstructured":"Mandl P (1974) Estimation and control of Markov chains. Adv Appl Probab 6:40?60","journal-title":"Adv Appl Probab"},{"key":"CR12","first-page":"203","volume":"20","author":"JAEE Nunen van","year":"1976","unstructured":"van Nunen JAEE (1976) A set of successive approximation methods for discounted Markovian decision problems. Z Oper Res 20:203?208","journal-title":"Z Oper Res"},{"key":"CR13","doi-asserted-by":"crossref","first-page":"1127","DOI":"10.1287\/mnsc.24.11.1127","volume":"24","author":"ML Puterman","year":"1978","unstructured":"Puterman ML, Shin MC (1978) Modified policy iteration algorithms for discounted Markov decision problems. Management Sci 24:1127?1137","journal-title":"Management Sci"},{"key":"CR14","series-title":"Lecture Notes in Pure and Applied Mathematics","volume-title":"Optimization: theory and algorithms","author":"M Sch\u00e4l","year":"1981","unstructured":"Sch\u00e4l M (1981) Estimation and control in discounted stochastic dynamic programming. In: Optimization: theory and algorithms. Lecture Notes in Pure and Applied Mathematics. Marcel Dekker, New York"},{"key":"CR15","first-page":"39","volume":"2","author":"M Sch\u00e4l","year":"1984","unstructured":"Sch\u00e4l M (1984) Asymptotic results for sequential Markov decision models under uncertainty. Statistics and Decisions 2:39?62","journal-title":"Statistics and Decisions"},{"key":"CR16","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1080\/17442508708833435","volume":"20","author":"M Sch\u00e4l","year":"1987","unstructured":"Sch\u00e4l M (1987) Estimation and control in discounted stochastic dynamic programming. Stochastics 20:51?71","journal-title":"Stochastics"},{"key":"CR17","doi-asserted-by":"crossref","first-page":"231","DOI":"10.1287\/moor.3.3.231","volume":"3","author":"W Whitt","year":"1978","unstructured":"Whitt W (1978) Approximations of dynamic programs. Math Oper Res 3:231?243","journal-title":"Math Oper Res"}],"container-title":["ZOR Zeitschrift f\ufffdr Operations Research Methods and Models of Operations Research"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/BF01415991.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/BF01415991\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/BF01415991","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,2]],"date-time":"2023-05-02T16:45:42Z","timestamp":1683045942000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/BF01415991"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[1991,11]]},"references-count":17,"journal-issue":{"issue":"6","published-print":{"date-parts":[[1991,11]]}},"alternative-id":["BF01415991"],"URL":"https:\/\/doi.org\/10.1007\/bf01415991","relation":{},"ISSN":["0340-9422","1432-5217"],"issn-type":[{"type":"print","value":"0340-9422"},{"type":"electronic","value":"1432-5217"}],"subject":[],"published":{"date-parts":[[1991,11]]}}}