{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T20:15:04Z","timestamp":1784837704292,"version":"3.55.0"},"reference-count":27,"publisher":"Elsevier","isbn-type":[{"value":"9781558603776","type":"print"}],"license":[{"start":{"date-parts":[[1995,1,1]],"date-time":"1995-01-01T00:00:00Z","timestamp":788918400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[1995]]},"DOI":"10.1016\/b978-1-55860-377-6.50052-9","type":"book-chapter","created":{"date-parts":[[2014,7,1]],"date-time":"2014-07-01T03:00:19Z","timestamp":1404183619000},"page":"362-370","source":"Crossref","is-referenced-by-count":271,"title":["Learning policies for partially observable environments: Scaling up"],"prefix":"10.1016","author":[{"given":"Michael L.","family":"Littman","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Anthony R.","family":"Cassandra","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Leslie Pack","family":"Kaelbling","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib1","doi-asserted-by":"crossref","first-page":"174","DOI":"10.1016\/0022-247X(65)90154-X","article-title":"Optimal control of Markov decision processes with incomplete state estimation","volume":"10","author":"Astrom","year":"1965","journal-title":"J. Math. Anal. Appl."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib2","series-title":"Dynamic Programming: Deterministic and Stochastic Models","author":"Bertsekas","year":"1987"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib3","unstructured":"Boutilier, C., Dearden, R., and Goldszmidt, M. (1995). Exploiting structure in policy construction. In Proceedings of the International Joint Conference on Artificial Intelligence."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib4","unstructured":"Cassandra, A. (1994). Optimal policies for partially observable Markov decision processes. Technical Report CS-94\u201314, Brown University, Department of Computer Science, Providence RI."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib5","unstructured":"Cassandra, A. R., Kaelbling, L. P., and Littman, M. L. (1994). Acting optimally in partially observable stochastic domains. In Proceedings of the Twelfth National Conference on Artificial Intelligence, Seattle, WA."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib6","unstructured":"Cheng, H.-T. (1988). Algorithms for Partially Observable Markov Decision Processes. PhD thesis, University of British Columbia, British Columbia, Canada."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib7","unstructured":"Chrisman, L. (1992). Reinforcement learning with perceptual aliasing: The perceptual distinctions approach. In Proc. Tenth National Conference on AI (AAAI)."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib8","series-title":"Robot Learning","article-title":"Rapid task learning for real robots","author":"Connell","year":"1993"},{"issue":"6","key":"10.1016\/B978-1-55860-377-6.50052-9_bib9","doi-asserted-by":"crossref","DOI":"10.1162\/neco.1994.6.6.1185","article-title":"On the convergence of stochastic iterative dynamic programming algorithms","volume":"6","author":"Jaakkola","year":"1994","journal-title":"Neural Computation"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib10","unstructured":"Kushmerick, N., Hanks, S., and Weld, D. (1993). An Algorithm for Probabilistic Planning. Technical Report 93\u201306-03, University of Washington Department of Computer Science and Engineering. To appear in Artificial Intelligence."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib11","doi-asserted-by":"crossref","unstructured":"Littman, M., Cassandra, A., and Kaelbling, L. (1995). Learning policies for partially observable environments: Scaling up. Technical Report CS-95\u201311, Brown University, Department of Computer Science, Providence RI.","DOI":"10.1016\/B978-1-55860-377-6.50052-9"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib12","unstructured":"Littman, M. L. (1994). The Witness algorithm: Solving partially observable Markov decision processes. Technical Report CS-94\u201340, Brown University, Department of Computer Science, Providence, RI."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib13","doi-asserted-by":"crossref","first-page":"47","DOI":"10.1007\/BF02055574","article-title":"A survey of algorithmic methods for partially observable Markov decision processes","volume":"28","author":"Lovejoy","year":"1991","journal-title":"Annals of Operations Research"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib14","unstructured":"McCallum, R. A. (1992). First results with utile distinction memory for reinforcement learning. Technical Report 446, Dept. Comp. Sci., Univ. Rochester. See also Proceedings of Machine Learning Conference 1993."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib15","article-title":"The parti-game algorithm for variable resolution reinforcement learning in multidimensional state spaces","volume":"6","author":"Moore","year":"1994"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib16","unstructured":"Nicholson, A. and Kaelbling, L. P. (1994). Toward approximate planning in very large stochastic domains. In Proceedings of the AAAI Spring Symposium on Decision Theoretic Planning, Stanford, California."},{"issue":"3","key":"10.1016\/B978-1-55860-377-6.50052-9_bib17","doi-asserted-by":"crossref","first-page":"441","DOI":"10.1287\/moor.12.3.441","article-title":"The complexity of Markov decision processes","volume":"12","author":"Papadimitriou","year":"1987","journal-title":"Mathematics of Operations Research"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib18","unstructured":"Parr, R. and Russell, S. (1995). Approximating optimal policies for partially observable stochastic domains. In Proceedings of the International Joint Conference on Artificial Intelligence."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib19","series-title":"Markov Decision Processes\u2014Discrete Stochastic Dynamic Programming","author":"Puterman","year":"1994"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib20","series-title":"Introduction to Stochastic Dynamic Programming","author":"Ross","year":"1983"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib21","unstructured":"Rumelhart, D. E., Hinton, G. E., and Williams, R. J. Learning internal representations by error backpropagation. In Rumelhart, D. E. and McClelland, J. L., editors, Parallel Distributed Processing: Explorations in the microstructures of cognition. Volume 1: Foundations, chapter 8. The MIT Press, Cambridge, MA."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib22","series-title":"Artificial Intelligence: A Modern Approach","author":"Russell","year":"1994"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib23","doi-asserted-by":"crossref","first-page":"1071","DOI":"10.1287\/opre.21.5.1071","article-title":"The optimal control of partially observable Markov processes over a finite horizon","volume":"21","author":"Smallwood","year":"1973","journal-title":"Operations Research"},{"issue":"2","key":"10.1016\/B978-1-55860-377-6.50052-9_bib24","doi-asserted-by":"crossref","DOI":"10.1287\/opre.26.2.282","article-title":"The optimal control of partially observable Markov processes over the infinite horizon: Discounted costs","volume":"26","author":"Sondik","year":"1978","journal-title":"Operations Research"},{"issue":"3","key":"10.1016\/B978-1-55860-377-6.50052-9_bib25","article-title":"Asynchronous stohcastic aproximation and Q-learning","volume":"16","author":"Tsitsikilis","year":"1994","journal-title":"Machine Learning"},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib26","unstructured":"Watkins, C. J. (1989). Learning with Delayed Rewards. PhD thesis, Cambridge University."},{"key":"10.1016\/B978-1-55860-377-6.50052-9_bib27","unstructured":"Williams, R. J. and Baird, L. C. I. (1993). Tight performance bounds on greedy policies based on imperfect value functions. Technical Report NU- CCS-93\u201313, Northeastern University, College of Computer Science, Boston, MA."}],"container-title":["Machine Learning Proceedings 1995"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:B9781558603776500529?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:B9781558603776500529?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2019,8,12]],"date-time":"2019-08-12T05:45:57Z","timestamp":1565588757000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/B9781558603776500529"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[1995]]},"ISBN":["9781558603776"],"references-count":27,"URL":"https:\/\/doi.org\/10.1016\/b978-1-55860-377-6.50052-9","relation":{},"subject":[],"published":{"date-parts":[[1995]]}}}