{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T08:35:49Z","timestamp":1774946149354,"version":"3.50.1"},"publisher-location":"Berlin, Heidelberg","reference-count":33,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783642334856","type":"print"},{"value":"9783642334863","type":"electronic"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2012]]},"DOI":"10.1007\/978-3-642-33486-3_7","type":"book-chapter","created":{"date-parts":[[2012,9,10]],"date-time":"2012-09-10T16:39:17Z","timestamp":1347295157000},"page":"99-115","source":"Crossref","is-referenced-by-count":12,"title":["Adaptive Planning for Markov Decision Processes with Uncertain Transition Models via Incremental Feature Dependency Discovery"],"prefix":"10.1007","author":[{"given":"N. Kemal","family":"Ure","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alborz","family":"Geramifard","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Girish","family":"Chowdhary","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jonathan P.","family":"How","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"issue":"1-2","key":"7_CR1","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1016\/0004-3702(94)00011-O","volume":"72","author":"S.P. Singh","year":"1995","unstructured":"Singh, S.P., Barto, A.G., Bradtke, S.J.: Learning to act using real-timedynamicprogramming. Artificial Intelligience\u00a072(1-2), 81\u2013138 (1995)","journal-title":"Artificial Intelligience"},{"key":"7_CR2","first-page":"19","volume-title":"Proceedings of the Proceedings of the Twenty-Fifth Conference Annual Conference on Uncertainty in Artificial Intelligence (UAI 2009)","author":"J. Asmuth","year":"2009","unstructured":"Asmuth, J., Li, L., Littman, M., Nouri, A., Wingate, D.: A Bayesian sampling approach to exploration in reinforcement learning. In: Proceedings of the Proceedings of the Twenty-Fifth Conference Annual Conference on Uncertainty in Artificial Intelligence (UAI 2009), pp. 19\u201326. AUAI Press, Corvallis (2009)"},{"key":"7_CR3","volume-title":"Dynamic Programming and Optimal Control","author":"D.P. Bertsekas","year":"2007","unstructured":"Bertsekas, D.P.: Dynamic Programming and Optimal Control, 3rd edn., vol.\u00a0I-II. Athena Scientific, Belmont (2007)","edition":"3"},{"key":"7_CR4","doi-asserted-by":"crossref","unstructured":"Bertuccelli, L.F., How, J.P.: Robust Markov decision processes using sigma point sampling. In: American Control Conference (ACC), June 11-13, pp. 5003\u20135008 (2008)","DOI":"10.1109\/ACC.2008.4587287"},{"key":"7_CR5","unstructured":"Bertuccelli, L.F.: Robust Decision-Making with Model Uncertainty in Aerospace Systems. PhD thesis, Massachusetts Institute of Technology, Department of Aeronautics and Astronautics, Cambridge MA (September 2008)"},{"key":"7_CR6","doi-asserted-by":"crossref","unstructured":"Bethke, B., How, J.P., Vian, J.: Multi-UAV Persistent Surveillance With Communication Constraints and Health Management. In: AIAA Guidance, Navigation, and Control Conference (GNC) (August 2009) (AIAA 2009-5654)","DOI":"10.2514\/6.2009-5654"},{"key":"7_CR7","first-page":"213","volume":"3","author":"R.I. Brafman","year":"2001","unstructured":"Brafman, R.I., Tennenholtz, M.: R-MAX - a general polynomial time algorithm for near-optimal reinforcement learning. Journal of Machine Learning Research (JMLR)\u00a03, 213\u2013231 (2001)","journal-title":"Journal of Machine Learning Research (JMLR)"},{"key":"7_CR8","unstructured":"Busoniu, L., Babuska, R., Schutter, B.D., Ernst, D.: Reinforcement Learning and Dynamic Programming Using Function Approximators. CRC Press (2010)"},{"key":"7_CR9","unstructured":"Delage, E., Mannor, S.: Percentile Optimization for Markov Decision Processes with Parameter Uncertainty. Subm. to Operations Research (2007)"},{"key":"7_CR10","doi-asserted-by":"crossref","unstructured":"Vasquez, T.F.D., Laugier, C.: Incremental Learning of Statistical Motion Patterns With Growing Hidden Markov Models. IEEE Transcations on Intelligent Transportation Systems\u00a010(3) (2009)","DOI":"10.1109\/TITS.2009.2020208"},{"key":"7_CR11","unstructured":"Fox, E.B.: Bayesian Nonparametric Learning of Complex Dynamical Phenomena. PhD thesis, Massachusetts Institute of Technology, Cambridge MA (December 2009)"},{"key":"7_CR12","unstructured":"Geramifard, A.: Practical Reinforcement Learning Using Representation Learning and Safe Exploration for Large Scale Markov Decision Processes. PhD thesis, Massachusetts Institute of Technology, Department of Aeronautics and Astronautics (February 2012)"},{"key":"7_CR13","first-page":"881","volume-title":"International Conference on Machine Learning (ICML)","author":"A. Geramifard","year":"2011","unstructured":"Geramifard, A., Doshi, F., Redding, J., Roy, N., How, J.: Online discovery of feature dependencies. In: Getoor, L., Scheffer, T. (eds.) International Conference on Machine Learning (ICML), pp. 881\u2013888. ACM, New York (2011)"},{"key":"7_CR14","unstructured":"Gullapalli, V., Barto, A.: Convergence of Indirect Adaptive Asynchronous Value Iteration Algorithms. In: Neural Information Processing Systems, NIPS (1994)"},{"issue":"2","key":"7_CR15","doi-asserted-by":"publisher","first-page":"257","DOI":"10.1287\/moor.1040.0129","volume":"30","author":"G. Iyengar","year":"2005","unstructured":"Iyengar, G.: Robust Dynamic Programming. Math. Oper. Res.\u00a030(2), 257\u2013280 (2005)","journal-title":"Math. Oper. Res."},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"Jilkov, V., Li, X.: Online Bayesian Estimation of Transition Probabilities for Markovian Jump Systems. IEEE Trans. on Signal Processing\u00a052(6) (2004)","DOI":"10.1109\/TSP.2004.827145"},{"issue":"4","key":"7_CR17","doi-asserted-by":"publisher","first-page":"383","DOI":"10.1007\/s10514-011-9248-x","volume":"31","author":"J. Joseph","year":"2011","unstructured":"Joseph, J., Doshi-Velez, F., Huang, A.S., Roy, N.: A Bayesian nonparametric approach to modeling motion patterns. Autonomous Robots\u00a031(4), 383\u2013400 (2011)","journal-title":"Autonomous Robots"},{"key":"7_CR18","unstructured":"Kushner, H.J., George Yin, G.: Convergence of indirect adaptive asynchronous value iteration algorithms. Springer (2003)"},{"issue":"2","key":"7_CR19","doi-asserted-by":"publisher","first-page":"543","DOI":"10.1109\/TSP.2007.907881","volume":"56","author":"W. Liu","year":"2008","unstructured":"Liu, W., Pokharel, P.P., Principe, J.C.: The kernel least-mean-square algorithm. IEEE Transactions on Signal Processing\u00a056(2), 543\u2013554 (2008)","journal-title":"IEEE Transactions on Signal Processing"},{"key":"7_CR20","first-page":"103","volume":"13","author":"A.W. Moore","year":"1993","unstructured":"Moore, A.W., Atkeson, C.G.: Prioritized sweeping: Reinforcement learning with less data and less time. Machine Learning\u00a013, 103\u2013130 (1993)","journal-title":"Machine Learning"},{"key":"7_CR21","doi-asserted-by":"crossref","unstructured":"Nigam, N., Kroo, I.: Persistent surveillance using multiple unmanned air vehicles. In: 2008 IEEE Aerospace Conference, pp. 1\u201314 (March 2008)","DOI":"10.1109\/AERO.2008.4526242"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Nilim, A., El Ghaoui, L.: Robust Solutions to Markov Decision Problems with Uncertain Transition Matrices. Operations Research\u00a053(5) (2005)","DOI":"10.1287\/opre.1050.0216"},{"key":"7_CR23","doi-asserted-by":"crossref","unstructured":"Puterman, M.L.: Markov Decision Processes: Stochastic Dynamic Programming. John Wiley and Sons (1994)","DOI":"10.1002\/9780470316887"},{"key":"7_CR24","volume-title":"Gaussian Processes for Machine Learning","author":"C. Rasmussen","year":"2006","unstructured":"Rasmussen, C., Williams, C.: Gaussian Processes for Machine Learning. MIT Press, Cambridge (2006)"},{"key":"7_CR25","doi-asserted-by":"crossref","unstructured":"Redding, J.D., Toksoz, T., Kemal Ure, N., Geramifard, A., How, J.P., Vavrina, M., Vian, J.: Persistent distributed multi-agent missions with automated battery management. In: AIAA Guidance, Navigation, and Control Conference (GNC) (August 2011) (AIAA-2011-6480)","DOI":"10.2514\/6.2011-6480"},{"key":"7_CR26","volume-title":"Artificial Intelligence: A Modern Approach","author":"S. Russell","year":"2003","unstructured":"Russell, S., Norvig, P.: Artificial Intelligence: A Modern Approach, 2nd edn. Prentice-Hall, Englewood Cliffs (2003)","edition":"2"},{"key":"7_CR27","doi-asserted-by":"crossref","unstructured":"Ryan, A., Hedrick, J.K.: A mode-switching path planner for uav-assisted search and rescue. In: 44th IEEE Conference on Decision and Control, 2005 and 2005 European Control Conference, CDC-ECC 2005, pp. 1471\u20131476 (December 2005)","DOI":"10.1109\/CDC.2005.1582366"},{"issue":"2","key":"7_CR28","doi-asserted-by":"publisher","first-page":"439","DOI":"10.1007\/BF02190104","volume":"91","author":"A. Shapiro","year":"1996","unstructured":"Shapiro, A., Wardi, Y.: Convergence Analysis of Gradient Descent Stochastic Algorithms. Journal of Optimization Theory and Applications\u00a091(2), 439\u2013454 (1996)","journal-title":"Journal of Optimization Theory and Applications"},{"key":"7_CR29","unstructured":"Sutton, R.S., Szepesvari, C., Geramifard, A., Bowling, M.: Dyna-style planning with linear function approximation and prioritized sweeping. In: Proceedings of the 24th Conference on Uncertainty in Artificial Intelligence, pp. 528\u2013536 (2008)"},{"key":"7_CR30","doi-asserted-by":"crossref","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press (1998)","DOI":"10.1109\/TNN.1998.712192"},{"issue":"3","key":"7_CR31","doi-asserted-by":"publisher","first-page":"127","DOI":"10.1049\/ecej:20010303","volume":"13","author":"T.C. Tozer","year":"2001","unstructured":"Tozer, T.C., Grace, D.: High-altitude platforms for wireless communications. Electronics Communication Engineering Journal\u00a013(3), 127\u2013137 (2001)","journal-title":"Electronics Communication Engineering Journal"},{"key":"7_CR32","unstructured":"Paul, E.: Constructive function approximation. In: Feature extraction, construction, and selection: A data-mining perspective (1998)"},{"key":"7_CR33","unstructured":"Yao, H., Sutton, R.S., Bhatnagar, S., Dongcui, D., Szepesv\u00e1ri, C.: Multi-step dyna planning for policy evaluation and control. In: Bengio, Y., Schuurmans, D., Lafferty, J.D., Williams, C.K.I., Culotta, A. (eds.) NIPS, pp. 2187\u20132195. Curran Associates, Inc. (2009)"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Discovery in Databases"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-33486-3_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,8]],"date-time":"2025-04-08T04:34:19Z","timestamp":1744086859000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-33486-3_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012]]},"ISBN":["9783642334856","9783642334863"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-33486-3_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2012]]}}}