{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T08:38:24Z","timestamp":1743064704646,"version":"3.40.3"},"publisher-location":"Berlin, Heidelberg","reference-count":28,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642198892"},{"type":"electronic","value":"9783642198908"}],"license":[{"start":{"date-parts":[[2011,1,1]],"date-time":"2011-01-01T00:00:00Z","timestamp":1293840000000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2011]]},"DOI":"10.1007\/978-3-642-19890-8_5","type":"book-chapter","created":{"date-parts":[[2011,3,7]],"date-time":"2011-03-07T09:40:58Z","timestamp":1299490858000},"page":"61-77","source":"Crossref","is-referenced-by-count":4,"title":["Towards Min Max Generalization in Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Raphael","family":"Fonteneau","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Susan A.","family":"Murphy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Louis","family":"Wehenkel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Damien","family":"Ernst","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"5_CR1","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1007\/BFb0109870","volume":"245","author":"A. Bemporad","year":"1999","unstructured":"Bemporad, A., Morari, M.: Robust model predictive control: A survey. Robustness in Identification and Control\u00a0245, 207\u2013226 (1999)","journal-title":"Robustness in Identification and Control"},{"key":"5_CR2","volume-title":"Neuro-Dynamic Programming","author":"D. Bertsekas","year":"1996","unstructured":"Bertsekas, D., Tsitsiklis, J.: Neuro-Dynamic Programming. Athena Scientific, Belmont (1996)"},{"key":"5_CR3","first-page":"369","volume-title":"Advances in Neural Information Processing Systems (NIPS 1995)","author":"J. Boyan","year":"1995","unstructured":"Boyan, J., Moore, A.: Generalization in reinforcement learning: Safely approximating the value function. In: Advances in Neural Information Processing Systems (NIPS 1995), vol.\u00a07, pp. 369\u2013376. MIT Press, Denver (1995)"},{"key":"5_CR4","doi-asserted-by":"crossref","unstructured":"Chakratovorty, S., Hyland, D.: Minimax reinforcement learning. In: Proceedings of AIAA Guidance, Navigation, and Control Conference and Exhibit, San Francisco, CA, USA (2003)","DOI":"10.2514\/6.2003-5718"},{"key":"5_CR5","first-page":"1679","volume":"9","author":"B.C. Cs\u00e1ji","year":"2008","unstructured":"Cs\u00e1ji, B.C., Monostori, L.: Value function based reinforcement learning in changing Markovian environments. Journal of Machine Learning Research\u00a09, 1679\u20131709 (2008)","journal-title":"Journal of Machine Learning Research"},{"key":"5_CR6","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-77974-2","volume-title":"Computational Geometry: Algorithms and Applications","author":"M. Berg De","year":"2008","unstructured":"De Berg, M., Cheong, O., Van Kreveld, M., Overmars, M.: Computational Geometry: Algorithms and Applications. Springer, Heidelberg (2008)"},{"key":"5_CR7","unstructured":"Delage, E., Mannor, S.: Percentile optimization for Markov decision processes with parameter uncertainty. Operations Research (2006)"},{"key":"5_CR8","unstructured":"Ernst, D.: Selecting concise sets of samples for a reinforcement learning agent. In: Proceedings of the Third International Conference on Computational Intelligence, Robotics and Autonomous Systems (CIRAS 2005), Singapore (2005)"},{"key":"5_CR9","first-page":"503","volume":"6","author":"D. Ernst","year":"2005","unstructured":"Ernst, D., Geurts, P., Wehenkel, L.: Tree-based batch mode reinforcement learning. Journal of Machine Learning Research\u00a06, 503\u2013556 (2005)","journal-title":"Journal of Machine Learning Research"},{"key":"5_CR10","doi-asserted-by":"publisher","first-page":"517","DOI":"10.1109\/TSMCB.2008.2007630","volume":"39","author":"D. Ernst","year":"2009","unstructured":"Ernst, D., Glavic, M., Capitanescu, F., Wehenkel, L.: Reinforcement learning versus model predictive control: a comparison on a power system problem. IEEE Transactions on Systems, Man, and Cybernetics - Part B: Cybernetics\u00a039, 517\u2013529 (2009)","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics - Part B: Cybernetics"},{"key":"5_CR11","doi-asserted-by":"crossref","unstructured":"Fonteneau, R., Murphy, S., Wehenkel, L., Ernst, D.: Inferring bounds on the performance of a control policy from a sample of trajectories. In: Proceedings of the 2009 IEEE Symposium on Adaptive Dynamic Programming and Reinforcement Learning (IEEE ADPRL 2009), Nashville, TN, USA (2009)","DOI":"10.1109\/ADPRL.2009.4927534"},{"key":"5_CR12","unstructured":"Fonteneau, R., Murphy, S., Wehenkel, L., Ernst, D.: A cautious approach to generalization in reinforcement learning. In: Proceedings of the Second International Conference on Agents and Artificial Intelligence (ICAART 2010), Valencia, Spain (2010)"},{"key":"5_CR13","unstructured":"Fonteneau, R., Murphy, S.A., Wehenkel, L., Ernst, D.: Computing bounds for kernel-based policy evaluation in reinforcement learning. Tech. rep., Arxiv (2010)"},{"key":"5_CR14","unstructured":"Gordon, G.: Approximate Solutions to Markov Decision Processes. Ph.D. thesis, Carnegie Mellon University (1999)"},{"key":"5_CR15","unstructured":"Ingersoll, J.: Theory of Financial Decision Making. Rowman and Littlefield Publishers, Inc. (1987)"},{"key":"5_CR16","first-page":"1107","volume":"4","author":"M. Lagoudakis","year":"2003","unstructured":"Lagoudakis, M., Parr, R.: Least-squares policy iteration. Jounal of Machine Learning Research\u00a04, 1107\u20131149 (2003)","journal-title":"Jounal of Machine Learning Research"},{"key":"5_CR17","doi-asserted-by":"crossref","unstructured":"Littman, M.L.: Markov games as a framework for multi-agent reinforcement learning. In: Proceedings of the Eleventh International Conference on Machine Learning (ICML 1994), New Brunswick, NJ, USA (1994)","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Mannor, S., Simester, D., Sun, P., Tsitsiklis, J.: Bias and variance in value function estimation. In: Proceedings of the Twenty-first International Conference on Machine Learning (ICML 2004), Banff, Alberta, Canada (2004)","DOI":"10.1145\/1015330.1015402"},{"issue":"2","key":"5_CR19","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1111\/1467-9868.00389","volume":"65","author":"S. Murphy","year":"2003","unstructured":"Murphy, S.: Optimal dynamic treatment regimes. Journal of the Royal Statistical Society, Series B\u00a065(2), 331\u2013366 (2003)","journal-title":"Journal of the Royal Statistical Society, Series B"},{"key":"5_CR20","doi-asserted-by":"publisher","first-page":"1455","DOI":"10.1002\/sim.2022","volume":"24","author":"S. Murphy","year":"2005","unstructured":"Murphy, S.: An experimental design for the development of adaptive treatment strategies. Statistics in Medicine\u00a024, 1455\u20131481 (2005)","journal-title":"Statistics in Medicine"},{"issue":"2-3","key":"5_CR21","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1023\/A:1017928328829","volume":"49","author":"D. Ormoneit","year":"2002","unstructured":"Ormoneit, D., Sen, S.: Kernel-based reinforcement learning. Machine Learning\u00a049(2-3), 161\u2013178 (2002)","journal-title":"Machine Learning"},{"key":"5_CR22","unstructured":"Qian, M., Murphy, S.: Performance guarantees for individualized treatment rules. Tech. Rep. 498, Department of Statistics, University of Michigan (2009)"},{"key":"5_CR23","series-title":"Lecture Notes in Artificial Intelligence","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1007\/11564096_32","volume-title":"Machine Learning: ECML 2005","author":"M. Riedmiller","year":"2005","unstructured":"Riedmiller, M.: Neural fitted Q iteration - first experiences with a data efficient neural reinforcement learning method. In: Gama, J., Camacho, R., Brazdil, P.B., Jorge, A.M., Torgo, L. (eds.) ECML 2005. LNCS (LNAI), vol.\u00a03720, pp. 317\u2013328. Springer, Heidelberg (2005)"},{"key":"5_CR24","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"417","DOI":"10.1007\/978-3-642-12842-4_53","volume-title":"Artificial Intelligence: Theories, Models and Applications","author":"M. Rovatous","year":"2010","unstructured":"Rovatous, M., Lagoudakis, M.: Minimax search and reinforcement learning for adversarial tetris. In: Konstantopoulos, S., Perantonis, S., Karkaletsis, V., Spyropoulos, C.D., Vouros, G. (eds.) SETN 2010. LNCS, vol.\u00a06040, pp. 417\u2013422. Springer, Heidelberg (2010)"},{"key":"5_CR25","volume-title":"Real and Complex Analysis","author":"W. Rudin","year":"1987","unstructured":"Rudin, W.: Real and Complex Analysis. McGraw-Hill, New York (1987)"},{"key":"5_CR26","first-page":"1038","volume-title":"Advances in Neural Information Processing Systems (NIPS 1996)","author":"R. Sutton","year":"1996","unstructured":"Sutton, R.: Generalization in reinforcement learning: Successful examples using sparse coding. In: Advances in Neural Information Processing Systems (NIPS 1996), vol.\u00a08, pp. 1038\u20131044. MIT Press, Denver (1996)"},{"key":"5_CR27","volume-title":"Reinforcement Learning","author":"R. Sutton","year":"1998","unstructured":"Sutton, R., Barto, A.: Reinforcement Learning. MIT Press, Cambridge (1998)"},{"issue":"2","key":"5_CR28","doi-asserted-by":"publisher","first-page":"260","DOI":"10.1109\/TIT.1967.1054010","volume":"13","author":"A. Viterbi","year":"1967","unstructured":"Viterbi, A.: Error bounds for convolutional codes and an asymptotically optimum decoding algorithm. IEEE Transactions on Information Theory\u00a013(2), 260\u2013269 (1967)","journal-title":"IEEE Transactions on Information Theory"}],"container-title":["Communications in Computer and Information Science","Agents and Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-19890-8_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,21]],"date-time":"2019-05-21T05:59:08Z","timestamp":1558418348000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-19890-8_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2011]]},"ISBN":["9783642198892","9783642198908"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-19890-8_5","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2011]]}}}