{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,16]],"date-time":"2026-05-16T03:59:45Z","timestamp":1778903985109,"version":"3.51.4"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2012,11,15]],"date-time":"2012-11-15T00:00:00Z","timestamp":1352937600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Ann Oper Res"],"published-print":{"date-parts":[[2013,9]]},"DOI":"10.1007\/s10479-012-1248-5","type":"journal-article","created":{"date-parts":[[2012,11,14]],"date-time":"2012-11-14T06:30:44Z","timestamp":1352874644000},"page":"383-416","source":"Crossref","is-referenced-by-count":23,"title":["Batch mode reinforcement learning based on the synthesis of artificial trajectories"],"prefix":"10.1007","volume":"208","author":[{"given":"Raphael","family":"Fonteneau","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Susan A.","family":"Murphy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Louis","family":"Wehenkel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Damien","family":"Ernst","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2012,11,15]]},"reference":[{"key":"1248_CR1","volume-title":"Advances in neural information processing systems (NIPS)","author":"A. Antos","year":"2007","unstructured":"Antos, A., Munos, R., & Szepesv\u00e1ri, C. (2007). Fitted Q-iteration in continuous action space MDPs. In Advances in neural information processing systems (NIPS) (Vol.\u00a020)."},{"key":"1248_CR2","volume-title":"Dynamic programming","author":"R. Bellman","year":"1957","unstructured":"Bellman, R. (1957). Dynamic programming. Princeton: Princeton University Press."},{"key":"1248_CR3","doi-asserted-by":"crossref","first-page":"151","DOI":"10.1007\/978-0-387-09695-7_15","volume-title":"Artificial intelligence in theory and practice II","author":"A. Bonarini","year":"2008","unstructured":"Bonarini, A., Caccia, C., Lazaric, A., & Restelli, M. (2008). Batch reinforcement learning for controlling a mobile wheeled pendulum robot. In Artificial intelligence in theory and practice II (pp. 151\u2013160)."},{"key":"1248_CR4","doi-asserted-by":"crossref","first-page":"233","DOI":"10.1023\/A:1017936530646","volume":"49","author":"J. Boyan","year":"2005","unstructured":"Boyan, J. (2005). Technical update: least-squares temporal difference learning. Machine Learning, 49, 233\u2013246.","journal-title":"Machine Learning"},{"key":"1248_CR5","first-page":"369","volume-title":"Advances in neural information processing systems (NIPS)","author":"J. Boyan","year":"1995","unstructured":"Boyan, J., & Moore, A. (1995). Generalization in reinforcement learning: safely approximating the value function. In Advances in neural information processing systems (NIPS) (Vol.\u00a07, pp. 369\u2013376). Denver: MIT Press."},{"key":"1248_CR6","first-page":"33","volume":"22","author":"S. Bradtke","year":"1996","unstructured":"Bradtke, S., & Barto, A. (1996). Linear least-squares algorithms for temporal difference learning. Machine Learning, 22, 33\u201357.","journal-title":"Machine Learning"},{"key":"1248_CR7","doi-asserted-by":"crossref","DOI":"10.1201\/9781439821091","volume-title":"Reinforcement learning and dynamic programming using function approximators","author":"L. Busoniu","year":"2010","unstructured":"Busoniu, L., Babuska, R., De Schutter, B., & Ernst, D. (2010). Reinforcement learning and dynamic programming using function approximators. London: Taylor & Francis\/CRC Press."},{"issue":"8","key":"1248_CR8","doi-asserted-by":"crossref","first-page":"1031","DOI":"10.1016\/j.conengprac.2006.02.011","volume":"15","author":"A. Castelletti","year":"2007","unstructured":"Castelletti, A., de Rigo, D., Rizzoli, A., Soncini-Sessa, R., & Weber, E. (2007). Neuro-dynamic programming for designing water reservoir network management policies. Control Engineering Practice, 15(8), 1031\u20131038.","journal-title":"Control Engineering Practice"},{"key":"1248_CR9","volume":"46","author":"A. Castelletti","year":"2010","unstructured":"Castelletti, A., Galelli, S., Restelli, M., & Soncini-Sessa, R. (2010). Tree-based reinforcement learning for optimal water reservoir operation. Water Resources Research, 46, W09507.","journal-title":"Water Resources Research"},{"key":"1248_CR10","unstructured":"Chakraborty, B., Strecher, V., & Murphy, S. (2008). Bias correction and confidence intervals for fitted Q-iteration. In Workshop on model uncertainty and risk in reinforcement learning (NIPS), Whistler, Canada."},{"key":"1248_CR11","unstructured":"Defourny, B., Ernst, D., & Wehenkel, L. (2008). Risk-aware decision making and dynamic programming. In Workshop on model uncertainty and risk in reinforcement learning (NIPS), Whistler, Canada."},{"key":"1248_CR12","first-page":"96","volume-title":"European conference on machine learning (ECML)","author":"D. Ernst","year":"2003","unstructured":"Ernst, D., Geurts, P., & Wehenkel, L. (2003). Iteratively extending time horizon reinforcement learning. In European conference on machine learning (ECML) (pp. 96\u2013107)."},{"key":"1248_CR13","first-page":"503","volume":"6","author":"D. Ernst","year":"2005","unstructured":"Ernst, D., Geurts, P., & Wehenkel, L. (2005). Tree-based batch mode reinforcement learning. Journal of Machine Learning Research, 6, 503\u2013556.","journal-title":"Journal of Machine Learning Research"},{"key":"1248_CR14","series-title":"Lecture notes in computer science","doi-asserted-by":"crossref","first-page":"446","DOI":"10.1007\/11821045_47","volume-title":"International workshop on intelligent computing in pattern analysis\/synthesis","author":"D. Ernst","year":"2006","unstructured":"Ernst, D., Mar\u00e9e, R., & Wehenkel, L. (2006a). Reinforcement learning with raw image pixels as state input (IWICPAS). In Lecture notes in computer science: Vol.\u00a04153. International workshop on intelligent computing in pattern analysis\/synthesis (pp. 446\u2013454)."},{"key":"1248_CR15","first-page":"65","volume-title":"Machine learning conference of Belgium and the Netherlands (BeNeLearn)","author":"D. Ernst","year":"2006","unstructured":"Ernst, D., Stan, G., Goncalves, J., & Wehenkel, L. (2006b). Clinical data based optimal STI strategies for HIV: a reinforcement learning approach. In Machine learning conference of Belgium and the Netherlands (BeNeLearn) (pp. 65\u201372)."},{"key":"1248_CR16","doi-asserted-by":"crossref","first-page":"517","DOI":"10.1109\/TSMCB.2008.2007630","volume":"39","author":"D. Ernst","year":"2009","unstructured":"Ernst, D., Glavic, M., Capitanescu, F., & Wehenkel, L. (2009). Reinforcement learning versus model predictive control: a comparison on a power system problem. IEEE Transactions on Systems, Man and Cybernetics. Part B. Cybernetics, 39, 517\u2013529.","journal-title":"IEEE Transactions on Systems, Man and Cybernetics. Part B. Cybernetics"},{"key":"1248_CR17","series-title":"Lecture notes in computer science","doi-asserted-by":"crossref","first-page":"55","DOI":"10.1007\/978-3-540-89722-4_5","volume-title":"Recent advances in reinforcement learning","author":"A. Farahmand","year":"2008","unstructured":"Farahmand, A., Ghavamzadeh, M., Szepesv\u00e1ri, C., & Mannor, S. (2008). Regularized fitted q-iteration: application to planning. In S. Girgin, M. Loth, R. Munos, P. Preux, & D. Ryabko (Eds.), Lecture notes in computer science: Vol.\u00a05323. Recent advances in reinforcement learning (pp. 55\u201368). Berlin\/Heidelberg: Springer."},{"key":"1248_CR18","unstructured":"Fonteneau, R. (2011). Contributions to batch mode reinforcement learning. Ph.D. thesis, University of Li\u00e8ge."},{"key":"1248_CR19","doi-asserted-by":"crossref","unstructured":"Fonteneau, R., Murphy, S., Wehenkel, L., & Ernst, D. (2009). Inferring bounds on the performance of a control policy from a sample of trajectories. In IEEE symposium on adaptive dynamic programming and reinforcement learning (ADPRL), Nashville, TN, USA.","DOI":"10.1109\/ADPRL.2009.4927534"},{"key":"1248_CR20","doi-asserted-by":"crossref","unstructured":"Fonteneau, R., Murphy, S., Wehenkel, L., & Ernst, D. (2010a). A cautious approach to generalization in reinforcement learning. In Second international conference on agents and artificial intelligence (ICAART), Valencia, Spain.","DOI":"10.1007\/978-3-642-19890-8_5"},{"key":"1248_CR21","unstructured":"Fonteneau, R., Murphy, S., Wehenkel, L., & Ernst, D. (2010b). Generating informative trajectories by using bounds on the return of control policies. In Workshop on active learning and experimental design 2010 (in conjunction with AISTATS 2010)."},{"key":"1248_CR22","series-title":"JMLR: W&CP","first-page":"217","volume-title":"Thirteenth international conference on artificial intelligence and statistics (AISTATS)","author":"R. Fonteneau","year":"2010","unstructured":"Fonteneau, R., Murphy, S., Wehenkel, L., & Ernst, D. (2010c). Model-free Monte Carlo-like policy evaluation. In JMLR: W&CP: Vol.\u00a09. Thirteenth international conference on artificial intelligence and statistics (AISTATS) (pp. 217\u2013224). Laguna: Chia."},{"key":"1248_CR23","series-title":"Communications in computer and information science (CCIS)","first-page":"61","volume-title":"Revised selected papers. agents and artificial intelligence: international conference (ICAART 2010)","author":"R. Fonteneau","year":"2010","unstructured":"Fonteneau, R., Murphy, S. A., Wehenkel, L., & Ernst, D. (2010d). Towards min max generalization in reinforcement learning. In Communications in computer and information science (CCIS): Vol.\u00a0129. Revised selected papers. agents and artificial intelligence: international conference (ICAART 2010), Valencia, Spain (pp. 61\u201377). Heidelberg: Springer."},{"key":"1248_CR24","first-page":"261","volume-title":"Twelfth international conference on machine learning (ICML)","author":"G. Gordon","year":"1995","unstructured":"Gordon, G. (1995). Stable function approximation in dynamic programming. In Twelfth international conference on machine learning (ICML) (pp. 261\u2013268)."},{"key":"1248_CR25","unstructured":"Gordon, G. (1999). Approximate solutions to Markov decision processes. Ph.D. thesis, Carnegie Mellon University."},{"key":"1248_CR26","unstructured":"Guez, A., Vincent, R., Avoli, M., & Pineau, J. (2008). Adaptive treatment of epilepsy via batch-mode reinforcement learning. In Innovative applications of artificial intelligence (IAAI)."},{"key":"1248_CR27","first-page":"1107","volume":"4","author":"M. Lagoudakis","year":"2003","unstructured":"Lagoudakis, M., & Parr, R. (2003). Least-squares policy iteration. Journal of Machine Learning Research, 4, 1107\u20131149.","journal-title":"Journal of Machine Learning Research"},{"key":"1248_CR28","unstructured":"Lange, S., & Riedmiller, M. (2010). Deep learning of visual control policies. In European symposium on artificial neural networks, computational intelligence and machine learning (ESANN), Brugge, Belgium."},{"key":"1248_CR29","unstructured":"Lazaric, A., Ghavamzadeh, M., & Munos, R. (2010a). Finite-sample analysis of least-squares policy iteration (Tech. Rep.). SEQUEL (INRIA) Lille\u2013Nord Europe."},{"key":"1248_CR30","first-page":"615","volume-title":"International conference on machine learning (ICML)","author":"A. Lazaric","year":"2010","unstructured":"Lazaric, A., Ghavamzadeh, M., & Munos, R. (2010b). Finite-sample analysis of LSTD. In International conference on machine learning (ICML) (pp. 615\u2013622)."},{"key":"1248_CR31","volume-title":"27th international conference on machine learning (ICML)","author":"T. Morimura","year":"2010","unstructured":"Morimura, T., Sugiyama, M., Kashima, H., Hachiya, H., & Tanaka, T. (2010a). Nonparametric return density estimation for reinforcement learning. In 27th international conference on machine learning (ICML), Haifa, Israel, June 21\u201325."},{"key":"1248_CR32","first-page":"368","volume-title":"26th conference on uncertainty in artificial intelligence (UAI)","author":"T. Morimura","year":"2010","unstructured":"Morimura, T., Sugiyama, M., Kashima, H., Hachiya, H., & Tanaka, T. (2010b). Parametric return density estimation for reinforcement learning. In 26th conference on uncertainty in artificial intelligence (UAI), Catalina Island, California, USA, Jul. 8\u201311 (pp. 368\u2013375)."},{"key":"1248_CR33","first-page":"815","volume":"9","author":"R. Munos","year":"2008","unstructured":"Munos, R., & Szepesv\u00e1ri, C. (2008). Finite-time bounds for fitted value iteration. Journal of Machine Learning Research, 9, 815\u2013857.","journal-title":"Journal of Machine Learning Research"},{"issue":"2","key":"1248_CR34","doi-asserted-by":"crossref","first-page":"331","DOI":"10.1111\/1467-9868.00389","volume":"65","author":"S. Murphy","year":"2003","unstructured":"Murphy, S. (2003). Optimal dynamic treatment regimes. Journal of the Royal Statistical Society. Series B, 65(2), 331\u2013366.","journal-title":"Journal of the Royal Statistical Society. Series B"},{"issue":"456","key":"1248_CR35","doi-asserted-by":"crossref","first-page":"1410","DOI":"10.1198\/016214501753382327","volume":"96","author":"S. Murphy","year":"2001","unstructured":"Murphy, S., Van Der Laan, M., & Robins, J. (2001). Marginal mean models for dynamic regimes. Journal of the American Statistical Association, 96(456), 1410\u20131423.","journal-title":"Journal of the American Statistical Association"},{"key":"1248_CR36","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1023\/A:1022192903948","volume":"13","author":"A. Nedi","year":"2003","unstructured":"Nedi, A., & Bertsekas, D. P. (2003). Least squares policy evaluation algorithms with linear function approximation. Discrete Event Dynamic Systems, 13, 79\u2013110. doi: 10.1023\/A:1022192903948 .","journal-title":"Discrete Event Dynamic Systems"},{"issue":"2\u20133","key":"1248_CR37","doi-asserted-by":"crossref","first-page":"161","DOI":"10.1023\/A:1017928328829","volume":"49","author":"D. Ormoneit","year":"2002","unstructured":"Ormoneit, D., & Sen, S. (2002). Kernel-based reinforcement learning. Machine Learning, 49(2\u20133), 161\u2013178.","journal-title":"Machine Learning"},{"key":"1248_CR38","first-page":"1","volume-title":"Third IEEE-RAS international conference on humanoid robots","author":"J. Peters","year":"2003","unstructured":"Peters, J., Vijayakumar, S., & Schaal, S. (2003). Reinforcement learning for humanoid robotics. In Third IEEE-RAS international conference on humanoid robots (pp. 1\u201320). Citeseer."},{"key":"1248_CR39","doi-asserted-by":"crossref","first-page":"73","DOI":"10.1109\/CIVTS.2011.5949533","volume-title":"Computational intelligence in vehicles and transportation systems (CIVTS), 2011 IEEE Symposium on","author":"O. Pietquin","year":"2011","unstructured":"Pietquin, O., Tango, F., & Aras, R. (2011). Batch reinforcement learning for optimizing longitudinal driving assistance strategies. In Computational intelligence in vehicles and transportation systems (CIVTS), 2011 IEEE Symposium on (pp. 73\u201379). Los Alamitos: IEEE Comput. Soc."},{"key":"1248_CR40","first-page":"317","volume-title":"Sixteenth European conference on machine learning (ECML)","author":"M. Riedmiller","year":"2005","unstructured":"Riedmiller, M. (2005). Neural fitted Q iteration\u2014first experiences with a data efficient neural reinforcement learning method. In Sixteenth European conference on machine learning (ECML), Porto, Portugal (pp.\u00a0317\u2013328)."},{"issue":"9\u201312","key":"1248_CR41","doi-asserted-by":"crossref","first-page":"1393","DOI":"10.1016\/0270-0255(86)90088-6","volume":"7","author":"J. Robins","year":"1986","unstructured":"Robins, J. (1986). A new approach to causal inference in mortality studies with a sustained exposure period\u2013application to control of the healthy worker survivor effect. Mathematical Modelling, 7(9\u201312), 1393\u20131512.","journal-title":"Mathematical Modelling"},{"key":"1248_CR42","first-page":"1038","volume-title":"Advances in neural information processing systems (NIPS)","author":"R. Sutton","year":"1996","unstructured":"Sutton, R. (1996). Generalization in reinforcement learning: successful examples using sparse coding. In Advances in neural information processing systems (NIPS) (Vol.\u00a08, pp. 1038\u20131044). Denver: MIT Press."},{"key":"1248_CR43","volume-title":"Reinforcement learning","author":"R. Sutton","year":"1998","unstructured":"Sutton, R., & Barto, A. (1998). Reinforcement learning. Cambridge: MIT Press."},{"key":"1248_CR44","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/ADPRL.2007.368162","volume-title":"IEEE symposium on approximate dynamic programming and reinforcement learning (ADPRL)","author":"S. Timmer","year":"2007","unstructured":"Timmer, S., & Riedmiller, M. (2007). Fitted Q iteration with CMACs. In IEEE symposium on approximate dynamic programming and reinforcement learning (ADPRL) (pp. 1\u20138). Los Alamitos: IEEE Comput. Soc."},{"key":"1248_CR45","first-page":"582","volume-title":"Control applications (CCA) & intelligent control (ISIC)","author":"S. Tognetti","year":"2009","unstructured":"Tognetti, S., Savaresi, S., Spelta, C., & Restelli, M. (2009). Batch reinforcement learning for semi-active suspension control. In Control applications (CCA) & intelligent control (ISIC) (pp. 582\u2013587). Los Alamitos: IEEE Comput. Soc."}],"container-title":["Annals of Operations Research"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10479-012-1248-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10479-012-1248-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10479-012-1248-5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,7,5]],"date-time":"2019-07-05T12:31:55Z","timestamp":1562329915000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10479-012-1248-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012,11,15]]},"references-count":45,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2013,9]]}},"alternative-id":["1248"],"URL":"https:\/\/doi.org\/10.1007\/s10479-012-1248-5","relation":{},"ISSN":["0254-5330","1572-9338"],"issn-type":[{"value":"0254-5330","type":"print"},{"value":"1572-9338","type":"electronic"}],"subject":[],"published":{"date-parts":[[2012,11,15]]}}}