{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T12:49:33Z","timestamp":1782478173913,"version":"3.54.5"},"publisher-location":"Cham","reference-count":131,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319994918","type":"print"},{"value":"9783319994925","type":"electronic"}],"license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-3-319-99492-5_13","type":"book-chapter","created":{"date-parts":[[2018,8,22]],"date-time":"2018-08-22T18:14:42Z","timestamp":1534961682000},"page":"298-328","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":47,"title":["From Reinforcement Learning to Deep Reinforcement Learning: An Overview"],"prefix":"10.1007","author":[{"given":"Forest","family":"Agostinelli","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guillaume","family":"Hocquet","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sameer","family":"Singh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pierre","family":"Baldi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2018,8,23]]},"reference":[{"key":"13_CR1","doi-asserted-by":"crossref","unstructured":"Abbeel, P., Ng, A.Y.: Apprenticeship learning via inverse reinforcement learning. In: Proceedings of the Twenty-First International Conference on Machine Learning, p. 1. ACM (2004)","DOI":"10.1145\/1015330.1015430"},{"issue":"12","key":"13_CR2","doi-asserted-by":"publisher","first-page":"i8","DOI":"10.1093\/bioinformatics\/btw243","volume":"32","author":"F Agostinelli","year":"2016","unstructured":"Agostinelli, F., Ceglia, N., Shahbaba, B., Sassone-Corsi, P., Baldi, P.: What time is it? deep learning approaches for circadian rhythms. Bioinformatics 32(12), i8\u2013i17 (2016)","journal-title":"Bioinformatics"},{"issue":"3","key":"13_CR3","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1109\/37.24809","volume":"9","author":"CW Anderson","year":"1989","unstructured":"Anderson, C.W.: Learning to control an inverted pendulum using neural networks. Control Syst. Mag. IEEE 9(3), 31\u201337 (1989)","journal-title":"Control Syst. Mag. IEEE"},{"key":"13_CR4","unstructured":"Andre, D., Russell, S.J.: State abstraction for programmable reinforcement learning agents. In: AAAI\/IAAI, pp. 119\u2013125 (2002)"},{"issue":"3","key":"13_CR5","doi-asserted-by":"publisher","first-page":"402","DOI":"10.1162\/neco.1993.5.3.402","volume":"5","author":"P Baldi","year":"1993","unstructured":"Baldi, P., Chauvin, Y.: Neural networks for fingerprint recognition. Neural Comput. 5(3), 402\u2013418 (1993)","journal-title":"Neural Comput."},{"key":"13_CR6","first-page":"575","volume":"4","author":"P Baldi","year":"2003","unstructured":"Baldi, P., Pollastri, G.: The principled design of large-scale recursive neural network architectures-DAG-RNNs and the protein structure prediction problem. J. Mach. Learn. Res. 4, 575\u2013602 (2003)","journal-title":"J. Mach. Learn. Res."},{"key":"13_CR7","doi-asserted-by":"publisher","first-page":"4308","DOI":"10.1038\/ncomms5308","volume":"5","author":"P Baldi","year":"2014","unstructured":"Baldi, P., Sadowski, P., Whiteson, D.: Searching for exotic particles in high-energy physics with deep learning. Nat. Commun. 5, 4308 (2014)","journal-title":"Nat. Commun."},{"key":"13_CR8","doi-asserted-by":"crossref","unstructured":"Bellemare, M.G., Ostrovski, G., Guez, A., Thomas, P.S., Munos, R.: Increasing the action gap: new operators for reinforcement learning. In: AAAI, pp. 1476\u20131483 (2016)","DOI":"10.1609\/aaai.v30i1.10303"},{"key":"13_CR9","doi-asserted-by":"crossref","unstructured":"Bellman, R.: The theory of dynamic programming. Technical report, DTIC Document (1954)","DOI":"10.2307\/1909830"},{"key":"13_CR10","unstructured":"Blundell, C., et al.: Model-free episodic control. arXiv preprint arXiv:1606.04460 (2016)"},{"key":"13_CR11","unstructured":"Boyan, J., Moore, A.W.: Generalization in reinforcement learning: safely approximating the value function. In: Advances in Neural Information Processing Systems, pp. 369\u2013376 (1995)"},{"key":"13_CR12","unstructured":"Boyan, J.A., Littman, M.L., et al.: Packet routing in dynamically changing networks: a reinforcement learning approach. In: Advances in Neural Information Processing Systems, pp. 671\u2013671 (1994)"},{"key":"13_CR13","first-page":"213","volume":"3","author":"RI Brafman","year":"2003","unstructured":"Brafman, R.I., Tennenholtz, M.: R-max-a general polynomial time algorithm for near-optimal reinforcement learning. J. Mach. Learn. Res. 3, 213\u2013231 (2003)","journal-title":"J. Mach. Learn. Res."},{"issue":"2","key":"13_CR14","doi-asserted-by":"publisher","first-page":"156","DOI":"10.1109\/TSMCC.2007.913919","volume":"38","author":"L. Busoniu","year":"2008","unstructured":"Busoniu, L., Babuska, R., De Schutter, B.: A comprehensive survey of multiagent reinforcement learning. IEEE Trans. Syst. Man Cybern. Part C Appl. Rev. 38(2), 156\u2013172 (2008)","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics, Part C (Applications and Reviews)"},{"key":"13_CR15","volume-title":"Reinforcement Learning and Dynamic Programming Using Function Approximators","author":"L Busoniu","year":"2010","unstructured":"Busoniu, L., Babuska, R., De Schutter, B., Ernst, D.: Reinforcement Learning and Dynamic Programming Using Function Approximators, vol. 39. CRC Press, Boca Raton (2010)"},{"key":"13_CR16","unstructured":"Cassandra, A.R., Kaelbling, L.P., Littman, M.L.: Acting optimally in partially observable stochastic domains. In: AAAI, vol. 94, p. 1023\u20131028 (1994)"},{"key":"13_CR17","unstructured":"Chiappa, S., Racaniere, S., Wierstra, D., Mohamed, S.: Recurrent environment simulators. arXiv preprint arXiv:1704.02254 (2017)"},{"key":"13_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1007\/978-3-540-75538-8_7","volume-title":"Computers and Games","author":"R Coulom","year":"2007","unstructured":"Coulom, R.: Efficient selectivity and backup operators in Monte-Carlo tree search. In: van den Herik, H.J., Ciancarini, P., Donkers, H.H.L.M.J. (eds.) CG 2006. LNCS, vol. 4630, pp. 72\u201383. Springer, Heidelberg (2007). https:\/\/doi.org\/10.1007\/978-3-540-75538-8_7"},{"key":"13_CR19","unstructured":"Crites, R., Barto, A.: Improving elevator performance using reinforcement learning. In: Advances in Neural Information Processing Systems, vol. 8. Citeseer (1996)"},{"key":"13_CR20","first-page":"396","volume-title":"Advances in Neural Information Processing Systems","author":"YL Cun","year":"1990","unstructured":"Cun, Y.L., et al.: Handwritten digit recognition with a back-propagation network. In: Touretzky, D. (ed.) Advances in Neural Information Processing Systems, pp. 396\u2013404. Morgan Kaufmann, San Mateo (1990)"},{"issue":"4","key":"13_CR21","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/BF02551274","volume":"2","author":"G Cybenko","year":"1989","unstructured":"Cybenko, G.: Approximation by superpositions of a sigmoidal function. Math. Control Signals Syst. (MCSS) 2(4), 303\u2013314 (1989)","journal-title":"Math. Control Signals Syst. (MCSS)"},{"key":"13_CR22","unstructured":"Dearden, R., Friedman, N., Andre, D.: Model based Bayesian exploration. In: Proceedings of the Fifteenth Conference on Uncertainty in Artificial Intelligence, pp. 150\u2013159. Morgan Kaufmann Publishers Inc. (1999)"},{"issue":"19","key":"13_CR23","doi-asserted-by":"publisher","first-page":"2449","DOI":"10.1093\/bioinformatics\/bts475","volume":"28","author":"Pietro Di Lena","year":"2012","unstructured":"Di Lena, P., Nagata, K., Baldi, P.: Deep architectures for protein contact map prediction. Bioinformatics 28, 2449\u20132457 (2012). https:\/\/doi.org\/10.1093\/bioinformatics\/bts475. First published online: July 30, 2012","journal-title":"Bioinformatics"},{"key":"13_CR24","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1007\/3-540-44914-0_2","volume-title":"Lecture Notes in Computer Science","author":"Thomas G. Dietterich","year":"2000","unstructured":"Dietterich, T.G.: An overview of MAXQ hierarchical reinforcement learning. In: Choueiry, B.Y., Walsh, T. (eds.) SARA 2000. LNCS (LNAI), vol. 1864, pp. 26\u201344. Springer, Heidelberg (2000). https:\/\/doi.org\/10.1007\/3-540-44914-0_2"},{"issue":"5","key":"13_CR25","doi-asserted-by":"publisher","first-page":"1207","DOI":"10.1109\/TSMCB.2008.925743","volume":"38","author":"D Dong","year":"2008","unstructured":"Dong, D., Chen, C., Li, H., Tarn, T.J.: Quantum reinforcement learning. IEEE Trans. Syst. Man Cybern. Part B Cybern. 38(5), 1207\u20131220 (2008)","journal-title":"IEEE Trans. Syst. Man Cybern. Part B Cybern."},{"key":"13_CR26","doi-asserted-by":"crossref","unstructured":"Dorigo, M., Gambardella, L.: Ant-Q: a reinforcement learning approach to the traveling salesman problem. In: Proceedings of ML-95, Twelfth International Conference on Machine Learning, pp. 252\u2013260 (2014)","DOI":"10.1016\/B978-1-55860-377-6.50039-6"},{"key":"13_CR27","unstructured":"Drake, A.W.: Observation of a Markov process through a noisy channel. Ph.D. thesis, Massachusetts Institute of Technology (1962)"},{"issue":"1\u20132","key":"13_CR28","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1023\/A:1007694015589","volume":"43","author":"S D\u017eeroski","year":"2001","unstructured":"D\u017eeroski, S., De Raedt, L., Driessens, K.: Relational reinforcement learning. Mach. Learn. 43(1\u20132), 7\u201352 (2001)","journal-title":"Mach. Learn."},{"issue":"7639","key":"13_CR29","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1038\/nature21056","volume":"542","author":"A Esteva","year":"2017","unstructured":"Esteva, A., et al.: Dermatologist-level classification of skin cancer with deep neural networks. Nature 542(7639), 115\u2013118 (2017)","journal-title":"Nature"},{"issue":"6","key":"13_CR30","doi-asserted-by":"publisher","first-page":"850","DOI":"10.1287\/opre.51.6.850.24925","volume":"51","author":"DP de Farias","year":"2003","unstructured":"de Farias, D.P., Van Roy, B.: The linear programming approach to approximate dynamic programming. Oper. Res. 51(6), 850\u2013865 (2003)","journal-title":"Oper. Res."},{"key":"13_CR31","unstructured":"Feng, Z., Zilberstein, S.: Region-based incremental pruning for POMDPs. In: Proceedings of the 20th Conference on Uncertainty in Artificial Intelligence, pp. 146\u2013153. AUAI Press (2004)"},{"key":"13_CR32","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1613\/jair.4992","volume":"57","author":"Y Goldberg","year":"2016","unstructured":"Goldberg, Y.: A primer on neural network models for natural language processing. J. Artif. Intell. Res. 57, 345\u2013420 (2016)","journal-title":"J. Artif. Intell. Res."},{"issue":"2","key":"13_CR33","doi-asserted-by":"publisher","first-page":"178","DOI":"10.1287\/ijoc.1080.0305","volume":"21","author":"A Gosavi","year":"2009","unstructured":"Gosavi, A.: Reinforcement learning: a tutorial survey and recent advances. INFORMS J. Comput. 21(2), 178\u2013192 (2009)","journal-title":"INFORMS J. Comput."},{"key":"13_CR34","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, A., Hinton, G.: Speech recognition with deep recurrent neural networks. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6645\u20136649. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"13_CR35","doi-asserted-by":"publisher","first-page":"399","DOI":"10.1613\/jair.1000","volume":"19","author":"C Guestrin","year":"2003","unstructured":"Guestrin, C., Koller, D., Parr, R., Venkataraman, S.: Efficient solution algorithms for factored MDPs. J. Artif. Intell. Res. 19, 399\u2013468 (2003)","journal-title":"J. Artif. Intell. Res."},{"key":"13_CR36","unstructured":"Guestrin, C., Lagoudakis, M., Parr, R.: Coordinated reinforcement learning. In: ICML, vol. 2, pp. 227\u2013234 (2002)"},{"key":"13_CR37","unstructured":"Hasselt, H.V.: Double q-learning. In: Advances in Neural Information Processing Systems, pp. 2613\u20132621 (2010)"},{"key":"13_CR38","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. arXiv preprint arXiv:1512.03385 (2015)","DOI":"10.1109\/CVPR.2016.90"},{"key":"13_CR39","volume-title":"The Organization of Behavior: A Neuropsychological Approach","author":"DO Hebb","year":"1949","unstructured":"Hebb, D.O.: The Organization of Behavior: A Neuropsychological Approach. Wiley, New York (1949)"},{"key":"13_CR40","unstructured":"Heinrich, J., Lanctot, M., Silver, D.: Fictitious self-play in extensive-form games. In: International Conference on Machine Learning (ICML), pp. 805\u2013813 (2015)"},{"key":"13_CR41","unstructured":"Heinrich, J., Silver, D.: Deep reinforcement learning from self-play in imperfect-information games. arXiv preprint arXiv:1603.01121 (2016)"},{"issue":"2","key":"13_CR42","doi-asserted-by":"publisher","first-page":"88","DOI":"10.1137\/0202009","volume":"2","author":"JH Holland","year":"1973","unstructured":"Holland, J.H.: Genetic algorithms and the optimal allocation of trials. SIAM J. Comput. 2(2), 88\u2013105 (1973)","journal-title":"SIAM J. Comput."},{"issue":"5","key":"13_CR43","doi-asserted-by":"publisher","first-page":"359","DOI":"10.1016\/0893-6080(89)90020-8","volume":"2","author":"K Hornik","year":"1989","unstructured":"Hornik, K., Stinchcombe, M., White, H.: Multilayer feedforward networks are universal approximators. Neural Netw. 2(5), 359\u2013366 (1989)","journal-title":"Neural Netw."},{"key":"13_CR44","unstructured":"Howard, R.A.: Dynamic programming and Markov processes (1960)"},{"key":"13_CR45","doi-asserted-by":"crossref","unstructured":"Hutter, M.: Feature reinforcement learning: Part I. Unstructured MDPs. J. Artif. Gen. Intell. 1(1), 3\u201324 (2009)","DOI":"10.2478\/v10229-011-0002-8"},{"key":"13_CR46","doi-asserted-by":"publisher","first-page":"237","DOI":"10.1613\/jair.301","volume":"4","author":"LP Kaelbling","year":"1996","unstructured":"Kaelbling, L.P., Littman, M.L., Moore, A.W.: Reinforcement learning: a survey. J. Artif. Intell. Res. 4, 237\u2013285 (1996)","journal-title":"J. Artif. Intell. Res."},{"key":"13_CR47","volume-title":"Principles of Neural Science","author":"ER Kandel","year":"2000","unstructured":"Kandel, E.R., Schwartz, J.H., Jessell, T.M.: Principles of Neural Science, vol. 4. McGraw-hill, New York (2000)"},{"issue":"9","key":"13_CR48","doi-asserted-by":"publisher","first-page":"2209","DOI":"10.1021\/ci200207y","volume":"51","author":"M Kayala","year":"2011","unstructured":"Kayala, M., Azencott, C., Chen, J., Baldi, P.: Learning to predict chemical reactions. J. Chem. Inf. Model. 51(9), 2209\u20132222 (2011)","journal-title":"J. Chem. Inf. Model."},{"issue":"10","key":"13_CR49","doi-asserted-by":"publisher","first-page":"2526","DOI":"10.1021\/ci3003039","volume":"52","author":"M Kayala","year":"2012","unstructured":"Kayala, M., Baldi, P.: Reactionpredictor: prediction of complex chemical reactions at the mechanistic level using machine learning. J. Chem. Inf. Model. 52(10), 2526\u20132540 (2012)","journal-title":"J. Chem. Inf. Model."},{"issue":"2\u20133","key":"13_CR50","doi-asserted-by":"publisher","first-page":"193","DOI":"10.1023\/A:1017932429737","volume":"49","author":"M Kearns","year":"2002","unstructured":"Kearns, M., Mansour, Y., Ng, A.Y.: A sparse sampling algorithm for near-optimal planning in large Markov decision processes. Mach. Learn. 49(2\u20133), 193\u2013208 (2002)","journal-title":"Mach. Learn."},{"issue":"6","key":"13_CR51","doi-asserted-by":"publisher","first-page":"851","DOI":"10.1007\/BF02743935","volume":"19","author":"SS Keerthi","year":"1994","unstructured":"Keerthi, S.S., Ravindran, B.: A tutorial survey of reinforcement learning. Sadhana 19(6), 851\u2013889 (1994)","journal-title":"Sadhana"},{"key":"13_CR52","doi-asserted-by":"publisher","first-page":"1238","DOI":"10.1177\/0278364913495721","volume":"32","author":"J Kober","year":"2013","unstructured":"Kober, J., Bagnell, J.A., Peters, J.: Reinforcement learning in robotics: a survey. Int. J. Robot. Res. 32, 1238\u20131274 (2013). p. 0278364913495721","journal-title":"Int. J. Robot. Res."},{"key":"13_CR53","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1007\/11871842_29","volume-title":"Machine Learning: ECML 2006","author":"L Kocsis","year":"2006","unstructured":"Kocsis, L., Szepesv\u00e1ri, C.: Bandit based Monte-Carlo planning. In: F\u00fcrnkranz, J., Scheffer, T., Spiliopoulou, M. (eds.) ECML 2006. LNCS (LNAI), vol. 4212, pp. 282\u2013293. Springer, Heidelberg (2006). https:\/\/doi.org\/10.1007\/11871842_29"},{"key":"13_CR54","unstructured":"Konda, V.R., Tsitsiklis, J.N.: Actor-critic algorithms. In: NIPS. 13, 1008\u20131014 (1999)"},{"key":"13_CR55","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems, pp. 1097\u20131105 (2012)"},{"key":"13_CR56","unstructured":"Lai, M.: Giraffe: Using deep reinforcement learning to play chess. arXiv preprint arXiv:1509.01549 (2015)"},{"key":"13_CR57","unstructured":"Leibfried, F., Kushman, N., Hofmann, K.: A deep learning approach for joint video frame and reward prediction in atari games. arXiv preprint arXiv:1611.07078 (2016)"},{"issue":"1","key":"13_CR58","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1109\/89.817450","volume":"8","author":"E Levin","year":"2000","unstructured":"Levin, E., Pieraccini, R., Eckert, W.: A stochastic model of human-machine interaction for learning dialog strategies. IEEE Trans. Speech Audio Process. 8(1), 11\u201323 (2000)","journal-title":"IEEE Trans. Speech Audio Process."},{"issue":"39","key":"13_CR59","first-page":"1","volume":"17","author":"S Levine","year":"2016","unstructured":"Levine, S., Finn, C., Darrell, T., Abbeel, P.: End-to-end training of deep visuomotor policies. J. Mach. Learn. Res. 17(39), 1\u201340 (2016)","journal-title":"J. Mach. Learn. Res."},{"key":"13_CR60","doi-asserted-by":"crossref","unstructured":"Levine, S., Pastor, P., Krizhevsky, A., Quillen, D.: Learning hand-eye coordination for robotic grasping with deep learning and large-scale data collection. In: International Symposium on Experimental Robotics (2016)","DOI":"10.1007\/978-3-319-50115-4_16"},{"key":"13_CR61","unstructured":"Lillicrap, T.P., et al.: Continuous control with deep reinforcement learning (2016)"},{"issue":"1","key":"13_CR62","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1109\/91.273126","volume":"2","author":"CT Lin","year":"1994","unstructured":"Lin, C.T., Lee, C.G.: Reinforcement structure\/parameter learning for neural-network-based fuzzy logic control systems. IEEE Trans. Fuzzy Syst. 2(1), 46\u201363 (1994)","journal-title":"IEEE Trans. Fuzzy Syst."},{"key":"13_CR63","doi-asserted-by":"crossref","unstructured":"Littman, M.L.: Markov games as a framework for multi-agent reinforcement learning. In: Proceedings of the Eleventh International Conference on Machine Learning, vol. 157, pp. 157\u2013163 (1994)","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"13_CR64","unstructured":"Littman, M.L.: Algorithms for sequential decision making. Ph.D. thesis, Brown University (1996)"},{"issue":"7","key":"13_CR65","doi-asserted-by":"publisher","first-page":"1563","DOI":"10.1021\/ci400187y","volume":"53","author":"A Lusci","year":"2013","unstructured":"Lusci, A., Pollastri, G., Baldi, P.: Deep architectures and deep learning in chemoinformatics: the prediction of aqueous solubility for drug-like molecules. J. Chem. Inf. Model. 53(7), 1563\u20131575 (2013)","journal-title":"J. Chem. Inf. Model."},{"key":"13_CR66","unstructured":"McGovern, A., Barto, A.G.: Automatic discovery of subgoals in reinforcement learning using diverse density. Computer Science Department Faculty Publication Series, p. 8 (2001)"},{"key":"13_CR67","unstructured":"Michie, D.: Trial and error. In: Science Survey, Part 2, pp. 129\u2013145 (1961)"},{"issue":"3","key":"13_CR68","doi-asserted-by":"publisher","first-page":"232","DOI":"10.1093\/comjnl\/6.3.232","volume":"6","author":"D Michie","year":"1963","unstructured":"Michie, D.: Experiments on the mechanization of game-learning part I. Characterization of the model and its parameters. Comput. J. 6(3), 232\u2013236 (1963)","journal-title":"Comput. J."},{"issue":"2","key":"13_CR69","first-page":"137","volume":"2","author":"D Michie","year":"1968","unstructured":"Michie, D., Chambers, R.A.: Boxes: an experiment in adaptive control. Mach. Intell. 2(2), 137\u2013152 (1968)","journal-title":"Mach. Intell."},{"issue":"1","key":"13_CR70","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1109\/JRPROC.1961.287775","volume":"49","author":"M Minsky","year":"1961","unstructured":"Minsky, M.: Steps toward artificial intelligence. Proc. IRE 49(1), 8\u201330 (1961)","journal-title":"Proc. IRE"},{"key":"13_CR71","unstructured":"Mnih, V., et al.: Asynchronous methods for deep reinforcement learning. In: International Conference on Machine Learning (ICML) (2016)"},{"issue":"7540","key":"13_CR72","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015)","journal-title":"Nature"},{"key":"13_CR73","unstructured":"Moody, J., Saffell, M.: Reinforcement learning for trading. In: Advances in Neural Information Processing Systems, pp. 917\u2013923 (1999)"},{"key":"13_CR74","doi-asserted-by":"publisher","first-page":"241","DOI":"10.1613\/jair.613","volume":"11","author":"DE Moriarty","year":"1999","unstructured":"Moriarty, D.E., Schultz, A.C., Grefenstette, J.J.: Evolutionary algorithms for reinforcement learning. J. Artif. Intell. Res. (JAIR) 11, 241\u2013276 (1999)","journal-title":"J. Artif. Intell. Res. (JAIR)"},{"key":"13_CR75","doi-asserted-by":"publisher","first-page":"629","DOI":"10.1016\/0743-1066(94)90035-3","volume":"19","author":"S Muggleton","year":"1994","unstructured":"Muggleton, S., De Raedt, L.: Inductive logic programming: theory and methods. J. Logic Program. 19, 629\u2013679 (1994)","journal-title":"J. Logic Program."},{"key":"13_CR76","unstructured":"Nair, A., et\u00a0al.: Massively parallel methods for deep reinforcement learning. arXiv preprint arXiv:1507.04296 (2015)"},{"key":"13_CR77","series-title":"Springer Tracts in Advanced Robotics","doi-asserted-by":"publisher","first-page":"363","DOI":"10.1007\/11552246_35","volume-title":"Experimental Robotics IX","author":"AY Ng","year":"2006","unstructured":"Ng, A.Y., et al.: Autonomous inverted helicopter flight via reinforcement learning. In: Ang, M.H., Khatib, O. (eds.) Experimental Robotics IX. STAR, vol. 21, pp. 363\u2013372. Springer, Heidelberg (2006). https:\/\/doi.org\/10.1007\/11552246_35"},{"key":"13_CR78","unstructured":"Ng, A.Y., Russell, S.J., et al.: Algorithms for inverse reinforcement learning. In: ICML, pp. 663\u2013670 (2000)"},{"key":"13_CR79","unstructured":"Oh, J., Guo, X., Lee, H., Lewis, R.L., Singh, S.: Action-conditional video prediction using deep networks in atari games. In: Advances in Neural Information Processing Systems, pp. 2863\u20132871 (2015)"},{"key":"13_CR80","unstructured":"Oh, J., Singh, S., Lee, H.: Value prediction network. In: Advances in Neural Information Processing Systems, pp. 6120\u20136130 (2017)"},{"issue":"2\u20133","key":"13_CR81","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1023\/A:1017928328829","volume":"49","author":"D Ormoneit","year":"2002","unstructured":"Ormoneit, D., Sen, \u015a.: Kernel-based reinforcement learning. Mach. Learn. 49(2\u20133), 161\u2013178 (2002)","journal-title":"Mach. Learn."},{"issue":"3","key":"13_CR82","doi-asserted-by":"publisher","first-page":"441","DOI":"10.1287\/moor.12.3.441","volume":"12","author":"CH Papadimitriou","year":"1987","unstructured":"Papadimitriou, C.H., Tsitsiklis, J.N.: The complexity of Markov decision processes. Math. Oper. Res. 12(3), 441\u2013450 (1987)","journal-title":"Math. Oper. Res."},{"key":"13_CR83","unstructured":"Parr, R., Russell, S.: Reinforcement learning with hierarchies of machines. In: Advances in Neural Information Processing Systems, pp. 1043\u20131049 (1998)"},{"key":"13_CR84","unstructured":"Pascanu, R., et al.: Learning model-based planning from scratch. arXiv preprint arXiv:1707.06170 (2017)"},{"key":"13_CR85","unstructured":"Pashenkova, E., Rish, I., Dechter, R.: Value iteration and policy iteration algorithms for Markov decision problem. In: AAAI 1996, Workshop on Structural Issues in Planning and Temporal Reasoning. Citeseer (1996)"},{"key":"13_CR86","unstructured":"Poupart, P., Boutilier, C.: VDCBPI: an approximate scalable algorithm for large POMDPs. In: Advances in Neural Information Processing Systems, pp. 1081\u20131088 (2004)"},{"key":"13_CR87","unstructured":"Powers, R., Shoham, Y.: New criteria and a new algorithm for learning in multi-agent systems. In: Advances in Neural Information Processing Systems, pp. 1089\u20131096 (2004)"},{"key":"13_CR88","unstructured":"Randl\u00f8v, J., Alstr\u00f8m, P.: Learning to drive a bicycle using reinforcement learning and shaping. In: ICML, vol. 98, pp. 463\u2013471. Citeseer (1998)"},{"key":"13_CR89","unstructured":"Ross, S.M.: Introduction to Stochastic Dynamic Programming. Academic press, Norwell (2014))"},{"key":"13_CR90","unstructured":"Rummery, G.A., Niranjan, M.: On-line Q-learning using connectionist systems. University of Cambridge, Department of Engineering (1994)"},{"key":"13_CR91","unstructured":"Rusu, A.A., et\u00a0al.: Policy distillation. In: International Conference on Learning Representations (ICLR) (2016)"},{"key":"13_CR92","unstructured":"Sadowski, P., Collado, J., Whiteson, D., Baldi, P.: Deep learning, dark knowledge, and dark matter. In: Journal of Machine Learning Research, Workshop and Conference Proceedings, vol. 42, pp. 81\u201397 (2015)"},{"issue":"6","key":"13_CR93","doi-asserted-by":"publisher","first-page":"601","DOI":"10.1147\/rd.116.0601","volume":"11","author":"A. L. Samuel","year":"1967","unstructured":"Samuel, A.L.: Some studies in machine learning using the game of checkers. II. Recent progress. IBM J. Res. Dev. 11(6), 601\u2013617 (1967)","journal-title":"IBM Journal of Research and Development"},{"issue":"2","key":"13_CR94","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1177\/105971239700600201","volume":"6","author":"Juan C. Santamaria","year":"1997","unstructured":"Santamar\u00eda, J.C., Sutton, R.S., Ram, A.: Experiments with reinforcement learning in problems with continuous state and action spaces. Adapt. Behav. 6(2), 163\u2013217 (1997)","journal-title":"Adaptive Behavior"},{"key":"13_CR95","unstructured":"Schaul, T., Horgan, D., Gregor, K., Silver, D.: Universal value function approximators. In: International Conference on Machine Learning (ICML), pp. 1312\u20131320 (2015)"},{"key":"13_CR96","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1016\/j.neunet.2014.09.003","volume":"61","author":"J\u00fcrgen Schmidhuber","year":"2015","unstructured":"Schmidhuber, J.: Deep learning in neural networks: an overview. Neural Netw. 61, 85\u2013117 (2015)","journal-title":"Neural Networks"},{"key":"13_CR97","unstructured":"Schulman, J., Moritz, P., Levine, S., Jordan, M., Abbeel, P.: High-dimensional continuous control using generalized advantage estimation. In: Proceedings of the International Conference on Learning Representations (ICLR) (2016)"},{"key":"13_CR98","unstructured":"Sherstov, A.A., Stone, P.: On continuous-action Q-learning via tile coding function approximation. Under Review (2004)"},{"key":"13_CR99","unstructured":"Silver, D., et\u00a0al.: The predictron: end-to-end learning and planning. arXiv preprint arXiv:1612.08810 (2016)"},{"issue":"7587","key":"13_CR100","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D., et al.: Mastering the game of go with deep neural networks and tree search. Nature 529(7587), 484\u2013489 (2016)","journal-title":"Nature"},{"key":"13_CR101","unstructured":"Silver, D., et\u00a0al.: Mastering chess and shogi by self-play with a general reinforcement learning algorithm. arXiv preprint arXiv:1712.01815 (2017)"},{"key":"13_CR102","unstructured":"Silver, D., Lever, G., Heess, N., Degris, T., Wierstra, D., Riedmiller, M.: Deterministic policy gradient algorithms. In: International Conference on Machine Learning (ICML) (2014)"},{"issue":"7676","key":"13_CR103","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver, D., et al.: Mastering the game of go without human knowledge. Nature 550(7676), 354 (2017)","journal-title":"Nature"},{"key":"13_CR104","unstructured":"Singh, S., Bertsekas, D.: Reinforcement learning for dynamic channel allocation in cellular telephone systems. In: Advances in Neural Information Processing Systems, pp. 974\u2013980 (1997)"},{"key":"13_CR105","doi-asserted-by":"crossref","unstructured":"Singh, S.P., Jaakkola, T.S., Jordan, M.I.: Learning without state-estimation in partially observable Markovian decision processes. In: ICML, pp. 284\u2013292 (1994)","DOI":"10.1016\/B978-1-55860-335-6.50042-8"},{"issue":"1\u20133","key":"13_CR106","first-page":"123","volume":"22","author":"SP Singh","year":"1996","unstructured":"Singh, S.P., Sutton, R.S.: Reinforcement learning with replacing eligibility traces. Mach. Learn. 22(1\u20133), 123\u2013158 (1996)","journal-title":"Mach. Learn."},{"key":"13_CR107","doi-asserted-by":"crossref","unstructured":"Socher, R., et\u00a0al.: Recursive deep models for semantic compositionality over a sentiment treebank. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing (EMNLP), vol. 1631, p. 1642. Citeseer (2013)","DOI":"10.18653\/v1\/D13-1170"},{"key":"13_CR108","doi-asserted-by":"crossref","unstructured":"Spaan, M.T., Spaan, M.T.: A point-based POMDP algorithm for robot planning. In: 2004 IEEE International Conference on Robotics and Automation, Proceedings, ICRA 2004, vol. 3, pp. 2399\u20132404. IEEE (2004)","DOI":"10.1109\/ROBOT.2004.1307420"},{"key":"13_CR109","unstructured":"Srivastava, R.K., Greff, K., Schmidhuber, J.: Training very deep networks. In: Advances in Neural Information Processing Systems, pp. 2368\u20132376 (2015)"},{"key":"13_CR110","unstructured":"Sutskever, I., Vinyals, O., Le, Q.V.: Sequence to sequence learning with neural networks. In: Advances in Neural Information Processing Systems, pp. 3104\u20133112 (2014)"},{"issue":"1","key":"13_CR111","first-page":"9","volume":"3","author":"RS Sutton","year":"1988","unstructured":"Sutton, R.S.: Learning to predict by the methods of temporal differences. Mach. Learn. 3(1), 9\u201344 (1988)","journal-title":"Mach. Learn."},{"key":"13_CR112","doi-asserted-by":"crossref","unstructured":"Sutton, R.S.: Integrated architectures for learning, planning, and reacting based on approximating dynamic programming. In: Machine Learning Proceedings 1990, pp. 216\u2013224. Elsevier (1990)","DOI":"10.1016\/B978-1-55860-141-3.50030-4"},{"key":"13_CR113","doi-asserted-by":"crossref","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (1998)","DOI":"10.1109\/TNN.1998.712192"},{"issue":"1-2","key":"13_CR114","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"Richard S. Sutton","year":"1999","unstructured":"Sutton, R.S., Precup, D., Singh, S.: Between MDPs and semi-MDPs: a framework for temporal abstraction in reinforcement learning. Artif. Intell. 112(1), 181\u2013211 (1999)","journal-title":"Artificial Intelligence"},{"key":"13_CR115","doi-asserted-by":"crossref","unstructured":"Szegedy, C., et\u00a0al.: Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1\u20139 (2015)","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"13_CR116","doi-asserted-by":"crossref","unstructured":"Taylor, M.E., Stone, P.: Cross-domain transfer for reinforcement learning. In: Proceedings of the 24th International Conference on Machine Learning, pp. 879\u2013886. ACM (2007)","DOI":"10.1145\/1273496.1273607"},{"issue":"3","key":"13_CR117","doi-asserted-by":"publisher","first-page":"58","DOI":"10.1145\/203330.203343","volume":"38","author":"G Tesauro","year":"1995","unstructured":"Tesauro, G.: Temporal difference learning and TD-Gammon. Commun. ACM 38(3), 58\u201368 (1995)","journal-title":"Commun. ACM"},{"key":"13_CR118","unstructured":"Thorndike, E.L.: Animal Intelligence: Experimental Studies. Transaction Publishers, New York (1965)"},{"issue":"5","key":"13_CR119","doi-asserted-by":"publisher","first-page":"674","DOI":"10.1109\/9.580874","volume":"42","author":"J.N. Tsitsiklis","year":"1997","unstructured":"Tsitsiklis, J.N., Van Roy, B.: An analysis of temporal-difference learning with function approximation. IEEE Trans. Autom. Control 42(5), 674\u2013690 (1997)","journal-title":"IEEE Transactions on Automatic Control"},{"key":"13_CR120","first-page":"66","volume":"10","author":"L Van Der Maaten","year":"2009","unstructured":"Van Der Maaten, L., Postma, E., Van den Herik, J.: Dimensionality reduction: a comparative. J. Mach. Learn. Res. 10, 66\u201371 (2009)","journal-title":"J. Mach. Learn. Res."},{"key":"13_CR121","doi-asserted-by":"crossref","unstructured":"Van Hasselt, H., Guez, A., Silver, D.: Deep reinforcement learning with double q-learning. In: AAAI, pp. 2094\u20132100 (2016)","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"13_CR122","unstructured":"Wang, X., Sandholm, T.: Reinforcement learning to play an optimal Nash equilibrium in team Markov games. In: Advances in Neural Information Processing Systems, pp. 1571\u20131578 (2002)"},{"issue":"3-4","key":"13_CR123","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1007\/BF00992698","volume":"8","author":"Christopher J. C. H. Watkins","year":"1992","unstructured":"Watkins, C.J., Dayan, P.: Q-learning. Mach. Learn. 8(3\u20134), 279\u2013292 (1992)","journal-title":"Machine Learning"},{"key":"13_CR124","unstructured":"Watter, M., Springenberg, J., Boedecker, J., Riedmiller, M.: Embed to control: a locally linear latent dynamics model for control from raw images. In: Advances in Neural Information Processing Systems, pp. 2746\u20132754 (2015)"},{"key":"13_CR125","unstructured":"Weber, T., et\u00a0al.: Imagination-augmented agents for deep reinforcement learning. arXiv preprint arXiv:1707.06203 (2017)"},{"issue":"3\u20134","key":"13_CR126","first-page":"229","volume":"8","author":"RJ Williams","year":"1992","unstructured":"Williams, R.J.: Simple statistical gradient-following algorithms for connectionist reinforcement learning. Mach. Learn. 8(3\u20134), 229\u2013256 (1992)","journal-title":"Mach. Learn."},{"key":"13_CR127","volume-title":"Advances in Neural Information Processing Systems 19","author":"L Wu","year":"2007","unstructured":"Wu, L., Baldi, P.: A scalable machine learning approach to go. In: Weiss, Y., Scholkopf, B., Editors, J.P. (eds.) NIPS 2006. MIT Press, Cambridge (2007)"},{"issue":"9","key":"13_CR128","doi-asserted-by":"publisher","first-page":"1392","DOI":"10.1016\/j.neunet.2008.02.002","volume":"21","author":"L Wu","year":"2008","unstructured":"Wu, L., Baldi, P.: Learning to play go using recursive neural networks. Neural Netw. 21(9), 1392\u20131400 (2008)","journal-title":"Neural Netw."},{"key":"13_CR129","unstructured":"Zhang, W., Dietterich, T.G.: High-performance job-shop scheduling with a time-delay td network. In: Advances in Neural Information Processing Systems, vol. 8, pp. 1024\u20131030 (1996)"},{"key":"13_CR130","unstructured":"Zhang, W.: Algorithms for partially observable Markov decision processes. Ph.D. thesis, Citeseer (2001)"},{"issue":"10","key":"13_CR131","doi-asserted-by":"publisher","first-page":"931","DOI":"10.1038\/nmeth.3547","volume":"12","author":"J Zhou","year":"2015","unstructured":"Zhou, J., Troyanskaya, O.G.: Predicting effects of noncoding variants with deep learning-based sequence model. Nat. Methods 12(10), 931\u2013934 (2015)","journal-title":"Nat. Methods"}],"container-title":["Lecture Notes in Computer Science","Braverman Readings in Machine Learning. Key Ideas from Inception to Current State"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-99492-5_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,6]],"date-time":"2025-07-06T13:38:54Z","timestamp":1751809134000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-99492-5_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018]]},"ISBN":["9783319994918","9783319994925"],"references-count":131,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-99492-5_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018]]},"assertion":[{"value":"23 August 2018","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}