{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T18:02:03Z","timestamp":1784656923164,"version":"3.55.0"},"publisher-location":"Cham","reference-count":45,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319569901","type":"print"},{"value":"9783319569918","type":"electronic"}],"license":[{"start":{"date-parts":[[2017,8,23]],"date-time":"2017-08-23T00:00:00Z","timestamp":1503446400000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-3-319-56991-8_32","type":"book-chapter","created":{"date-parts":[[2017,8,22]],"date-time":"2017-08-22T03:57:16Z","timestamp":1503374236000},"page":"426-440","source":"Crossref","is-referenced-by-count":241,"title":["Deep Reinforcement Learning: An Overview"],"prefix":"10.1007","author":[{"given":"Seyed Sajad","family":"Mousavi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael","family":"Schukat","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Enda","family":"Howley","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2017,8,23]]},"reference":[{"key":"32_CR1","doi-asserted-by":"crossref","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y Lecun","year":"1998","unstructured":"Lecun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proc. IEEE 86, 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"key":"32_CR2","doi-asserted-by":"crossref","first-page":"1238","DOI":"10.1177\/0278364913495721","volume":"32","author":"J Kober","year":"2013","unstructured":"Kober, J., Bagnell, J.A., Peters, J.: Reinforcement learning in robotics: a survey. Int. J. Robot. Res. 32, 1238\u20131274 (2013)","journal-title":"Int. J. Robot. Res."},{"key":"32_CR3","unstructured":"Vengerov, D.: A reinforcement learning approach to dynamic resource allocation. Sun Microsystems, Inc. (2005)"},{"key":"32_CR4","doi-asserted-by":"crossref","first-page":"341","DOI":"10.1023\/A:1025696116075","volume":"13","author":"AG Barto","year":"2003","unstructured":"Barto, A.G., Mahadevan, S.: Recent advances in hierarchical reinforcement learning. Discrete Event Dyn. Syst. 13, 341\u2013379 (2003)","journal-title":"Discrete Event Dyn. Syst."},{"key":"32_CR5","doi-asserted-by":"crossref","first-page":"118","DOI":"10.1016\/j.asoc.2014.08.071","volume":"25","author":"SS Mousavi","year":"2014","unstructured":"Mousavi, S.S., Ghazanfari, B., Mozayani, N., Jahed-Motlagh, M.R.: Automatic abstraction controller in reinforcement learning agent via automata. Appl. Soft Comput. 25, 118\u2013128 (2014)","journal-title":"Appl. Soft Comput."},{"key":"32_CR6","unstructured":"Sutton, R.S., David, A.M., Satinder, P.S., Mansour, Y.: Policy Gradient Methods for Reinforcement Learning with Function Approximation, pp. 1057\u20131063 (2000)"},{"key":"32_CR7","doi-asserted-by":"crossref","unstructured":"Mattner, J., Lange, S., Riedmiller, M.: Learn to swing up and balance a real pole based on raw visual input data. In: Huang, T., Zeng, Z., Li, C., Leung, C.S., (eds.) Neural Information Processing: 19th International Conference, ICONIP 2012, Doha, Qatar, 12\u201315 November 2012, Proceedings, Part V, pp. 126\u2013133. Springer, Heidelberg (2012)","DOI":"10.1007\/978-3-642-34500-5_16"},{"key":"32_CR8","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Graves, A., Antonoglou, I., Wierstra, D., et al.: Playing atari with deep reinforcement learning. In: NIPS Deep Learning Workshop (2013)"},{"key":"32_CR9","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Rusu, A.A., Veness, J., Bellemare, M.G., et al.: Human-level control through deep reinforcement learning. Nature 518, 529\u2013533 (2015)","journal-title":"Nature"},{"key":"32_CR10","doi-asserted-by":"crossref","first-page":"353","DOI":"10.1007\/s13218-015-0356-1","volume":"29","author":"W B\u00f6hmer","year":"2015","unstructured":"B\u00f6hmer, W., Springenberg, J.T., Boedecker, J., Riedmiller, M., Obermayer, K.: Autonomous learning of state representations for control: an emerging field aims to autonomously learn state representations for reinforcement learning agents from their real-world sensor observations. KI - K\u00fcnstliche Intelligenz 29, 353\u2013362 (2015)","journal-title":"KI - K\u00fcnstliche Intelligenz"},{"key":"32_CR11","unstructured":"Levine, S., Fin, C., Darre, T., Abbee, P.: End-to-End training of deep visuomotor policies. arXiv:1504.00702 (2015)"},{"key":"32_CR12","doi-asserted-by":"crossref","first-page":"237","DOI":"10.1613\/jair.301","volume":"4","author":"LP Kaelbling","year":"1996","unstructured":"Kaelbling, L.P., Littman, M.L., Moore, A.W.: Reinforcement learning: a survey. J. Artif. Intell. Res. (JAIR) 4, 237\u2013285 (1996)","journal-title":"J. Artif. Intell. Res. (JAIR)"},{"key":"32_CR13","volume-title":"Introduction to Reinforcement Learning","author":"RS Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Introduction to Reinforcement Learning. MIT Press, Cambridge (1998)"},{"key":"32_CR14","doi-asserted-by":"crossref","unstructured":"Riedmiller, M.: Neural fitted Q iteration \u2013 first experiences with a data efficient neural reinforcement learning method. In: Gama, J., Camacho, R., Brazdil, P.B., Jorge, A.M., Torgo, L., (eds.) Machine Learning: ECML 2005: 16th European Conference on Machine Learning, Porto, Portugal, 3\u20137 October 2005, Proceedings, pp. 317\u2013328. Springer, Heidelberg (2005)","DOI":"10.1007\/11564096_32"},{"key":"32_CR15","unstructured":"Oh, J., Guo, X., Lee, H., Lewis, R.L., Singh, S.: Action-conditional video prediction using deep networks in Atari games, pp. 2845\u20132853 (2015)"},{"key":"32_CR16","doi-asserted-by":"crossref","first-page":"1798","DOI":"10.1109\/TPAMI.2013.50","volume":"35","author":"Y Bengio","year":"2013","unstructured":"Bengio, Y.: Representation learning: a review and new perspectives. IEEE Trans. Pattern Anal. Mach. Intell. 35, 1798\u20131828 (2013)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"32_CR17","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1561\/2200000006","volume":"2","author":"Y Bengio","year":"2009","unstructured":"Bengio, Y.: Learning deep architectures for AI. Found. Trends Mach. Learn. 2, 1\u2013127 (2009)","journal-title":"Found. Trends Mach. Learn."},{"key":"32_CR18","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Pascal, L., Dan, P., Larochelle, H.: Greedy Layer-Wise Training of Deep Networks, pp. 153\u2013160 (2007)","DOI":"10.7551\/mitpress\/7503.003.0024"},{"key":"32_CR19","doi-asserted-by":"crossref","unstructured":"Vincent, P., Larochelle, H., Bengio, Y., Manzagol, P.-A.: Extracting and composing robust features with denoising autoencoders. Presented at the Proceedings of the 25th International Conference on Machine learning, Helsinki, Finland (2008)","DOI":"10.1145\/1390156.1390294"},{"key":"32_CR20","first-page":"3371","volume":"11","author":"P Vincent","year":"2010","unstructured":"Vincent, P., Larochelle, H., Lajoie, I., Bengio, Y., Manzagol, P.-A.: Stacked denoising autoencoders: learning useful representations in a deep network with a local denoising criterion. J. Mach. Learn. Res. 11, 3371\u20133408 (2010)","journal-title":"J. Mach. Learn. Res."},{"key":"32_CR21","doi-asserted-by":"crossref","first-page":"541","DOI":"10.1162\/neco.1989.1.4.541","volume":"1","author":"Y LeCun","year":"1989","unstructured":"LeCun, Y., Boser, B., Denker, J.S., Henderson, D., Howard, R.E., Hubbard, W., et al.: Backpropagation applied to handwritten zip code recognition. Neural Comput. 1, 541\u2013551 (1989)","journal-title":"Neural Comput."},{"key":"32_CR22","unstructured":"LeCun, Y., Bengio, Y.: Convolutional networks for images, speech, and time series. In: Michael, A.A., (ed.) The Handbook of Brain Theory and Neural Networks, pp. 255\u2013258. MIT Press (1998)"},{"key":"32_CR23","doi-asserted-by":"crossref","unstructured":"Scherer, D., M\u00fcller, A., Behnke, S.: Evaluation of pooling operations in convolutional architectures for object recognition. In: Diamantaras, K., Duch, W., Iliadis, L.S., (eds.) Artificial Neural Networks \u2013 ICANN 2010: 20th International Conference, Thessaloniki, Greece, 15\u201318 September 2010, Proceedings, Part III, pp. 92\u2013101. Springer, Heidelberg (2010)","DOI":"10.1007\/978-3-642-15825-4_10"},{"key":"32_CR24","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Kai, L., Li, F.-F.: ImageNet: a large-scale hierarchical image database. In: IEEE Conference on Computer Vision and Pattern Recognition, 2009, CVPR 2009, pp. 248\u2013255 (2009)"},{"key":"32_CR25","doi-asserted-by":"crossref","unstructured":"Deng, L., Hinton, G., Kingsbury, B.: New types of deep neural network learning for speech recognition and related applications: an overview. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8599\u20138603 (2013)","DOI":"10.1109\/ICASSP.2013.6639344"},{"key":"32_CR26","doi-asserted-by":"crossref","first-page":"157","DOI":"10.1109\/72.279181","volume":"5","author":"Y Bengio","year":"1994","unstructured":"Bengio, Y., Simard, P., Frasconi, P.: Learning long-term dependencies with gradient descent is difficult. IEEE Trans. Neural Netw. 5, 157\u2013166 (1994)","journal-title":"IEEE Trans. Neural Netw."},{"key":"32_CR27","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9, 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"32_CR28","doi-asserted-by":"crossref","first-page":"215","DOI":"10.1162\/neco.1994.6.2.215","volume":"6","author":"G Tesauro","year":"1994","unstructured":"Tesauro, G.: TD-Gammon, a self-teaching backgammon program, achieves master-level play. Neural Comput. 6, 215\u2013219 (1994)","journal-title":"Neural Comput."},{"key":"32_CR29","doi-asserted-by":"crossref","unstructured":"Riedmiller, M., Braun, H.: A direct adaptive method for faster backpropagation learning: the RPROP algorithm. In: IEEE International Conference on Neural Networks, 1993, vol. 1, pp. 586\u2013591 (1993)","DOI":"10.1109\/ICNN.1993.298623"},{"key":"32_CR30","first-page":"253","volume":"47","author":"MG Bellemare","year":"2013","unstructured":"Bellemare, M.G., Naddaf, Y., Veness, J., Bowling, M.: The arcade learning environment: an evaluation platform for general agents. J. Artif. Int. Res. 47, 253\u2013279 (2013)","journal-title":"J. Artif. Int. Res."},{"key":"32_CR31","unstructured":"Lin, L.-J.: Reinforcement learning for robots using neural networks. Carnegie Mellon University (1993)"},{"key":"32_CR32","unstructured":"Guo, X., Singh, S., Lee, H., Lewis, R.L., Wang, X.: Deep learning for real-time atari game play using offline monte-carlo tree search planning, pp. 3338\u20133346 (2014)"},{"key":"32_CR33","doi-asserted-by":"crossref","unstructured":"Kocsis, L., Szepesv\u00e1ri, C.: Bandit based monte-carlo planning. Presented at the Proceedings of the 17th European Conference on Machine Learning, Berlin, Germany (2006)","DOI":"10.1007\/11871842_29"},{"key":"32_CR34","doi-asserted-by":"crossref","unstructured":"Gr\u00fcttner, M., Sehnke, F., Schaul, T., Schmidhuber, J.: Multi-dimensional deep memory Atari-Go players for parameter exploring policy gradients. In: Diamantaras, K., Duch, W., Iliadis, L.S., (eds.) Artificial Neural Networks \u2013 ICANN 2010: 20th International Conference, Thessaloniki, Greece, 15\u201318 September 2010, Proceedings, Part II, pp. 114\u2013123. Springer, Heidelberg (2010)","DOI":"10.1007\/978-3-642-15822-3_14"},{"key":"32_CR35","unstructured":"Hochreiter, S., Bengio, Y., Frasconi, P., Schmidhuber, J.: Gradient flow in recurrent nets: the difficulty of learning long-term dependencies. In: Kremer, S.C., Kolen, J.F., (eds.) A Field Guide to Dynamical Recurrent Neural Networks (2001)"},{"issue":"4","key":"32_CR36","doi-asserted-by":"crossref","first-page":"551","DOI":"10.1016\/j.neunet.2009.12.004","volume":"23","author":"F Sehnke","year":"2010","unstructured":"Sehnke, F., Osendorfer, C., R\u00fcckstie\u00df, T., Graves, A., Peters, J., Schmidhuber, J.: Parameter-exploring policy gradients. Neural Netw. 23(4), 551\u2013559 (2010)","journal-title":"Neural Netw."},{"key":"32_CR37","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1023\/A:1015059928466","volume":"1","author":"H-G Beyer","year":"2002","unstructured":"Beyer, H.-G., Schwefel, H.-P.: Evolution strategies \u2013 a comprehensive introduction. Nat. Comput. 1, 3\u201352 (2002)","journal-title":"Nat. Comput."},{"key":"32_CR38","unstructured":"Clark, C., Storkey, A.: Teaching deep convolutional neural networks to play Go, arXiv preprint arXiv:1412.3409 (2014)"},{"key":"32_CR39","doi-asserted-by":"crossref","unstructured":"Koutn\u00ed, J., Cuccu, G., Schmidhuber, J., Gomez, F.: Evolving large-scale neural networks for vision-based reinforcement learning. In: Proceedings of the Genetic and Evolutionary Computation Conference, Amsterdam, pp. 1061\u20131068 (2013)","DOI":"10.1145\/2463372.2463509"},{"key":"32_CR40","doi-asserted-by":"crossref","unstructured":"Lange, S., Riedmiller, M.: Deep auto-encoder neural networks in reinforcement learning. In: The 2010 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20138 (2010)","DOI":"10.1109\/IJCNN.2010.5596468"},{"key":"32_CR41","doi-asserted-by":"crossref","unstructured":"Lange, S., Riedmiller, M., Voigtlaender, A.: Autonomous reinforcement learning on raw visual input data in a real world application. In: International Joint Conference on Neural Networks, pp. 1\u20138 (2012)","DOI":"10.1109\/IJCNN.2012.6252823"},{"key":"32_CR42","doi-asserted-by":"crossref","first-page":"161","DOI":"10.1023\/A:1017928328829","volume":"49","author":"D Ormoneit","year":"2002","unstructured":"Ormoneit, D., Sen, \u015a.: Kernel-based reinforcement learning. Mach. Learn. 49, 161\u2013178 (2002)","journal-title":"Mach. Learn."},{"key":"32_CR43","doi-asserted-by":"crossref","first-page":"85","DOI":"10.1016\/j.neunet.2014.09.003","volume":"61","author":"J Schmidhuber","year":"2015","unstructured":"Schmidhuber, J.: Deep learning in neural networks: an overview. Neural Netw. 61, 85\u2013117 (2015)","journal-title":"Neural Netw."},{"key":"32_CR44","unstructured":"Bakker, B., Zhumatiy, V., Gruener, G., Schmidhuber, J.: A robot that reinforcement-learns to identify and memorize important previous observations. In: 2003 IEEE\/RSJ International Conference on Intelligent Robots and Systems, 2003, (IROS 2003), Proceedings, vol. 1, pp. 430\u2013435 (2003)"},{"key":"32_CR45","unstructured":"Hausknecht, M., Stone, P.: Deep recurrent Q-learning for partially observable MDPs, arXiv preprint arXiv:1507.06527v3 (2015)"}],"container-title":["Lecture Notes in Networks and Systems","Proceedings of SAI Intelligent Systems Conference (IntelliSys) 2016"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-56991-8_32","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,25]],"date-time":"2023-08-25T05:17:55Z","timestamp":1692940675000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-56991-8_32"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,8,23]]},"ISBN":["9783319569901","9783319569918"],"references-count":45,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-56991-8_32","relation":{},"ISSN":["2367-3370","2367-3389"],"issn-type":[{"value":"2367-3370","type":"print"},{"value":"2367-3389","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017,8,23]]}}}