{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,29]],"date-time":"2025-11-29T08:01:41Z","timestamp":1764403301962,"version":"3.40.3"},"publisher-location":"Cham","reference-count":35,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031496615"},{"type":"electronic","value":"9783031496622"}],"license":[{"start":{"date-parts":[[2023,12,15]],"date-time":"2023-12-15T00:00:00Z","timestamp":1702598400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,12,15]],"date-time":"2023-12-15T00:00:00Z","timestamp":1702598400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-49662-2_6","type":"book-chapter","created":{"date-parts":[[2023,12,14]],"date-time":"2023-12-14T13:03:00Z","timestamp":1702558980000},"page":"96-120","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Reinforcement Learning Algorithms: Categorization and Structural Properties"],"prefix":"10.1007","author":[{"given":"Kenneth","family":"Schr\u00f6der","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexander","family":"Kastius","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rainer","family":"Schlosser","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,12,15]]},"reference":[{"key":"6_CR1","unstructured":"Achiam, J.: Spinning up in deep reinforcement learning (2018). https:\/\/spinningup.openai.com\/"},{"issue":"1","key":"6_CR2","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1162\/089976600300015961","volume":"12","author":"K Doya","year":"2000","unstructured":"Doya, K.: Reinforcement learning in continuous time and space. Neural Comput. 12(1), 219\u2013245 (2000)","journal-title":"Neural Comput."},{"key":"6_CR3","unstructured":"Eysenbach, B., Levine, S.: Maximum entropy RL (provably) solves some robust RL problems. arXiv preprint arXiv:2103.06257 (2021)"},{"key":"6_CR4","unstructured":"Fakoor, R., Chaudhari, P., Smola, A.J.: P3O: policy-on policy-off policy optimization. In: Uncertainty in Artificial Intelligence, pp. 1017\u20131027. PMLR (2020)"},{"key":"6_CR5","unstructured":"Fujimoto, S., et al.: Addressing function approximation error in actor-critic methods. In: ICML, pp. 1587\u20131596. PMLR (2018)"},{"key":"6_CR6","unstructured":"G\u00e9ron, A.: Hands-on Machine Learning with Scikit-Learn, Keras, and TensorFlow: Concepts, Tools, and Techniques to Build Intelligent Systems. O\u2019Reilly Media, Inc., Sebastopol (2019)"},{"key":"6_CR7","unstructured":"Haarnoja, T., et al.: Soft actor-critic algorithms and applications. arXiv preprint arXiv:1812.05905 (2018)"},{"key":"6_CR8","unstructured":"Haarnoja, T., et al.: Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: ICML, pp. 1861\u20131870. PMLR (2018)"},{"key":"6_CR9","unstructured":"Hasselt, H.: Double q-learning. Adv. Neural Inf. Process. Syst. 23 (2010)"},{"key":"6_CR10","doi-asserted-by":"publisher","unstructured":"Henderson, P., et al.: Deep reinforcement learning that matters. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32, no. 1 (2018). https:\/\/doi.org\/10.1609\/aaai.v32i1.11694, https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/11694","DOI":"10.1609\/aaai.v32i1.11694"},{"key":"6_CR11","unstructured":"Hussenot, L., et al.: Hyperparameter selection for imitation learning. In: ICML, pp. 4511\u20134522. PMLR (2021)"},{"key":"6_CR12","unstructured":"Kirk, D.E.: Optimal Control Theory: An Introduction. Courier Corporation, Chelmsford (2004)"},{"issue":"13","key":"6_CR13","doi-asserted-by":"publisher","first-page":"3521","DOI":"10.1073\/pnas.1611835114","volume":"114","author":"J Kirkpatrick","year":"2017","unstructured":"Kirkpatrick, J., et al.: Overcoming catastrophic forgetting in neural networks. Proc. Natl. Acad. Sci. 114(13), 3521\u20133526 (2017)","journal-title":"Proc. Natl. Acad. Sci."},{"key":"6_CR14","doi-asserted-by":"crossref","unstructured":"Liessner, R., et al.: Hyperparameter optimization for deep reinforcement learning in vehicle energy management. In: ICAART (2), pp. 134\u2013144 (2019)","DOI":"10.5220\/0007364701340144"},{"key":"6_CR15","unstructured":"Lillicrap, T.P., et al.: Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 (2015)"},{"key":"6_CR16","unstructured":"Mahmood, A.R., et al.: Benchmarking reinforcement learning algorithms on real-world robots. In: Conference on Robot Learning, pp. 561\u2013591. PMLR (2018)"},{"key":"6_CR17","unstructured":"Mankowitz, D.J., et al.: Robust reinforcement learning for continuous control with model misspecification. arXiv preprint arXiv:1906.07516 (2019)"},{"key":"6_CR18","unstructured":"McFarlane, R.: A survey of exploration strategies in reinforcement learning. McGill University (2018)"},{"issue":"1","key":"6_CR19","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1109\/JRPROC.1961.287775","volume":"49","author":"M Minsky","year":"1961","unstructured":"Minsky, M.: Steps toward artificial intelligence. Proc. IRE 49(1), 8\u201330 (1961)","journal-title":"Proc. IRE"},{"key":"6_CR20","unstructured":"Mnih, V., et al.: Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 (2013)"},{"key":"6_CR21","unstructured":"Nikishin, E., et al.: Improving stability in deep reinforcement learning with weight averaging. In: Uncertainty in Artificial Intelligence Workshop on Uncertainty in Deep Learning (2018)"},{"key":"6_CR22","unstructured":"Ren, Z., Zhu, G., Hu, H., Han, B., Chen, J., Zhang, C.: On the estimation bias in double q-learning. Adv. Neural Inf. Process. Syst. 34 (2021)"},{"key":"6_CR23","unstructured":"Schaul, T., et al.: Prioritized experience replay. arXiv preprint arXiv:1511.05952 (2015)"},{"key":"6_CR24","doi-asserted-by":"crossref","unstructured":"Schr\u00f6der, K., Kastius, A., Schlosser, R.: Welcome to the jungle: a conceptual comparison of reinforcement learning algorithms. In: Proceedings of the 12th International Conference on Operations Research and Enterprise Systems - Volume 1: ICORES, pp. 143\u2013150 (2023)","DOI":"10.5220\/0011626700003396"},{"key":"6_CR25","unstructured":"Schulman, J., et al.: High-dimensional continuous control using generalized advantage estimation. arXiv preprint arXiv:1506.02438 (2015)"},{"key":"6_CR26","unstructured":"Schulman, J., et al.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"6_CR27","unstructured":"Silver, D.: Lectures on reinforcement learning (2015). https:\/\/www.davidsilver.uk\/teaching\/"},{"key":"6_CR28","unstructured":"Silver, D., et al.: Mastering chess and shogi by self-play with a general reinforcement learning algorithm. arXiv preprint arXiv:1712.01815 (2017)"},{"key":"6_CR29","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (2018). https:\/\/www.andrew.cmu.edu\/course\/10-703\/textbook\/BartoSutton.pdf"},{"key":"6_CR30","unstructured":"Sutton, R.S., McAllester, D., Singh, S., Mansour, Y.: Policy gradient methods for reinforcement learning with function approximation. Adv. Neural Inf. Process. Syst. 12 (1999)"},{"key":"6_CR31","unstructured":"Weaver, L., Tao, N.: The optimal reward baseline for gradient-based reinforcement learning. arXiv preprint arXiv:1301.2315 (2013)"},{"key":"6_CR32","unstructured":"Weng, J., et al.: Tianshou: a highly modularized deep reinforcement learning library. arXiv preprint arXiv:2107.14171 (2021)"},{"key":"6_CR33","unstructured":"Weng, L.: Policy gradient algorithms (2018). lilianweng.github.io, https:\/\/lilianweng.github.io\/posts\/2018-04-08-policy-gradient\/"},{"key":"6_CR34","unstructured":"Weng, L.: Exploration strategies in deep reinforcement learning (2020). https:\/\/lilianweng.github.io\/"},{"key":"6_CR35","unstructured":"Yildiz, C., et al.: Continuous-time model-based reinforcement learning. In: ICML, pp. 12009\u201312018. PMLR (2021). https:\/\/youtu.be\/PIouASLg_-g"}],"container-title":["Communications in Computer and Information Science","Operations Research and Enterprise Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-49662-2_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,14]],"date-time":"2023-12-14T13:04:11Z","timestamp":1702559051000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-49662-2_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,15]]},"ISBN":["9783031496615","9783031496622"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-49662-2_6","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2023,12,15]]},"assertion":[{"value":"15 December 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICORES","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Operations Research and Enterprise Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lisbon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Portugal","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 February 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 February 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icores2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icores.scitevents.org\/?y=2023","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}