{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T14:11:33Z","timestamp":1742911893197,"version":"3.40.3"},"publisher-location":"Cham","reference-count":49,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031779404"},{"type":"electronic","value":"9783031779411"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-77941-1_10","type":"book-chapter","created":{"date-parts":[[2025,1,26]],"date-time":"2025-01-26T08:44:45Z","timestamp":1737881085000},"page":"123-139","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Evidence on\u00a0the\u00a0Regularisation Properties of\u00a0Maximum-Entropy Reinforcement Learning"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-3382-6081","authenticated-orcid":false,"given":"R\u00e9my Hosseinkhan","family":"Boucher","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0130-0545","authenticated-orcid":false,"given":"Onofrio","family":"Semeraro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lionel","family":"Mathelin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,27]]},"reference":[{"key":"10_CR1","unstructured":"Agarwal, A., Jiang, N., Kakade, S.M.: Reinforcement learning: Theory and algorithms (2019)"},{"key":"10_CR2","unstructured":"Agarwal, R., Schwarzer, M., Castro, P.S., Courville, A.C., Bellemare, M.: Deep reinforcement learning at the edge of the statistical precipice. In: Advances in Neural Information Processing Systems, vol. 34 (2021)"},{"key":"10_CR3","unstructured":"Ahmed, Z., Le\u00a0Roux, N., Norouzi, M., Schuurmans, D.: Understanding the impact of entropy on policy optimization. In: Chaudhuri, K., Salakhutdinov, R. (eds.) Proceedings of the 36th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a097, pp. 151\u2013160. PMLR (09\u201315 Jun 2019) (2019)"},{"key":"10_CR4","doi-asserted-by":"publisher","unstructured":"Amari, S.i.: Natural Gradient Works Efficiently in Learning. Neural Comput. 10(2), 251\u2013276 (1998). https:\/\/doi.org\/10.1162\/089976698300017746, https:\/\/doi.org\/10.1162\/089976698300017746","DOI":"10.1162\/089976698300017746"},{"key":"10_CR5","doi-asserted-by":"crossref","unstructured":"Bucci, M.A., Semeraro, O., Allauzen, A., Wisniewski, G., Cordier, L., Mathelin, L.: Control of chaotic systems by deep reinforcement learning. Proceedings of the Royal Society A: Mathematical, Physical and Engineering Sciences 475(2231), 20190351 (2019). https:\/\/royalsocietypublishing.org\/doi\/abs\/10.1098\/rspa.2019.0351","DOI":"10.1098\/rspa.2019.0351"},{"key":"10_CR6","unstructured":"Cassandra, A.R.: Exact and Approximate Algorithms for Partially Observable Markov Decision Processes. Ph.D. thesis, Brown University (1998)"},{"key":"10_CR7","doi-asserted-by":"crossref","unstructured":"Chaudhari, P., Choromanska, A., Soatto, S., LeCun, Y., Baldassi, C., Borgs, C., Chayes, J., Sagun, L., Zecchina, R.: Entropy-sgd: biasing gradient descent into wide valleys*. J. Stat. Mech. Theory Exp. 2019(12), 124018 (2019)","DOI":"10.1088\/1742-5468\/ab39d9"},{"key":"10_CR8","unstructured":"Colas, C., Sigaud, O., Oudeyer, P.Y.: How many random seeds? Statistical power analysis in deep reinforcement learning experiments (2018)"},{"key":"10_CR9","unstructured":"Cover, T.M., Thomas, J.A.: Elements of Information Theory 2nd edn. (Wiley Series in Telecommunications and Signal Processing). Wiley-Interscience (2006)"},{"key":"10_CR10","unstructured":"Deisenroth, M.P., Peters, J.: Solving nonlinear continuous state-action-observation POMDPs for mechanical systems with gaussian noise. In: European Workshop on Reinforcement Learning (2012)"},{"key":"10_CR11","unstructured":"Derman, E., Geist, M., Mannor, S.: Twice regularized mdps and the equivalence between robustness and regularization. In: Ranzato, M., Beygelzimer, A., Dauphin, Y., Liang, P., Vaughan, J.W. (eds.) Advances in Neural Information Processing Systems, vol.\u00a034, pp. 22274\u201322287. Curran Associates, Inc. (2021)"},{"key":"10_CR12","unstructured":"Dinh, L., Pascanu, R., Bengio, S., Bengio, Y.: Sharp minima can generalize for deep nets. In: Precup, D., Teh, Y.W. (eds.) Proceedings of the 34th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a070, pp. 1019\u20131028. PMLR (06\u201311 Aug 2017) (2017)"},{"key":"10_CR13","doi-asserted-by":"crossref","unstructured":"Doyle, J.: Robust and optimal control. In: Proceedings of 35th IEEE Conference on Decision and Control, vol.\u00a02, pp. 1595\u20131598 (1996)","DOI":"10.1109\/CDC.1996.572756"},{"key":"10_CR14","unstructured":"Eysenbach, B., Levine, S.: Maximum entropy RL (provably) solves some robust RL problems. In: International Conference on Learning Representations (2022)"},{"key":"10_CR15","unstructured":"Gogianu, F., Berariu, T., Rosca, M.C., Clopath, C., Busoniu, L., Pascanu, R.: Spectral normalisation for deep reinforcement learning: An optimisation perspective. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 3734\u20133744. PMLR (18\u201324 Jul 2021) (2021)"},{"key":"10_CR16","unstructured":"Golowich, N., Rakhlin, A., Shamir, O.: Size-independent sample complexity of neural networks. In: Bubeck, S., Perchet, V., Rigollet, P. (eds.) Proceedings of the 31st Conference On Learning Theory. Proceedings of Machine Learning Research, vol.\u00a075, pp. 297\u2013299. PMLR (06\u201309 Jul 2018) (2018)"},{"key":"10_CR17","unstructured":"Haarnoja, T., Tang, H., Abbeel, P., Levine, S.: Reinforcement learning with deep energy-based policies. In: Precup, D., Teh, Y.W. (eds.) Proceedings of the 34th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a070, pp. 1352\u20131361. PMLR (06\u201311 Aug 2017) (2017)"},{"key":"10_CR18","unstructured":"Haarnoja, T., Zhou, A., Abbeel, P., Levine, S.: Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: Dy, J., Krause, A. (eds.) Proceedings of the 35th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a080, pp. 1861\u20131870. PMLR (10\u201315 Jul 2018) (2018)"},{"issue":"6","key":"10_CR19","doi-asserted-by":"publisher","first-page":"2451","DOI":"10.1214\/aos\/1030741081","volume":"25","author":"D Haussler","year":"1997","unstructured":"Haussler, D., Opper, M.: Mutual information, metric entropy and cumulative relative entropy risk. Ann. Stat. 25(6), 2451\u20132492 (1997)","journal-title":"Ann. Stat."},{"key":"10_CR20","doi-asserted-by":"crossref","unstructured":"Henderson, P., Islam, R., Bachman, P., Pineau, J., Precup, D., Meger, D.: Deep reinforcement learning that matters. In: Proceedings of the Thirty-Second AAAI Conference on Artificial Intelligence and Thirtieth Innovative Applications of Artificial Intelligence Conference and Eighth AAAI Symposium on Educational Advances in Artificial Intelligence. AAAI 2018\/IAAI 2018\/EAAI 2018, AAAI Press (2018)","DOI":"10.1609\/aaai.v32i1.11694"},{"key":"10_CR21","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-0729-0","volume-title":"Discrete-Time Markov Control Processes: Basic Optimality Criteria","author":"O Hern\u00e1ndez-Lerma","year":"1996","unstructured":"Hern\u00e1ndez-Lerma, O., Lasserre, J.B.: Discrete-Time Markov Control Processes: Basic Optimality Criteria, 1st edn. Springer, New York (1996). https:\/\/doi.org\/10.1007\/978-1-4612-0729-0","edition":"1"},{"key":"10_CR22","doi-asserted-by":"crossref","unstructured":"Hochreiter, S., Schmidhuber, J.: Flat Minima. Neural Comput. 9(1), 1\u201342 (1997)","DOI":"10.1162\/neco.1997.9.1.1"},{"key":"10_CR23","unstructured":"Jastrzebski, S., Arpit, D., Astrand, O., Kerg, G.B., Wang, H., Xiong, C., Socher, R., Cho, K., Geras, K.J.: Catastrophic fisher explosion: Early phase fisher matrix impacts generalization. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 4772\u20134784. PMLR (18\u201324 Jul 2021) (2021)"},{"key":"10_CR24","unstructured":"Kakade, S., Langford, J.: Approximately optimal approximate reinforcement learning. In: Proceedings of the Nineteenth International Conference on Machine Learning, pp. 267\u2013274. ICML 2002, Morgan Kaufmann Publishers Inc., San Francisco, CA, USA (2002)"},{"key":"10_CR25","unstructured":"Kakade, S.M.: A natural policy gradient. In: Dietterich, T., Becker, S., Ghahramani, Z. (eds.) Advances in Neural Information Processing Systems, vol.\u00a014. MIT Press (2001)"},{"key":"10_CR26","unstructured":"Karakida, R., Akaho, S., Amari, S.: Universal statistics of fisher information in deep neural networks: Mean field approach. In: Chaudhuri, K., Sugiyama, M. (eds.) Proceedings of the Twenty-Second International Conference on Artificial Intelligence and Statistics. Proceedings of Machine Learning Research, vol.\u00a089, pp. 1032\u20131041. PMLR (16\u201318 Apr 2019) (2019)"},{"key":"#cr-split#-10_CR27.1","unstructured":"Keskar, N., Nocedal, J., Tang, P., Mudigere, D., Smelyanskiy, M.: On large-batch training for deep learning: Generalization gap and sharp minima (2017), 5th International Conference on Learning Representations, ICLR 2017"},{"key":"#cr-split#-10_CR27.2","unstructured":"Conference date: 24-04-2017 Through 26-04-2017 (2017)"},{"key":"10_CR28","doi-asserted-by":"crossref","unstructured":"Lecarpentier, E., Abel, D., Asadi, K., Jinnai, Y., Rachelson, E., Littman, M.L.: Lipschitz lifelong reinforcement learning (2021)","DOI":"10.1609\/aaai.v35i9.17006"},{"key":"10_CR29","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"203","DOI":"10.1007\/978-3-540-45167-9_16","volume-title":"Learning Theory and Kernel Machines","author":"D McAllester","year":"2003","unstructured":"McAllester, D.: Simplified PAC-Bayesian margin bounds. In: Sch\u00f6lkopf, B., Warmuth, M.K. (eds.) COLT-Kernel 2003. LNCS (LNAI), vol. 2777, pp. 203\u2013215. Springer, Heidelberg (2003). https:\/\/doi.org\/10.1007\/978-3-540-45167-9_16"},{"key":"10_CR30","unstructured":"Miyato, T., Kataoka, T., Koyama, M., Yoshida, Y.: Spectral normalization for generative adversarial networks (2018)"},{"key":"10_CR31","unstructured":"Mulayoff, R., Michaeli, T.: Unique properties of flat minima in deep networks. In: Proceedings of the 37th International Conference on Machine Learning. ICML 2020, JMLR.org (2020)"},{"key":"10_CR32","unstructured":"Neu, G., Jonsson, A., G\u00f3mez, V.: A unified view of entropy-regularized Markov decision processes. CoRR abs\/1705.07798 (2017). http:\/\/arxiv.org\/abs\/1705.07798"},{"key":"10_CR33","unstructured":"Neyshabur, B., Bhojanapalli, S., Mcallester, D., Srebro, N.: Exploring generalization in deep learning. In: Guyon, I., Luxburg, U.V., Bengio, S., Wallach, H., Fergus, R., Vishwanathan, S., Garnett, R. (eds.) Advances in Neural Information Processing Systems. vol.\u00a030. Curran Associates, Inc. (2017)"},{"key":"10_CR34","unstructured":"Neyshabur, B., Tomioka, R., Srebro, N.: Norm-based capacity control in neural networks. In: Gr\u00fcnwald, P., Hazan, E., Kale, S. (eds.) Proceedings of The 28th Conference on Learning Theory. Proceedings of Machine Learning Research, vol.\u00a040, pp. 1376\u20131401. PMLR, Paris, France (03\u201306 Jul 2015) (2015)"},{"key":"10_CR35","unstructured":"Nilim, A., Ghaoui, L.: Robustness in Markov decision problems with uncertain transition matrices. In: Thrun, S., Saul, L., Sch\u00f6lkopf, B. (eds.) Advances in Neural Information Processing Systems, vol.\u00a016. MIT Press (2003)"},{"issue":"268","key":"10_CR36","first-page":"1","volume":"22","author":"A Raffin","year":"2021","unstructured":"Raffin, A., Hill, A., Gleave, A., Kanervisto, A., Ernestus, M., Dormann, N.: Stable-baselines3: reliable reinforcement learning implementations. J. Mach. Learn. Res. 22(268), 1\u20138 (2021)","journal-title":"J. Mach. Learn. Res."},{"key":"10_CR37","unstructured":"Sagun, L., Bottou, L., LeCun, Y.: Eigenvalues of the hessian in deep learning: singularity and beyond (2017)"},{"key":"10_CR38","doi-asserted-by":"crossref","unstructured":"Satia, J.K., Lave, R.E.: Markovian decision processes with uncertain transition probabilities. Oper. Res. 21(3), 728\u2013740 (1973). http:\/\/www.jstor.org\/stable\/169381","DOI":"10.1287\/opre.21.3.728"},{"key":"10_CR39","unstructured":"Schulman, J., Levine, S., Abbeel, P., Jordan, M., Moritz, P.: Trust region policy optimization. In: Bach, F., Blei, D. (eds.) Proceedings of the 32nd International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a037, pp. 1889\u20131897. PMLR, Lille, France (07\u201309 Jul 2015) (2015)"},{"key":"10_CR40","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. CoRR (2017)"},{"key":"10_CR41","unstructured":"Shen, Z., Ribeiro, A., Hassani, H., Qian, H., Mi, C.: Hessian aided policy gradient. In: Chaudhuri, K., Salakhutdinov, R. (eds.) Proceedings of the 36th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a097, pp. 5729\u20135738. PMLR (6 2019)"},{"key":"10_CR42","unstructured":"Sigaud, O., Buffet, O.: Markov Decision Processes in Artificial Intelligence. Wiley (2010)"},{"issue":"1","key":"10_CR43","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1007\/BF02169423","volume":"1","author":"TL Vincent","year":"1991","unstructured":"Vincent, T.L., Yu, J.: Control of a chaotic system. Dyn. Control 1(1), 35\u201352 (1991)","journal-title":"Dyn. Control"},{"issue":"3","key":"10_CR44","doi-asserted-by":"publisher","first-page":"241","DOI":"10.1080\/09540099108946587","volume":"3","author":"RJ Williams","year":"1991","unstructured":"Williams, R.J., Peng, J., Li, H.: Function optimization using connectionist reinforcement learning algorithms. Connect. Sci. 3(3), 241\u2013268 (1991)","journal-title":"Connect. Sci."},{"key":"10_CR45","unstructured":"Xie, Z., Sato, I., Sugiyama, M.: A diffusion theory for deep learning dynamics: stochastic gradient descent exponentially favors flat minima (2021)"},{"key":"10_CR46","unstructured":"Yoshida, Y., Miyato, T.: Spectral norm regularization for improving the generalizability of deep learning (2017)"},{"key":"10_CR47","unstructured":"Zhao, Y., Zhang, H., Hu, X.: Penalizing gradient norm for efficiently improving generalization in deep learning. In: Chaudhuri, K., Jegelka, S., Song, L., Szepesvari, C., Niu, G., Sabato, S. (eds.) Proceedings of the 39th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0162, pp. 26982\u201326992. PMLR (17\u201323 Jul 2022) (2022)"},{"key":"10_CR48","unstructured":"Zhou, K., Doyle, J., Glover, K.: Robust and Optimal Control. Feher\/Prentice Hall Digital and, Prentice Hall (1996)"}],"container-title":["Communications in Computer and Information Science","Optimization and Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-77941-1_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,26]],"date-time":"2025-01-26T08:45:04Z","timestamp":1737881104000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-77941-1_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031779404","9783031779411"],"references-count":49,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-77941-1_10","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"27 January 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"OLA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Optimization and Learning","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Dubrovnik","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Croatia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 May 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 May 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ola2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ola2024.sciencesconf.org","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}