{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T20:58:45Z","timestamp":1775681925690,"version":"3.50.1"},"reference-count":48,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476110"],"award-info":[{"award-number":["62476110"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476109"],"award-info":[{"award-number":["62476109"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62206108"],"award-info":[{"award-number":["62206108"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2341229"],"award-info":[{"award-number":["U2341229"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2021ZD0112500"],"award-info":[{"award-number":["2021ZD0112500"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Key Research and Development Project of Jilin Province","award":["20240304200SF"],"award-info":[{"award-number":["20240304200SF"]}]},{"name":"International Cooperation Project of Jilin Province","award":["20220402009GH"],"award-info":[{"award-number":["20220402009GH"]}]},{"DOI":"10.13039\/501100004543","name":"China Scholarship Council","doi-asserted-by":"publisher","award":["202206170041"],"award-info":[{"award-number":["202206170041"]}],"id":[{"id":"10.13039\/501100004543","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1109\/tnnls.2025.3626050","type":"journal-article","created":{"date-parts":[[2025,11,20]],"date-time":"2025-11-20T18:43:08Z","timestamp":1763664188000},"page":"1851-1863","source":"Crossref","is-referenced-by-count":0,"title":["Universal Stabilization for Maximum Entropy Optimization in Reinforcement Learning"],"prefix":"10.1109","volume":"37","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5685-8506","authenticated-orcid":false,"given":"Xing","family":"Chen","sequence":"first","affiliation":[{"name":"School of Artificial Intelligence, Jilin University, Changchun, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-0073-123X","authenticated-orcid":false,"given":"Yewen","family":"Li","sequence":"additional","affiliation":[{"name":"College of Computing and Data Science, Nanyang Technological University, Jurong West, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7391-0334","authenticated-orcid":false,"given":"Xiaofeng","family":"Cao","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Jilin University, Changchun, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7835-9556","authenticated-orcid":false,"given":"Hechang","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Jilin University, Changchun, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1258-1845","authenticated-orcid":false,"given":"Hengshuai","family":"Yao","sequence":"additional","affiliation":[{"name":"Department of Computing Science, University of Alberta, Edmonton, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7064-7438","authenticated-orcid":false,"given":"Bo","family":"An","sequence":"additional","affiliation":[{"name":"College of Computing and Data Science, Nanyang Technological University, Jurong West, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2697-8093","authenticated-orcid":false,"given":"Yi","family":"Chang","sequence":"additional","affiliation":[{"name":"Engineering Research Center of Knowledge-Driven Human-Machine Intelligence, MOE, Changchun, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"2171","article-title":"Reinforcement learning with deep energy-based policies","volume-title":"Proc. 34th Int. Conf. Mach. Learn. (ICML)","author":"Haarnoja"},{"key":"ref2","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref3","first-page":"2681","article-title":"Provably efficient maximum entropy exploration","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Hazan"},{"key":"ref4","first-page":"467","article-title":"Maximum entropy models and stochastic optimality theory","author":"J\u00e4ger","year":"2007","journal-title":"Architectures, Rules, and Preferences: Variations on Themes By Joan W. Bresnan"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1103\/RevModPhys.89.015002"},{"key":"ref6","article-title":"Maximum entropy RL (provably) solves some robust RL problems","author":"Eysenbach","year":"2021","journal-title":"arXiv:2103.06257"},{"key":"ref7","first-page":"151","article-title":"Understanding the impact of entropy on policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ahmed"},{"key":"ref8","first-page":"7553","article-title":"Maximum entropy-regularized multi-goal reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhao"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.6021"},{"key":"ref10","first-page":"60384","article-title":"Multi-modal inverse constrained reinforcement learning from a mixture of demonstrations","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Qiao"},{"key":"ref11","first-page":"2593","article-title":"Maximum entropy GFlowNets with soft Q-learning","volume-title":"Proc. 27th Int. Conf. Artif. Intell. Statist.","author":"Mohammadpour"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3236361"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-024-00829-3"},{"key":"ref14","first-page":"34161","article-title":"Fast rates for maximum entropy exploration","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Tiapkin"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1812.05905"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1980.1056144"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/SSCI47803.2020.9308468"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref19","volume-title":"Introducing Roboschool","year":"2017"},{"key":"ref20","article-title":"OpenAI gym","author":"Brockman","year":"2016","journal-title":"arXiv:1606.01540"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992698"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"ref24","first-page":"2613","article-title":"Double Q-learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"23","author":"Hasselt"},{"key":"ref25","first-page":"2587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. 35th Int. Conf. Mach. Learn. (ICML)","volume":"4","author":"Fujimoto"},{"key":"ref26","first-page":"1995","article-title":"Dueling network architectures for deep reinforcement learning","volume-title":"Proc. Int. Conf. Int. Conf. Mach. Learn.","volume":"48","author":"Wang"},{"key":"ref27","first-page":"202","article-title":"Taming the noise in reinforcement learning via soft updates","volume-title":"Proc. 32nd Conf. Uncertainty Artif. Intell.","author":"Fox"},{"key":"ref28","first-page":"2976","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"5","author":"Haarnoja"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3174051"},{"key":"ref30","first-page":"267","article-title":"Approximately optimal approximate reinforcement learning","volume-title":"Proc. 19th Int. Conf. Mach. Learn.","author":"Kakade"},{"key":"ref31","article-title":"Trust region policy optimization","author":"Schulman","year":"2015","journal-title":"arXiv:1502.05477"},{"key":"ref32","first-page":"22","article-title":"Constrained policy optimization","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Achiam"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.3044196"},{"key":"ref34","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv:1707.06347"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v24i1.7727"},{"issue":"93","key":"ref36","first-page":"1","article-title":"Hierarchical relative entropy policy search","volume":"17","author":"Daniel","year":"2016","journal-title":"J. Mach. Learn. Res."},{"key":"ref37","article-title":"Maximum a Posteriori policy optimisation","author":"Abdolmaleki","year":"2018","journal-title":"arXiv:1806.06920"},{"key":"ref38","article-title":"V-MPO: On-policy maximum a Posteriori policy optimization for discrete and continuous control","author":"Francis Song","year":"2019","journal-title":"arXiv:1909.12238"},{"key":"ref39","first-page":"5737","article-title":"Variational inference with tail-adaptive f-divergence","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Wang"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.6144"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i6.25864"},{"key":"ref42","article-title":"Modeling purposeful adaptive behavior with the principle of maximum causal entropy","author":"Ziebart","year":"2010"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00041"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/tnn.1998.712192"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2025.3537087"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3082568"},{"key":"ref47","article-title":"High-dimensional continuous control using generalized advantage estimation","author":"Schulman","year":"2015","journal-title":"arXiv:1506.02438"},{"key":"ref48","first-page":"307","article-title":"Improving stability in deep reinforcement learning with weight averaging","volume-title":"Proc. Uncertainty Artif. Intell. Workshop Uncertainty Deep Learn.","author":"Nikishin"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/5962385\/11475606\/11261545.pdf?arnumber=11261545","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T20:06:54Z","timestamp":1775678814000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11261545\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":48,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2025.3626050","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]}}}