{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T15:34:48Z","timestamp":1772120088419,"version":"3.50.1"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,10,13]],"date-time":"2024-10-13T00:00:00Z","timestamp":1728777600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,13]],"date-time":"2024-10-13T00:00:00Z","timestamp":1728777600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1007\/s13042-024-02399-7","type":"journal-article","created":{"date-parts":[[2024,10,13]],"date-time":"2024-10-13T19:02:37Z","timestamp":1728846157000},"page":"2417-2429","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Off-policy asymptotic and adaptive maximum entropy deep reinforcement learning"],"prefix":"10.1007","volume":"16","author":[{"given":"Huihui","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xu","family":"Han","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,13]]},"reference":[{"issue":"7540","key":"2399_CR1","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"key":"2399_CR2","unstructured":"Mnih V, Badia AP, Mirza M, Graves A, Lillicrap T, Harley T, Silver D, Kavukcuoglu K (2016) Asynchronous methods for deep reinforcement learning. In: International Conference on Machine Learning, pp. 1928\u20131937"},{"key":"2399_CR3","unstructured":"Haarnoja T, Zhou A, Abbeel P, Levine S (2018) Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor, 1861\u20131870. PMLR"},{"key":"2399_CR4","unstructured":"Heess N, Wayne G, Silver D, Lillicrap T, Tassa Y, Erez T (2015) Learning continuous control policies by stochastic value gradients. arXiv preprint arXiv:1510.09142"},{"key":"2399_CR5","unstructured":"Ziebart BD, Maas AL, Bagnell JA, Dey AK, et al.(2008) Maximum entropy inverse reinforcement learning. In: Aaai, vol. 8, pp. 1433\u20131438. Chicago, IL, USA"},{"key":"2399_CR6","unstructured":"Ziebart BD (2010) modeling purposeful adaptive behavior with the principle of maximum causal entropy. Carnegie Mellon University"},{"key":"2399_CR7","unstructured":"Kumar A, Fu J, Soh M, Tucker G, Levine S (2019) Stabilizing off-policy q-learning via bootstrapping error reduction. Advances in Neural Information Processing Systems 32"},{"key":"2399_CR8","unstructured":"Kumar A, Zhou A, Tucker G, Levine S (2020) Conservative q-learning for offline reinforcement learning. arXiv preprint arXiv:2006.04779"},{"key":"2399_CR9","unstructured":"Fujimoto S, Meger D, Precup D (2019) Off-policy deep reinforcement learning without exploration, 2052\u20132062. PMLR"},{"key":"2399_CR10","unstructured":"Wu Y, Tucker G, Nachum O (2019) Behavior regularized offline reinforcement learning. arXiv preprint arXiv:1911.11361"},{"key":"2399_CR11","unstructured":"Wang C, Ross K (2019) Boosting soft actor-critic: Emphasizing recent experience without forgetting the past. arXiv preprint arXiv:1906.04009"},{"key":"2399_CR12","unstructured":"Martin JB, Chekroun R, Moutarde F (2021) Learning from demonstrations with sacr2: Soft actor-critic with reward relabeling. arXiv preprint arXiv:2110.14464"},{"key":"2399_CR13","doi-asserted-by":"crossref","unstructured":"Duan J, Guan Y, Li SE, Ren Y, Sun Q, Cheng B (2021) Distributional soft actor-critic: Off-policy reinforcement learning for addressing value estimation errors. IEEE Trans Neural Netw Learn Syst","DOI":"10.1109\/TNNLS.2021.3082568"},{"key":"2399_CR14","doi-asserted-by":"crossref","unstructured":"Ren Y, Duan J, Li SE, Guan Y, Sun Q (2020) Improving generalization of reinforcement learning with minimax distributional soft actor-critic. In: 2020 IEEE 23rd International Conference on Intelligent Transportation Systems (ITSC), pp. 1\u20136. IEEE","DOI":"10.1109\/ITSC45102.2020.9294300"},{"key":"2399_CR15","unstructured":"Ma X, Xia L, Zhou Z, Yang J, Zhao Q (2020) Dsac: Distributional soft actor critic for risk-sensitive reinforcement learning. arXiv preprint arXiv:2004.14547"},{"key":"2399_CR16","unstructured":"Duan J, Ren Y, Zhang F, Guan Y, Yu D, Li SE, Cheng B, Zhao L (2021) Encoding distributional soft actor-critic for autonomous driving in multi-lane scenarios. arXiv preprint arXiv:2109.05540"},{"key":"2399_CR17","unstructured":"Akimov D (2019) Distributed soft actor-critic with multivariate reward representation and knowledge distillation. arXiv preprint arXiv:1911.13056"},{"key":"2399_CR18","unstructured":"Hou Z, Zhang K, Wan Y, Li D, Fu C, Yu H (2020) Off-policy maximum entropy reinforcement learning: soft actor-critic with advantage weighted mixture policy (sac-awmp). arXiv preprint arXiv:2002.02829"},{"key":"2399_CR19","unstructured":"Ward PN, Smofsky A, Bose AJ (2019) Improving exploration in soft-actor-critic with normalizing flows policies. arXiv preprint arXiv:1906.02771"},{"key":"2399_CR20","unstructured":"Haarnoja T, Zhou A, Hartikainen K, Tucker G, Ha S, Tan J, Kumar V, Zhu H, Gupta A, Abbeel P, et al. (2018) Soft actor-critic algorithms and applications. arXiv preprint arXiv:1812.05905"},{"key":"2399_CR21","unstructured":"Wang Y, Ni T (2020) Meta-sac: Auto-tune the entropy temperature of soft actor-critic via metagradient. arXiv preprint arXiv:2007.01932"},{"key":"2399_CR22","doi-asserted-by":"crossref","unstructured":"Zeng X, Peng H, Li A (2023) Effective and stable role-based multi-agent collaboration by structural information principles. In: Proceedings of the AAAI Conference on Artificial Intelligence 37:11772\u201311780","DOI":"10.1609\/aaai.v37i10.26390"},{"key":"2399_CR23","unstructured":"Zeng X, Peng H, Su D, Li A (2024) Effective reinforcement learning based on structural information principles. arXiv preprint arXiv:2404.09760"},{"key":"2399_CR24","doi-asserted-by":"crossref","unstructured":"Zeng X, Peng H, Li A, Liu C, He L, Yu PS (2023) Hierarchical state abstraction based on structural information principles. arXiv preprint arXiv:2304.12000","DOI":"10.24963\/ijcai.2023\/506"},{"issue":"3","key":"2399_CR25","doi-asserted-by":"publisher","first-page":"58","DOI":"10.1145\/203330.203343","volume":"38","author":"G Tesauro","year":"1995","unstructured":"Tesauro G (1995) Temporal difference learning and td-gammon. Commun ACM 38(3):58\u201368","journal-title":"Commun ACM"},{"key":"2399_CR26","unstructured":"Lillicrap TP, Hunt JJ, Pritzel A, Heess N, Erez T, Tassa Y, Silver D, Wierstra D (2015) Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971"},{"key":"2399_CR27","unstructured":"Jaques N, Ghandeharioun A, Shen JH, Ferguson C, Lapedriza A, Jones N, Gu S, Picard R (2019) Way off-policy batch deep reinforcement learning of implicit human preferences in dialog. arXiv preprint arXiv:1907.00456"},{"key":"2399_CR28","unstructured":"Schulman J, Levine S, Abbeel P, Jordan MI, Moritz P (2015) Trust region policy optimization. In: Icml 37:1889\u20131897"},{"key":"2399_CR29","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O (2017) Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347"},{"key":"2399_CR30","unstructured":"Watkins CJCH (1989) Learning from delayed rewards"},{"key":"2399_CR31","doi-asserted-by":"crossref","unstructured":"Todorov E, Erez T, Tassa Y (2012) Mujoco: A physics engine for model-based control. In: 2012 IEEE\/RSJ International Conference on Intelligent Robots and Systems, pp. 5026\u20135033. IEEE","DOI":"10.1109\/IROS.2012.6386109"},{"key":"2399_CR32","unstructured":"Brockman G, Cheung V, Pettersson L, Schneider J, Schulman J, Tang J, Zaremba W (2016) Openai gym. arXiv preprint arXiv:1606.01540"},{"key":"2399_CR33","unstructured":"Fujimoto S, Van Hoof H, Meger D (2018) Addressing function approximation error in actor-critic methods 80:1587\u20131596"},{"key":"2399_CR34","unstructured":"Duan Y, Chen X, Houthooft R, Schulman J, Abbeel P (2016) Benchmarking deep reinforcement learning for continuous control. In: International Conference on Machine Learning, pp. 1329\u20131338. PMLR"},{"key":"2399_CR35","unstructured":"Christodoulou P (2019) Soft actor-critic for discrete action settings. arXiv preprint arXiv:1910.07207"},{"key":"2399_CR36","unstructured":"Kingma DP, Ba J (2014) Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"issue":"3","key":"2399_CR37","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1214\/aoms\/1177729586","volume":"22","author":"H Robbins","year":"1951","unstructured":"Robbins H, Monro S (1951) A stochastic approximation method. Ann Math Stat 22(3):400\u2013407","journal-title":"Ann Math Stat"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-024-02399-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-024-02399-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-024-02399-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,7]],"date-time":"2025-04-07T03:30:25Z","timestamp":1743996625000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-024-02399-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,13]]},"references-count":37,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["2399"],"URL":"https:\/\/doi.org\/10.1007\/s13042-024-02399-7","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-4351146\/v1","asserted-by":"object"}]},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"value":"1868-8071","type":"print"},{"value":"1868-808X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,13]]},"assertion":[{"value":"30 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare they have no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The authors declare that this work is in compliance with the ethical standards of this journal.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with the ethical standards"}}]}}