{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T05:53:42Z","timestamp":1784613222011,"version":"3.55.0"},"reference-count":42,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000183","name":"Army Research Office","doi-asserted-by":"publisher","award":["W911NF2110103"],"award-info":[{"award-number":["W911NF2110103"]}],"id":[{"id":"10.13039\/100000183","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000006","name":"Office of Naval Research","doi-asserted-by":"publisher","award":["N000142212474"],"award-info":[{"award-number":["N000142212474"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2024,4]]},"DOI":"10.1109\/tai.2023.3297988","type":"journal-article","created":{"date-parts":[[2023,7,27]],"date-time":"2023-07-27T13:55:35Z","timestamp":1690466135000},"page":"1563-1572","source":"Crossref","is-referenced-by-count":7,"title":["Generalized Maximum Entropy Reinforcement Learning via Reward Shaping"],"prefix":"10.1109","volume":"5","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2507-3180","authenticated-orcid":false,"given":"Feng","family":"Tao","sequence":"first","affiliation":[{"name":"Volvo Car Technology USA LLC, Sunnyvale, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2960-0787","authenticated-orcid":false,"given":"Mingkang","family":"Wu","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering, University of Texas, San Antonio, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3383-0185","authenticated-orcid":false,"given":"Yongcan","family":"Cao","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering, University of Texas, San Antonio, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"151","article-title":"Understanding the impact of entropy on policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ahmed","year":"2019"},{"key":"ref2","first-page":"12","article-title":"Robot learning from demonstration","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Atkeson","year":"1997"},{"key":"ref3","article-title":"Dota 2 with large scale deep reinforcement learning","author":"Berner","year":"2019"},{"key":"ref4","article-title":"Openai gym","author":"Brockman","year":"2016"},{"key":"ref5","article-title":"Go-explore: A new approach for hard-exploration problems","author":"Ecoffet","year":"2019"},{"key":"ref6","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","volume-title":"Proc. Thirteenth Int. Conf. Artif. Intell. Statist.","author":"Glorot","year":"2010"},{"key":"ref7","first-page":"1352","article-title":"Reinforcement learning with deep energy-based policies","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja","year":"2017"},{"key":"ref8","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja","year":"2018"},{"key":"ref9","first-page":"2613","article-title":"Double Q-learning","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst,","author":"Hasselt","year":"2010"},{"key":"ref10","first-page":"2681","article-title":"Provably efficient maximum entropy exploration","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Hazan","year":"2019"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794127"},{"key":"ref12","first-page":"2469","article-title":"Policy optimization with demonstrations","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kang","year":"2018"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913495721"},{"key":"ref14","first-page":"1008","article-title":"Actor-critic algorithms","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Konda","year":"2000"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-022-01816-5"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.2478\/cait-2012-0021"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.32657\/10356\/90191"},{"key":"ref18","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mnih","year":"2016"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref20","first-page":"278","article-title":"Policy invariance under reward transformations: Theory and application to reward shaping","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ng","year":"1999"},{"key":"ref21","first-page":"93","article-title":"How can we define intrinsic motivation?","volume-title":"Proc. 8th Int. Conf. Epigenetic Robot.: Model. Cogn. Develop. Robotic Syst.","author":"Oudeyer","year":"2008"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/s11768-011-0313-y"},{"key":"ref23","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schulman","year":"2015"},{"key":"ref24","article-title":"High-dimensional continuous control using generalized advantage estimation","author":"Schulman","year":"2015"},{"key":"ref25","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/475"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/tnn.1998.712192"},{"key":"ref29","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Sutton","year":"2000"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1162\/neco.2006.18.12.2936"},{"key":"ref31","article-title":"Source codes for generalized maximum entropy reinforcement learning via reward shaping","author":"Tao","year":"2023"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.2139\/ssrn.3316387"},{"key":"ref34","article-title":"Exploring model-based planning with policy networks","author":"Wang","year":"2019"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992698"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref37","first-page":"5940","article-title":"A regularized approach to sparse optimal policy in reinforcement learning","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Yang","year":"2019"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/JSAIT.2021.3081108"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1137\/19M1288012"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i1.16156"},{"key":"ref41","first-page":"4644","article-title":"On learning intrinsic rewards for policy gradient methods","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Zheng","year":"2018"},{"key":"ref42","volume-title":"Modeling Purposeful Adaptive Behavior With the Principle of Maximum Causal Entropy","author":"Ziebart","year":"2010"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9078688\/10495741\/10196410.pdf?arnumber=10196410","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T01:09:22Z","timestamp":1755911362000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10196410\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4]]},"references-count":42,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tai.2023.3297988","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,4]]}}}