{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T14:07:58Z","timestamp":1784210878352,"version":"3.55.0"},"reference-count":53,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000006","name":"Office of Naval Research","doi-asserted-by":"publisher","award":["N00014-23-1-2532"],"award-info":[{"award-number":["N00014-23-1-2532"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100004351","name":"Cisco Research Award","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100004351","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1109\/tnnls.2023.3305983","type":"journal-article","created":{"date-parts":[[2023,9,13]],"date-time":"2023-09-13T17:39:09Z","timestamp":1694626749000},"page":"17549-17558","source":"Crossref","is-referenced-by-count":12,"title":["Hierarchical Adversarial Inverse Reinforcement Learning"],"prefix":"10.1109","volume":"35","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7708-5247","authenticated-orcid":false,"given":"Jiayu","family":"Chen","sequence":"first","affiliation":[{"name":"School of Industrial Engineering, Purdue University, West Lafayette, IN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3010-8090","authenticated-orcid":false,"given":"Tian","family":"Lan","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, George Washington University, Washington, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9131-4723","authenticated-orcid":false,"given":"Vaneet","family":"Aggarwal","sequence":"additional","affiliation":[{"name":"School of Industrial Engineering, Purdue University, West Lafayette, IN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1609\/icaps.v31i1.15998"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2019.2931830"},{"key":"ref3","article-title":"Visual foresight: Model-based deep reinforcement learning for vision-based robotic control","author":"Ebert","year":"2018","journal-title":"arXiv:1812.00568"},{"key":"ref4","article-title":"Continuous control with deep reinforcement learning","author":"Lillicrap","year":"2015","journal-title":"arXiv:1509.02971"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2019.2927869"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3121870"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2022.105116"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/S0004-3702(99)00052-1"},{"key":"ref9","first-page":"1","article-title":"Directed-info GAIL: Learning hierarchical policies from unsegmented demonstrations using directed information","volume-title":"Proc. 7th Int. Conf. Learn. Represent.","author":"Sharma"},{"key":"ref10","first-page":"5097","article-title":"Adversarial option-aware hierarchical imitation learning","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Jing"},{"key":"ref11","first-page":"1","article-title":"Generative adversarial imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ho"},{"key":"ref12","first-page":"303","article-title":"Causality, feedback and directed information","volume-title":"Proc. Int. Symp. Inf. Theory Appl.","author":"Massey"},{"key":"ref13","first-page":"627","article-title":"A reduction of imitation learning and structured prediction to no-regret online learning","volume-title":"Proc. 14th Int. Conf. Artif. Intell. Statist.","author":"Ross"},{"key":"ref14","article-title":"Learning robust rewards with adversarial inverse reinforcement learning","author":"Fu","year":"2017","journal-title":"arXiv:1710.11248"},{"key":"ref15","first-page":"80","article-title":"Augmenting GAIL with bc for sample efficient imitation learning","volume-title":"Proc. Conf. Robot Learn.","author":"Jena"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9560907"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/79.543975"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2008.10.024"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3213246"},{"key":"ref21","first-page":"663","article-title":"Algorithms for inverse reinforcement learning","volume-title":"Proc. 7th Int. Conf. Mach. Learn.","author":"Ng"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2543000"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1991.3.1.88"},{"key":"ref24","first-page":"1433","article-title":"Maximum entropy inverse reinforcement learning","volume-title":"Proc. 33rd AAAI Conf. Artif. Intell.","author":"Ziebart"},{"key":"ref25","first-page":"19","article-title":"Nonlinear inverse reinforcement learning with Gaussian processes","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Levine"},{"key":"ref26","article-title":"Maximum entropy deep inverse reinforcement learning","author":"Wulfmeier","year":"2015","journal-title":"arXiv:1507.04888"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3114612"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3106635"},{"key":"ref29","first-page":"2923","article-title":"Hierarchical imitation and reinforcement learning","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","volume":"80","author":"Le"},{"key":"ref30","first-page":"3418","article-title":"Compile: Compositional imitation learning and execution","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","volume":"97","author":"Kipf"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-016-5580-x"},{"key":"ref32","first-page":"883","article-title":"Provable hierarchical imitation learning via EM","volume-title":"Proc. 24th Int. Conf. Artif. Intell. Statist.","volume":"130","author":"Zhang"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1023\/A:1007469218079"},{"key":"ref34","article-title":"A primer on maximum causal entropy inverse reinforcement learning","author":"Gleave","year":"2022","journal-title":"arXiv:2203.11409"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.10916"},{"key":"ref36","first-page":"2010","article-title":"DAC: The double actor-critic architecture for learning options","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref37","first-page":"1","article-title":"The skill-action architecture: Learning abstract action embeddings for reinforcement learning","volume-title":"Proc. 9th Int. Conf. Learn. Represent.","author":"Li"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11775"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2018.2805379"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3087733"},{"key":"ref41","first-page":"30406","article-title":"Scalable multi-agent covering option discovery based on Kronecker graphs","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Chen"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"ref44","volume-title":"Elements of Information Theory","author":"Cover","year":"1999"},{"key":"ref45","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref46","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv:1707.06347"},{"key":"ref47","article-title":"OpenAI gym","volume":"abs\/1606.01540","author":"Brockman","year":"2016","journal-title":"CoRR"},{"key":"ref48","article-title":"Multi-task hierarchical adversarial inverse reinforcement learning","author":"Chen","year":"2023","journal-title":"arXiv:2305.12633"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9197020"},{"key":"ref50","first-page":"15486","article-title":"Stacked capsule autoencoders","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kosiorek"},{"key":"ref51","article-title":"Three tutorial lectures on entropy and counting","author":"Galvin","year":"2014","journal-title":"arXiv:1406.7872"},{"key":"ref52","first-page":"2980","article-title":"A recurrent latent variable model for sequential data","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chung"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1007\/BF02418571"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/10772360\/10250822.pdf?arnumber=10250822","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T19:07:36Z","timestamp":1733252856000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10250822\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":53,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2023.3305983","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12]]}}}