{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T16:29:21Z","timestamp":1781022561026,"version":"3.54.1"},"reference-count":59,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2023,4,1]],"date-time":"2023-04-01T00:00:00Z","timestamp":1680307200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,4,1]],"date-time":"2023-04-01T00:00:00Z","timestamp":1680307200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,4,1]],"date-time":"2023-04-01T00:00:00Z","timestamp":1680307200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62176133"],"award-info":[{"award-number":["62176133"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61836004"],"award-info":[{"award-number":["61836004"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Tsinghua-Guoqiang Research Program","award":["2019GQG0006"],"award-info":[{"award-number":["2019GQG0006"]}]},{"name":"National Key Research and Development Program of China","award":["2021ZD0200300"],"award-info":[{"award-number":["2021ZD0200300"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2023,4,1]]},"DOI":"10.1109\/tpami.2022.3192418","type":"journal-article","created":{"date-parts":[[2022,7,19]],"date-time":"2022-07-19T19:31:13Z","timestamp":1658259073000},"page":"4152-4166","source":"Crossref","is-referenced-by-count":31,"title":["Adjacency Constraint for Efficient Hierarchical Reinforcement Learning"],"prefix":"10.1109","volume":"45","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9687-5263","authenticated-orcid":false,"given":"Tianren","family":"Zhang","sequence":"first","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3181-6881","authenticated-orcid":false,"given":"Shangqi","family":"Guo","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9173-2984","authenticated-orcid":false,"given":"Tian","family":"Tan","sequence":"additional","affiliation":[{"name":"Department of Civil and Environmental Engineering, Stanford University, Stanford, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4907-7354","authenticated-orcid":false,"given":"Xiaolin","family":"Hu","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Technology, Institute for Artificial Intelligence, Beijing National Research Center for Information Science and Technology, State Key Laboratory of Intelligent Technology and Systems, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4813-2494","authenticated-orcid":false,"given":"Feng","family":"Chen","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/S0004-3702(99)00052-1"},{"key":"ref2","article-title":"Temporal abstraction in reinforcement learning","author":"Precup","year":"2000"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1023\/A:1022140919877"},{"key":"ref4","first-page":"271","article-title":"Feudal reinforcement learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Dayan"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/3116.003.0027"},{"key":"ref6","first-page":"3675","article-title":"Hierarchical deep reinforcement learning: Integrating temporal abstraction and intrinsic motivation","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Kulkarni"},{"key":"ref7","first-page":"3540","article-title":"FeUdal networks for hierarchical reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Vezhnevets"},{"key":"ref8","first-page":"3307","article-title":"Data-efficient hierarchical reinforcement learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Nachum"},{"key":"ref9","article-title":"Learning multi-level hierarchies with hindsight","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Levy"},{"key":"ref10","article-title":"Near-optimal representation learning for hierarchical reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Nachum"},{"key":"ref11","first-page":"1940","article-title":"Mapping state space using landmarks for universal goal reaching","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Huang"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2021.3069005"},{"key":"ref13","first-page":"3566","article-title":"Learn what not to learn: Action elimination with deep reinforcement learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Zahavy"},{"key":"ref14","article-title":"Q-learning in enormous action spaces via amortized approximate maximization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"van de Wiele"},{"key":"ref15","first-page":"5243","article-title":"What can I do here? A theory of affordances in reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Khetarpal"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref17","article-title":"Stochastic neural networks for hierarchical reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Florensa"},{"key":"ref18","first-page":"21579","article-title":"Generating adjacency-constrained subgoals in hierarchical reinforcement learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1002\/SERIES1345"},{"issue":"51","key":"ref20","first-page":"1563","article-title":"Near-optimal regret bounds for reinforcement learning","volume":"11","author":"Jaksch","year":"2010","journal-title":"J. Mach. Learn. Res."},{"key":"ref21","first-page":"1184","article-title":"Optimistic posterior sampling for reinforcement learning: Worst-case regret bounds","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Agrawal"},{"key":"ref22","first-page":"576","article-title":"Exploration-exploitation in MDPs with options","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Fruit"},{"key":"ref23","first-page":"3692","article-title":"POLITEX: Regret bounds for policy iteration using expert prediction","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Abbasi-Yadkori"},{"key":"ref24","first-page":"3007","article-title":"Learning infinite-horizon average-reward MDPs with linear function approximation","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Wei"},{"key":"ref25","article-title":"Continuous control with deep reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lillicrap"},{"key":"ref26","article-title":"Understanding domain randomization for sim-to-real transfer","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Chen"},{"key":"ref27","first-page":"5048","article-title":"Hindsight experience replay","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Andrychowicz"},{"key":"ref28","article-title":"Self-supervised learning of image embedding for continuous control","author":"Florensa","year":"2019"},{"key":"ref29","first-page":"15220","article-title":"Search on the replay buffer: Bridging planning and reinforcement learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Eysenbach"},{"key":"ref30","article-title":"Dynamical distance learning for semi-supervised and unsupervised skill discovery","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hartikainen"},{"key":"ref31","first-page":"1312","article-title":"Universal value function approximators","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schaul"},{"key":"ref32","article-title":"Temporal difference models: Model-free deep RL for model-based control","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Pong"},{"key":"ref33","article-title":"Semi-parametric topological memory for navigation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Savinov"},{"key":"ref34","article-title":"Episodic curiosity through reachability","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Savinov"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2006.100"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-020-03157-9"},{"key":"ref37","first-page":"6306","article-title":"Neural discrete representation learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Oord"},{"key":"ref38","first-page":"1329","article-title":"Benchmarking deep reinforcement learning for continuous control","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Duan"},{"key":"ref39","article-title":"Multi-goal reinforcement learning: Challenging robotics environments and request for research","author":"Plappert","year":"2018"},{"key":"ref40","first-page":"1582","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref41","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mnih"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8460528"},{"key":"ref43","first-page":"361","article-title":"Automatic discovery of subgoals in reinforcement learning using diverse density","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"McGovern"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/1102351.1102454"},{"key":"ref45","article-title":"Deep successor reinforcement learning","author":"Kulkarni","year":"2016"},{"key":"ref46","first-page":"17","article-title":"Unsupervised methods for subgoal discovery during intrinsic motivation in model-free hierarchical reinforcement learning","volume-title":"Proc. 2nd Workshop Knowl. Extraction Games Co-Located 33rd AAAI Conf. Artif. Intell.","author":"Rafati"},{"key":"ref47","first-page":"5837","article-title":"Composable planning with attributes","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhang"},{"key":"ref48","first-page":"14814","article-title":"Planning with goal-conditioned policies","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Nasiriany"},{"key":"ref49","first-page":"1514","article-title":"Automatic goal generation for reinforcement learning agents","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Florensa"},{"key":"ref50","first-page":"9209","article-title":"Visual reinforcement learning with imagined goals","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Nair"},{"key":"ref51","first-page":"13464","article-title":"Exploration via hindsight goal generation","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Ren"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1993.5.4.613"},{"key":"ref53","first-page":"162","article-title":"Metrics for finite Markov decision processes","volume-title":"Proc. 20th Conf. Uncertainty Artif. Intell.","author":"Ferns"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i06.6564"},{"key":"ref55","article-title":"Why does hierarchy (sometimes) work so well in reinforcement learning?","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst. Workshop","author":"Nachum"},{"key":"ref56","article-title":"\u03b2-VAE: Learning basic visual concepts with a constrained variational framework","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Higgins"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.2307\/j.ctt4cgngj.10"},{"key":"ref58","first-page":"8024","article-title":"PyTorch: An imperative style, high-performance deep learning library","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Paszke"},{"key":"ref59","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kingma"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/10061515\/09833270.pdf?arnumber=9833270","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T09:34:14Z","timestamp":1725960854000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9833270\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,4,1]]},"references-count":59,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2022.3192418","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,4,1]]}}}