{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T23:33:09Z","timestamp":1780356789779,"version":"3.54.1"},"reference-count":52,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T00:00:00Z","timestamp":1743465600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T00:00:00Z","timestamp":1743465600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T00:00:00Z","timestamp":1743465600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62176056"],"award-info":[{"award-number":["62176056"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["92370132"],"award-info":[{"award-number":["92370132"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Young Elite Scientists Sponsorship Program by the China Association for Science and Technology","award":["2021QNRC001"],"award-info":[{"award-number":["2021QNRC001"]}]},{"name":"Southeast University Big Data Computing Center"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Emerg. Top. Comput. Intell."],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1109\/tetci.2025.3529902","type":"journal-article","created":{"date-parts":[[2025,1,28]],"date-time":"2025-01-28T13:52:28Z","timestamp":1738072348000},"page":"1292-1306","source":"Crossref","is-referenced-by-count":7,"title":["MENTOR: Guiding Hierarchical Reinforcement Learning With Human Feedback and Dynamic Distance Constraint"],"prefix":"10.1109","volume":"9","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-5149-4450","authenticated-orcid":false,"given":"Xinglin","family":"Zhou","sequence":"first","affiliation":[{"name":"Southeast University-Monash University Joint Graduate School, Southeast University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yifu","family":"Yuan","sequence":"additional","affiliation":[{"name":"College of Intelligence and Computing, Tianjin University, Tianjin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7727-9669","authenticated-orcid":false,"given":"Shaofu","family":"Yang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Southeast University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0422-8235","authenticated-orcid":false,"given":"Jianye","family":"Hao","sequence":"additional","affiliation":[{"name":"College of Intelligence and Computing, Tianjin University, Tianjin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"10092","article-title":"SQIL: Imitation learning via reinforcement learning with sparse rewards","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Reddy","year":"2020"},{"key":"ref2","article-title":"Leveraging demonstrations for deep reinforcement learning on robotics problems with sparse rewards","author":"Vecerik","year":"2017"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TETCI.2023.3303251"},{"key":"ref4","first-page":"4377","article-title":"Learning multi-level hierarchies with hindsight","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Levy","year":"2019"},{"key":"ref5","first-page":"3307","article-title":"Data-efficient hierarchical reinforcement learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Nachum","year":"2018"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.3390\/make4010009"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TETCI.2023.3309738"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1613\/jair.639"},{"key":"ref9","first-page":"1043","article-title":"Reinforcement learning with hierarchies of machines","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Parr","year":"1997"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3192418"},{"key":"ref11","first-page":"5227","article-title":"Diversity is all you need: Learning skills without a reward function","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Eysenbach","year":"2019"},{"key":"ref12","first-page":"27225","article-title":"Controllability-aware unsupervised skill discovery","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Park","year":"2023"},{"key":"ref13","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","author":"Ahn","year":"2022"},{"key":"ref14","first-page":"1317","article-title":"Explore, discover and learn: Unsupervised discovery of state-covering skills","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Campos","year":"2020"},{"key":"ref15","first-page":"5048","article-title":"Hindsight experience replay","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Andrychowicz","year":"2017"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-021-02726-3"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.10916"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11831"},{"key":"ref19","first-page":"7114","article-title":"Option discovery using deep skill chaining","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Bagaria","year":"2020"},{"key":"ref20","first-page":"1312","article-title":"Universal value function approximators","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schaul","year":"2015"},{"key":"ref21","first-page":"13207","article-title":"Dynamics-aware unsupervised discovery of skills","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Sharma","year":"2020"},{"key":"ref22","article-title":"Hierarchical empowerment: Towards tractable empowerment-based skill-learning","author":"Levy","year":"2023"},{"key":"ref23","first-page":"3120","article-title":"Finding options that minimize planning time","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Jinnai","year":"2019"},{"key":"ref24","first-page":"26091","article-title":"Deep hierarchical planning from pixels","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Hafner","year":"2022"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3243618"},{"key":"ref26","first-page":"28336","article-title":"Landmark-guided subgoal generation in hierarchical reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kim","year":"2021"},{"key":"ref27","article-title":"Hierarchical reinforcement learning with adversarially guided subgoals","author":"Wang","year":"2022"},{"key":"ref28","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brown","year":"2020"},{"key":"ref29","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ouyang","year":"2022"},{"key":"ref30","first-page":"6152","article-title":"PEBBLE: Feedback-efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Lee","year":"2021"},{"key":"ref31","first-page":"1","article-title":"B-PREF: Benchmarking preference-based reinforcement learning","volume-title":"Proc. Neural Inf. Process. Syst. Track Datasets Benchmarks","author":"Lee","year":"2021"},{"key":"ref32","first-page":"1","article-title":"SURF: Semi-supervised reward learning with data augmentation for feedback-efficient preference-based reinforcement learning","volume-title":"Proc. 10th Int. Conf. Learn. Representations","author":"Park","year":"2022"},{"key":"ref33","first-page":"53728","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume-title":"Proc. 37th Int. Conf. Neural Inf. Process. Syst.","author":"Rafailov","year":"2023"},{"key":"ref34","first-page":"1","article-title":"Safe RLHF: Safe reinforcement learning from human feedback","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Dai","year":"2024"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00200"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3063927"},{"issue":"58","key":"ref37","first-page":"1","article-title":"Estimation from pairwise comparisons: Sharp minimax bounds with topology dependence","volume":"17","author":"Shah","year":"2016","journal-title":"J. Mach. Learn. Res."},{"key":"ref38","first-page":"968","article-title":"Efficient and optimal algorithms for contextual dueling bandits under realizability","volume-title":"Proc. Int. Conf. Algorithmic Learn. Theory","author":"Saha","year":"2022"},{"key":"ref39","first-page":"1","article-title":"Exploration by random network distillation","volume-title":"Proc. 7th Int. Conf. Learn. Representations","author":"Burda","year":"2019"},{"key":"ref40","first-page":"1479","article-title":"Unifying count-based exploration and intrinsic motivation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Bellemare","year":"2016"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.2307\/2334029"},{"key":"ref43","article-title":"Automatic curriculum learning through value disagreement","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang","year":"2020"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3390051"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2024\/556"},{"key":"ref46","first-page":"7057","article-title":"Dynamical distance learning for semi-supervised and unsupervised skill discovery","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Hartikainen","year":"2020"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1007\/s10589-022-00358-y"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1038\/sj.jors.2600425"},{"key":"ref49","first-page":"6820","article-title":"On the global convergence rates of softmax policy gradient methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mei","year":"2020"},{"key":"ref50","article-title":"Multi-goal reinforcement learning: Challenging robotics environments and request for research","author":"Plappert","year":"2018"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.126620"},{"key":"ref52","first-page":"1","article-title":"Learning to reach goals via iterated supervised learning","volume-title":"Proc. 9th Int. Conf. Learn. Representations","author":"Ghosh","year":"2021"}],"container-title":["IEEE Transactions on Emerging Topics in Computational Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7433297\/10939044\/10856553.pdf?arnumber=10856553","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,26]],"date-time":"2025-11-26T19:06:05Z","timestamp":1764183965000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10856553\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4]]},"references-count":52,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tetci.2025.3529902","relation":{},"ISSN":["2471-285X"],"issn-type":[{"value":"2471-285X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,4]]}}}