{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T11:40:03Z","timestamp":1755862803671,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":23,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,2,2]],"date-time":"2024-02-02T00:00:00Z","timestamp":1706832000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China (NSFC)","award":["U19A2059"],"award-info":[{"award-number":["U19A2059"]}]},{"name":"Sichuan Science and Technology Program","award":["No. 206999977"],"award-info":[{"award-number":["No. 206999977"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,2,2]]},"DOI":"10.1145\/3651671.3651714","type":"proceedings-article","created":{"date-parts":[[2024,6,7]],"date-time":"2024-06-07T18:55:50Z","timestamp":1717786550000},"page":"29-35","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Past Data-Driven Adaptation in Hierarchical Reinforcement Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-2751-7607","authenticated-orcid":false,"given":"Sijie","family":"Zhang","sequence":"first","affiliation":[{"name":"School of Computer Science &amp; Engineering, University of Electronic Science and Technology of China: Chengdu, CN, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3712-2349","authenticated-orcid":false,"given":"Aiguo","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Computer Science &amp; Engineering, University of Electronic Science and Technology of China: Chengdu, CN, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5338-8648","authenticated-orcid":false,"given":"Xincen","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Computer Science &amp; Engineering, University of Electronic Science and Technology of China: Chengdu, CN, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1667-9190","authenticated-orcid":false,"given":"Tianzi","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computer Science &amp; Engineering, University of Electronic Science and Technology of China: Chengdu, CN, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,6,7]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Hindsight Experience Replay. CoRR abs\/1707.01495","author":"Andrychowicz Marcin","year":"2017","unstructured":"Marcin Andrychowicz, Filip Wolski, Alex Ray, Jonas Schneider, Rachel Fong, Peter Welinder, Bob McGrew, Josh Tobin, Pieter Abbeel, and Wojciech Zaremba. 2017. Hindsight Experience Replay. CoRR abs\/1707.01495 (2017). arXiv:1707.01495http:\/\/arxiv.org\/abs\/1707.01495"},{"key":"e_1_3_2_1_2_1","volume-title":"Hindsight experience replay. Advances in neural information processing systems 30","author":"Andrychowicz Marcin","year":"2017","unstructured":"Marcin Andrychowicz, Filip Wolski, Alex Ray, Jonas Schneider, Rachel Fong, Peter Welinder, Bob McGrew, Josh Tobin, OpenAI Pieter\u00a0Abbeel, and Wojciech Zaremba. 2017. Hindsight experience replay. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_3_1","volume-title":"Proc. of the 8-th Conf. on Intelligent Autonomous Systems. 438\u2013445","author":"Bakker Bram","year":"2004","unstructured":"Bram Bakker, J\u00fcrgen Schmidhuber, 2004. Hierarchical reinforcement learning based on subgoal discovery and subpolicy specialization. In Proc. of the 8-th Conf. on Intelligent Autonomous Systems. 438\u2013445."},{"key":"e_1_3_2_1_4_1","volume-title":"Prioritized sequence experience replay. arXiv preprint arXiv:1905.12726","author":"Brittain Marc","year":"2019","unstructured":"Marc Brittain, Josh Bertram, Xuxi Yang, and Peng Wei. 2019. Prioritized sequence experience replay. arXiv preprint arXiv:1905.12726 (2019)."},{"key":"e_1_3_2_1_5_1","volume-title":"Possibility Before Utility: Learning And Using Hierarchical Affordances. arXiv preprint arXiv:2203.12686","author":"Costales Robby","year":"2022","unstructured":"Robby Costales, Shariq Iqbal, and Fei Sha. 2022. Possibility Before Utility: Learning And Using Hierarchical Affordances. arXiv preprint arXiv:2203.12686 (2022)."},{"key":"e_1_3_2_1_6_1","first-page":"67","article-title":"The theory of affordances","volume":"1","author":"Gibson J","year":"1977","unstructured":"James\u00a0J Gibson. 1977. The theory of affordances. Hilldale, USA 1, 2 (1977), 67\u201382.","journal-title":"Hilldale, USA"},{"key":"e_1_3_2_1_7_1","volume-title":"Dealing with Sparse Rewards in Reinforcement Learning. CoRR abs\/1910.09281","author":"Hare Joshua","year":"2019","unstructured":"Joshua Hare. 2019. Dealing with Sparse Rewards in Reinforcement Learning. CoRR abs\/1910.09281 (2019). arXiv:1910.09281http:\/\/arxiv.org\/abs\/1910.09281"},{"key":"e_1_3_2_1_8_1","volume-title":"Mapping state space using landmarks for universal goal reaching. Advances in Neural Information Processing Systems 32","author":"Huang Zhiao","year":"2019","unstructured":"Zhiao Huang, Fangchen Liu, and Hao Su. 2019. Mapping state space using landmarks for universal goal reaching. Advances in Neural Information Processing Systems 32 (2019)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.3390\/make4010009"},{"key":"e_1_3_2_1_10_1","volume-title":"Model-Based Reinforcement Learning for Atari. CoRR abs\/1903.00374","author":"Kaiser Lukasz","year":"2019","unstructured":"Lukasz Kaiser, Mohammad Babaeizadeh, Piotr Milos, Blazej Osinski, Roy\u00a0H. Campbell, Konrad Czechowski, Dumitru Erhan, Chelsea Finn, Piotr Kozakowski, Sergey Levine, Ryan Sepassi, George Tucker, and Henryk Michalewski. 2019. Model-Based Reinforcement Learning for Atari. CoRR abs\/1903.00374 (2019). arXiv:1903.00374http:\/\/arxiv.org\/abs\/1903.00374"},{"key":"e_1_3_2_1_11_1","volume-title":"International Conference on Machine Learning. PMLR, 5243\u20135253","author":"Khetarpal Khimya","year":"2020","unstructured":"Khimya Khetarpal, Zafarali Ahmed, Gheorghe Comanici, David Abel, and Doina Precup. 2020. What can I do here? A Theory of Affordances in Reinforcement Learning. In International Conference on Machine Learning. PMLR, 5243\u20135253."},{"key":"e_1_3_2_1_12_1","first-page":"28336","article-title":"Landmark-guided subgoal generation in hierarchical reinforcement learning","volume":"34","author":"Kim Junsu","year":"2021","unstructured":"Junsu Kim, Younggyo Seo, and Jinwoo Shin. 2021. Landmark-guided subgoal generation in hierarchical reinforcement learning. Advances in Neural Information Processing Systems 34 (2021), 28336\u201328349.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_13_1","volume-title":"Hierarchical deep reinforcement learning: Integrating temporal abstraction and intrinsic motivation. Advances in neural information processing systems 29","author":"Kulkarni D","year":"2016","unstructured":"Tejas\u00a0D Kulkarni, Karthik Narasimhan, Ardavan Saeedi, and Josh Tenenbaum. 2016. Hierarchical deep reinforcement learning: Integrating temporal abstraction and intrinsic motivation. Advances in neural information processing systems 29 (2016)."},{"key":"e_1_3_2_1_14_1","first-page":"18560","article-title":"Discor: Corrective feedback in reinforcement learning via distribution correction","volume":"33","author":"Kumar Aviral","year":"2020","unstructured":"Aviral Kumar, Abhishek Gupta, and Sergey Levine. 2020. Discor: Corrective feedback in reinforcement learning via distribution correction. Advances in Neural Information Processing Systems 33 (2020), 18560\u201318572.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2021.1004141"},{"key":"e_1_3_2_1_16_1","unstructured":"Amy McGovern and Andrew\u00a0G Barto. 2001. Automatic discovery of subgoals in reinforcement learning using diverse density. (2001)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.330110009"},{"key":"e_1_3_2_1_18_1","volume-title":"Prioritized experience replay. arXiv preprint arXiv:1511.05952","author":"Schaul Tom","year":"2015","unstructured":"Tom Schaul, John Quan, Ioannis Antonoglou, and David Silver. 2015. Prioritized experience replay. arXiv preprint arXiv:1511.05952 (2015)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/1102351.1102454"},{"key":"e_1_3_2_1_20_1","volume-title":"Learning for Dynamics and Control Conference. PMLR, 110\u2013123","author":"Sinha Samarth","year":"2022","unstructured":"Samarth Sinha, Jiaming Song, Animesh Garg, and Stefano Ermon. 2022. Experience replay with likelihood-free importance weights. In Learning for Dynamics and Control Conference. PMLR, 110\u2013123."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45622-8_16"},{"key":"e_1_3_2_1_22_1","volume-title":"International Conference on Machine Learning. PMLR, 3540\u20133549","author":"Vezhnevets Alexander\u00a0Sasha","year":"2017","unstructured":"Alexander\u00a0Sasha Vezhnevets, Simon Osindero, Tom Schaul, Nicolas Heess, Max Jaderberg, David Silver, and Koray Kavukcuoglu. 2017. Feudal networks for hierarchical reinforcement learning. In International Conference on Machine Learning. PMLR, 3540\u20133549."},{"key":"e_1_3_2_1_23_1","volume-title":"International Conference on Machine Learning. PMLR, 10070\u201310080","author":"Wang Che","year":"2020","unstructured":"Che Wang, Yanqiu Wu, Quan Vuong, and Keith Ross. 2020. Striving for simplicity and performance in off-policy DRL: Output normalization and non-uniform sampling. In International Conference on Machine Learning. PMLR, 10070\u201310080."}],"event":{"name":"ICMLC 2024: 2024 16th International Conference on Machine Learning and Computing","acronym":"ICMLC 2024","location":"Shenzhen China"},"container-title":["Proceedings of the 2024 16th International Conference on Machine Learning and Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3651671.3651714","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3651671.3651714","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T11:20:53Z","timestamp":1755861653000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3651671.3651714"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,2]]},"references-count":23,"alternative-id":["10.1145\/3651671.3651714","10.1145\/3651671"],"URL":"https:\/\/doi.org\/10.1145\/3651671.3651714","relation":{},"subject":[],"published":{"date-parts":[[2024,2,2]]},"assertion":[{"value":"2024-06-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}