{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T21:54:02Z","timestamp":1785275642432,"version":"3.55.0"},"reference-count":32,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T00:00:00Z","timestamp":1737417600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T00:00:00Z","timestamp":1737417600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,1,21]]},"DOI":"10.1109\/sii59315.2025.10870988","type":"proceedings-article","created":{"date-parts":[[2025,2,12]],"date-time":"2025-02-12T18:17:07Z","timestamp":1739384227000},"page":"29-36","source":"Crossref","is-referenced-by-count":2,"title":["Leveraging Symbolic Models in Reinforcement Learning for Multi-skill Chaining<sup>*<\/sup>"],"prefix":"10.1109","author":[{"given":"Wenhao","family":"Lu","sequence":"first","affiliation":[{"name":"Chalmers University of Technology,Faculty of Electrical Engineering,Gothenburg,Sweden,SE-412 96"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Maximilian","family":"Diehl","sequence":"additional","affiliation":[{"name":"Chalmers University of Technology,Faculty of Electrical Engineering,Gothenburg,Sweden,SE-412 96"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jonas","family":"Sj\u00f6berg","sequence":"additional","affiliation":[{"name":"Chalmers University of Technology,Faculty of Electrical Engineering,Gothenburg,Sweden,SE-412 96"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Karinne","family":"Ramirez-Amaro","sequence":"additional","affiliation":[{"name":"Chalmers University of Technology,Faculty of Electrical Engineering,Gothenburg,Sweden,SE-412 96"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9812140"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.32657\/10356\/90191"},{"key":"ref3","article-title":"Proximal policy optimization algorithms","volume-title":"CoRR","author":"Schulman","year":"2017"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00447"},{"key":"ref5","volume-title":"Automated Planning: Theory and Practice","author":"Ghallab","year":"2004"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2014.6943079"},{"key":"ref7","first-page":"406","article-title":"Adversarial skill chaining for long-horizon robot manipulation via terminal state regularization","volume-title":"Conference on Robot Learning","author":"Lee"},{"key":"ref8","article-title":"Multi-skill mobile manipulation for object rearrangement","volume-title":"The Eleventh International Conference on Learning Representations","author":"Gu"},{"key":"ref9","first-page":"1312","article-title":"Universal value function approximators","volume-title":"Proceedings of the 32nd International Conference on Machine Learning","volume":"37","author":"Schaul"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196619"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/IROS51168.2021.9636781"},{"key":"ref12","article-title":"Pddl-the planning domain definition language","author":"Ghallab","year":"1998"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-77949-0_6"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/675"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33012970"},{"key":"ref16","article-title":"Openai gym","author":"Brockman","year":"2016"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/case59546.2024.10711595"},{"key":"ref18","first-page":"188","article-title":"Accelerating reinforcement learning with learned skill priors","volume-title":"Conference on robot learning","author":"Pertsch"},{"key":"ref19","first-page":"2095","article-title":"Residual skill policies: Learning an adaptable skill-based action space for reinforcement learning for robotics","volume-title":"Conference on Robot Learning","author":"Rana"},{"issue":"1","key":"ref20","first-page":"172","article-title":"Hierarchical reinforcement learning: A survey and open research challenges","volume-title":"Machine Learning and Knowledge Extraction","volume":"4","author":"Hutsebaut-Buysse","year":"2022"},{"key":"ref21","first-page":"3540","article-title":"Feudal networks for hierarchical reinforcement learning","volume-title":"International conference on machine learning","author":"Vezhnevets"},{"key":"ref22","first-page":"1851","article-title":"Latent space policies for hierarchical reinforcement learning","volume-title":"International Conference on Machine Learning","author":"Haarnoja"},{"key":"ref23","volume-title":"Artificial Intelligence: A Modern Approach","author":"Russell","year":"2010"},{"key":"ref24","first-page":"726","article-title":"Transporter networks: Rearranging the visual world for robotic manipulation","volume-title":"Conference on Robot Learning","author":"Zeng"},{"key":"ref25","first-page":"251","article-title":"Habitat 2.0: Training home assistants to rearrange their habitat","volume":"34","author":"Szot","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref26","article-title":"Composing complex skills by learning transition policies","volume-title":"Proceedings of International Conference on Learning Representations","author":"Lee"},{"issue":"3","key":"ref27","first-page":"189","article-title":"Strips: A new approach to the application of theorem proving to problem solving","volume-title":"Artificial Intelligence","volume":"2","author":"Fikes","year":"1971"},{"key":"ref28","article-title":"Curiosity-driven multi-criteria hindsight experience replay","volume-title":"CoRR","author":"Lanier","year":"2019"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1705"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/3583136"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref32","article-title":"Multi-goal reinforcement learning: Challenging robotics environments and request for research","volume-title":"CoRR","author":"Plappert","year":"2018"}],"event":{"name":"2025 IEEE\/SICE International Symposium on System Integration (SII)","location":"Munich, Germany","start":{"date-parts":[[2025,1,21]]},"end":{"date-parts":[[2025,1,24]]}},"container-title":["2025 IEEE\/SICE International Symposium on System Integration (SII)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10870372\/10870581\/10870988.pdf?arnumber=10870988","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,20]],"date-time":"2025-02-20T19:49:03Z","timestamp":1740080943000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10870988\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,21]]},"references-count":32,"URL":"https:\/\/doi.org\/10.1109\/sii59315.2025.10870988","relation":{},"subject":[],"published":{"date-parts":[[2025,1,21]]}}}