{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T17:40:43Z","timestamp":1779385243274,"version":"3.53.1"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,8,4]],"date-time":"2023-08-04T00:00:00Z","timestamp":1691107200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["IIS-1909702"],"award-info":[{"award-number":["IIS-1909702"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000183","name":"Army Research Office","doi-asserted-by":"publisher","award":["W911NF21-1-0198"],"award-info":[{"award-number":["W911NF21-1-0198"]}],"id":[{"id":"10.13039\/100000183","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,8,6]]},"DOI":"10.1145\/3580305.3599506","type":"proceedings-article","created":{"date-parts":[[2023,8,4]],"date-time":"2023-08-04T18:13:58Z","timestamp":1691172838000},"page":"3513-3524","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Skill Disentanglement for Imitation Learning from Suboptimal Demonstrations"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4504-7809","authenticated-orcid":false,"given":"Tianxiang","family":"Zhao","sequence":"first","affiliation":[{"name":"The Pennsylvania State University, State College, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2480-448X","authenticated-orcid":false,"given":"Wenchao","family":"Yu","sequence":"additional","affiliation":[{"name":"NEC-Labs America, Princeton, NJ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3448-4878","authenticated-orcid":false,"given":"Suhang","family":"Wang","sequence":"additional","affiliation":[{"name":"The Pennsylvania State University, State College, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7305-1496","authenticated-orcid":false,"given":"Lu","family":"Wang","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0940-6595","authenticated-orcid":false,"given":"Xiang","family":"Zhang","sequence":"additional","affiliation":[{"name":"The Pennsylvania State University, State College, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5111-3716","authenticated-orcid":false,"given":"Yuncong","family":"Chen","sequence":"additional","affiliation":[{"name":"NEC-Labs America, Princeton, NJ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4396-5139","authenticated-orcid":false,"given":"Yanchi","family":"Liu","sequence":"additional","affiliation":[{"name":"NEC-Labs America, Princeton, NJ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5456-626X","authenticated-orcid":false,"given":"Wei","family":"Cheng","sequence":"additional","affiliation":[{"name":"NEC-Labs America, Princeton, NJ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9363-738X","authenticated-orcid":false,"given":"Haifeng","family":"Chen","sequence":"additional","affiliation":[{"name":"NEC-Labs America, Princeton, NJ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,8,4]]},"reference":[{"key":"e_1_3_2_2_1_1","article-title":"Deep reinforcement learning for autonomous driving: A survey","author":"Kiran B. R.","year":"2021","unstructured":"B. R. Kiran , I. Sobh , V. Talpaert , P. Mannion , A. A. Al Sallab , S. Yogamani , and P. P\u00e9rez , \" Deep reinforcement learning for autonomous driving: A survey ,\" IEEE Transactions on Intelligent Transportation Systems , 2021 . B. R. Kiran, I. Sobh, V. Talpaert, P. Mannion, A. A. Al Sallab, S. Yogamani, and P. P\u00e9rez, \"Deep reinforcement learning for autonomous driving: A survey,\" IEEE Transactions on Intelligent Transportation Systems, 2021.","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"e_1_3_2_2_2_1","first-page":"332","volume-title":"Towards adaptive social behavior generation for assistive robots using reinforcement learning,\" in Proceedings of the 2017 ACM\/IEEE International Conference on Human-Robot Interaction","author":"Hemminghaus J.","year":"2017","unstructured":"J. Hemminghaus and S. Kopp , \" Towards adaptive social behavior generation for assistive robots using reinforcement learning,\" in Proceedings of the 2017 ACM\/IEEE International Conference on Human-Robot Interaction , 2017 , pp. 332 -- 340 . J. Hemminghaus and S. Kopp, \"Towards adaptive social behavior generation for assistive robots using reinforcement learning,\" in Proceedings of the 2017 ACM\/IEEE International Conference on Human-Robot Interaction, 2017, pp. 332--340."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6514"},{"key":"e_1_3_2_2_4_1","volume-title":"Transmart: A practical interactive machine translation system,\" arXiv preprint arXiv:2105.13072","author":"Huang G.","year":"2021","unstructured":"G. Huang , L. Liu , X. Wang , L. Wang , H. Li , Z. Tu , C. Huang , and S. Shi , \" Transmart: A practical interactive machine translation system,\" arXiv preprint arXiv:2105.13072 , 2021 . G. Huang, L. Liu, X. Wang, L. Wang, H. Li, Z. Tu, C. Huang, and S. Shi, \"Transmart: A practical interactive machine translation system,\" arXiv preprint arXiv:2105.13072, 2021."},{"key":"e_1_3_2_2_5_1","unstructured":"D. Pomerleau \"An autonomous land vehicle in a neural network \" Advances in Neural Information Processing Systems vol. 1 1998.  D. Pomerleau \"An autonomous land vehicle in a neural network \" Advances in Neural Information Processing Systems vol. 1 1998."},{"key":"e_1_3_2_2_6_1","volume-title":"Learning robust rewards with adverserial inverse reinforcement learning,\" in International Conference on Learning Representations","author":"Fu J.","year":"2018","unstructured":"J. Fu , K. Luo , and S. Levine , \" Learning robust rewards with adverserial inverse reinforcement learning,\" in International Conference on Learning Representations , 2018 . J. Fu, K. Luo, and S. Levine, \"Learning robust rewards with adverserial inverse reinforcement learning,\" in International Conference on Learning Representations, 2018."},{"key":"e_1_3_2_2_7_1","volume-title":"Generative adversarial imitation learning,\" Advances in neural information processing systems","author":"Ho J.","year":"2016","unstructured":"J. Ho and S. Ermon , \" Generative adversarial imitation learning,\" Advances in neural information processing systems , vol. 29 , 2016 . J. Ho and S. Ermon, \"Generative adversarial imitation learning,\" Advances in neural information processing systems, vol. 29, 2016."},{"key":"e_1_3_2_2_8_1","first-page":"3406","volume-title":"IEEE","author":"Pinto L.","year":"2016","unstructured":"L. Pinto and A. Gupta , \" Supersizing self-supervision: Learning to grasp from 50k tries and 700 robot hours,\" in 2016 IEEE international conference on robotics and automation (ICRA) . IEEE , 2016 , pp. 3406 -- 3413 . L. Pinto and A. Gupta, \"Supersizing self-supervision: Learning to grasp from 50k tries and 700 robot hours,\" in 2016 IEEE international conference on robotics and automation (ICRA). IEEE, 2016, pp. 3406--3413."},{"key":"e_1_3_2_2_9_1","first-page":"24","volume-title":"PMLR","author":"Xu H.","year":"2022","unstructured":"H. Xu , X. Zhan , H. Yin , and H. Qin , \" Discriminator-weighted offline imitation learning from suboptimal demonstrations,\" in International Conference on Machine Learning . PMLR , 2022 , pp. 24 725--24 742. H. Xu, X. Zhan, H. Yin, and H. Qin, \"Discriminator-weighted offline imitation learning from suboptimal demonstrations,\" in International Conference on Machine Learning. PMLR, 2022, pp. 24 725--24 742."},{"key":"e_1_3_2_2_10_1","first-page":"1048","article-title":"Scaling robot supervision to hundreds of hours with roboturk: Robotic manipulation dataset through human reasoning and dexterity,\" in 2019 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","author":"Mandlekar A.","year":"2019","unstructured":"A. Mandlekar , J. Booher , M. Spero , A. Tung , A. Gupta , Y. Zhu , A. Garg , S. Savarese , and L. Fei-Fei , \" Scaling robot supervision to hundreds of hours with roboturk: Robotic manipulation dataset through human reasoning and dexterity,\" in 2019 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS) . IEEE , 2019 , pp. 1048 -- 1055 . A. Mandlekar, J. Booher, M. Spero, A. Tung, A. Gupta, Y. Zhu, A. Garg, S. Savarese, and L. Fei-Fei, \"Scaling robot supervision to hundreds of hours with roboturk: Robotic manipulation dataset through human reasoning and dexterity,\" in 2019 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS). IEEE, 2019, pp. 1048--1055.","journal-title":"IEEE"},{"key":"e_1_3_2_2_11_1","volume-title":"Confidence-aware imitation learning from demonstrations with varying optimality,\" Advances in Neural Information Processing Systems","author":"Zhang S.","year":"2021","unstructured":"S. Zhang , Z. Cao , D. Sadigh , and Y. Sui , \" Confidence-aware imitation learning from demonstrations with varying optimality,\" Advances in Neural Information Processing Systems , vol. 34 , 2021 . S. Zhang, Z. Cao, D. Sadigh, and Y. Sui, \"Confidence-aware imitation learning from demonstrations with varying optimality,\" Advances in Neural Information Processing Systems, vol. 34, 2021."},{"key":"e_1_3_2_2_12_1","volume-title":"Imitation learning by estimating expertise of demonstrators,\" arXiv preprint arXiv:2202.01288","author":"Beliaev M.","year":"2022","unstructured":"M. Beliaev , A. Shih , S. Ermon , D. Sadigh , and R. Pedarsani , \" Imitation learning by estimating expertise of demonstrators,\" arXiv preprint arXiv:2202.01288 , 2022 . M. Beliaev, A. Shih, S. Ermon, D. Sadigh, and R. Pedarsani, \"Imitation learning by estimating expertise of demonstrators,\" arXiv preprint arXiv:2202.01288, 2022."},{"key":"e_1_3_2_2_13_1","volume-title":"Dynamics-aware unsupervised discovery of skills,\" in International Conference on Learning Representations","author":"Sharma A.","year":"2019","unstructured":"A. Sharma , S. Gu , S. Levine , V. Kumar , and K. Hausman , \" Dynamics-aware unsupervised discovery of skills,\" in International Conference on Learning Representations , 2019 . A. Sharma, S. Gu, S. Levine, V. Kumar, and K. Hausman, \"Dynamics-aware unsupervised discovery of skills,\" in International Conference on Learning Representations, 2019."},{"key":"e_1_3_2_2_14_1","volume-title":"dissertation","author":"Ma Y. J.","year":"2020","unstructured":"Y. J. Ma , \" From adversarial imitation learning to robust batch imitation learning,\" Ph. D. dissertation , 2020 . Y. J. Ma, \"From adversarial imitation learning to robust batch imitation learning,\" Ph.D. dissertation, 2020."},{"key":"e_1_3_2_2_15_1","first-page":"1785","article-title":"Adversarial cooperative imitation learning for dynamic treatment regimes","volume":"2020","author":"Wang L.","year":"2020","unstructured":"L. Wang , W. Yu , X. He , W. Cheng , M. R. Ren , W. Wang , B. Zong , H. Chen , and H. Zha , \" Adversarial cooperative imitation learning for dynamic treatment regimes ,\" in Proceedings of The Web Conference 2020 , 2020 , pp. 1785 -- 1795 . L. Wang, W. Yu, X. He, W. Cheng, M. R. Ren, W. Wang, B. Zong, H. Chen, and H. Zha, \"Adversarial cooperative imitation learning for dynamic treatment regimes,\" in Proceedings of The Web Conference 2020, 2020, pp. 1785--1795.","journal-title":"Proceedings of The Web Conference"},{"key":"e_1_3_2_2_16_1","first-page":"1433","volume-title":"Dey et al., \"Maximum entropy inverse reinforcement learning.\" in Aaai","author":"Ziebart B. D.","year":"2008","unstructured":"B. D. Ziebart , A. L. Maas , J. A. Bagnell , A. K. Dey et al., \"Maximum entropy inverse reinforcement learning.\" in Aaai , vol. 8 . Chicago, IL , USA , 2008 , pp. 1433 -- 1438 . B. D. Ziebart, A. L. Maas, J. A. Bagnell, A. K. Dey et al., \"Maximum entropy inverse reinforcement learning.\" in Aaai, vol. 8. Chicago, IL, USA, 2008, pp. 1433--1438."},{"key":"e_1_3_2_2_17_1","volume-title":"Trail: Near-optimal imitation learning with suboptimal data,\" in International Conference on Learning Representations","author":"Yang M.","year":"2021","unstructured":"M. Yang , S. Levine , and O. Nachum , \" Trail: Near-optimal imitation learning with suboptimal data,\" in International Conference on Learning Representations , 2021 . M. Yang, S. Levine, and O. Nachum, \"Trail: Near-optimal imitation learning with suboptimal data,\" in International Conference on Learning Representations, 2021."},{"key":"e_1_3_2_2_18_1","volume-title":"Behavioral cloning from noisy demonstrations,\" in International Conference on Learning Representations","author":"Sasaki F.","year":"2020","unstructured":"F. Sasaki and R. Yamashina , \" Behavioral cloning from noisy demonstrations,\" in International Conference on Learning Representations , 2020 . F. Sasaki and R. Yamashina, \"Behavioral cloning from noisy demonstrations,\" in International Conference on Learning Representations, 2020."},{"key":"e_1_3_2_2_19_1","volume-title":"Preference-learning based inverse reinforcement learning for dialog control,\" in Thirteenth Annual Conference of the International Speech Communication Association","author":"Sugiyama H.","year":"2012","unstructured":"H. Sugiyama , T. Meguro , and Y. Minami , \" Preference-learning based inverse reinforcement learning for dialog control,\" in Thirteenth Annual Conference of the International Speech Communication Association , 2012 . H. Sugiyama, T. Meguro, and Y. Minami, \"Preference-learning based inverse reinforcement learning for dialog control,\" in Thirteenth Annual Conference of the International Speech Communication Association, 2012."},{"key":"e_1_3_2_2_20_1","volume-title":"Model-free preference-based reinforcement learning,\" in Thirtieth AAAI Conference on Artificial Intelligence","author":"Wirth C.","year":"2016","unstructured":"C. Wirth , J. F\u00fcrnkranz , and G. Neumann , \" Model-free preference-based reinforcement learning,\" in Thirtieth AAAI Conference on Artificial Intelligence , 2016 . C. Wirth, J. F\u00fcrnkranz, and G. Neumann, \"Model-free preference-based reinforcement learning,\" in Thirtieth AAAI Conference on Artificial Intelligence, 2016."},{"key":"e_1_3_2_2_21_1","first-page":"6818","volume-title":"PMLR","author":"Wu H.","year":"2019","unstructured":"Y.- H. Wu , N. Charoenphakdee , H. Bao , V. Tangkaratt , and M. Sugiyama , \" Imitation learning from imperfect demonstration,\" in International Conference on Machine Learning . PMLR , 2019 , pp. 6818 -- 6827 . Y.-H.Wu, N. Charoenphakdee, H. Bao, V. Tangkaratt, and M. Sugiyama, \"Imitation learning from imperfect demonstration,\" in International Conference on Machine Learning. PMLR, 2019, pp. 6818--6827."},{"key":"e_1_3_2_2_22_1","volume-title":"Feudal reinforcement learning,\" Advances in neural information processing systems","author":"Dayan P.","year":"1992","unstructured":"P. Dayan and G. E. Hinton , \" Feudal reinforcement learning,\" Advances in neural information processing systems , vol. 5 , 1992 . P. Dayan and G. E. Hinton, \"Feudal reinforcement learning,\" Advances in neural information processing systems, vol. 5, 1992."},{"issue":"1","key":"e_1_3_2_2_23_1","first-page":"2","article-title":"Between mdps and semi-mdps: A framework for temporal abstraction in reinforcement learning","volume":"112","author":"Sutton R. S.","year":"1999","unstructured":"R. S. Sutton , D. Precup , and S. Singh , \" Between mdps and semi-mdps: A framework for temporal abstraction in reinforcement learning ,\" Artificial intelligence , vol. 112 , no. 1 -- 2 , pp. 181--211, 1999 . R. S. Sutton, D. Precup, and S. Singh, \"Between mdps and semi-mdps: A framework for temporal abstraction in reinforcement learning,\" Artificial intelligence, vol. 112, no. 1--2, pp. 181--211, 1999.","journal-title":"Artificial intelligence"},{"key":"e_1_3_2_2_24_1","first-page":"3540","volume-title":"PMLR","author":"Vezhnevets A. S.","year":"2017","unstructured":"A. S. Vezhnevets , S. Osindero , T. Schaul , N. Heess , M. Jaderberg , D. Silver , and K. Kavukcuoglu , \" Feudal networks for hierarchical reinforcement learning,\" in International Conference on Machine Learning . PMLR , 2017 , pp. 3540 -- 3549 . A. S. Vezhnevets, S. Osindero, T. Schaul, N. Heess, M. Jaderberg, D. Silver, and K. Kavukcuoglu, \"Feudal networks for hierarchical reinforcement learning,\" in International Conference on Machine Learning. PMLR, 2017, pp. 3540--3549."},{"key":"e_1_3_2_2_25_1","volume-title":"The option-critic architecture,\" in Proceedings of the AAAI Conference on Artificial Intelligence","author":"Bacon P.-L.","unstructured":"P.-L. Bacon , J. Harb , and D. Precup , \" The option-critic architecture,\" in Proceedings of the AAAI Conference on Artificial Intelligence , vol. 31 , no. 1, 2017. P.-L. Bacon, J. Harb, and D. Precup, \"The option-critic architecture,\" in Proceedings of the AAAI Conference on Artificial Intelligence, vol. 31, no. 1, 2017."},{"key":"e_1_3_2_2_26_1","volume-title":"Diversity is all you need: Learning skills without a reward function,\" arXiv preprint arXiv:1802.06070","author":"Eysenbach B.","year":"2018","unstructured":"B. Eysenbach , A. Gupta , J. Ibarz , and S. Levine , \" Diversity is all you need: Learning skills without a reward function,\" arXiv preprint arXiv:1802.06070 , 2018 . B. Eysenbach, A. Gupta, J. Ibarz, and S. Levine, \"Diversity is all you need: Learning skills without a reward function,\" arXiv preprint arXiv:1802.06070, 2018."},{"key":"e_1_3_2_2_27_1","first-page":"1317","volume-title":"PMLR","author":"Campos V.","year":"2020","unstructured":"V. Campos , A. Trott , C. Xiong , R. Socher , X. Gir\u00f3-i Nieto , and J. Torres , \" Explore, discover and learn: Unsupervised discovery of state-covering skills,\" in International Conference on Machine Learning . PMLR , 2020 , pp. 1317 -- 1327 . V. Campos, A. Trott, C. Xiong, R. Socher, X. Gir\u00f3-i Nieto, and J. Torres, \"Explore, discover and learn: Unsupervised discovery of state-covering skills,\" in International Conference on Machine Learning. PMLR, 2020, pp. 1317--1327."},{"key":"e_1_3_2_2_28_1","volume-title":"Flexible option learning,\" Advances in Neural Information Processing Systems","author":"Klissarov M.","year":"2021","unstructured":"M. Klissarov and D. Precup , \" Flexible option learning,\" Advances in Neural Information Processing Systems , vol. 34 , 2021 . M. Klissarov and D. Precup, \"Flexible option learning,\" Advances in Neural Information Processing Systems, vol. 34, 2021."},{"key":"e_1_3_2_2_29_1","volume-title":"Discovery of options via meta-learned subgoals,\" Advances in Neural Information Processing Systems","author":"Veeriah V.","year":"2021","unstructured":"V. Veeriah , T. Zahavy , M. Hessel , Z. Xu , J. Oh , I. Kemaev , H. P. van Hasselt , D. Silver , and S. Singh , \" Discovery of options via meta-learned subgoals,\" Advances in Neural Information Processing Systems , vol. 34 , 2021 . V. Veeriah, T. Zahavy, M. Hessel, Z. Xu, J. Oh, I. Kemaev, H. P. van Hasselt, D. Silver, and S. Singh, \"Discovery of options via meta-learned subgoals,\" Advances in Neural Information Processing Systems, vol. 34, 2021."},{"key":"e_1_3_2_2_30_1","volume-title":"Reinforcement learning with hierarchies of machines,\" Advances in neural information processing systems","author":"Parr R.","year":"1997","unstructured":"R. Parr and S. Russell , \" Reinforcement learning with hierarchies of machines,\" Advances in neural information processing systems , vol. 10 , 1997 . R. Parr and S. Russell, \"Reinforcement learning with hierarchies of machines,\" Advances in neural information processing systems, vol. 10, 1997."},{"key":"e_1_3_2_2_31_1","first-page":"1504","volume-title":"Semi-supervised drifted stream learning with short lookback,\" in Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","author":"Ren W.","year":"2022","unstructured":"W. Ren , P. Wang , X. Li , C. E. Hughes , and Y. Fu , \" Semi-supervised drifted stream learning with short lookback,\" in Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining , 2022 , pp. 1504 -- 1513 . W. Ren, P. Wang, X. Li, C. E. Hughes, and Y. Fu, \"Semi-supervised drifted stream learning with short lookback,\" in Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 2022, pp. 1504--1513."},{"key":"e_1_3_2_2_32_1","first-page":"553","article-title":"Hierarchical skills for efficient exploration","volume":"34","author":"Gehring J.","year":"2021","unstructured":"J. Gehring , G. Synnaeve , A. Krause , and N. Usunier , \" Hierarchical skills for efficient exploration ,\" Advances in Neural Information Processing Systems , vol. 34 , pp. 11 553 -- 511 564, 2021 . J. Gehring, G. Synnaeve, A. Krause, and N. Usunier, \"Hierarchical skills for efficient exploration,\" Advances in Neural Information Processing Systems, vol. 34, pp. 11 553--11 564, 2021.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_33_1","first-page":"299","article-title":"The utility of temporal abstraction in reinforcement learning","author":"Jong N. K.","year":"2008","unstructured":"N. K. Jong , T. Hester , and P. Stone , \" The utility of temporal abstraction in reinforcement learning .\" in AAMAS (1). Citeseer , 2008 , pp. 299 -- 306 . N. K. Jong, T. Hester, and P. Stone, \"The utility of temporal abstraction in reinforcement learning.\" in AAMAS (1). Citeseer, 2008, pp. 299--306.","journal-title":"AAMAS (1). Citeseer"},{"key":"e_1_3_2_2_34_1","volume-title":"Hierarchical few-shot imitation with skill transition models,\" in International Conference on Learning Representations","author":"Hakhamaneshi K.","year":"2021","unstructured":"K. Hakhamaneshi , R. Zhao , A. Zhan , P. Abbeel , and M. Laskin , \" Hierarchical few-shot imitation with skill transition models,\" in International Conference on Learning Representations , 2021 . K. Hakhamaneshi, R. Zhao, A. Zhan, P. Abbeel, and M. Laskin, \"Hierarchical few-shot imitation with skill transition models,\" in International Conference on Learning Representations, 2021."},{"key":"e_1_3_2_2_35_1","volume-title":"Categorical reparametrization with gumble-softmax,\" in International Conference on Learning Representations (ICLR","author":"Jang E.","year":"2017","unstructured":"E. Jang , S. Gu , and B. Poole , \" Categorical reparametrization with gumble-softmax,\" in International Conference on Learning Representations (ICLR 2017 ). OpenReview . net, 2017. E. Jang, S. Gu, and B. Poole, \"Categorical reparametrization with gumble-softmax,\" in International Conference on Learning Representations (ICLR 2017). OpenReview. net, 2017."},{"key":"e_1_3_2_2_36_1","first-page":"1423","volume-title":"PMLR","author":"Celik O.","year":"2022","unstructured":"O. Celik , D. Zhou , G. Li , P. Becker , and G. Neumann , \" Specializing versatile skill libraries using local mixture of experts,\" in Conference on Robot Learning . PMLR , 2022 , pp. 1423 -- 1433 . O. Celik, D. Zhou, G. Li, P. Becker, and G. Neumann, \"Specializing versatile skill libraries using local mixture of experts,\" in Conference on Robot Learning. PMLR, 2022, pp. 1423--1433."},{"key":"e_1_3_2_2_37_1","first-page":"531","volume-title":"PMLR","author":"Belghazi M. I.","year":"2018","unstructured":"M. I. Belghazi , A. Baratin , S. Rajeshwar , S. Ozair , Y. Bengio , A. Courville , and D. Hjelm , \" Mutual information neural estimation,\" in International conference on machine learning . PMLR , 2018 , pp. 531 -- 540 . M. I. Belghazi, A. Baratin, S. Rajeshwar, S. Ozair, Y. Bengio, A. Courville, and D. Hjelm, \"Mutual information neural estimation,\" in International conference on machine learning. PMLR, 2018, pp. 531--540."},{"key":"e_1_3_2_2_38_1","volume-title":"Learning deep representations by mutual information estimation and maximization,\" in International Conference on Learning Representations","author":"Hjelm R. D.","year":"2018","unstructured":"R. D. Hjelm , A. Fedorov , S. Lavoie-Marchildon , K. Grewal , P. Bachman , A. Trischler , and Y. Bengio , \" Learning deep representations by mutual information estimation and maximization,\" in International Conference on Learning Representations , 2018 . R. D. Hjelm, A. Fedorov, S. Lavoie-Marchildon, K. Grewal, P. Bachman, A. Trischler, and Y. Bengio, \"Learning deep representations by mutual information estimation and maximization,\" in International Conference on Learning Representations, 2018."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539597.3570421"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-020-05877-5"},{"key":"e_1_3_2_2_41_1","first-page":"1028","article-title":"Exploring edge disentanglement for node classification","volume":"2022","author":"Zhao T.","year":"2022","unstructured":"T. Zhao , X. Zhang , and S. Wang , \" Exploring edge disentanglement for node classification ,\" in Proceedings of the ACM Web Conference 2022 , 2022 , pp. 1028 -- 1036 . T. Zhao, X. Zhang, and S. Wang, \"Exploring edge disentanglement for node classification,\" in Proceedings of the ACM Web Conference 2022, 2022, pp. 1028--1036.","journal-title":"Proceedings of the ACM Web Conference"},{"key":"e_1_3_2_2_42_1","volume-title":"Minimalistic gridworld environment for openai gym,\" https:\/\/github.com\/maximecb\/gym-minigrid","author":"Chevalier-Boisvert M.","year":"2018","unstructured":"M. Chevalier-Boisvert , L. Willems , and S. Pal , \" Minimalistic gridworld environment for openai gym,\" https:\/\/github.com\/maximecb\/gym-minigrid , 2018 . M. Chevalier-Boisvert, L. Willems, and S. Pal, \"Minimalistic gridworld environment for openai gym,\" https:\/\/github.com\/maximecb\/gym-minigrid, 2018."},{"key":"e_1_3_2_2_43_1","volume-title":"Predicting medications from diagnostic codes with recurrent neural networks","author":"Bajor J. M.","year":"2016","unstructured":"J. M. Bajor and T. A. Lasko , \" Predicting medications from diagnostic codes with recurrent neural networks ,\" 2016 . J. M. Bajor and T. A. Lasko, \"Predicting medications from diagnostic codes with recurrent neural networks,\" 2016."},{"key":"e_1_3_2_2_44_1","volume-title":"The imitation library for imitation learning and inverse reinforcement learning,\" https:\/\/github.com\/ HumanCompatibleAI\/imitation","author":"Wang S.","year":"2020","unstructured":"S. Wang , S. Toyer , A. Gleave , and S. Emmons , \" The imitation library for imitation learning and inverse reinforcement learning,\" https:\/\/github.com\/ HumanCompatibleAI\/imitation , 2020 . S. Wang, S. Toyer, A. Gleave, and S. Emmons, \"The imitation library for imitation learning and inverse reinforcement learning,\" https:\/\/github.com\/ HumanCompatibleAI\/imitation, 2020."},{"key":"e_1_3_2_2_45_1","volume-title":"Modeling interaction via the principle of maximum causal entropy,\" in ICML","author":"Ziebart B. D.","year":"2010","unstructured":"B. D. Ziebart , J. A. Bagnell , and A. K. Dey , \" Modeling interaction via the principle of maximum causal entropy,\" in ICML , 2010 . B. D. Ziebart, J. A. Bagnell, and A. K. Dey, \"Modeling interaction via the principle of maximum causal entropy,\" in ICML, 2010."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0031-3203(96)00142-2"},{"key":"e_1_3_2_2_47_1","volume-title":"Adam: A method for stochastic optimization,\" in ICLR (Poster)","author":"Kingma D. P.","year":"2015","unstructured":"D. P. Kingma and J. Ba , \" Adam: A method for stochastic optimization,\" in ICLR (Poster) , 2015 . D. P. Kingma and J. Ba, \"Adam: A method for stochastic optimization,\" in ICLR (Poster), 2015."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3488560.3498535"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1001\/jama.2016.0287"}],"event":{"name":"KDD '23: The 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Long Beach CA USA","acronym":"KDD '23","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3580305.3599506","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3580305.3599506","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3580305.3599506","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:37:52Z","timestamp":1750178272000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3580305.3599506"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,4]]},"references-count":49,"alternative-id":["10.1145\/3580305.3599506","10.1145\/3580305"],"URL":"https:\/\/doi.org\/10.1145\/3580305.3599506","relation":{},"subject":[],"published":{"date-parts":[[2023,8,4]]},"assertion":[{"value":"2023-08-04","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}