{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T07:10:02Z","timestamp":1755846602394,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,12,23]],"date-time":"2022-12-23T00:00:00Z","timestamp":1671753600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,12,23]]},"DOI":"10.1145\/3579654.3579753","type":"proceedings-article","created":{"date-parts":[[2023,3,14]],"date-time":"2023-03-14T16:09:40Z","timestamp":1678810180000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Offline Imitation Learning Using Reward-free Exploratory Data"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2312-4247","authenticated-orcid":false,"given":"Hao","family":"Wang","sequence":"first","affiliation":[{"name":"College of Computer, National University of Defense Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7587-8905","authenticated-orcid":false,"given":"Dawei","family":"Feng","sequence":"additional","affiliation":[{"name":"National Laboratory for Parallel and Distributed Processing, National University of Defense Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1236-8318","authenticated-orcid":false,"given":"Bo","family":"Ding","sequence":"additional","affiliation":[{"name":"National Laboratory for Parallel and Distributed Processing, National University of Defense Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6457-0625","authenticated-orcid":false,"given":"Wei","family":"Li","sequence":"additional","affiliation":[{"name":"Independent Researcher, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,3,14]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"e_1_3_2_1_2_1","volume-title":"International conference on machine learning. PMLR, 214\u2013223","author":"Arjovsky Martin","year":"2017","unstructured":"Martin Arjovsky, Soumith Chintala, and L\u00e9on Bottou. 2017. Wasserstein generative adversarial networks. In International conference on machine learning. PMLR, 214\u2013223."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Michael Bain and Claude Sammut. 1995. A Framework for Behavioural Cloning.. In Machine Intelligence 15. 103\u2013129.","DOI":"10.1093\/oso\/9780198538677.003.0006"},{"key":"e_1_3_2_1_4_1","unstructured":"Yuri Burda Harrison Edwards Amos Storkey and Oleg Klimov. 2018. Exploration by random network distillation. arXiv preprint arXiv:1810.12894(2018)."},{"key":"e_1_3_2_1_5_1","first-page":"9912","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","volume":"33","author":"Caron Mathilde","year":"2020","unstructured":"Mathilde Caron, Ishan Misra, Julien Mairal, Priya Goyal, Piotr Bojanowski, and Armand Joulin. 2020. Unsupervised learning of visual features by contrasting cluster assignments. Advances in Neural Information Processing Systems 33 (2020), 9912\u20139924.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_6_1","unstructured":"Kamil Ciosek. 2021. Imitation learning by reinforcement learning. arXiv preprint arXiv:2108.04763(2021)."},{"key":"e_1_3_2_1_7_1","unstructured":"Benjamin Eysenbach Abhishek Gupta Julian Ibarz and Sergey Levine. 2018. Diversity is all you need: Learning skills without a reward function. arXiv preprint arXiv:1802.06070(2018)."},{"key":"e_1_3_2_1_8_1","unstructured":"Justin Fu Aviral Kumar Ofir Nachum George Tucker and Sergey Levine. 2020. D4rl: Datasets for deep data-driven reinforcement learning. arXiv preprint arXiv:2004.07219(2020)."},{"key":"e_1_3_2_1_9_1","volume-title":"A minimalist approach to offline reinforcement learning. Advances in neural information processing systems 34","author":"Fujimoto Scott","year":"2021","unstructured":"Scott Fujimoto and Shixiang\u00a0Shane Gu. 2021. A minimalist approach to offline reinforcement learning. Advances in neural information processing systems 34 (2021), 20132\u201320145."},{"key":"e_1_3_2_1_10_1","volume-title":"International conference on machine learning. PMLR, 1587\u20131596","author":"Fujimoto Scott","year":"2018","unstructured":"Scott Fujimoto, Herke Hoof, and David Meger. 2018. Addressing function approximation error in actor-critic methods. In International conference on machine learning. PMLR, 1587\u20131596."},{"key":"e_1_3_2_1_11_1","volume-title":"International conference on machine learning. PMLR","author":"Fujimoto Scott","year":"2019","unstructured":"Scott Fujimoto, David Meger, and Doina Precup. 2019. Off-policy deep reinforcement learning without exploration. In International conference on machine learning. PMLR, 2052\u20132062."},{"key":"e_1_3_2_1_12_1","first-page":"7248","article-title":"Rl unplugged: A suite of benchmarks for offline reinforcement learning","volume":"33","author":"Gulcehre Caglar","year":"2020","unstructured":"Caglar Gulcehre, Ziyu Wang, Alexander Novikov, Thomas Paine, Sergio G\u00f3mez, Konrad Zolna, Rishabh Agarwal, Josh\u00a0S Merel, Daniel\u00a0J Mankowitz, Cosmin Paduraru, 2020. Rl unplugged: A suite of benchmarks for offline reinforcement learning. Advances in Neural Information Processing Systems 33 (2020), 7248\u20137259.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_13_1","volume-title":"Generative adversarial imitation learning. Advances in neural information processing systems 29","author":"Ho Jonathan","year":"2016","unstructured":"Jonathan Ho and Stefano Ermon. 2016. Generative adversarial imitation learning. Advances in neural information processing systems 29 (2016)."},{"key":"e_1_3_2_1_14_1","unstructured":"Naveen Kodali Jacob Abernethy James Hays and Zsolt Kira. 2017. On convergence and stability of gans. arXiv preprint arXiv:1705.07215(2017)."},{"key":"e_1_3_2_1_15_1","unstructured":"Ilya Kostrikov Ofir Nachum and Jonathan Tompson. 2019. Imitation learning via off-policy distribution matching. arXiv preprint arXiv:1912.05032(2019)."},{"key":"e_1_3_2_1_16_1","first-page":"1179","article-title":"Conservative q-learning for offline reinforcement learning","volume":"33","author":"Kumar Aviral","year":"2020","unstructured":"Aviral Kumar, Aurick Zhou, George Tucker, and Sergey Levine. 2020. Conservative q-learning for offline reinforcement learning. Advances in Neural Information Processing Systems 33 (2020), 1179\u20131191.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_17_1","volume-title":"Reinforcement learning with augmented data. Advances in neural information processing systems 33","author":"Laskin Misha","year":"2020","unstructured":"Misha Laskin, Kimin Lee, Adam Stooke, Lerrel Pinto, Pieter Abbeel, and Aravind Srinivas. 2020. Reinforcement learning with augmented data. Advances in neural information processing systems 33 (2020), 19884\u201319895."},{"key":"e_1_3_2_1_18_1","volume-title":"URLB: Unsupervised reinforcement learning benchmark. arXiv preprint arXiv:2110.15191(2021).","author":"Laskin Michael","year":"2021","unstructured":"Michael Laskin, Denis Yarats, Hao Liu, Kimin Lee, Albert Zhan, Kevin Lu, Catherine Cang, Lerrel Pinto, and Pieter Abbeel. 2021. URLB: Unsupervised reinforcement learning benchmark. arXiv preprint arXiv:2110.15191(2021)."},{"key":"e_1_3_2_1_19_1","unstructured":"Lisa Lee Benjamin Eysenbach Emilio Parisotto Eric Xing Sergey Levine and Ruslan Salakhutdinov. 2019. Efficient exploration via state marginal matching. arXiv preprint arXiv:1906.05274(2019)."},{"key":"e_1_3_2_1_20_1","volume-title":"International Conference on Machine Learning. PMLR, 6736\u20136747","author":"Liu Hao","year":"2021","unstructured":"Hao Liu and Pieter Abbeel. 2021. Aps: Active pretraining with successor features. In International Conference on Machine Learning. PMLR, 6736\u20136747."},{"key":"e_1_3_2_1_21_1","first-page":"18459","article-title":"Behavior from the void: Unsupervised active pre-training","volume":"34","author":"Liu Hao","year":"2021","unstructured":"Hao Liu and Pieter Abbeel. 2021. Behavior from the void: Unsupervised active pre-training. Advances in Neural Information Processing Systems 34 (2021), 18459\u201318473.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"e_1_3_2_1_23_1","volume-title":"International conference on machine learning. PMLR, 5062\u20135071","author":"Pathak Deepak","year":"2019","unstructured":"Deepak Pathak, Dhiraj Gandhi, and Abhinav Gupta. 2019. Self-supervised exploration via disagreement. In International conference on machine learning. PMLR, 5062\u20135071."},{"key":"e_1_3_2_1_24_1","volume-title":"Sqil: Imitation learning via reinforcement learning with sparse rewards. arXiv preprint arXiv:1905.11108(2019).","author":"Reddy Siddharth","year":"2019","unstructured":"Siddharth Reddy, Anca\u00a0D Dragan, and Sergey Levine. 2019. Sqil: Imitation learning via reinforcement learning with sparse rewards. arXiv preprint arXiv:1905.11108(2019)."},{"key":"e_1_3_2_1_25_1","volume-title":"David Budden, Abbas Abdolmaleki","author":"Tassa Yuval","year":"2018","unstructured":"Yuval Tassa, Yotam Doron, Alistair Muldal, Tom Erez, Yazhe Li, Diego de\u00a0Las Casas, David Budden, Abbas Abdolmaleki, Josh Merel, Andrew Lefrancq, 2018. Deepmind control suite. arXiv preprint arXiv:1801.00690(2018)."},{"key":"e_1_3_2_1_26_1","volume-title":"Reinforcement Learning: An Introduction. In 2006 International Conference on Artificial Intelligence: 50 Years\u2019 Achievements, Future Directions and Social Impacts.","author":"Wang R.","year":"2006","unstructured":"R. Wang. 2006. Reinforcement Learning: An Introduction. In 2006 International Conference on Artificial Intelligence: 50 Years\u2019 Achievements, Future Directions and Social Impacts."},{"key":"e_1_3_2_1_27_1","first-page":"7768","article-title":"Critic regularized regression","volume":"33","author":"Wang Ziyu","year":"2020","unstructured":"Ziyu Wang, Alexander Novikov, Konrad Zolna, Josh\u00a0S Merel, Jost\u00a0Tobias Springenberg, Scott\u00a0E Reed, Bobak Shahriari, Noah Siegel, Caglar Gulcehre, Nicolas Heess, 2020. Critic regularized regression. Advances in Neural Information Processing Systems 33 (2020), 7768\u20137778.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_28_1","unstructured":"Denis Yarats David Brandfonbrener Hao Liu Michael Laskin Pieter Abbeel Alessandro Lazaric and Lerrel Pinto. 2022. Don\u2019t Change the Algorithm Change the Data: Exploratory Data for Offline Reinforcement Learning. arXiv preprint arXiv:2201.13425(2022)."},{"key":"e_1_3_2_1_29_1","volume-title":"International Conference on Machine Learning. PMLR, 11920\u201311931","author":"Yarats Denis","year":"2021","unstructured":"Denis Yarats, Rob Fergus, Alessandro Lazaric, and Lerrel Pinto. 2021. Reinforcement learning with prototypical representations. In International Conference on Machine Learning. PMLR, 11920\u201311931."},{"key":"e_1_3_2_1_30_1","unstructured":"Konrad Zolna Alexander Novikov Ksenia Konyushkova Caglar Gulcehre Ziyu Wang Yusuf Aytar Misha Denil Nando de Freitas and Scott Reed. 2020. Offline learning from demonstrations and unlabeled experience. arXiv preprint arXiv:2011.13885(2020)."}],"event":{"name":"ACAI 2022: 2022 5th International Conference on Algorithms, Computing and Artificial Intelligence","acronym":"ACAI 2022","location":"Sanya China"},"container-title":["Proceedings of the 2022 5th International Conference on Algorithms, Computing and Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3579654.3579753","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3579654.3579753","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T06:56:49Z","timestamp":1755845809000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3579654.3579753"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,23]]},"references-count":30,"alternative-id":["10.1145\/3579654.3579753","10.1145\/3579654"],"URL":"https:\/\/doi.org\/10.1145\/3579654.3579753","relation":{},"subject":[],"published":{"date-parts":[[2022,12,23]]},"assertion":[{"value":"2023-03-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}