{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:23:15Z","timestamp":1785543795179,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,24]],"date-time":"2024-08-24T00:00:00Z","timestamp":1724457600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,25]]},"DOI":"10.1145\/3637528.3672059","type":"proceedings-article","created":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T04:54:55Z","timestamp":1724561695000},"page":"2608-2617","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Offline Imitation Learning with Model-based Reverse Augmentation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8107-114X","authenticated-orcid":false,"given":"Jie-Jing","family":"Shao","sequence":"first","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0587-0267","authenticated-orcid":false,"given":"Hao-Sen","family":"Shi","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8965-1288","authenticated-orcid":false,"given":"Lan-Zhe","family":"Guo","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, School of Intelligence Science and Technology, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7727-4304","authenticated-orcid":false,"given":"Yu-Feng","family":"Li","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,8,24]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"Gaon An Seungyong Moon Jang-Hyun Kim and Hyun Oh Song. 2021. Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble. In Advances in Neural Information Processing Systems 34. Virtual Event 7436--7447."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2021.103500"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2022.3227738"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1177\/02783649221078031"},{"key":"e_1_3_2_2_5_1","volume-title":"Virtual Event","author":"Brown Tom B.","year":"2020","unstructured":"Tom B. Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel M. Ziegler, Jeffrey Wu, Clemens Winter, Christopher Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. 2020. Language Models are Few-Shot Learners. In Advances in Neural Information Processing Systems 33. Virtual Event, 1877--1901."},{"key":"e_1_3_2_2_6_1","unstructured":"Jonathan D. Chang Masatoshi Uehara Dhruv Sreenivas Rahul Kidambi and Wen Sun. 2021. Mitigating Covariate Shift in Imitation Learning via Offline Data With Partial Coverage. In Advances in Neural Information Processing Systems 34. Virtual Event 965--979."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s41315-019-00103-5"},{"key":"e_1_3_2_2_8_1","volume-title":"Offline Reinforcement Learning for Autonomous Driving with Real World Driving Data. In 25th IEEE International Conference on Intelligent Transportation Systems","author":"Fang Xing","year":"2022","unstructured":"Xing Fang, Qichao Zhang, Yinfeng Gao, and Dongbin Zhao. 2022. Offline Reinforcement Learning for Autonomous Driving with Real World Driving Data. In 25th IEEE International Conference on Intelligent Transportation Systems. Macau, China, 3417--3422."},{"key":"e_1_3_2_2_9_1","volume-title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning. CoRR","author":"Fu Justin","year":"2020","unstructured":"Justin Fu, Aviral Kumar, Ofir Nachum, George Tucker, and Sergey Levine. 2020. D4RL: Datasets for Deep Data-Driven Reinforcement Learning. CoRR, Vol. abs\/2004.07219 (2020). showeprint[arXiv]2004.07219"},{"key":"e_1_3_2_2_10_1","volume-title":"Virtual Event","author":"Fujimoto Scott","year":"2021","unstructured":"Scott Fujimoto and Shixiang Shane Gu. 2021. A Minimalist Approach to Offline Reinforcement Learning. In Advances in Neural Information Processing Systems 34. Virtual Event, 20132--20145."},{"key":"e_1_3_2_2_11_1","volume-title":"Proceedings of the 36th International Conference on Machine Learning","author":"Fujimoto Scott","year":"2019","unstructured":"Scott Fujimoto, David Meger, and Doina Precup. 2019. Off-Policy Deep Reinforcement Learning without Exploration. In Proceedings of the 36th International Conference on Machine Learning. Long Beach, CA, 2052--2062."},{"key":"e_1_3_2_2_12_1","volume-title":"Recall Traces: Backtracking Models for Efficient Reinforcement Learning. In 7th International Conference on Learning Representations","author":"Goyal Anirudh","year":"2019","unstructured":"Anirudh Goyal, Philemon Brakel, William Fedus, Soumye Singhal, Timothy Lillicrap, Sergey Levine, Hugo Larochelle, and Yoshua Bengio. 2019. Recall Traces: Backtracking Models for Efficient Reinforcement Learning. In 7th International Conference on Learning Representations. New Orleans, LA."},{"key":"e_1_3_2_2_13_1","unstructured":"Lan-Zhe Guo Yi-Ge Zhang Zhi-Fan Wu Jie-Jing Shao and Yu-Feng Li. 2022. Robust Semi-Supervised Learning when Not All Classes have Labels. In Advances in Neural Information Processing Systems 35. 3305--3317."},{"key":"e_1_3_2_2_14_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning. 3897--3906","author":"Guo Lan-Zhe","year":"2020","unstructured":"Lan-Zhe Guo, Zhenyu Zhang, Yuan Jiang, Yu-Feng Li, and Zhi-Hua Zhou. 2020. Safe Deep Semi-Supervised Learning for Unseen-Class Unlabeled Data. In Proceedings of the 37th International Conference on Machine Learning. 3897--3906."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1037\/0096-3445.128.1.3"},{"key":"e_1_3_2_2_16_1","volume-title":"The Provable Benefit of Unsupervised Data Sharing for Offline Reinforcement Learning. In The 11th International Conference on Learning Representations. Kigali, Rwanda.","author":"Hu Hao","year":"2023","unstructured":"Hao Hu, Yiqin Yang, Qianchuan Zhao, and Chongjie Zhang. 2023. The Provable Benefit of Unsupervised Data Sharing for Offline Reinforcement Learning. In The 11th International Conference on Learning Representations. Kigali, Rwanda."},{"key":"e_1_3_2_2_17_1","volume-title":"Offline inverse reinforcement learning. arXiv preprint arXiv:2106.05068","author":"Jarboui Firas","year":"2021","unstructured":"Firas Jarboui and Vianney Perchet. 2021. Offline inverse reinforcement learning. arXiv preprint arXiv:2106.05068 (2021)."},{"key":"e_1_3_2_2_18_1","unstructured":"Daniel Jarrett Ioana Bica and Mihaela van der Schaar. 2020. Strictly Batch Imitation Learning by Energy-based Distribution Matching. In Advances in Neural Information Processing Systems 33. 7354--7365."},{"key":"e_1_3_2_2_19_1","volume-title":"Hauptmann","author":"Jiang Lu","year":"2014","unstructured":"Lu Jiang, Deyu Meng, Shoou-I Yu, Zhen-Zhong Lan, Shiguang Shan, and Alexander G. Hauptmann. 2014. Self-Paced Learning with Diversity. In Advances in Neural Information Processing Systems 27. Montreal, Canada, 2078--2086."},{"key":"e_1_3_2_2_20_1","volume-title":"DemoDICE: Offline Imitation Learning with Supplementary Imperfect Demonstrations. In The 10th International Conference on Learning Representations. Virtual Event.","author":"Kim Geon-Hyeong","year":"2022","unstructured":"Geon-Hyeong Kim, Seokin Seo, Jongmin Lee, Wonseok Jeon, HyeongJoo Hwang, Hongseok Yang, and Kee-Eung Kim. 2022. DemoDICE: Offline Imitation Learning with Supplementary Imperfect Demonstrations. In The 10th International Conference on Learning Representations. Virtual Event."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2022.103829"},{"key":"e_1_3_2_2_22_1","volume-title":"Offline Reinforcement Learning with Implicit Q-Learning. In The Tenth International Conference on Learning Representations. Virtual Event.","author":"Kostrikov Ilya","year":"2022","unstructured":"Ilya Kostrikov, Ashvin Nair, and Sergey Levine. 2022. Offline Reinforcement Learning with Implicit Q-Learning. In The Tenth International Conference on Learning Representations. Virtual Event."},{"key":"e_1_3_2_2_23_1","volume-title":"Advances in Neural Information Processing Systems 23.","author":"Kumar M. Pawan","unstructured":"M. Pawan Kumar, Benjamin Packer, and Daphne Koller. 2010. Self-Paced Learning for Latent Variable Models. In Advances in Neural Information Processing Systems 23. Vancouver, Canada, 1189--1197."},{"key":"e_1_3_2_2_24_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning. 5618--5627","author":"Lai Hang","year":"2020","unstructured":"Hang Lai, Jian Shen, Weinan Zhang, and Yong Yu. 2020. Bidirectional Model-based Policy Optimization. In Proceedings of the 37th International Conference on Machine Learning. 5618--5627."},{"key":"e_1_3_2_2_25_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning. 5757--5766","author":"Lee Kimin","year":"2020","unstructured":"Kimin Lee, Younggyo Seo, Seunghyun Lee, Honglak Lee, and Jinwoo Shin. 2020. Context-aware Dynamics Model for Generalization in Model-Based Reinforcement Learning. In Proceedings of the 37th International Conference on Machine Learning. 5757--5766."},{"key":"e_1_3_2_2_26_1","volume-title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems. CoRR","author":"Levine Sergey","year":"2020","unstructured":"Sergey Levine, Aviral Kumar, George Tucker, and Justin Fu. 2020. Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems. CoRR, Vol. abs\/2005.01643 (2020)."},{"key":"e_1_3_2_2_27_1","unstructured":"Minghuan Liu Hanye Zhao Zhengyu Yang Jian Shen Weinan Zhang Li Zhao and Tie-Yan Liu. 2021. Curriculum Offline Imitating Learning. In Advances in Neural Information Processing Systems 34. 6266--6277."},{"key":"e_1_3_2_2_28_1","volume-title":"Optimal Transport for Offline Imitation Learning. In The 11th International Conference on Learning Representations.","author":"Luo Yicheng","year":"2023","unstructured":"Yicheng Luo, Zhengyao Jiang, Samuel Cohen, Edward Grefenstette, and Marc Peter Deisenroth. 2023. Optimal Transport for Offline Imitation Learning. In The 11th International Conference on Learning Representations."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2017.05.043"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2022.3144867"},{"key":"e_1_3_2_2_31_1","volume-title":"Accelerating Online Reinforcement Learning with Offline Datasets. CoRR","author":"Nair Ashvin","year":"2020","unstructured":"Ashvin Nair, Murtaza Dalal, Abhishek Gupta, and Sergey Levine. 2020. Accelerating Online Reinforcement Learning with Offline Datasets. CoRR, Vol. abs\/2006.09359 (2020)."},{"key":"e_1_3_2_2_32_1","volume-title":"Proceedings of the 17th International Conference on Machine Learning","author":"Andrew","unstructured":"Andrew Y. Ng and Stuart Russell. 2000. Algorithms for Inverse Reinforcement Learning. In Proceedings of the 17th International Conference on Machine Learning. Stanford, CA, 663--670."},{"key":"e_1_3_2_2_33_1","volume-title":"ALVINN: An Autonomous Land Vehicle in a Neural Network. In Advances in Neural Information Processing Systems 1.","author":"Pomerleau Dean","year":"1988","unstructured":"Dean Pomerleau. 1988. ALVINN: An Autonomous Land Vehicle in a Neural Network. In Advances in Neural Information Processing Systems 1. Denver, CO, 305--313."},{"key":"e_1_3_2_2_34_1","unstructured":"Rongjun Qin Xingyuan Zhang Songyi Gao Xiong-Hui Chen Zewen Li Weinan Zhang and Yang Yu. 2022. NeoRL: A Near Real-World Benchmark for Offline Reinforcement Learning. In Advances in Neural Information Processing Systems 35. Virtual Event."},{"key":"e_1_3_2_2_35_1","unstructured":"Nived Rajaraman Lin F. Yang Jiantao Jiao and Kannan Ramchandran. 2020. Toward the Fundamental Limits of Imitation Learning. In Advances in Neural Information Processing Systems 33. Virtual Event."},{"key":"e_1_3_2_2_36_1","volume-title":"9th International Conference on Learning Representations. Virtual Event.","author":"Sasaki Fumihiro","year":"2021","unstructured":"Fumihiro Sasaki and Ryota Yamashina. 2021. Behavioral Cloning from Noisy Demonstrations. In 9th International Conference on Learning Representations. Virtual Event."},{"key":"e_1_3_2_2_37_1","volume-title":"LOG: Active Model Adaptation for Label-Efficient OOD Generalization. In Advances in Neural Information Processing Systems 35.","author":"Shao Jie-Jing","year":"2022","unstructured":"Jie-Jing Shao, Lan-Zhe Guo, Xiao-Wen Yang, and Yu-Feng Li. 2022. LOG: Active Model Adaptation for Label-Efficient OOD Generalization. In Advances in Neural Information Processing Systems 35. New Orleans, LA."},{"key":"e_1_3_2_2_38_1","unstructured":"Jie-Jing Shao Hao-Sen Shi Tian Xu Lan-Zhe Guo Yang Yu and Yu-Feng Li. 2023. Offline Imitation Learning without Auxiliary High-quality Behavior Data. (2023). https:\/\/openreview.net\/forum?id=7fxzVTSgZC"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-022-06237-1"},{"key":"e_1_3_2_2_40_1","volume-title":"6th Conference on Robot Learning","author":"Sinha Samarth","year":"2021","unstructured":"Samarth Sinha, Ajay Mandlekar, and Animesh Garg. 2021. S4RL: Surprisingly Simple Self-Supervision for Offline Reinforcement Learning in Robotics. In 6th Conference on Robot Learning. London, UK, 907--917."},{"key":"e_1_3_2_2_41_1","unstructured":"Kihyuk Sohn Honglak Lee and Xinchen Yan. 2015. Learning Structured Output Representation using Deep Conditional Generative Models. In Advances in Neural Information Processing Systems 28. Montreal Canada 3483--3491."},{"key":"e_1_3_2_2_42_1","article-title":"Covariate shift adaptation by importance weighted cross validation","volume":"8","author":"Sugiyama Masashi","year":"2007","unstructured":"Masashi Sugiyama, Matthias Krauledat, and Klaus-Robert M\u00fcller. 2007. Covariate shift adaptation by importance weighted cross validation. Journal of Machine Learning Research, Vol. 8, 5 (2007).","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_2_43_1","unstructured":"Richard S Sutton Andrew G Barto et al. 1998. Introduction to reinforcement learning. Vol. 135. MIT press Cambridge."},{"key":"e_1_3_2_2_44_1","volume-title":"7th Conference on Robot Learning.","author":"Wang Chen","year":"2023","unstructured":"Chen Wang, Linxi Fan, Jiankai Sun, Ruohan Zhang, Li Fei-Fei, Danfei Xu, Yuke Zhu, and Anima Anandkumar. 2023. MimicPlay: Long-Horizon Imitation Learning by Watching Human Play. In 7th Conference on Robot Learning."},{"key":"e_1_3_2_2_45_1","unstructured":"Jianhao Wang Wenzhe Li Haozhe Jiang Guangxiang Zhu Siyuan Li and Chongjie Zhang. 2021. Offline Reinforcement Learning with Reverse Model-based Imagination. In Advances in Neural Information Processing Systems 34. 29420--29432."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF01066989"},{"key":"e_1_3_2_2_47_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning","author":"Xu Haoran","year":"2022","unstructured":"Haoran Xu, Xianyuan Zhan, Honglei Yin, and Huiling Qin. 2022. Discriminator-Weighted Offline Imitation Learning from Suboptimal Demonstrations. In Proceedings of the 39th International Conference on Machine Learning. Baltimore, MD, 24725--24742."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3096966"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477600"},{"key":"e_1_3_2_2_50_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning","author":"Yu Tianhe","year":"2022","unstructured":"Tianhe Yu, Aviral Kumar, Yevgen Chebotar, Karol Hausman, Chelsea Finn, and Sergey Levine. 2022. How to Leverage Unlabeled Data in Offline Reinforcement Learning. In Proceedings of the 39th International Conference on Machine Learning. Baltimore, MD, 25611--25635."},{"key":"e_1_3_2_2_51_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning.","author":"Yue Sheng","year":"2024","unstructured":"Sheng Yue, Jiani Liu, Xingyuan Hua, Ju Ren, Sen Lin, Junshan Zhang, and Yaoxue Zhang. 2024. How to Leverage Diverse Demonstrations in Offline Imitation Learning. In Proceedings of the 41st International Conference on Machine Learning."},{"key":"e_1_3_2_2_52_1","volume-title":"CLARE: Conservative Model-Based Reward Learning for Offline Inverse Reinforcement Learning. In The 11th International Conference on Learning Representations. Kigali, Rwanda.","author":"Yue Sheng","year":"2023","unstructured":"Sheng Yue, Guanbo Wang, Wei Shao, Zhaofeng Zhang, Sen Lin, Ju Ren, and Junshan Zhang. 2023. CLARE: Conservative Model-Based Reward Learning for Offline Inverse Reinforcement Learning. In The 11th International Conference on Learning Representations. Kigali, Rwanda."},{"key":"e_1_3_2_2_53_1","first-page":"1","article-title":"Adaptivity and Non-stationarity: Problem-dependent Dynamic Regret for Online Convex Optimization","volume":"25","author":"Zhao Peng","year":"2024","unstructured":"Peng Zhao, Yu-Jie Zhang, Lijun Zhang, and Zhi-Hua Zhou. 2024. Adaptivity and Non-stationarity: Problem-dependent Dynamic Regret for Online Convex Optimization. Journal of Machine Learning Research, Vol. 25, 98 (2024), 1 -- 52.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_2_54_1","first-page":"1433","article-title":"Maximum Entropy Inverse Reinforcement Learning. In Proceedings of the 23rd AAAI Conference on Artificial Intelligence","author":"Ziebart Brian D.","year":"2008","unstructured":"Brian D. Ziebart, Andrew L. Maas, J. Andrew Bagnell, and Anind K. Dey. 2008. Maximum Entropy Inverse Reinforcement Learning. In Proceedings of the 23rd AAAI Conference on Artificial Intelligence. Chicago, IL, 1433--1438.","journal-title":"Chicago"}],"event":{"name":"KDD '24: The 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Barcelona Spain","acronym":"KDD '24","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3672059","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3637528.3672059","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:04:23Z","timestamp":1750291463000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3672059"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,24]]},"references-count":54,"alternative-id":["10.1145\/3637528.3672059","10.1145\/3637528"],"URL":"https:\/\/doi.org\/10.1145\/3637528.3672059","relation":{},"subject":[],"published":{"date-parts":[[2024,8,24]]},"assertion":[{"value":"2024-08-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}