{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,11]],"date-time":"2026-05-11T20:12:14Z","timestamp":1778530334961,"version":"3.51.4"},"reference-count":59,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2024CSJZN00300"],"award-info":[{"award-number":["2024CSJZN00300"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["623B2049"],"award-info":[{"award-number":["623B2049"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Jiangsu Science Foundation","award":["BK20243039"],"award-info":[{"award-number":["BK20243039"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1109\/tpami.2026.3657578","type":"journal-article","created":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T05:48:07Z","timestamp":1769492887000},"page":"6380-6392","source":"Crossref","is-referenced-by-count":0,"title":["Adversarial Imitation Learning With General Function Approximation: Theoretical Analysis and Practical Algorithms"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9409-448X","authenticated-orcid":false,"given":"Tian","family":"Xu","sequence":"first","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhilong","family":"Zhang","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zexuan","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-9804-8967","authenticated-orcid":false,"given":"Ruishuo","family":"Chen","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yihao","family":"Sun","sequence":"additional","affiliation":[{"name":"Mila-Quebec Artificial Intelligence Institute, Montr&#x00E9;al, QC, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1052-5447","authenticated-orcid":false,"given":"Yang","family":"Yu","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref3","first-page":"12519","article-title":"When to trust your model: Model-based policy optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Janner","year":"2019"},{"key":"ref4","first-page":"1052","article-title":"Generative adversarial user model for reinforcement learning based recommendation system","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Chen","year":"2019"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014902"},{"key":"ref6","first-page":"2679","article-title":"OpenVLA: An open-source vision-language-action model","volume-title":"Proc. Conf. Robot Learn.","author":"Kim","year":"2024"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2025.xxi.010"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1991.3.1.88"},{"key":"ref9","first-page":"627","article-title":"A reduction of imitation learning and structured prediction to no-regret online learning","volume-title":"Proc. 14th Int. Conf. Artif. Intell. Statist.","author":"Ross","year":"2011"},{"key":"ref10","article-title":"Disagreement-regularized imitation learning","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Brantley","year":"2020"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/687"},{"key":"ref12","article-title":"Discriminator-actor-critic: Addressing sample inefficiency and reward bias in adversarial imitation learning","volume-title":"Proc. 7th Int. Conf. Learn. Representations","author":"Kostrikov","year":"2019"},{"key":"ref13","first-page":"8510","article-title":"Offline imitation learning with a misspecified simulator","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Jiang","year":"2020"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-66723-8_19"},{"key":"ref15","first-page":"1259","article-title":"A divergence minimization perspective on imitation learning methods","volume-title":"Proc. 3rd Annu. Conf. Robot Learn.","author":"Ghasemipour","year":"2019"},{"key":"ref16","first-page":"4028","article-title":"IQ-learn: Inverse soft-Q learning for imitation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Garg","year":"2021"},{"key":"ref17","article-title":"Transferable reward learning by dynamics-agnostic discriminator ensemble","author":"Luo","year":"2022"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0810"},{"key":"ref19","first-page":"120602","article-title":"Is behavior cloning all you need? Understanding horizon in imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Foster","year":"2024"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20798"},{"key":"ref21","first-page":"2367","article-title":"Provably efficient adversarial imitation learning with unknown transitions","volume-title":"Proc. 39th Conf. Uncertainty Artif. Intell.","author":"Xu","year":"2023"},{"key":"ref22","first-page":"14094","article-title":"Provably efficient generative adversarial imitation learning for online and offline setting with linear function approximation","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","author":"Liu","year":"2022"},{"key":"ref23","first-page":"49471","article-title":"Imitation learning in discounted linear MDPs without exploration assumptions","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","author":"Viano","year":"2024"},{"key":"ref24","first-page":"15737","article-title":"Error bounds of imitating policies and environments","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xu","year":"2020"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2026.3673238"},{"key":"ref26","first-page":"2914","article-title":"Toward the fundamental limits of imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Rajaraman","year":"2020"},{"key":"ref27","article-title":"Exploration in deep reinforcement learning: A comprehensive survey","author":"Yang","year":"2021"},{"key":"ref28","first-page":"21380","article-title":"From dirichlet to rubin: Optimistic exploration in RL without bonuses","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","author":"Tiapkin","year":"2022"},{"key":"ref29","first-page":"22151","article-title":"Maximize to explore: One objective function fusing estimation, planning, and exploration","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Liu","year":"2024"},{"key":"ref30","article-title":"A posterior sampling framework for interactive decision making","author":"Zhong","year":"2022"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"ref32","first-page":"1449","article-title":"A game-theoretic approach to apprenticeship learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Syed","year":"2007"},{"key":"ref33","first-page":"6036","article-title":"Provably efficient imitation learning from observation alone","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Sun","year":"2019"},{"key":"ref34","first-page":"1325","article-title":"On the value of interaction and function approximation in imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Rajaraman","year":"2021"},{"key":"ref35","first-page":"7077","article-title":"Minimax optimal online imitation learning via replay estimation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Swamy","year":"2022"},{"key":"ref36","first-page":"24309","article-title":"Proximal point imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Viano","year":"2022"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3096966"},{"key":"ref38","first-page":"4565","article-title":"Generative adversarial imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ho","year":"2016"},{"key":"ref39","article-title":"Imitation learning via off-policy distribution matching","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Kostrikov","year":"2020"},{"key":"ref40","first-page":"390","article-title":"End-to-end differentiable adversarial imitation learning","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Baram","year":"2017"},{"key":"ref41","first-page":"1777","article-title":"Efficient imitation learning with conservative world models","volume-title":"th Annu. Learn. Dyn. Control Conf.","author":"Kolev","year":"2024"},{"key":"ref42","first-page":"42428","article-title":"Hybrid inverse reinforcement learning","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","author":"Ren","year":"2024"},{"key":"ref43","first-page":"1466","article-title":"Model-based reinforcement learning and the Eluder dimension","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Osband","year":"2014"},{"key":"ref44","first-page":"13406","article-title":"Bellman Eluder dimension: New rich classes of RL problems, and sample-efficient algorithms","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Jin","year":"2021"},{"key":"ref45","first-page":"661","article-title":"Efficient reductions for imitation learning","volume-title":"Proc. 13 rd Int. Conf. Artif. Intell. Statist.","author":"Ross","year":"2010"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1017\/9781108627771"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1561\/9781680831719"},{"key":"ref48","first-page":"845","article-title":"Online non-convex learning: Following the perturbed leader is optimal","volume-title":"Proc. 31st Int. Conf. Algorithmic Learn. Theory","author":"Suggala","year":"2020"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-007-5038-2"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1287\/moor.2022.1309"},{"key":"ref51","first-page":"214","article-title":"Wasserstein generative adversarial networks","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Arjovsky","year":"2017"},{"key":"ref52","first-page":"1856","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Haarnoja","year":"2018"},{"key":"ref53","first-page":"3852","article-title":"Adversarially trained actor critic for offline reinforcement learning","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","author":"Cheng","year":"2022"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0060"},{"key":"ref55","article-title":"A note on target Q-learning for solving finite MDPs with a generative oracle","author":"Li","year":"2022"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.2307\/jj.20206644.79"},{"key":"ref57","article-title":"Deepmind control suite","author":"Tassa","year":"2018"},{"key":"ref58","article-title":"Mastering visual continuous control: Improved data-augmented reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Yarats","year":"2021"},{"key":"ref59","first-page":"33299","article-title":"Inverse reinforcement learning without reinforcement learning","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Swamy","year":"2023"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/34\/11512030\/11363667-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11512030\/11363667.pdf?arnumber=11363667","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,11]],"date-time":"2026-05-11T19:46:25Z","timestamp":1778528785000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11363667\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":59,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2026.3657578","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]}}}