{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T20:17:49Z","timestamp":1783455469081,"version":"3.55.0"},"reference-count":57,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Fundamental Research Program for Young Scholars"},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["623B2049"],"award-info":[{"award-number":["623B2049"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Research and Development Program of China","award":["2024CSJZN00300"],"award-info":[{"award-number":["2024CSJZN00300"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62495093"],"award-info":[{"award-number":["62495093"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Jiangsu Science Foundation","award":["BK20243039"],"award-info":[{"award-number":["BK20243039"]}]},{"name":"Guangdong Major Project of Basic and Applied Basic Research","award":["2023B0303000001"],"award-info":[{"award-number":["2023B0303000001"]}]},{"name":"Guangdong Provincial Key Laboratory of Big Data Computing"},{"name":"National Key Research and Development Project","award":["2022YFA1003900"],"award-info":[{"award-number":["2022YFA1003900"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1109\/tpami.2026.3673238","type":"journal-article","created":{"date-parts":[[2026,3,12]],"date-time":"2026-03-12T20:36:39Z","timestamp":1773347799000},"page":"8919-8935","source":"Crossref","is-referenced-by-count":2,"title":["Understanding Adversarial Imitation Learning in Small Sample Regime: A Stage-Coupled Analysis"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9409-448X","authenticated-orcid":false,"given":"Tian","family":"Xu","sequence":"first","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0449-002X","authenticated-orcid":false,"given":"Ziniu","family":"Li","sequence":"additional","affiliation":[{"name":"Shenzhen Research Institute of Big Data, Chinese University of Hong Kong, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1052-5447","authenticated-orcid":false,"given":"Yang","family":"Yu","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3995-914X","authenticated-orcid":false,"given":"Zhi-Quan","family":"Luo","sequence":"additional","affiliation":[{"name":"School of Science and Engineering, Chinese University of Hong Kong, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2008.10.024"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3054912"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1561\/2300000053"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1991.3.1.88"},{"key":"ref5","first-page":"627","article-title":"A reduction of imitation learning and structured prediction to no-regret online learning","volume-title":"Proc. 14th Int. Conf. Artif. Intell. Statist.","author":"Ross"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"issue":"39","key":"ref7","first-page":"1","article-title":"End-to-end training of deep visuomotor policies","volume":"17","author":"Levine","year":"2016","journal-title":"J. Mach. Learn. Res."},{"key":"ref8","volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"Puterman","year":"2014"},{"key":"ref9","first-page":"1259","article-title":"A divergence minimization perspective on imitation learning methods","volume-title":"Proc. 3rd Annu. Conf. Robot Learn.","author":"Ghasemipour"},{"key":"ref10","first-page":"15737","article-title":"Error bounds of imitating policies and environments","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xu"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3096966"},{"key":"ref12","first-page":"2914","article-title":"Toward the fundamental limits of imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Rajaraman"},{"key":"ref13","first-page":"4565","article-title":"Generative adversarial imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ho"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/tnn.1998.712192"},{"key":"ref15","article-title":"Learning robust rewards with adverserial inverse reinforcement learning","volume-title":"Proc. 6th Int. Conf. Learn. Representations","author":"Fu"},{"key":"ref16","article-title":"Discriminator-actor-critic: Addressing sample inefficiency and reward bias in adversarial imitation learning","volume-title":"Proc. 7th Int. Conf. Learn. Representations","author":"Kostrikov"},{"key":"ref17","article-title":"Imitation learning via off-policy distribution matching","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Kostrikov"},{"key":"ref18","article-title":"Disagreement-regularized imitation learning","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Brantley"},{"key":"ref19","first-page":"6036","article-title":"Provably efficient imitation learning from observation alone","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Sun"},{"key":"ref20","article-title":"On computation and generalization of generative adversarial imitation learning","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Wang"},{"key":"ref21","first-page":"11044","article-title":"Generative adversarial imitation learning with neural network parameterization: Global optimality and convergence rate","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Zhang"},{"key":"ref22","article-title":"Provably breaking the quadratic error compounding barrier in imitation learning, optimally","author":"Rajaraman","year":"2021"},{"key":"ref23","first-page":"10022","article-title":"Of moments and matching: A game-theoretic framework for closing the imitation gap","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Swamy"},{"key":"ref24","first-page":"14094","article-title":"Learning from demonstration: Provably efficient adversarial policy imitation with linear function approximation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Liu","year":"2022"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"ref26","first-page":"1449","article-title":"A game-theoretic approach to apprenticeship learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Syed"},{"key":"ref27","first-page":"661","article-title":"Efficient reductions for imitation learning","volume-title":"Proc. 13 rd Int. Conf. Artif. Intell. Statist.","author":"Ross"},{"key":"ref28","first-page":"2253","article-title":"A reduction from apprenticeship learning to classification","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Syed"},{"key":"ref29","first-page":"1325","article-title":"On the value of interaction and function approximation in imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Rajaraman"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390286"},{"key":"ref31","first-page":"1433","article-title":"Maximum entropy inverse reinforcement learning","volume-title":"Proc. 23rd AAAI Conf. Artif. Intell.","author":"Ziebart"},{"key":"ref32","first-page":"10925","article-title":"Intrinsic reward driven imitation learning via generative model","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Yu"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.65109\/TTCY1842"},{"key":"ref34","article-title":"Primal wasserstein imitation learning","volume-title":"Proc. 9th Int. Conf. Learn. Representations","author":"Dadashi"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.65109\/CEGM4710"},{"key":"ref36","article-title":"On the global convergence of imitation learning: A case for linear quadratic regulator","author":"Cai","year":"2019"},{"issue":"31","key":"ref37","first-page":"1","article-title":"Sample-efficient adversarial imitation learning","volume":"25","author":"Jung","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3834"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(89)90020-8"},{"key":"ref40","volume-title":"Functional Analysis","author":"Yosida","year":"2012"},{"key":"ref41","first-page":"14656","article-title":"What matters for adversarial imitation learning?","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Orsini"},{"key":"ref42","article-title":"Inequalities for the l1 deviation of the empirical distribution","author":"Weissman","year":"2003"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2015.2478816"},{"key":"ref44","first-page":"1066","article-title":"On learning distributions from their samples","volume-title":"Proc. 28th Conf. Learn. Theory","author":"Kamath"},{"key":"ref45","volume-title":"Dynamic Programming and Optimal Control: Volume I","author":"Bertsekas","year":"2012"},{"key":"ref46","article-title":"More efficient adversarial imitation learning algorithms with known and unknown transitions","author":"Xu","year":"2021"},{"key":"ref47","first-page":"4870","article-title":"Reward-free exploration for reinforcement learning","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Jin"},{"key":"ref48","first-page":"7599","article-title":"Fast active learning for pure exploration in reinforcement learning","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"M\u00e9nard"},{"key":"ref49","first-page":"3812","article-title":"Infogail: Interpretable imitation learning from visual demonstrations","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Li"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v29i1.9590"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.2307\/2333344"},{"key":"ref52","first-page":"895","article-title":"Concentration inequalities for the missing mass and for histogram rule error","volume":"4","author":"McAllester","year":"2003","journal-title":"J. Mach. Learn. Res."},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-66723-8_19"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1561\/2200000018"},{"key":"ref55","first-page":"1856","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref56","article-title":"Rethinking valuedice: Does it really improve performance?","volume-title":"Proc. 11st Int. Conf. Learn. Representations","author":"Li"},{"key":"ref57","article-title":"A modern introduction to online learning","author":"Orabona","year":"2019"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11595778\/11433103.pdf?arnumber=11433103","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T19:45:16Z","timestamp":1783453516000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11433103\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":57,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2026.3673238","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8]]}}}