{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T12:41:06Z","timestamp":1766061666524,"version":"3.48.0"},"reference-count":36,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iros60139.2025.11247025","type":"proceedings-article","created":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T18:54:45Z","timestamp":1764269685000},"page":"583-590","source":"Crossref","is-referenced-by-count":0,"title":["ACGD: Visual Multitask Policy Learning with Asymmetric Critic Guided Distillation"],"prefix":"10.1109","author":[{"given":"Krishnan","family":"Srinivasan","sequence":"first","affiliation":[{"name":"Stanford University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Xu","sequence":"additional","affiliation":[{"name":"NVIDIA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Henry","family":"Ang","sequence":"additional","affiliation":[{"name":"Stanford University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eric","family":"Heiden","sequence":"additional","affiliation":[{"name":"NVIDIA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dieter","family":"Fox","sequence":"additional","affiliation":[{"name":"NVIDIA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jeannette","family":"Bohg","sequence":"additional","affiliation":[{"name":"Stanford University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Animesh","family":"Garg","sequence":"additional","affiliation":[{"name":"NVIDIA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"year":"2023","author":"Caggiano","article-title":"Myodex: A generalizable prior for dexterous manipulation","key":"ref1"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1126\/scirobotics.adc9244"},{"volume-title":"Conference on Robot Learning","author":"Chen","article-title":"A system for general in-hand object re-orientation","key":"ref3"},{"doi-asserted-by":"publisher","key":"ref4","DOI":"10.1109\/3DV62453.2024.00016"},{"year":"2024","author":"Zhou","article-title":"Autonomous improvement of instruction following skills via foundation models","key":"ref5"},{"year":"2024","author":"Sikchi","article-title":"Dual rl: Unification and new methods for reinforcement and imitation learning","key":"ref6"},{"year":"2021","author":"Mandlekar","article-title":"What matters in learning from offline human demonstrations for robot manipulation","key":"ref7"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1109\/icra48891.2023.10160216"},{"volume-title":"Thirty-sixth Conference on Neural Information Processing Systems Datasets and Benchmarks Track","author":"Chen","article-title":"Towards human-level bimanual dexterous manipulation with reinforcement learning","key":"ref9"},{"year":"2011","author":"Ross","article-title":"A reduction of imitation learning and structured prediction to no-regret online learning","key":"ref10"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.1109\/ICRA48891.2023.10161147"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1109\/CVPR52733.2024.00054"},{"year":"2023","author":"Zakka","article-title":"Robopianist: Dexterous piano playing with deep reinforcement learning","key":"ref13"},{"year":"2019","author":"Kurenkov","article-title":"Ac-teach: A bayesian actor-critic method for policy learning with an ensemble of suboptimal teachers","key":"ref14"},{"key":"ref15","doi-asserted-by":"crossref","first-page":"2468","DOI":"10.1109\/IROS47612.2022.9981126","article-title":"How to spend your robot time: Bridging kickstarting and offline reinforcement learning for vision-based robotic manipulation","volume-title":"2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","author":"Lee","year":"2022"},{"year":"2024","author":"Tirumala","article-title":"Learning robot soccer from egocentric vision with deep reinforcement learning","key":"ref16"},{"year":"2023","author":"Agarwal","article-title":"Dexterous functional grasping","key":"ref17"},{"year":"2021","author":"Huang","article-title":"Generalization in dexterous manipulation via geometry-aware multi-task learning","key":"ref18"},{"year":"2024","author":"Hansen","article-title":"Td-mpc2: Scalable, robust world models for continuous control","key":"ref19"},{"doi-asserted-by":"publisher","key":"ref20","DOI":"10.15607\/RSS.2023.XIX.016"},{"year":"2024","author":"Fu","article-title":"Mobile aloha: Learning bimanual mobile manipulation with low-cost whole-body teleoperation","key":"ref21"},{"volume-title":"8th Annual Conference on Robot Learning","author":"Zhao","article-title":"Aloha unleashed: A simple recipe for robot dexterity","key":"ref22"},{"doi-asserted-by":"publisher","key":"ref23","DOI":"10.1109\/CVPR52729.2023.00459"},{"year":"2022","author":"Jia","article-title":"Improving policy optimization with generalist-specialist learning","key":"ref24"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.1126\/scirobotics.abc5986"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.52202\/079017-4484"},{"year":"2024","author":"Lee","article-title":"Behavior generation with latent actions","key":"ref27"},{"year":"2022","author":"Nair","article-title":"R3m: A universal visual representation for robot manipulation","key":"ref28"},{"year":"2021","author":"Makoviychuk","article-title":"Isaac gym: High performance gpu-based physics simulation for robot learning","key":"ref29"},{"year":"2025","author":"Yin","article-title":"Rapidly adapting policies to the real world via simulation-guided fine-tuning","key":"ref30"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.1109\/ICRA57147.2024.10611293"},{"year":"2023","author":"Ha","article-title":"Scaling up and distilling down: Language-guided robot skill acquisition","key":"ref32"},{"year":"2021","author":"Yu","article-title":"Meta-world: A benchmark and evaluation for multi-task and meta reinforcement learning","key":"ref33"},{"year":"2022","author":"Vittorio","article-title":"Myosuite \u2013 a contact-rich simulation suite for musculoskeletal motor control","key":"ref34"},{"key":"ref35","first-page":"1348","article-title":"Action-quantized offline reinforcement learning for robotic skill learning","volume-title":"Conference on Robot Learning","author":"Luo"},{"volume-title":"International Conference on Learning Representations","author":"Mazzaglia","article-title":"Choreographer: Learning and adapting skills in imagination","key":"ref36"}],"event":{"name":"2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","start":{"date-parts":[[2025,10,19]]},"location":"Hangzhou, China","end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11245651\/11245652\/11247025.pdf?arnumber=11247025","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T12:37:37Z","timestamp":1766061457000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11247025\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":36,"URL":"https:\/\/doi.org\/10.1109\/iros60139.2025.11247025","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}