{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,7]],"date-time":"2025-07-07T11:27:36Z","timestamp":1751887656453,"version":"3.28.0"},"reference-count":55,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,5,13]]},"DOI":"10.1109\/icra57147.2024.10610569","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T17:51:05Z","timestamp":1723139465000},"page":"17208-17215","source":"Crossref","is-referenced-by-count":1,"title":["Domain Adaptation of Visual Policies with a Single Demonstration"],"prefix":"10.1109","author":[{"given":"Weiyao","family":"Wang","sequence":"first","affiliation":[{"name":"Johns Hopkins University,Department of Computer Science,Baltimore,MD,USA,21218"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gregory D.","family":"Hager","sequence":"additional","affiliation":[{"name":"Johns Hopkins University,Department of Computer Science,Baltimore,MD,USA,21218"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"year":"2018","author":"Kalashnikov","article-title":"Qt-opt: Scalable deep reinforcement learning for vision-based robotic manipulation","key":"ref1"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1109\/ECMR.2019.8870964"},{"key":"ref3","first-page":"3680","article-title":"Stabilizing deep q-learning with convnets and vision transformers under data augmentation","volume":"34","author":"Hansen","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"doi-asserted-by":"publisher","key":"ref4","DOI":"10.1109\/IROS.2012.6386109"},{"year":"2021","author":"Makoviychuk","article-title":"Isaac gym: High performance gpu-based physics simulation for robot learning","key":"ref5"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1109\/SSCI47803.2020.9308468"},{"key":"ref7","first-page":"2048","article-title":"Leveraging procedural generation to benchmark reinforcement learning","volume-title":"International conference on machine learning","author":"Cobbe"},{"issue":"1","key":"ref8","first-page":"2096","article-title":"Domain-adversarial training of neural networks","volume":"17","author":"Ganin","year":"2016","journal-title":"The journal of machine learning research"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.1109\/CVPR.2017.316"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1109\/IROS.2017.8202133"},{"year":"2021","author":"Chen","article-title":"Understanding domain randomization for sim-to-real transfer","key":"ref11"},{"year":"2021","author":"Fan","article-title":"Secant: Self-expert cloning for zero-shot generalization of visual policies","key":"ref12"},{"key":"ref13","first-page":"627","article-title":"A reduction of imitation learning and structured prediction to no-regret online learning","volume-title":"Proceedings of the fourteenth international conference on artificial intelligence and statistics. JMLR Workshop and Conference Proceedings","author":"Ross"},{"key":"ref14","article-title":"Generative adversarial imitation learning","volume":"29","author":"Ho","year":"2016","journal-title":"Advances in neural information processing systems"},{"year":"2015","author":"Rusu","article-title":"Policy distillation","key":"ref15"},{"year":"2015","author":"Parisotto","article-title":"Actor-mimic: Deep multitask and transfer reinforcement learning","key":"ref16"},{"key":"ref17","first-page":"297","article-title":"A system for general in-hand object re-orientation","volume-title":"Conference on Robot Learning","author":"Chen"},{"year":"2022","author":"Lai","article-title":"Sim-to-real transfer for quadrupedal locomotion via terrain transformer","key":"ref18"},{"key":"ref19","first-page":"2071","article-title":"Transformers for one-shot visual imitation","volume-title":"Conference on Robot Learning","author":"Dasari"},{"doi-asserted-by":"publisher","key":"ref20","DOI":"10.1109\/ICRA46639.2022.9812450"},{"key":"ref21","first-page":"24631","article-title":"Prompting decision transformer for few-shot policy generalization","volume-title":"International Conference on Machine Learning","author":"Xu"},{"doi-asserted-by":"publisher","key":"ref22","DOI":"10.1109\/ICRA.2016.7487173"},{"key":"ref23","article-title":"Adaptive auxiliary task weighting for reinforcement learning","volume":"32","author":"Lin","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref24","first-page":"1","article-title":"On the effect of auxiliary tasks on representation dynamics","volume-title":"International Conference on Artificial Intelligence and Statistics","author":"Lyle"},{"year":"2022","author":"He","article-title":"Reinforcement learning with automated auxiliary loss search","key":"ref25"},{"year":"2020","author":"Kostrikov","article-title":"Image augmentation is all you need: Regularizing deep reinforcement learning from pixels","key":"ref26"},{"doi-asserted-by":"publisher","key":"ref27","DOI":"10.1109\/ICRA48506.2021.9561103"},{"year":"2022","author":"Ma","article-title":"A comprehensive survey of data augmentation in visual reinforcement learning","key":"ref28"},{"volume-title":"Thirty-sixth Conference on Neural Information Processing Systems (NeurIPS 2022)","author":"Bertoin","article-title":"Look where you look! saliency-guided q-networks for visual rl tasks","key":"ref29"},{"doi-asserted-by":"publisher","key":"ref30","DOI":"10.1126\/scirobotics.abc5986"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.1126\/scirobotics.abk2822"},{"year":"2021","author":"Lim","article-title":"Planar robot casting with real2sim2real self-supervised learning","key":"ref32"},{"doi-asserted-by":"publisher","key":"ref33","DOI":"10.1109\/ICRA46639.2022.9811651"},{"doi-asserted-by":"publisher","key":"ref34","DOI":"10.15607\/RSS.2017.XIII.048"},{"doi-asserted-by":"publisher","key":"ref35","DOI":"10.15607\/RSS.2021.XVII.011"},{"year":"2020","author":"Hansen","article-title":"Self-supervised policy adaptation during deployment","key":"ref36"},{"doi-asserted-by":"publisher","key":"ref37","DOI":"10.1109\/CVPR42600.2020.01117"},{"year":"2020","author":"Ho","article-title":"Retinagan: An object-aware approach to sim-to-real transfer","key":"ref38"},{"year":"2021","author":"Yoneda","article-title":"Invariance through inference","key":"ref39"},{"key":"ref40","first-page":"5331","article-title":"Efficient off-policy meta-reinforcement learning via probabilistic context variables","volume-title":"International conference on machine learning","author":"Rakelly"},{"doi-asserted-by":"publisher","key":"ref41","DOI":"10.1109\/TPAMI.2021.3079209"},{"doi-asserted-by":"publisher","key":"ref42","DOI":"10.1145\/3386252"},{"key":"ref43","first-page":"1126","article-title":"Model-agnostic meta-learning for fast adaptation of deep networks","volume-title":"International conference on machine learning","author":"Finn"},{"key":"ref44","article-title":"One-shot imitation learning","volume":"30","author":"Duan","year":"2017","journal-title":"Advances in neural information processing systems"},{"year":"2019","author":"Kirsch","article-title":"Improving generalization in meta reinforcement learning using learned objectives","key":"ref45"},{"doi-asserted-by":"publisher","key":"ref46","DOI":"10.1109\/IROS45743.2020.9341571"},{"doi-asserted-by":"publisher","key":"ref47","DOI":"10.48550\/ARXIV.1706.03762"},{"doi-asserted-by":"publisher","key":"ref48","DOI":"10.1016\/j.aiopen.2022.10.001"},{"doi-asserted-by":"publisher","key":"ref49","DOI":"10.1145\/3560815"},{"year":"2022","author":"Wei","article-title":"Chain of thought prompting elicits reasoning in large language models","key":"ref50"},{"key":"ref51","first-page":"15084","article-title":"Decision transformer: Reinforcement learning via sequence modeling","volume":"34","author":"Chen","year":"2021","journal-title":"Advances in neural information processing systems"},{"volume-title":"Reinforcement learning: An introduction.","year":"2018","author":"Sutton","key":"ref52"},{"year":"2018","author":"Haarnoja","article-title":"Soft actor-critic algorithms and applications","key":"ref53"},{"year":"2021","author":"Stone","article-title":"The distracting control suite\u2013a challenging benchmark for reinforcement learning from pixels","key":"ref54"},{"year":"2018","author":"Tassa","article-title":"Deepmind control suite","key":"ref55"}],"event":{"name":"2024 IEEE International Conference on Robotics and Automation (ICRA)","start":{"date-parts":[[2024,5,13]]},"location":"Yokohama, Japan","end":{"date-parts":[[2024,5,17]]}},"container-title":["2024 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10609961\/10609862\/10610569.pdf?arnumber=10610569","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,10]],"date-time":"2024-08-10T05:51:10Z","timestamp":1723269070000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10610569\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,13]]},"references-count":55,"URL":"https:\/\/doi.org\/10.1109\/icra57147.2024.10610569","relation":{},"subject":[],"published":{"date-parts":[[2024,5,13]]}}}