{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T14:25:26Z","timestamp":1784643926727,"version":"3.55.0"},"reference-count":48,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2022YFA1004000"],"award-info":[{"award-number":["2022YFA1004000"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62136008"],"award-info":[{"award-number":["62136008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"International Partnership Program of the Chinese Academy of Sciences","award":["104GJHZ2022013GC"],"award-info":[{"award-number":["104GJHZ2022013GC"]}]},{"DOI":"10.13039\/501100001809","name":"Excellent Youth Program of State Key Laboratory of Multimodal Artificial Intelligence Systems","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Syst. Man Cybern, Syst."],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1109\/tsmc.2025.3541926","type":"journal-article","created":{"date-parts":[[2025,2,28]],"date-time":"2025-02-28T13:50:40Z","timestamp":1740750640000},"page":"3601-3613","source":"Crossref","is-referenced-by-count":4,"title":["Cross-Domain Random Pretraining With Prototypes for Reinforcement Learning"],"prefix":"10.1109","volume":"55","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7857-3058","authenticated-orcid":false,"given":"Xin","family":"Liu","sequence":"first","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9356-0610","authenticated-orcid":false,"given":"Yaran","family":"Chen","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2559-9585","authenticated-orcid":false,"given":"Haoran","family":"Li","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2425-936X","authenticated-orcid":false,"given":"Boyu","family":"Li","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8218-9633","authenticated-orcid":false,"given":"Dongbin","family":"Zhao","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref2","first-page":"1","article-title":"Dream to control: Learning behaviors by latent imagination","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Hafner"},{"key":"ref3","first-page":"1","article-title":"Continuous control with deep reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Lillicrap"},{"key":"ref4","first-page":"19884","article-title":"Reinforcement learning with augmented data","volume-title":"Proc. 34th Adv. Neural Inf. Process. Syst.","author":"Laskin"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2023.3275128"},{"issue":"1","key":"ref6","first-page":"1334","article-title":"End-to-end training of deep visuomotor policies","volume":"17","author":"Levine","year":"2016","journal-title":"J. Mach. Learn. Res."},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2020.2967936"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3067028"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2023.3270444"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/353"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/439"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2022.3143201"},{"key":"ref13","first-page":"18459","article-title":"Behavior from the void: Unsupervised active pre-training","volume-title":"Proc. 35th Adv. Neural Inf. Process. Syst.","author":"Liu"},{"key":"ref14","first-page":"11920","article-title":"Reinforcement learning with prototypical representations","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Yarats"},{"key":"ref15","article-title":"DeepMind control suite","author":"Tassa","year":"2018","journal-title":"arXiv:1801.00690"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/461"},{"key":"ref17","first-page":"1126","article-title":"Model-agnostic meta-learning for fast adaptation of deep networks","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Finn"},{"key":"ref18","first-page":"16043","article-title":"CtrlFormer: Learning transferable state representation for visual control via transformer","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mu"},{"key":"ref19","article-title":"A generalist agent","author":"Reed","year":"2022","journal-title":"arXiv:2205.06175"},{"key":"ref20","first-page":"1","article-title":"Diversity is all you need: Learning skills without a reward function","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Eysenbach"},{"key":"ref21","first-page":"5062","article-title":"Self-supervised exploration via disagreement","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Pathak"},{"key":"ref22","first-page":"34478","article-title":"Unsupervised reinforcement learning with contrastive intrinsic control","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Laskin"},{"key":"ref23","first-page":"39183","article-title":"Behavior contrastive learning for unsupervised skill discovery","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Yang"},{"key":"ref24","first-page":"1","article-title":"Learning to navigate in complex environments","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Mirowski"},{"key":"ref25","first-page":"1","article-title":"Reinforcement learning with unsupervised auxiliary tasks","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Jaderberg"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.2307\/j.ctt4cgngj.10"},{"key":"ref27","first-page":"1","article-title":"Data-efficient reinforcement learning with self-predictive representations","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Schwarzer"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref29","first-page":"9912","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","volume-title":"Proc. 34th Adv. Neural Inf. Process. Syst.","author":"Caron"},{"key":"ref30","first-page":"1","article-title":"Sinkhorn distances: Lightspeed computation of optimal transport","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Cuturi"},{"key":"ref31","first-page":"4956","article-title":"DreamerPro: Reconstruction-free model-based reinforcement learning with prototypical representations","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","author":"Deng"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2024.3396525"},{"key":"ref33","first-page":"4182","article-title":"Data-efficient image recognition with contrastive predictive coding","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Henaff"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref35","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proc. NAACL-HLT","author":"Devlin"},{"key":"ref36","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. 34th Adv. Neural Inf. Process. Syst.","author":"Brown"},{"key":"ref37","first-page":"9870","article-title":"Decoupling representation learning from reinforcement learning","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Stooke"},{"key":"ref38","article-title":"Masked visual pre-training for motor control","author":"Xiao","year":"2022","journal-title":"arXiv:2203.06173"},{"key":"ref39","first-page":"12686","article-title":"Pretraining representations for data-efficient reinforcement learning","volume-title":"Proc. 35th Adv. Neural Inf. Process. Syst.","author":"Schwarzer"},{"key":"ref40","first-page":"1","article-title":"Model-based reinforcement learning for Atari","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kaiser"},{"key":"ref41","first-page":"1","article-title":"URLB: Unsupervised reinforcement learning benchmark","volume-title":"Proc. 35th Conf. Neural Inf. Process. Syst. Datasets Benchmarks Track (Round2)","author":"Laskin"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1016\/S0004-3702(98)00023-X"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1512\/iumj.1957.6.56038"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1080\/01966324.2003.10737616"},{"key":"ref46","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref47","first-page":"1","article-title":"Image augmentation is all you need: Regularizing deep reinforcement learning from pixels","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Yarats"},{"key":"ref48","first-page":"1","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kingma"}],"container-title":["IEEE Transactions on Systems, Man, and Cybernetics: Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6221021\/10966471\/10908358.pdf?arnumber=10908358","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,13]],"date-time":"2025-11-13T18:44:41Z","timestamp":1763059481000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10908358\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5]]},"references-count":48,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tsmc.2025.3541926","relation":{},"ISSN":["2168-2216","2168-2232"],"issn-type":[{"value":"2168-2216","type":"print"},{"value":"2168-2232","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5]]}}}