{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,8]],"date-time":"2026-01-08T04:44:06Z","timestamp":1767847446553,"version":"3.49.0"},"reference-count":67,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Brain Korea 21 FOUR"},{"name":"Ministry of Science and ICT (MSIT) in Korea"},{"name":"Institute for Information Communication Technology Planning and Evaluation","award":["IITP-2020-0-01749"],"award-info":[{"award-number":["IITP-2020-0-01749"]}]},{"name":"Korean Government","award":["RS-2022-00144190"],"award-info":[{"award-number":["RS-2022-00144190"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1109\/tnnls.2024.3439261","type":"journal-article","created":{"date-parts":[[2024,8,14]],"date-time":"2024-08-14T13:39:12Z","timestamp":1723642752000},"page":"8814-8827","source":"Crossref","is-referenced-by-count":6,"title":["Masked and Inverse Dynamics Modeling for Data-Efficient Reinforcement Learning"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1309-0602","authenticated-orcid":false,"given":"Young","family":"Jae Lee","sequence":"first","affiliation":[{"name":"Department of Industrial and Management Engineering, Korea University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4773-6467","authenticated-orcid":false,"given":"Jaehoon","family":"Kim","sequence":"additional","affiliation":[{"name":"Department of Industrial and Management Engineering, Korea University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3950-3065","authenticated-orcid":false,"given":"Young","family":"Joon Park","sequence":"additional","affiliation":[{"name":"LG AI Research, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0649-9909","authenticated-orcid":false,"given":"Mingu","family":"Kwak","sequence":"additional","affiliation":[{"name":"School of Industrial and Systems Engineering, Georgia Institute of Technology, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2205-8516","authenticated-orcid":false,"given":"Seoung","family":"Bum Kim","sequence":"additional","affiliation":[{"name":"Department of Industrial and Management Engineering, Korea University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1613\/jair.3912"},{"key":"ref2","article-title":"DeepMind control suite","author":"Tassa","year":"2018","journal-title":"arXiv:1801.00690"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.2307\/j.ctt4cgngj.10"},{"key":"ref4","first-page":"1","article-title":"Image augmentation is all you need: Regularizing deep reinforcement learning from pixels","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Yarats"},{"key":"ref5","first-page":"1","article-title":"Data-efficient reinforcement learning with self-predictive representations","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Schwarzer"},{"key":"ref6","first-page":"1","article-title":"Return-based contrastive representation learning for reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Liu"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3176413"},{"key":"ref8","first-page":"9870","article-title":"Decoupling representation learning from reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Stooke"},{"key":"ref9","first-page":"18459","article-title":"Behavior from the void: Unsupervised active pre-training","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Liu"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN55064.2022.9892416"},{"key":"ref11","first-page":"11920","article-title":"Reinforcement learning with prototypical representations","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Yarats"},{"key":"ref12","first-page":"1","article-title":"Pretrained encoders are all you need","volume-title":"Proc. ICML Workshop Unsupervised Reinforcement Learn.","author":"Khan"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.278"},{"key":"ref14","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018","journal-title":"arXiv:1810.04805"},{"key":"ref15","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. NIPS","author":"Brown"},{"key":"ref16","first-page":"1","article-title":"Cross-lingual language model pretraining","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Conneau"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref18","first-page":"1","article-title":"BEit: BERT pre-training of image transformers","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Bao"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2018.07.006"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref21","first-page":"25117","article-title":"Mask-based latent reconstruction for reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Yu"},{"key":"ref22","first-page":"1","article-title":"Large-scale study of curiosity-driven learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Burda"},{"key":"ref23","first-page":"1","article-title":"Self-supervised policy adaptation during deployment","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Hansen"},{"key":"ref24","first-page":"1332","article-title":"Masked world models for visual control","volume-title":"Proc. Conf. Robot Learn.","author":"Seo"},{"key":"ref25","first-page":"1","article-title":"Loss is its own reward: Self-supervision for reinforcement learning","volume-title":"Proc. ICML 2017 Workshop","author":"Shelhamer"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1812.05905"},{"key":"ref28","first-page":"1","article-title":"Model based reinforcement learning for Atari","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kaiser"},{"key":"ref29","first-page":"25476","article-title":"Mastering Atari games with limited data","volume-title":"Proc. Adv. neural Inf. Process. Syst.","volume":"34","author":"Ye"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref31","first-page":"19884","article-title":"Reinforcement learning with augmented data","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Laskin"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561103"},{"key":"ref33","first-page":"1","article-title":"An image is worth 16\u00d716 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Dosovitskiy"},{"key":"ref34","first-page":"37607","article-title":"Masked trajectory models for prediction, representation, and control","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wu"},{"key":"ref35","first-page":"3875","article-title":"Bootstrap latent-predictive representations for multitask reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Guo"},{"key":"ref36","first-page":"5276","article-title":"Playvirtual: Augmenting cycle-consistent virtual trajectories for reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Yu"},{"key":"ref37","first-page":"1","article-title":"Planning from pixels using inverse dynamics models","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Paster"},{"key":"ref38","first-page":"23178","article-title":"ISO-dream: Isolating and leveraging noncontrollable visual dynamics in world models","volume-title":"Proc. Adv. neural Inf. Process. Syst.","volume":"35","author":"Pan"},{"key":"ref39","first-page":"1","article-title":"Transformers are sample-efficient world models","volume-title":"Proc. 11th Int. Conf. Learn. Represent.","author":"Micheli"},{"issue":"1","key":"ref40","first-page":"1","article-title":"Guaranteed discovery of control-endogenous latent states with multi-step inverse models","volume":"1","author":"Lamb","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref41","first-page":"1","article-title":"Agent-controller representations: Principled offline RL with rich exogenous information","volume-title":"Proc. Adv. Neural Inf. Process. Syst. Workshop Offline RL","author":"Islam"},{"key":"ref42","article-title":"Playing Atari with deep reinforcement learning","author":"Mnih","year":"2013","journal-title":"arXiv:1312.5602"},{"key":"ref43","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"ref46","first-page":"1995","article-title":"Dueling network architectures for deep reinforcement learning","volume-title":"Proc. Int. Conf. Int. Conf. Mach. Learn.","volume":"48","author":"Wang"},{"key":"ref47","article-title":"Prioritized experience replay","author":"Schaul","year":"2015","journal-title":"arXiv:1511.05952"},{"key":"ref48","first-page":"1","article-title":"Noisy networks for exploration","volume-title":"Proc. Int. Conf. Represent. Learn.","author":"Fortunato"},{"key":"ref49","first-page":"449","article-title":"A distributional perspective on reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bellemare"},{"key":"ref50","first-page":"1433","article-title":"Maximum entropy inverse reinforcement learning","volume":"8","author":"Ziebart","year":"2008","journal-title":"Aaai"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"ref52","article-title":"A cookbook of self-supervised learning","author":"Balestriero","year":"2023","journal-title":"arXiv:2304.12210"},{"key":"ref53","first-page":"3015","article-title":"Whitening for self-supervised representation learning","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","volume":"139","author":"Ermolov"},{"key":"ref54","first-page":"29304","article-title":"Deep reinforcement learning at the edge of the statistical precipice","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Agarwal"},{"key":"ref55","article-title":"Layer normalization","author":"Lei Ba","year":"2016","journal-title":"arXiv:1607.06450"},{"key":"ref56","article-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2014","journal-title":"arXiv:1412.6980"},{"key":"ref57","first-page":"1","article-title":"When to use parametric models in reinforcement learning?","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Van Hasselt"},{"key":"ref58","volume-title":"Do recent advancements in model-based deep reinforcement learning really improve data efficiency?","author":"Kielak","year":"2020"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/s101070100263"},{"key":"ref60","first-page":"2555","article-title":"Learning latent dynamics for planning from pixels","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Hafner"},{"key":"ref61","first-page":"1","article-title":"Dream to control: Learning behaviors by latent imagination","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Hafner"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i12.17276"},{"key":"ref63","first-page":"741","article-title":"Stochastic latent actor-critic: Deep reinforcement learning with a latent variable model","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst. (NeurIPS)","author":"Lee"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3497510"},{"key":"ref65","first-page":"1","article-title":"Mastering visual continuous control: Improved data-augmented reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Yarats"},{"key":"ref66","first-page":"2784","article-title":"Stabilizing off-policy deep reinforcement learning from pixels","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Cetin"},{"key":"ref67","first-page":"1","article-title":"Taco: Temporal latent action-driven contrastive loss for visual reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Zheng"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/5962385\/10982361\/10636769.pdf?arnumber=10636769","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,5]],"date-time":"2025-12-05T18:39:29Z","timestamp":1764959969000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10636769\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5]]},"references-count":67,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2024.3439261","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5]]}}}