{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T05:33:21Z","timestamp":1782970401174,"version":"3.54.5"},"reference-count":48,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2024,9,1]],"date-time":"2024-09-01T00:00:00Z","timestamp":1725148800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,9,1]],"date-time":"2024-09-01T00:00:00Z","timestamp":1725148800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,9,1]],"date-time":"2024-09-01T00:00:00Z","timestamp":1725148800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62136008"],"award-info":[{"award-number":["62136008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62293541"],"award-info":[{"award-number":["62293541"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005089","name":"Beijing Natural Science Foundation","doi-asserted-by":"publisher","award":["4232056"],"award-info":[{"award-number":["4232056"]}],"id":[{"id":"10.13039\/501100005089","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1109\/tai.2024.3379969","type":"journal-article","created":{"date-parts":[[2024,3,21]],"date-time":"2024-03-21T15:03:13Z","timestamp":1711033393000},"page":"4364-4375","source":"Crossref","is-referenced-by-count":12,"title":["Enhancing Reinforcement Learning via Transformer-Based State Predictive Representations"],"prefix":"10.1109","volume":"5","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-0822-1377","authenticated-orcid":false,"given":"Minsong","family":"Liu","sequence":"first","affiliation":[{"name":"National Key Laboratory for Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5384-423X","authenticated-orcid":false,"given":"Yuanheng","family":"Zhu","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9356-0610","authenticated-orcid":false,"given":"Yaran","family":"Chen","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8218-9633","authenticated-orcid":false,"given":"Dongbin","family":"Zhao","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3215788"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TG.2020.3022698"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2023.3268612"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i12.17300"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2023.3299899"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2019.2927869"},{"key":"ref7","article-title":"Representation learning with contrastive predictive coding","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Oord","year":"2021"},{"key":"ref8","article-title":"Data-efficient reinforcement learning with self-predictive representations","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Schwarzer","year":"2021"},{"key":"ref9","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"ref10","first-page":"741","article-title":"Stochastic latent actor-critic: Deep reinforcement learning with a latent variable model","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Lee","year":"2020"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3309608"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3078573"},{"key":"ref13","article-title":"Transdreamer: Reinforcement learning with transformer world models","volume-title":"Proc. Deep RL Workshop NeurIPS","author":"Chen","year":"2021"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561103"},{"key":"ref16","article-title":"Deepmind control suite","author":"Tassa","year":"2018"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"ref18","article-title":"Model-based reinforcement learning for Atari","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kaiser","year":"2020"},{"key":"ref19","article-title":"Image augmentation is all you need: Regularizing deep reinforcement learning from pixels","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kostrikov","year":"2021"},{"key":"ref20","article-title":"Is a good representation sufficient for sample efficient reinforcement learning?","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Du","year":"2020"},{"key":"ref21","first-page":"19884","article-title":"Reinforcement learning with augmented data","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Laskin","year":"2020"},{"key":"ref22","article-title":"Dream to control: Learning behaviors by latent imagination","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hafner","year":"2020"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2023.3283488"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.2307\/j.ctt4cgngj.10"},{"key":"ref25","article-title":"Return-based contrastive representation learning for reinforcement learning","author":"Liu","year":"2021","journal-title":"Int. Conf. Learn. Representations"},{"key":"ref26","article-title":"Observational overfitting in reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Song","year":"2020"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref28","first-page":"18459","article-title":"Behavior from the void: Unsupervised active pre-training","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Liu","year":"2021"},{"key":"ref29","first-page":"9870","article-title":"Decoupling representation learning from reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Stooke","year":"2021"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2021.3121663"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i12.17276"},{"key":"ref32","article-title":"Contrastive behavioral similarity embeddings for generalization in reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Agarwal","year":"2021"},{"key":"ref33","first-page":"5276","article-title":"Playvirtual: Augmenting cycle-consistent virtual trajectories for reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Yu","year":"2021"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2022.3218940"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3497510"},{"key":"ref36","first-page":"12116","article-title":"Do vision transformers see like convolutional neural networks?","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Raghu","year":"2021"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2022.3221912"},{"key":"ref38","article-title":"CtrlFormer: Learning transferable state representation for visual control via transformer","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mu","year":"2022"},{"key":"ref39","first-page":"7487","article-title":"Stabilizing transformers for reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Parisotto","year":"2020"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3176413"},{"key":"ref41","first-page":"25117","article-title":"Mask-based latent reconstruction for reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Yu","year":"2022"},{"key":"ref42","article-title":"Soft actor-critic algorithms and applications","author":"Haarnoja","year":"2018"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.3041469"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CEC.2012.6252962"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3264730"},{"key":"ref47","article-title":"Do recent advancements in model-based deep reinforcement learning really improve data efficiency?","author":"Kielak","year":"2020"},{"key":"ref48","first-page":"29304","article-title":"Deep reinforcement learning at the edge of the statistical precipice","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Agarwal","year":"2021"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9078688\/10673734\/10477774.pdf?arnumber=10477774","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T01:09:28Z","timestamp":1755911368000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10477774\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9]]},"references-count":48,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/tai.2024.3379969","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9]]}}}