{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T17:27:34Z","timestamp":1779384454495,"version":"3.53.1"},"reference-count":28,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2022]]},"DOI":"10.1109\/tpami.2022.3185549","type":"journal-article","created":{"date-parts":[[2022,6,23]],"date-time":"2022-06-23T19:36:17Z","timestamp":1656012977000},"page":"1-17","source":"Crossref","is-referenced-by-count":52,"title":["Meta-Reinforcement Learning in Non-Stationary and Dynamic Environments"],"prefix":"10.1109","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0896-2517","authenticated-orcid":false,"given":"Zhenshan","family":"Bing","sequence":"first","affiliation":[{"name":"Department of Informatics, Technical University of Munich, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2348-6280","authenticated-orcid":false,"given":"David","family":"Lerch","sequence":"additional","affiliation":[{"name":"Department of Informatics, Technical University of Munich, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0359-7810","authenticated-orcid":false,"given":"Kai","family":"Huang","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4840-076X","authenticated-orcid":false,"given":"Alois","family":"Knoll","sequence":"additional","affiliation":[{"name":"Department of Informatics, Technical University of Munich, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Solving Rubiks Cube with a robot hand","year":"2019"},{"key":"ref2","article-title":"Evolutionary principles in self-referential learning. (On learning how to learn: The meta hook.)","author":"Schmidhuber","year":"1987"},{"key":"ref3","article-title":"RL$^{2}$2: Fast reinforcement learning via slow reinforcement learning","author":"Duan","year":"2016"},{"key":"ref4","article-title":"Learning to reinforcement learn","author":"Wang","year":"2016"},{"key":"ref5","article-title":"Model-agnostic meta-learning for fast adaptation of deep networks","author":"Finn","year":"2017"},{"key":"ref6","article-title":"Efficient off-policy meta-reinforcement learning via probabilistic context variables","author":"Rakelly","year":"2019"},{"key":"ref7","article-title":"Auto-encoding variational bayes","author":"Kingma","year":"2013"},{"key":"ref8","first-page":"2976","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref9","article-title":"Context-based meta-reinforcement learning with structured latent space","volume-title":"Proc. Conf. Neural Inf. Process. Syst.","author":"Ren"},{"key":"ref10","article-title":"Learning context-aware task reasoning for efficient meta-reinforcement learning","author":"Wang","year":"2020"},{"key":"ref11","article-title":"Learning to adapt in dynamic, real-world environments through meta-reinforcement learning","author":"Nagabandi","year":"2018"},{"key":"ref12","article-title":"Deep online learning via meta-learning: Continual adaptation for model-based RL","author":"Nagabandi","year":"2018"},{"key":"ref13","article-title":"A simple neural attentive meta-learner","author":"Mishra","year":"2017"},{"key":"ref14","article-title":"Meta-world: A benchmark and evaluation for multi-task and meta reinforcement learning","author":"Yu","year":"2019"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref16","article-title":"ProMP: Proximal meta-policy search","author":"Rothfuss","year":"2018"},{"key":"ref17","article-title":"Meta reinforcement learning as task inference","author":"Humplik","year":"2019"},{"key":"ref18","article-title":"OpenAI Gym","author":"Brockman","year":"2016"},{"key":"ref19","first-page":"2850","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. 33rd Int. Conf. Mach. Learn.","author":"Mnih"},{"key":"ref20","article-title":"WaveNet: A generative model for raw audio","author":"van den Oord","year":"2016"},{"key":"ref21","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"ref22","article-title":"Continuous adaptation via meta-learning in nonstationary and competitive environments","author":"Al-Shedivat","year":"2017"},{"key":"ref23","article-title":"Meta-reinforcement learning of structured exploration strategies","author":"Gupta","year":"2018"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/387"},{"key":"ref25","article-title":"Learning an embedding space for transferable robot skills","volume-title":"Proc. 6th Int. Conf. Learn. Representations","author":"Hausman"},{"key":"ref26","article-title":"Deep variational reinforcement learning for POMDPs","author":"Igl","year":"2018"},{"key":"ref27","first-page":"12 767","article-title":"LatentGNN: Learning efficient non-local relations for visual recognition","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Zhang"},{"key":"ref28","article-title":"Variational dropout and the local reparameterization trick","author":"Kingma","year":"2015"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/4359286\/09804728.pdf?arnumber=9804728","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T05:44:54Z","timestamp":1706766294000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9804728\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/tpami.2022.3185549","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]}}}