{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T15:46:36Z","timestamp":1783611996158,"version":"3.55.0"},"reference-count":37,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2023,3,1]],"date-time":"2023-03-01T00:00:00Z","timestamp":1677628800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,3,1]],"date-time":"2023-03-01T00:00:00Z","timestamp":1677628800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,3,1]],"date-time":"2023-03-01T00:00:00Z","timestamp":1677628800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100002855","name":"National Science and Technology Major Project of the Ministry of Science and Technology of China","doi-asserted-by":"publisher","award":["2018AAA0101604"],"award-info":[{"award-number":["2018AAA0101604"]}],"id":[{"id":"10.13039\/501100002855","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61906106"],"award-info":[{"award-number":["61906106"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61803321"],"award-info":[{"award-number":["61803321"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62022048"],"award-info":[{"award-number":["62022048"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Beijing Academy of Artificial Intelligence"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2023,3]]},"DOI":"10.1109\/tnnls.2021.3105407","type":"journal-article","created":{"date-parts":[[2021,8,31]],"date-time":"2021-08-31T19:56:56Z","timestamp":1630439816000},"page":"1454-1464","source":"Crossref","is-referenced-by-count":15,"title":["Meta-Reinforcement Learning With Dynamic Adaptiveness Distillation"],"prefix":"10.1109","volume":"34","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9825-8186","authenticated-orcid":false,"given":"Hangkai","family":"Hu","sequence":"first","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7251-0988","authenticated-orcid":false,"given":"Gao","family":"Huang","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0699-1904","authenticated-orcid":false,"given":"Xiang","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7361-9283","authenticated-orcid":false,"given":"Shiji","family":"Song","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref3","article-title":"Playing Atari with deep reinforcement learning","volume-title":"Proc. NIPS Deep Learn. Workshop","author":"Mnih"},{"key":"ref4","first-page":"1126","article-title":"Model-agnostic meta-learning for fast adaptation of deep networks","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","volume":"70","author":"Finn"},{"key":"ref5","article-title":"On first-order meta-learning algorithms","volume-title":"arXiv:1803.02999","author":"Nichol","year":"2018"},{"key":"ref6","first-page":"5302","article-title":"Meta-reinforcement learning of structured exploration strategies","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Gupta"},{"key":"ref7","article-title":"RL\u00b2: Fast reinforcement learning via slow reinforcement learning","volume-title":"arXiv:1611.02779","author":"Duan","year":"2016"},{"key":"ref8","article-title":"Meta-SGD: Learning to learn quickly for few-shot learning","volume-title":"arXiv:1707.09835","author":"Li","year":"2017"},{"key":"ref9","article-title":"A simple neural attentive meta-learner","volume-title":"arXiv:1707.03141","author":"Mishra","year":"2017"},{"key":"ref10","article-title":"Meta reinforcement learning with latent variable Gaussian processes","volume-title":"arXiv:1803.07551","author":"S\u00e6mundsson","year":"2018"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/387"},{"key":"ref12","article-title":"Efficient off-policy meta-reinforcement learning via probabilistic context variables","volume-title":"arXiv:1903.08254","author":"Rakelly","year":"2019"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref14","article-title":"Evolutionary principles in self-referential learning, or on learning how to learn: The meta-meta-\u2026 hook","author":"Schmidhuber","year":"1987"},{"issue":"1","key":"ref15","doi-asserted-by":"crossref","first-page":"105","DOI":"10.1023\/A:1007383707642","article-title":"Shifting inductive bias with success-story algorithm, adaptive Levin search, and incremental self-improvement","volume":"28","author":"Schmidhuber","year":"1997","journal-title":"Mach. Learn."},{"key":"ref16","first-page":"1","article-title":"Learning to adapt in dynamic, real-world environments through meta-reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Nagabandi"},{"key":"ref17","article-title":"Learning to learn: Meta-critic networks for sample efficient learning","volume-title":"arXiv:1706.09529","author":"Sung","year":"2017"},{"key":"ref18","article-title":"Meta-Q-learning","volume-title":"arXiv:1910.00125","author":"Fakoor","year":"2019"},{"key":"ref19","first-page":"5400","article-title":"Evolved policy gradients","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Houthooft"},{"key":"ref20","article-title":"Model-based reinforcement learning via meta-policy optimization","volume-title":"arXiv:1809.05214","author":"Clavera","year":"2018"},{"key":"ref21","article-title":"MGHRL: Meta goal-generation for hierarchical reinforcement learning","volume-title":"arXiv:1909.13607","author":"Fu","year":"2019"},{"key":"ref22","article-title":"Learning to reinforcement learn","volume-title":"arXiv:1611.05763","author":"Wang","year":"2016"},{"key":"ref23","article-title":"ProMP: Proximal meta-policy search","volume-title":"arXiv:1810.06784","author":"Rothfuss","year":"2018"},{"key":"ref24","first-page":"1471","article-title":"Unifying count-based exploration and intrinsic motivation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Bellemare"},{"key":"ref25","article-title":"Count-based exploration with neural density models","volume-title":"arXiv:1703.01310","author":"Ostrovski","year":"2017"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5955"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"ref28","article-title":"Exploration by random network distillation","volume-title":"arXiv:1810.12894","author":"Burda","year":"2018"},{"key":"ref29","article-title":"Information-directed exploration for deep reinforcement learning","volume-title":"arXiv:1812.07544","author":"Nikolov","year":"2018"},{"key":"ref30","article-title":"Contingency-aware exploration in reinforcement learning","volume-title":"arXiv:1811.01483","author":"Choi","year":"2018"},{"key":"ref31","article-title":"Some considerations on learning to explore via meta-reinforcement learning","volume-title":"arXiv:1803.01118","author":"Stadie","year":"2018"},{"key":"ref32","first-page":"4618","article-title":"Fast efficient hyperparameter tuning for policy gradient methods","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Paul"},{"key":"ref33","article-title":"Guided meta-policy search","volume-title":"arXiv:1904.00956","author":"Mendonca","year":"2019"},{"key":"ref34","article-title":"Prioritized experience replay","volume-title":"arXiv:1511.05952","author":"Schaul","year":"2015"},{"issue":"1","key":"ref35","first-page":"347","article-title":"Experience selection in deep reinforcement learning for control","volume":"19","author":"De Bruin","year":"2018","journal-title":"J. Mach. Learn. Res."},{"key":"ref36","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"arXiv:1801.01290","author":"Haarnoja","year":"2018"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/10056374\/09525812.pdf?arnumber=9525812","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,11]],"date-time":"2024-01-11T23:38:28Z","timestamp":1705016308000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9525812\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3]]},"references-count":37,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2021.3105407","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,3]]}}}