{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T22:13:06Z","timestamp":1740175986162,"version":"3.37.3"},"reference-count":31,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62073033"],"award-info":[{"award-number":["62073033"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Robot. Autom. Lett."],"published-print":{"date-parts":[[2022,10]]},"DOI":"10.1109\/lra.2022.3214057","type":"journal-article","created":{"date-parts":[[2022,10,12]],"date-time":"2022-10-12T19:38:50Z","timestamp":1665603530000},"page":"12251-12258","source":"Crossref","is-referenced-by-count":2,"title":["APD: Learning Diverse Behaviors for Reinforcement Learning Through Unsupervised Active Pre-Training"],"prefix":"10.1109","volume":"7","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5868-8758","authenticated-orcid":false,"given":"Kailin","family":"Zeng","sequence":"first","affiliation":[{"name":"School of Artificial Intelligence, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8519-4259","authenticated-orcid":false,"given":"QiYuan","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Mechatronics Engineering, Harbin Institute of Technology, Harbin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8591-8843","authenticated-orcid":false,"given":"Bin","family":"Liang","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9386-5825","authenticated-orcid":false,"given":"Jun","family":"Yang","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/IVS.2011.5940562"},{"key":"ref2","first-page":"1995","article-title":"Dueling network architectures for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn","author":"Wang","year":"2016"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref4","first-page":"5055","article-title":"Hindsight experience replay","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Andrychowicz","year":"2017"},{"key":"ref5","first-page":"1117","article-title":"Vime: Variational information maximizing exploration","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"29","author":"Houthooft","year":"2016"},{"key":"ref6","first-page":"1","article-title":"Diversity is all you need: Learning skills without a reward function","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Eysenbach","year":"2019"},{"key":"ref7","first-page":"6736","article-title":"APS: Active pretraining with successor features","volume-title":"Proc. Int. Conf. Mach. Learn","author":"Liu","year":"2021"},{"article-title":"Efficient exploration via state marginal matching","year":"2019","author":"Lee","key":"ref8"},{"key":"ref9","article-title":"Learning more skills through optimistic exploration","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Strouse","year":"2022"},{"key":"ref10","first-page":"1","article-title":"# exploration: A study of count-based exploration for deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Tang","year":"2017"},{"key":"ref11","first-page":"1479","article-title":"Unifying count-based exploration and intrinsic motivation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"29","author":"Bellemare","year":"2016"},{"key":"ref12","first-page":"2721","article-title":"Count-based exploration with neural density models","volume-title":"Proc. Int. Conf. Mach. Learn","author":"Ostrovski","year":"2017"},{"key":"ref13","first-page":"2249","article-title":"An empirical evaluation of thompson sampling","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"24","author":"Chapelle","year":"2011"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1561\/9781680834710"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"ref16","first-page":"17007","article-title":"Dynamic bottleneck for robust self-supervised exploration","volume":"34","author":"Bai","year":"2021","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"ref17","first-page":"18459","article-title":"Behavior from the void: Unsupervised active pre-training","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Liu","year":"2021"},{"key":"ref18","article-title":"Fast task inference with variational intrinsic successor features","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hansen","year":"2020"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2022.04.009"},{"key":"ref20","article-title":"Divide-and-conquer reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ghosh","year":"2018"},{"key":"ref21","first-page":"1423","article-title":"Specializing versatile skill libraries using local mixture of experts","volume-title":"Proc. Conf. Robot Learn.","author":"Celik","year":"2022"},{"key":"ref22","article-title":"URLB: Unsupervised reinforcement learning benchmark","volume-title":"Proc. NeurIPS Datasets Benchmarks","author":"Laskin","year":"2021"},{"volume-title":"Reinforcement Learning: An Introduction","year":"2018","author":"Sutton","key":"ref23"},{"key":"ref24","article-title":"Continuous control with deep reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lillicrap","year":"2015"},{"key":"ref25","first-page":"1317","article-title":"Explore, discover and learn: Unsupervised discovery of state-covering skills","volume-title":"Proc. Int. Conf. Mach. Learn","author":"Campos","year":"2020"},{"key":"ref26","first-page":"11920","article-title":"Reinforcement learning with prototypical representations","volume-title":"Proc. Int. Conf. Mach. Learn","author":"Yarats","year":"2021"},{"key":"ref27","article-title":"Learning subgoal representations with slow dynamics","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Li","year":"2020"},{"key":"ref28","first-page":"5639","article-title":"CURL: Contrastive unsupervised representations for reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Laskin","year":"2020"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.12.094"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/j.simpa.2020.100022"},{"article-title":"Openai gym","year":"2016","author":"Brockman","key":"ref31"}],"container-title":["IEEE Robotics and Automation Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7083369\/9831196\/09917354.pdf?arnumber=9917354","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,24]],"date-time":"2024-01-24T03:42:25Z","timestamp":1706067745000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9917354\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10]]},"references-count":31,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/lra.2022.3214057","relation":{},"ISSN":["2377-3766","2377-3774"],"issn-type":[{"type":"electronic","value":"2377-3766"},{"type":"electronic","value":"2377-3774"}],"subject":[],"published":{"date-parts":[[2022,10]]}}}