{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T19:27:50Z","timestamp":1775503670701,"version":"3.50.1"},"reference-count":39,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"A*STAR AME Young Individual Research","award":["A2084c0156"],"award-info":[{"award-number":["A2084c0156"]}]},{"name":"MTC Individual Research Grants","award":["M22K2c0079"],"award-info":[{"award-number":["M22K2c0079"]}]},{"name":"ANR-NRF Joint Grant","award":["NRF2021-NRF-ANR003 HM Science"],"award-info":[{"award-number":["NRF2021-NRF-ANR003 HM Science"]}]},{"name":"Nanyang Technological University, Singapore, through the SUG-NAP Grant"},{"name":"MOE AcRF Tier 1 Funding","award":["RG13\/23"],"award-info":[{"award-number":["RG13\/23"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1109\/tnnls.2023.3317628","type":"journal-article","created":{"date-parts":[[2023,10,3]],"date-time":"2023-10-03T17:53:57Z","timestamp":1696355637000},"page":"18553-18564","source":"Crossref","is-referenced-by-count":11,"title":["Sampling Efficient Deep Reinforcement Learning Through Preference-Guided Stochastic Exploration"],"prefix":"10.1109","volume":"35","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7212-027X","authenticated-orcid":false,"given":"Wenhui","family":"Huang","sequence":"first","affiliation":[{"name":"School of Mechanical and Aerospace Engineering, Nanyang Technological University, Jurong West, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8434-1181","authenticated-orcid":false,"given":"Cong","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Nanyang Technological University, Jurong West, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7336-4492","authenticated-orcid":false,"given":"Jingda","family":"Wu","sequence":"additional","affiliation":[{"name":"School of Mechanical and Aerospace Engineering, Nanyang Technological University, Jurong West, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9818-0879","authenticated-orcid":false,"given":"Xiangkun","family":"He","sequence":"additional","affiliation":[{"name":"School of Mechanical and Aerospace Engineering, Nanyang Technological University, Jurong West, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Nanyang Technological University, Jurong West, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6897-4512","authenticated-orcid":false,"given":"Chen","family":"Lv","sequence":"additional","affiliation":[{"name":"School of Mechanical and Aerospace Engineering, Nanyang Technological University, Jurong West, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3103642"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-021-04357-7"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00946"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.2969483"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.2977924"},{"key":"ref6","article-title":"Goal-guided transformer-enabled reinforcement learning for efficient autonomous navigation","author":"Huang","year":"2023","journal-title":"arXiv:2301.00362"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3142822"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3177685"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2022.3209399"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2023.123225"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-022-3629-1"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref13","first-page":"799","article-title":"Autonomous helicopter flight via reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"16","author":"Kim"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3098985"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3108034"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"ref17","first-page":"1995","article-title":"Dueling network architectures for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wang"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.23919\/ChiCC.2018.8483478"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11791"},{"key":"ref20","article-title":"Noisy networks for exploration","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Fortunato"},{"key":"ref21","first-page":"1352","article-title":"Reinforcement learning with deep energy-based policies","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref22","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2021.3054625"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1812.05905"},{"key":"ref25","first-page":"449","article-title":"A distributional perspective on reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bellemare"},{"key":"ref26","first-page":"1096","article-title":"Implicit quantile networks for distributional reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Dabney"},{"key":"ref27","article-title":"Maxmin Q-learning: Controlling the estimation bias of Q-learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Lan"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1038\/nature14539"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"ref30","article-title":"Parameter space noise for exploration","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Plappert"},{"key":"ref31","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"12","author":"Sutton"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2009.07.008"},{"key":"ref33","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992699"},{"key":"ref35","first-page":"525","article-title":"Multi-task learning as multi-objective optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Sener"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729694"},{"key":"ref37","volume-title":"The Principles of Learning and Behavior","author":"Domjan","year":"2015"},{"key":"ref38","article-title":"OpenAI gym","author":"Brockman","year":"2016","journal-title":"arXiv:1606.01540"},{"key":"ref39","article-title":"UCB exploration via Q-ensembles","author":"Chen","year":"2017","journal-title":"arXiv:1706.01502"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/10772360\/10269149.pdf?arnumber=10269149","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T19:09:36Z","timestamp":1733252976000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10269149\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":39,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2023.3317628","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12]]}}}