{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,13]],"date-time":"2026-02-13T21:36:19Z","timestamp":1771018579036,"version":"3.50.1"},"reference-count":49,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62222107"],"award-info":[{"award-number":["62222107"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Yangtze River Delta Science and Technology Innovation Community Joint Research","award":["BK2024CSJZN00300"],"award-info":[{"award-number":["BK2024CSJZN00300"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Veh. Technol."],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1109\/tvt.2025.3597537","type":"journal-article","created":{"date-parts":[[2025,8,11]],"date-time":"2025-08-11T17:43:07Z","timestamp":1754934187000},"page":"1754-1766","source":"Crossref","is-referenced-by-count":0,"title":["Cognitive Escape Reinforcement Learning for Complex Decision Making"],"prefix":"10.1109","volume":"75","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-9168-1221","authenticated-orcid":false,"given":"Shijin","family":"Zhao","sequence":"first","affiliation":[{"name":"College of Electronic and Information Engineering, Nanjing University of Aeronautics and Astronautics, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9521-5084","authenticated-orcid":false,"given":"Qihui","family":"Wu","sequence":"additional","affiliation":[{"name":"College of Electronic and Information Engineering, Nanjing University of Aeronautics and Astronautics, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6880-6244","authenticated-orcid":false,"given":"Fuhui","family":"Zhou","sequence":"additional","affiliation":[{"name":"College of Artificial Intelligence, Nanjing University of Aeronautics and Astronautics, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1923-589X","authenticated-orcid":false,"given":"Hao","family":"Zhang","sequence":"additional","affiliation":[{"name":"College of Electronic and Information Engineering, Nanjing University of Aeronautics and Astronautics, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6685-221X","authenticated-orcid":false,"given":"Yang","family":"Huang","sequence":"additional","affiliation":[{"name":"College of Electronic and Information Engineering, Nanjing University of Aeronautics and Astronautics, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2932-5709","authenticated-orcid":false,"given":"Kai-Kuang","family":"Ma","sequence":"additional","affiliation":[{"name":"College of Electronic and Information Engineering, Nanjing University of Aeronautics and Astronautics, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-021-00431-x"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1126\/science.add4679"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-023-06004-9"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.ado5888"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161023"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2023.3330703"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-020-2939-8"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2023.3292368"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2023.3314929"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-021-3449-x"},{"key":"ref12","first-page":"1282","article-title":"Quantifying generalization in reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Cobbe","year":"2019"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3195549"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3178128"},{"key":"ref15","first-page":"2063","article-title":"Transfer learning for related reinforcement learning tasks via image-to-image translation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Gamrian","year":"2019"},{"key":"ref16","first-page":"12951","article-title":"On the importance of exploration for generalization in reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Jiang","year":"2023"},{"key":"ref17","first-page":"5027","article-title":"Improving exploration in evolution strategies for deep reinforcement learning via a population of novelty-seeking agents","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Conti","year":"2018"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i14.17457"},{"key":"ref19","first-page":"2784","article-title":"Stabilizing off-policy deep reinforcement learning from pixels","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Cetin","year":"2022"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TASE.2023.3342419"},{"key":"ref21","first-page":"13978","article-title":"Generalization in reinforcement learning with selective noise injection and information bottleneck","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Igl","year":"2019"},{"key":"ref22","article-title":"Efficient deep reinforcement learning requires regulating overfitting","author":"Li","year":"2023"},{"key":"ref23","first-page":"5402","article-title":"Automatic data augmentation for generalization in reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Raileanu","year":"2021"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561103"},{"key":"ref25","article-title":"Revisiting data augmentation in deep reinforcement learning","author":"Hu","year":"2024"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref27","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mnih","year":"2016"},{"key":"ref28","first-page":"12972","article-title":"A study of global and episodic bonuses for exploration in contextual MDPs","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Henaff","year":"2023"},{"key":"ref29","first-page":"49455","article-title":"To the max: Reinventing reward in reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Veviurko","year":"2024"},{"key":"ref30","first-page":"2750","article-title":"Exploration: A study of count-based exploration for deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Tang","year":"2017"},{"key":"ref31","first-page":"2721","article-title":"Count-based exploration with neural density models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ostrovski","year":"2017"},{"key":"ref32","first-page":"37631","article-title":"Exploration via elliptical episodic bonuses","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Henaff","year":"2022"},{"key":"ref33","first-page":"22594","article-title":"Flipping coins to estimate pseudocounts for exploration in reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lobel","year":"2023"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"ref35","first-page":"31855","article-title":"Byol-Explore: Exploration by bootstrapped prediction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Guo","year":"2022"},{"key":"ref36","first-page":"15220","article-title":"How to stay curious while avoiding noisy TVS using aleatoric uncertainty estimation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mavor-Parker","year":"2022"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1016\/j.tics.2017.11.005"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1038\/nn.4244"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1038\/nrn1178"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-022-31675-9"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-022-31440-y"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1038\/nn.4382"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuron.2014.12.053"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989381"},{"key":"ref45","volume-title":"The Art of Computer Programming","author":"Knuth","year":"1997"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"ref47","first-page":"1995","article-title":"Dueling network architectures for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wang","year":"2016"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2006.100"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1038\/nature09116"}],"container-title":["IEEE Transactions on Vehicular Technology"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/25\/11395176\/11122290.pdf?arnumber=11122290","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,13]],"date-time":"2026-02-13T20:50:26Z","timestamp":1771015826000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11122290\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2]]},"references-count":49,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tvt.2025.3597537","relation":{},"ISSN":["0018-9545","1939-9359"],"issn-type":[{"value":"0018-9545","type":"print"},{"value":"1939-9359","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2]]}}}