{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T04:47:46Z","timestamp":1785818866493,"version":"3.56.0"},"reference-count":72,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62206098"],"award-info":[{"award-number":["62206098"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100021171","name":"Guangdong Basic and Applied Basic Research Foundation","doi-asserted-by":"crossref","award":["2023A1515012896"],"award-info":[{"award-number":["2023A1515012896"]}],"id":[{"id":"10.13039\/501100021171","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Introduced Innovative Research and Development Team of Guangdong","award":["2023ZT10L145"],"award-info":[{"award-number":["2023ZT10L145"]}]},{"name":"Guangdong Regional Joint Foundation Key Project","award":["2022B1515120076"],"award-info":[{"award-number":["2022B1515120076"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Evol. Computat."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1109\/tevc.2025.3627631","type":"journal-article","created":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T17:12:07Z","timestamp":1761930727000},"page":"1761-1775","source":"Crossref","is-referenced-by-count":0,"title":["Evolutionary Reinforcement Learning With Late-Start Evolution and Clustering Archive"],"prefix":"10.1109","volume":"30","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-7222-6857","authenticated-orcid":false,"given":"Qiuting","family":"Cai","sequence":"first","affiliation":[{"name":"South China University of Technology","place":["Guangzhou, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0950-9968","authenticated-orcid":false,"given":"Ya-Hui","family":"Jia","sequence":"additional","affiliation":[{"name":"South China University of Technology","place":["Guangzhou, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4512-1764","authenticated-orcid":false,"given":"Kaitong","family":"Zheng","sequence":"additional","affiliation":[{"name":"South China University of Technology","place":["Guangzhou, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7844-5187","authenticated-orcid":false,"given":"Shiqi","family":"Ou","sequence":"additional","affiliation":[{"name":"South China University of Technology","place":["Guangzhou, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0843-5802","authenticated-orcid":false,"given":"Wei-Neng","family":"Chen","sequence":"additional","affiliation":[{"name":"South China University of Technology","place":["Guangzhou, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1561\/9781680835397"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/YAC51587.2020.9337653"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/nature14539"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"key":"ref5","article-title":"Playing Atari with deep reinforcement learning","author":"Mnih","year":"2013","journal-title":"arXiv:1312.5602"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3148435"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2022.3197298"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3128075"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/tevc.2024.3395699"},{"key":"ref10","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schulman"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2022.3199213"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2005.850290"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3283523"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3264540"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2022.3199045"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2021.3079985"},{"key":"ref17","article-title":"Deep neuroevolution: Genetic algorithms are a competitive alternative for training deep neural networks for reinforcement learning","author":"Such","year":"2017","journal-title":"arXiv:1712.06567"},{"key":"ref18","first-page":"1","article-title":"Evolution-guided policy gradient in reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Khadka"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5728"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i12.29289"},{"key":"ref21","article-title":"CEM-RL: Combining evolutionary and gradient-based methods for policy search","author":"Pourchot","year":"2018","journal-title":"arXiv:1810.01222"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.65109\/ulxn2173"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/s10479-005-5724-z"},{"key":"ref24","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref25","first-page":"16828","article-title":"The primacy bias in deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Nikishin"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3583131.3590512"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2024.3443913"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1162\/isal_a_00338"},{"key":"ref29","article-title":"Improving deep policy gradients with value function search","author":"Marchesini","year":"2023","journal-title":"arXiv:2302.10145"},{"key":"ref30","article-title":"Population based training of neural networks","author":"Jaderberg","year":"2017","journal-title":"arXiv:1711.09846"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3205455.3205486"},{"key":"ref32","article-title":"Deep reinforcement learning for robotic manipulation tasks using a genetic algorithm-based function optimizer","volume-title":"Encyclopedia with Semantic Computing and Robotic Intelligence","author":"Sehgal","year":"2023"},{"key":"ref33","article-title":"Sample-efficient automated deep reinforcement learning","author":"Franke","year":"2020","journal-title":"arXiv:2009.01555"},{"key":"ref34","article-title":"Go-explore: A new approach for hard-exploration problems","author":"Ecoffet","year":"2019","journal-title":"arXiv:1901.10995"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/205"},{"key":"ref36","first-page":"1","article-title":"Deep reinforcement learning in a handful of trials using probabilistic dynamics models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Chua"},{"key":"ref37","first-page":"2555","article-title":"Learning latent dynamics for planning from pixels","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Hafner"},{"key":"ref38","first-page":"277","article-title":"Model-predictive control via cross-entropy and gradient-based optimization","volume-title":"Proc. Learn. Dyn. Control","author":"Bharadhwaj"},{"key":"ref39","article-title":"Symbolic regression via neural-guided genetic programming population seeding","author":"Mundhenk","year":"2021","journal-title":"arXiv:2111.00053"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2021.106836"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1016\/j.swevo.2023.101236"},{"key":"ref42","article-title":"QD-RL: Efficient mixing of quality and diversity in reinforcement learning","author":"Cideron","year":"2020","journal-title":"arXiv:2006.08505"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3583131.3590503"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.4249\/scholarpedia.1482"},{"key":"ref45","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4757-4321-0","volume":"133","author":"Rubinstein","year":"2004","journal-title":"The Cross-Entropy Method: A Unified Approach to Combinatorial Optimization, Monte-Carlo Simulation, and Machine Learning"},{"key":"ref46","article-title":"The CMA evolution strategy: A tutorial","author":"Hansen","year":"2016","journal-title":"arXiv:1604.00772"},{"key":"ref47","article-title":"Continuous control with deep reinforcement learning","author":"Lillicrap","year":"2015","journal-title":"arXiv:1509.02971"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1007\/s12293-022-00375-8"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CEC55065.2022.9870209"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICNN.1995.488968"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-30111-7_49"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i18.30079"},{"key":"ref53","first-page":"3341","article-title":"Collaborative evolutionary reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Khadka"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-64583-0_26"},{"key":"ref55","first-page":"1","article-title":"Genetic soft updates for policy evolution in deep reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Marchesini"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.65109\/RJQI2619"},{"key":"ref57","first-page":"17455","article-title":"Cooperative heterogeneous deep reinforcement learning","volume":"33","author":"Zheng","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.65109\/zlpc7942"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2022.10.134"},{"key":"ref60","first-page":"1","article-title":"ERL-re 2: Efficient evolutionary reinforcement learning with shared state representation and individual policy representation","volume-title":"Proc. 11th Int. Conf. Learn. Represent.","author":"Jianye"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.3233\/faia240878"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2020.3046569"},{"key":"ref63","first-page":"716","article-title":"A survey and analysis of diversity measures in genetic programming","volume-title":"Proc. 4th Annu. Conf. Genet. Evol. Comput.","author":"Burke"},{"key":"ref64","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref65","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv:1707.06347"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref67","article-title":"Dota 2 with large scale deep reinforcement learning","author":"Berner","year":"2019","journal-title":"arXiv:1912.06680"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/TSG.2024.3399705"},{"key":"ref69","first-page":"1","article-title":"Evolutionary diversity optimization with clustering-based selection for reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Wang"},{"key":"ref70","first-page":"1","article-title":"PGPS: Coupling policy gradient with population-based search","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kim"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-020-00712-x"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1080\/14786451.2015.1100196"}],"container-title":["IEEE Transactions on Evolutionary Computation"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/4235\/11635978\/11223076.pdf?arnumber=11223076","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T04:40:11Z","timestamp":1785818411000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11223076\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":72,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tevc.2025.3627631","relation":{},"ISSN":["1089-778X","1089-778X","1941-0026"],"issn-type":[{"value":"1089-778X","type":"print"},{"value":"1089-778X","type":"print"},{"value":"1941-0026","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8]]}}}