{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T04:41:41Z","timestamp":1760676101825,"version":"build-2065373602"},"reference-count":77,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T00:00:00Z","timestamp":1760659200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T00:00:00Z","timestamp":1760659200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Front. Comput. Sci."],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s11704-025-41167-w","type":"journal-article","created":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T02:49:35Z","timestamp":1760669375000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Diversity from human feedback"],"prefix":"10.1007","volume":"20","author":[{"given":"Ren-Jian","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ke","family":"Xue","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu-Tong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peng","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao-Bo","family":"Fu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang","family":"Fu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chao","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,10,17]]},"reference":[{"key":"41167_CR1","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1145\/2001576.2001606","volume-title":"Proceedings of the 13th Annual Conference on Genetic and Evolutionary Computation","author":"J Lehman","year":"2011","unstructured":"Lehman J, Stanley K O. Evolving a diversity of virtual creatures through novelty search and local competition. In: Proceedings of the 13th Annual Conference on Genetic and Evolutionary Computation. 2011, 211\u2013218"},{"key":"41167_CR2","first-page":"5032","volume-title":"Proceedings of the 32nd International Conference on Neural Information Processing Systems","author":"E Conti","year":"2018","unstructured":"Conti E, Madhavan V, Such F P, Lehman J, Stanley K O, Clune J. Improving exploration in evolution strategies for deep reinforcement learning via a population of novelty-seeking agents. In: Proceedings of the 32nd International Conference on Neural Information Processing Systems. 2018, 5032\u20135043"},{"key":"41167_CR3","volume-title":"Proceedings of the 7th International Conference on Learning Representations","author":"B Eysenbach","year":"2019","unstructured":"Eysenbach B, Gupta A, Ibarz J, Levine S. Diversity is all you need: learning skills without a reward function. In: Proceedings of the 7th International Conference on Learning Representations. 2019"},{"key":"41167_CR4","first-page":"1515","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing System","author":"J Parker-Holder","year":"2020","unstructured":"Parker-Holder J, Pacchiano A, Choromanski K, Roberts S. Effective diversity in population based reinforcement learning. In: Proceedings of the 34th International Conference on Neural Information Processing System. 2020, 1515"},{"key":"41167_CR5","volume-title":"Proceedings of the 11th International Conference on Learning Representations","author":"F Chalumeau","year":"2023","unstructured":"Chalumeau F, Boige R, Lim B, Mac\u00e9 V, Allard M, Flajolet A, Cully A, Pierrot T. Neuroevolution is a competitive alternative to reinforcement learning for skill discovery. In: Proceedings of the 11th International Conference on Learning Representations. 2023"},{"key":"41167_CR6","volume-title":"Proceedings of the 11th International Conference on Learning Representations","author":"S Wu","year":"2023","unstructured":"Wu S, Yao J, Fu H, Tian Y, Qian C, Yang Y, Fu Q, Wei Y. Quality-similar diversity via population based reinforcement learning. In: Proceedings of the 11th International Conference on Learning Representations. 2023"},{"key":"41167_CR7","first-page":"2964","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"J Yao","year":"2023","unstructured":"Yao J, Liu W, Fu H, Yang Y, McAleer S, Fu Q, Yang W. Policy space diversity for non-transitive games. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. 2023, 2964"},{"issue":"1","key":"41167_CR8","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1016\/j.inffus.2004.04.004","volume":"6","author":"G Brown","year":"2005","unstructured":"Brown G, Wyatt J, Harris R, Yao X. Diversity creation methods: a survey and categorisation. Information Fusion, 2005, 6(1): 5\u201320","journal-title":"Information Fusion"},{"key":"41167_CR9","doi-asserted-by":"publisher","DOI":"10.1201\/b12207","volume-title":"Ensemble Methods: Foundations and Algorithms","author":"Z-H Zhou","year":"2012","unstructured":"Zhou Z-H. Ensemble Methods: Foundations and Algorithms. New York: CRC Press, 2012"},{"issue":"2","key":"41167_CR10","first-page":"23","volume":"50","author":"H M Gomes","year":"2017","unstructured":"Gomes H M, Barddal J P, Enembreck F, Bifet A. A survey on ensemble learning for data stream classification. ACM Computing Surveys (CSUR), 2017, 50(2): 23","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"41167_CR11","first-page":"569","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems","author":"G An","year":"2021","unstructured":"An G, Moon S, Kim J-H, Song H O. Uncertainty-based offline reinforcement learning with diversified Q-ensemble. In: Proceedings of the 35th International Conference on Neural Information Processing Systems. 2021, 569"},{"issue":"6","key":"41167_CR12","doi-asserted-by":"publisher","first-page":"3545","DOI":"10.1007\/s10994-023-06429-3","volume":"113","author":"Y-X He","year":"2024","unstructured":"He Y-X, Wu Y-C, Qian C, Zhou Z-H. Margin distribution and structural diversity guided ensemble pruning. Machine Learning, 2024, 113(6): 3545\u20133567","journal-title":"Machine Learning"},{"key":"41167_CR13","first-page":"768","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems","author":"M C Fontaine","year":"2021","unstructured":"Fontaine M C, Nikolaidis S. Differentiable quality diversity. In: Proceedings of the 35th International Conference on Neural Information Processing Systems. 2021, 768"},{"issue":"1","key":"41167_CR14","doi-asserted-by":"publisher","first-page":"181304","DOI":"10.1007\/s11704-022-2385-x","volume":"18","author":"H Tao","year":"2024","unstructured":"Tao H, Long C, Xiao C. CRD-CGAN: category-consistent and relativistic constraints for diverse text-to-image generation. Frontiers of Computer Science, 2024, 18(1): 181304","journal-title":"Frontiers of Computer Science"},{"key":"41167_CR15","volume-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems","author":"M Samvelyan","year":"2024","unstructured":"Samvelyan M, Raparthy S C, Lupu A, Hambro E, Markosyan A H, Bhatt M, Mao Y, Jiang M, Parker-Holder J, Foerster J, Rockt\u00e4schel T, Raileanu R. Rainbow teaming: open-ended generation of diverse adversarial prompts. In: Proceedings of the 38th International Conference on Neural Information Processing Systems. 2024"},{"issue":"3","key":"41167_CR16","first-page":"11","volume":"2","author":"A Do","year":"2022","unstructured":"Do A, Guo M, Neumann A, Neumann F. Analysis of evolutionary diversity optimization for permutation problems. ACM Transactions on Evolutionary Learning, 2022, 2(3): 11","journal-title":"ACM Transactions on Evolutionary Learning"},{"key":"41167_CR17","first-page":"250","volume-title":"Proceedings of the 17th International Conference on Parallel Problem Solving from Nature","author":"A Nikfarjam","year":"2022","unstructured":"Nikfarjam A, Moosavi A, Neumann A, Neumann F. Computing high-quality solutions for the patient admission scheduling problem using evolutionary diversity optimisation. In: Proceedings of the 17th International Conference on Parallel Problem Solving from Nature. 2022, 250\u2013264"},{"key":"41167_CR18","first-page":"1336","volume-title":"Proceedings of the 1st Annual Conference on Genetic and Evolutionary Computation","author":"C C Maley","year":"1999","unstructured":"Maley C C. Four steps toward open-ended evolution. In: Proceedings of the 1st Annual Conference on Genetic and Evolutionary Computation. 1999, 1336\u20131343"},{"issue":"2","key":"41167_CR19","doi-asserted-by":"publisher","first-page":"167","DOI":"10.1142\/S1469026803000914","volume":"3","author":"R K Standish","year":"2003","unstructured":"Standish R K. Open-ended artificial evolution. International Journal of Computational Intelligence and Applications, 2003, 3(2): 167\u2013175","journal-title":"International Journal of Computational Intelligence and Applications"},{"key":"41167_CR20","first-page":"73","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems","author":"X Liu","year":"2021","unstructured":"Liu X, Jia H, Wen Y, Hu Y, Chen Y, Fan C, Hu Z, Yang Y. Towards unifying behavioral and response diversity for open-ended learning in zero-sum games. In: Proceedings of the 35th International Conference on Neural Information Processing Systems. 2021, 73"},{"key":"41167_CR21","first-page":"4940","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"M Jiang","year":"2021","unstructured":"Jiang M, Grefenstette E, Rockt\u00e4schel T. Prioritized level replay. In: Proceedings of the 38th International Conference on Machine Learning. 2021, 4940\u20134950"},{"key":"41167_CR22","first-page":"145","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing System","author":"M Jiang","year":"2021","unstructured":"Jiang M, Dennis M, Parker-Holder J, Foerster J, Grefenstette E, Rockt\u00e4schel T. Replay-guided adversarial environment design. In: Proceedings of the 35th International Conference on Neural Information Processing System. 2021, 145"},{"key":"41167_CR23","first-page":"5922","volume-title":"Proceedings of the 35th AAAI Conference on Artificial Intelligence","author":"M C Fontaine","year":"2021","unstructured":"Fontaine M C, Liu R, Khalifa A, Modi J, Togelius J, Hoover A K, Nikolaidis S. Illuminating Mario scenes in the latent space of a generative adversarial network. In: Proceedings of the 35th AAAI Conference on Artificial Intelligence. 2021, 5922\u20135930"},{"key":"41167_CR24","first-page":"2737","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"V Bhatt","year":"2022","unstructured":"Bhatt V, Tjanaka B, Fontaine M C, Nikolaidis S. Deep surrogate assisted generation of environments. In: Proceedings of the 36th International Conference on Neural Information Processing Systems. 2022, 2737"},{"key":"41167_CR25","first-page":"17473","volume-title":"Proceedings of the 39th International Conference on Machine Learning","author":"J Parker-Holder","year":"2022","unstructured":"Parker-Holder J, Jiang M, Dennis M, Samvelyan M, Foerster J, Grefenstette E, Rockt\u00e4schel T. Evolving curricula with regret-based environment design. In: Proceedings of the 39th International Conference on Machine Learning. 2022, 17473\u201317498"},{"key":"41167_CR26","first-page":"5503","volume-title":"Proceedings of the 32nd International Joint Conference on Artificial Intelligence","author":"Y Zhang","year":"2023","unstructured":"Zhang Y, Fontaine M C, Bhatt V, Nikolaidis S, Li J. Multi-robot coordination and layout design for automated warehousing. In: Proceedings of the 32nd International Joint Conference on Artificial Intelligence. 2023, 5503\u20135511"},{"key":"41167_CR27","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4471-0793-4","volume-title":"Combining Artificial Neural Nets","author":"A J C Sharkey","year":"1999","unstructured":"Sharkey A J C. Combining Artificial Neural Nets. Springer, 1999"},{"key":"41167_CR28","first-page":"19731","volume-title":"Proceedings of the 39th International Conference on Machine Learning","author":"H Sheikh","year":"2022","unstructured":"Sheikh H, Frisbee K, Phielipp M. DNS: determinantal point process based neural network sampler for ensemble reinforcement learning. In: Proceedings of the 39th International Conference on Machine Learning. 2022, 19731\u201319746"},{"key":"41167_CR29","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-13-5956-9","volume-title":"Evolutionary Learning: Advances in Theories and Algorithms","author":"Z-H Zhou","year":"2019","unstructured":"Zhou Z-H, Yu Y, Qian C. Evolutionary Learning: Advances in Theories and Algorithms. Singapore: Springer, 2019"},{"issue":"2","key":"41167_CR30","doi-asserted-by":"publisher","first-page":"120102","DOI":"10.1007\/s11432-023-3895-3","volume":"67","author":"P Yang","year":"2024","unstructured":"Yang P, Zhang L, Liu H, Li G. Reducing idleness in financial cloud services via multi-objective evolutionary reinforcement learning based load balancer. Science China Information Sciences, 2024, 67(2): 120102","journal-title":"Science China Information Sciences"},{"key":"41167_CR31","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"K Xue","year":"2024","unstructured":"Xue K, Wang R-J, Li P, Li D, Hao J, Qian C. Sample-efficient quality-diversity by cooperative coevolution. In: Proceedings of the 12th International Conference on Learning Representations. 2024"},{"key":"41167_CR32","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"R-J Wang","year":"2024","unstructured":"Wang R-J, Xue K, Guan C, Qian C. Quality-diversity with limited resources. In: Proceedings of the 41st International Conference on Machine Learning. 2024"},{"key":"41167_CR33","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2024.3443913","volume-title":"IEEE Transactions on Evolutionary Computation","author":"P Li","year":"2024","unstructured":"Li P, Hao J, Tang H, Fu X, Zhen Y, Tang K. Bridging evolutionary algorithms and reinforcement learning: a comprehensive survey on hybrid algorithms. IEEE Transactions on Evolutionary Computation, 2024, DOI: https:\/\/doi.org\/10.1109\/TEVC.2024.3443913"},{"key":"41167_CR34","doi-asserted-by":"publisher","first-page":"40","DOI":"10.3389\/frobt.2016.00040","volume":"3","author":"J K Pugh","year":"2016","unstructured":"Pugh J K, Soros L B, Stanley K O. Quality diversity: a new frontier for evolutionary computation. Frontiers in Robotics and AI, 2016, 3: 40","journal-title":"Frontiers in Robotics and AI"},{"key":"41167_CR35","first-page":"991","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"W Fu","year":"2023","unstructured":"Fu W, Du W, Li J, Chen S, Zhang J, Wu Y. Iteratively learn diverse strategies with state distance information. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. 2023, 991"},{"key":"41167_CR36","doi-asserted-by":"publisher","first-page":"965","DOI":"10.1145\/2001576.2001708","volume-title":"Proceedings of the 13th Annual Conference on Genetic and Evolutionary Computation","author":"S Kistemaker","year":"2011","unstructured":"Kistemaker S, Whiteson S. Critical factors in the performance of novelty search. In: Proceedings of the 13th Annual Conference on Genetic and Evolutionary Computation. 2011, 965\u2013972"},{"issue":"6","key":"41167_CR37","doi-asserted-by":"publisher","first-page":"1539","DOI":"10.1109\/TEVC.2022.3159855","volume":"26","author":"L Grillotti","year":"2022","unstructured":"Grillotti L, Cully A. Unsupervised behavior discovery with quality-diversity optimization. IEEE Transactions on Evolutionary Computation, 2022, 26(6): 1539\u20131552","journal-title":"IEEE Transactions on Evolutionary Computation"},{"key":"41167_CR38","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1145\/3377930.3390232","volume-title":"Proceedings of the 2020 Genetic and Evolutionary Computation Conference","author":"M C Fontaine","year":"2020","unstructured":"Fontaine M C, Togelius J, Nikolaidis S, Hoover A K. Covariance matrix adaptation for the rapid illumination of behavior space. In: Proceedings of the 2020 Genetic and Evolutionary Computation Conference. 2020, 94\u2013102"},{"key":"41167_CR39","unstructured":"Mouret J B, Clune J. Illuminating search spaces by mapping elites. 2015, arXiv preprint arXiv: 1504.04909"},{"key":"41167_CR40","doi-asserted-by":"publisher","first-page":"866","DOI":"10.1145\/3449639.3459304","volume-title":"Proceedings of the Genetic and Evolutionary Computation Conference","author":"O Nilsson","year":"2021","unstructured":"Nilsson O, Cully A. Policy gradient assisted MAP-Elites. In: Proceedings of the Genetic and Evolutionary Computation Conference. 2021, 866\u2013875"},{"key":"41167_CR41","unstructured":"Vinyals O, Ewalds T, Bartunov S, Georgiev P, Vezhnevets A S, et al. StarCraft II: a new challenge for reinforcement learning. 2017, arXiv preprint arXiv: 1708.04782"},{"key":"41167_CR42","unstructured":"OpenAI, Berner C, Brockman G, Chan B, Cheung V, et al. Dota 2 with large scale deep reinforcement learning. 2019, arXiv preprint arXiv: 1912.06680"},{"issue":"7553","key":"41167_CR43","doi-asserted-by":"publisher","first-page":"503","DOI":"10.1038\/nature14422","volume":"521","author":"A Cully","year":"2015","unstructured":"Cully A, Clune J, Tarapore D, Mouret J-B. Robots that can adapt like animals. Nature, 2015, 521(7553): 503\u2013507","journal-title":"Nature"},{"key":"41167_CR44","first-page":"2023","volume-title":"Transactions on Machine Learning Research","author":"B Lim","year":"2023","unstructured":"Lim B, Allard M, Grillotti L, Cully A. Accelerated quality-diversity through massive parallelism. Transactions on Machine Learning Research, 2023, 2023"},{"issue":"1","key":"41167_CR45","first-page":"108","volume":"25","author":"F Chalumeau","year":"2024","unstructured":"Chalumeau F, Lim B, Boige R, Allard M, Grillotti L, Flageat M, Mac\u00e9 V, Richard G, Flajolet A, Pierrot T, Cully A. QDax: a library for quality-diversity and population-based algorithms with hardware acceleration. The Journal of Machine Learning Research, 2024, 25(1): 108","journal-title":"The Journal of Machine Learning Research"},{"key":"41167_CR46","unstructured":"Wang R-J, Xue K, Wang Y, Yang P, Fu H, Fu Q, Qian C. Diversity from human feedback. 2023, arXiv preprint arXiv: 2310.06648v1"},{"key":"41167_CR47","unstructured":"Ding L, Zhang J, Clune J, Spector L, Lehman J. Quality diversity through human feedback. 2023, arXiv preprint arXiv:2310.12103v1"},{"key":"41167_CR48","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"L Ding","year":"2024","unstructured":"Ding L, Zhang J, Clune J, Spector L, Lehman J. Quality diversity through human feedback: towards open-ended diversity-driven optimization. In: Proceedings of the 41st International Conference on Machine Learning. 2024"},{"key":"41167_CR49","doi-asserted-by":"publisher","first-page":"109","DOI":"10.1007\/978-3-030-66515-9_4","volume-title":"Black Box Optimization, Machine Learning, and No-Free Lunch Theorems","author":"K Chatzilygeroudis","year":"2021","unstructured":"Chatzilygeroudis K, Cully A, Vassiliades V, Mouret J-B. Quality-diversity optimization: a novel branch of stochastic optimization. In: Pardalos P M, Rasskazova V, Vrahatis M N, eds. Black Box Optimization, Machine Learning, and No-Free Lunch Theorems. Cham: Springer, 2021, 109\u2013135"},{"issue":"2","key":"41167_CR50","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1109\/TEVC.2017.2704781","volume":"22","author":"A Cully","year":"2018","unstructured":"Cully A, Demiris Y. Quality and diversity optimization: a unifying modular framework. IEEE Transactions on Evolutionary Computation, 2018, 22(2): 245\u2013259","journal-title":"IEEE Transactions on Evolutionary Computation"},{"key":"41167_CR51","volume-title":"Proceedings of the 10th International Conference on Learning Representations","author":"Y Wang","year":"2022","unstructured":"Wang Y, Xue K, Qian C. Evolutionary diversity optimization with clustering-based selection for reinforcement learning. In: Proceedings of the 10th International Conference on Learning Representations. 2022"},{"key":"41167_CR52","first-page":"4335","volume-title":"Proceedings of the 32nd International Joint Conference on Artificial Intelligence","author":"R-J Wang","year":"2023","unstructured":"Wang R-J, Xue K, Shang H, Qian C, Fu H, Fu Q. Multi-objective optimization-based selection for quality-diversity by non-surrounded-dominated sorting. In: Proceedings of the 32nd International Joint Conference on Artificial Intelligence. 2023, 4335\u20134343"},{"key":"41167_CR53","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1145\/3377930.3390217","volume-title":"Proceedings of the 2020 Genetic and Evolutionary Computation Conference","author":"C Colas","year":"2020","unstructured":"Colas C, Madhavan V, Huizinga J, Clune J. Scaling MAP-Elites to deep neuroevolution. In: Proceedings of the 2020 Genetic and Evolutionary Computation Conference. 2020, 67\u201375"},{"key":"41167_CR54","doi-asserted-by":"publisher","first-page":"1075","DOI":"10.1145\/3512290.3528845","volume-title":"Proceedings of the Genetic and Evolutionary Computation Conference","author":"T Pierrot","year":"2022","unstructured":"Pierrot T, Mac\u00e9 V, Chalumeau F, Flajolet A, Cideron G, Beguir K, Cully A, Sigaud O, Perrin-Gilbert N. Diversity policy gradient for sample efficient quality-diversity optimization. In: Proceedings of the Genetic and Evolutionary Computation Conference. 2022, 1075\u20131083"},{"key":"41167_CR55","first-page":"6994","volume-title":"Proceedings of the 33rd International Joint Conference on Artificial Intelligence","author":"C Qian","year":"2024","unstructured":"Qian C, Xue K, Wang R-J. Quality-diversity algorithms can provably be helpful for optimization. In: Proceedings of the 33rd International Joint Conference on Artificial Intelligence. 2024, 6994\u20137002"},{"key":"41167_CR56","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1145\/3321707.3321804","volume-title":"Proceedings of the Genetic and Evolutionary Computation Conference","author":"A Cully","year":"2019","unstructured":"Cully A. Autonomous skill discovery with quality-diversity and unsupervised descriptors. In: Proceedings of the Genetic and Evolutionary Computation Conference. 2019, 81\u201389"},{"key":"41167_CR57","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1145\/3512290.3528837","volume-title":"Proceedings of the Genetic and Evolutionary Computation Conference","author":"L Grillotti","year":"2022","unstructured":"Grillotti L, Cully A. Relevance-guided unsupervised discovery of abilities with quality-diversity algorithms. In: Proceedings of the Genetic and Evolutionary Computation Conference. 2022, 77\u201385"},{"key":"41167_CR58","volume-title":"Proceedings of the 8th International Conference on Learning Representations","author":"A Sharma","year":"2020","unstructured":"Sharma A, Gu S, Levine S, Kumar V, Hausman K. Dynamics-aware unsupervised discovery of skills. In: Proceedings of the 8th International Conference on Learning Representations. 2020"},{"key":"41167_CR59","first-page":"1","volume-title":"Proceedings of 2019 IEEE Conference on Games","author":"A Alvarez","year":"2019","unstructured":"Alvarez A, Dahlskog S, Font J, Togelius J. Empowering quality diversity in dungeon design with interactive constrained MAP-Elites. In: Proceedings of 2019 IEEE Conference on Games. 2019, 1\u20138"},{"issue":"2","key":"41167_CR60","doi-asserted-by":"publisher","first-page":"202","DOI":"10.1109\/TG.2020.3046133","volume":"14","author":"A Alvarez","year":"2022","unstructured":"Alvarez A, Dahlskog S, Font J, Togelius J. Interactive constrained MAP-Elites: analysis and evaluation of the expressiveness of the feature dimensions. IEEE Transactions on Games, 2022, 14(2): 202\u2013211","journal-title":"IEEE Transactions on Games"},{"issue":"136","key":"41167_CR61","first-page":"4945","volume":"18","author":"C Wirth","year":"2017","unstructured":"Wirth C, Akrour R, Neumann G, F\u00fcrnkranz J. A survey of preference-based reinforcement learning methods. The Journal of Machine Learning Research, 2017, 18(136): 4945\u20134990","journal-title":"The Journal of Machine Learning Research"},{"key":"41167_CR62","first-page":"4299","volume-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","author":"P F Christiano","year":"2017","unstructured":"Christiano P F, Leike J, Brown T B, Martic M, Legg S, Amodei D. Deep reinforcement learning from human preferences. In: Proceedings of the 31st International Conference on Neural Information Processing Systems. 2017, 4299\u20134307"},{"key":"41167_CR63","first-page":"2023","volume-title":"Transactions on Machine Learning Research","author":"S Casper","year":"2023","unstructured":"Casper S, Davies X, Shi C, Gilbert T K, Scheurer J, Rando J, Freedman R, Korbak T, Lindner D, Freire P, Wang T, Marks S, S\u00e9gerie C-R, Carroll M, Peng A, Christoffersen P J K, Damani M, Slocum S, Anwar U, Siththaranjan A, Nadeau M, Michaud E J, Pfau J, Krasheninnikov D, Chen X, Langosco L, Hase P, Biyik E, Dragan A D, Krueger D, Sadigh D, Hadfield-Menell D. Open problems and fundamental limitations of reinforcement learning from human feedback. Transactions on Machine Learning Research, 2023, 2023"},{"key":"41167_CR64","first-page":"7444","volume-title":"Proceedings of the 36th International Conference on Machine Learning","author":"M Zhang","year":"2019","unstructured":"Zhang M, Vikram S, Smith L M, Abbeel P, Johnson M J, Levine S. SOLAR: deep structured representations for model-based reinforcement learning. In: Proceedings of the 36th International Conference on Machine Learning. 2019, 7444\u20137453"},{"key":"41167_CR65","first-page":"253","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing System","author":"N Stiennon","year":"2020","unstructured":"Stiennon N, Ouyang L, Wu J, Ziegler D M, Lowe R, Voss C, Radford A, Amodei D, Christiano P. Learning to summarize from human feedback. In: Proceedings of the 34th International Conference on Neural Information Processing System. 2020, 253"},{"key":"41167_CR66","first-page":"2011","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"L Ouyang","year":"2022","unstructured":"Ouyang L, Wu J, Jiang X, Almeida D, Wainwright C L, Mishkin P, Zhang C, Agarwal S, Slama K, Ray A, Schulman J, Hilton J, Kelton F, Miller L, Simens M, Askell A, Welinder P, Christiano P, Leike J, Lowe R. Training language models to follow instructions with human feedback. In: Proceedings of the 36th International Conference on Neural Information Processing Systems. 2022, 2011"},{"key":"41167_CR67","unstructured":"Open-AI, Achiam J, Adler S, Agarwal S, Ahmad L, et al. GPT-4 technical report. 2023, arXiv preprint arXiv: 2303.08774"},{"key":"41167_CR68","unstructured":"Gemini Team Google. Gemini: A family of highly capable multimodal models. 2023, arXiv preprint arXiv:2312.11805"},{"key":"41167_CR69","first-page":"1135","volume-title":"Proceedings of the 2023 International Conference on Autonomous Agents and Multiagent Systems","author":"M Hussonnois","year":"2023","unstructured":"Hussonnois M, Karimpanal T G, Rana S. Controlled diversity with preference: towards learning a diverse set of desired skills. In: Proceedings of the 2023 International Conference on Autonomous Agents and Multiagent Systems. 2023, 1135\u20131143"},{"issue":"8","key":"41167_CR70","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J. Long short-term memory. Neural Computation, 1997, 9(8): 1735\u20131780","journal-title":"Neural Computation"},{"key":"41167_CR71","first-page":"334","volume-title":"Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing","author":"C Dyer","year":"2015","unstructured":"Dyer C, Ballesteros M, Ling W, Matthews A, Smith N A. Transition-based dependency parsing with stack long short-term memory. In: Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing. 2015, 334\u2013343"},{"issue":"3\u20134","key":"41167_CR72","first-page":"324","volume":"39","author":"R A Bradley","year":"1952","unstructured":"Bradley R A, Terry M E. Rank analysis of incomplete block designs: I. The method of paired comparisons. Biometrika, 1952, 39(3\u20134): 324\u2013345","journal-title":"Biometrika"},{"key":"41167_CR73","volume-title":"Proceedings of the 32nd International Conference on Neural Information Processing Systems","author":"B Ibarz","year":"2018","unstructured":"Ibarz B, Leike J, Pohlen T, Irving G, Legg S, Amodei D. Reward learning from human preferences and demonstrations in Atari. In: Proceedings of the 32nd International Conference on Neural Information Processing Systems. 2018"},{"key":"41167_CR74","first-page":"6152","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"K Lee","year":"2021","unstructured":"Lee K, Smith L M, Abbeel P. PEBBLE: feedback-efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training. In: Proceedings of the 38th International Conference on Machine Learning. 2021, 6152\u20136163"},{"key":"41167_CR75","unstructured":"van den Oord A, Li Y, Vinyals O. Representation learning with contrastive predictive coding. 2018, arXiv preprint arXiv: 1807.03748"},{"key":"41167_CR76","volume-title":"Proceedings of the 8th International Conference on Learning Representations","author":"A P Badia","year":"2020","unstructured":"Badia A P, Sprechmann P, Vitvitskyi A, Guo Z D, Piot B, Kapturowski S, Tieleman O, Arjovsky M, Pritzel A, Bolt A, Blundell C. Never give up: learning directed exploration strategies. In: Proceedings of the 8th International Conference on Learning Representations. 2020"},{"key":"41167_CR77","first-page":"507","volume-title":"Proceedings of the 37th International Conference on Machine Learning","author":"A P Badia","year":"2020","unstructured":"Badia A P, Piot B, Kapturowski S, Sprechmann P, Vitvitskyi A, Guo Z D, Blundell C. Agent57: outperforming the Atari human benchmark. In: Proceedings of the 37th International Conference on Machine Learning. 2020, 507\u2013517"}],"container-title":["Frontiers of Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-025-41167-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11704-025-41167-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-025-41167-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T04:04:41Z","timestamp":1760673881000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11704-025-41167-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,17]]},"references-count":77,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["41167"],"URL":"https:\/\/doi.org\/10.1007\/s11704-025-41167-w","relation":{},"ISSN":["2095-2228","2095-2236"],"issn-type":[{"type":"print","value":"2095-2228"},{"type":"electronic","value":"2095-2236"}],"subject":[],"published":{"date-parts":[[2025,10,17]]},"assertion":[{"value":"30 October 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 March 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 October 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare that they have no competing interests or financial conflicts to disclose.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"2002320"}}