{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T18:16:05Z","timestamp":1783707365312,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T00:00:00Z","timestamp":1783641600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,13]]},"DOI":"10.1145\/3795095.3805197","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T17:44:29Z","timestamp":1783705469000},"page":"289-299","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Distributional Value Estimation Without Target Networks for Robust Quality-Diversity"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8115-2887","authenticated-orcid":false,"given":"Behrad","family":"Koohy","sequence":"first","affiliation":[{"name":"luffy.ai, Oxford, United Kingdom"},{"name":"University of Southampton, Southampton, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8765-1962","authenticated-orcid":false,"given":"Jamie","family":"Bayne","sequence":"additional","affiliation":[{"name":"luffy.ai, Oxford, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,10]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3712256.3726310"},{"key":"e_1_3_2_1_2_1","volume-title":"International conference on machine learning. PMLR, 342\u2013350","author":"Balduzzi David","year":"2017","unstructured":"David Balduzzi, Marcus Frean, Lennox Leary, JP Lewis, Kurt Wan-Duo Ma, and Brian McWilliams. 2017. The shattered gradients problem: If resnets are the answer, then what is the question?. In International conference on machine learning. PMLR, 342\u2013350."},{"key":"e_1_3_2_1_3_1","volume-title":"International conference on machine learning. PMLR, 449\u2013458","author":"Bellemare Marc G","year":"2017","unstructured":"Marc G Bellemare, Will Dabney, and R\u00e9mi Munos. 2017. A distributional perspective on reinforcement learning. In International conference on machine learning. PMLR, 449\u2013458."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/14207.001.0001"},{"key":"e_1_3_2_1_5_1","volume-title":"Crossq: Batch normalization in deep reinforcement learning for greater sample efficiency and simplicity. arXiv preprint arXiv:1902.05605","author":"Bhatt Aditya","year":"2019","unstructured":"Aditya Bhatt, Daniel Palenicek, Boris Belousov, Max Argus, Artemij Amiranashvili, Thomas Brox, and Jan Peters. 2019. Crossq: Batch normalization in deep reinforcement learning for greater sample efficiency and simplicity. arXiv preprint arXiv:1902.05605 (2019)."},{"key":"e_1_3_2_1_6_1","first-page":"1","article-title":"QDax: A library for quality-diversity and population-based algorithms with hardware acceleration","volume":"25","author":"Chalumeau Felix","year":"2024","unstructured":"Felix Chalumeau, Bryan Lim, Raphael Boige, Maxime Allard, Luca Grillotti, Manon Flageat, Valentin Mac\u00e9, Guillaume Richard, Arthur Flajolet, Thomas Pierrot, et al. 2024. QDax: A library for quality-diversity and population-based algorithms with hardware acceleration. Journal of Machine Learning Research 25, 108 (2024), 1\u201316.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_7_1","volume-title":"Randomized ensembled double q-learning: Learning fast without a model. arXiv preprint arXiv:2101.05982","author":"Chen Xinyue","year":"2021","unstructured":"Xinyue Chen, Che Wang, Zijian Zhou, and Keith Ross. 2021. Randomized ensembled double q-learning: Learning fast without a model. arXiv preprint arXiv:2101.05982 (2021)."},{"key":"e_1_3_2_1_8_1","volume-title":"Beyond the rainbow: High performance deep reinforcement learning on a desktop pc. arXiv preprint arXiv:2411.03820","author":"Clark Tyler","year":"2024","unstructured":"Tyler Clark, Mark Towers, Christine Evers, and Jonathon Hare. 2024. Beyond the rainbow: High performance deep reinforcement learning on a desktop pc. arXiv preprint arXiv:2411.03820 (2024)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11791"},{"key":"e_1_3_2_1_10_1","volume-title":"Deep Reinforcement Learning Workshop NeurIPS","author":"D'Oro Pierluca","year":"2022","unstructured":"Pierluca D'Oro, Max Schwarzer, Evgenii Nikishin, Pierre-Luc Bacon, Marc G Bellemare, and Aaron Courville. 2022. Sample-efficient reinforcement learning by breaking the replay ratio barrier. In Deep Reinforcement Learning Workshop NeurIPS 2022."},{"key":"e_1_3_2_1_11_1","volume-title":"Diversity is all you need: Learning skills without a reward function. arXiv preprint arXiv:1802.06070","author":"Eysenbach Benjamin","year":"2018","unstructured":"Benjamin Eysenbach, Abhishek Gupta, Julian Ibarz, and Sergey Levine. 2018. Diversity is all you need: Learning skills without a reward function. arXiv preprint arXiv:1802.06070 (2018)."},{"key":"e_1_3_2_1_12_1","volume-title":"Yevgen Chebotar, Ted Xiao, Alex Irpan, Sergey Levine, Pablo Samuel Castro, Aleksandra Faust, et al.","author":"Farebrother Jesse","year":"2024","unstructured":"Jesse Farebrother, Jordi Orbay, Quan Vuong, Adrien Ali Ta\u00efga, Yevgen Chebotar, Ted Xiao, Alex Irpan, Sergey Levine, Pablo Samuel Castro, Aleksandra Faust, et al. 2024. Stop regressing: Training value functions via classification for scalable deep rl. arXiv preprint arXiv:2403.03950 (2024)."},{"key":"e_1_3_2_1_13_1","volume-title":"International conference on machine learning. PMLR, 3061\u20133071","author":"Fedus William","year":"2020","unstructured":"William Fedus, Prajit Ramachandran, Rishabh Agarwal, Yoshua Bengio, Hugo Larochelle, Mark Rowland, and Will Dabney. 2020. Revisiting fundamentals of experience replay. In International conference on machine learning. PMLR, 3061\u20133071."},{"key":"e_1_3_2_1_14_1","volume-title":"Brax-a differentiable physics engine for large scale rigid body simulation. arXiv preprint arXiv:2106.13281","author":"Freeman C Daniel","year":"2021","unstructured":"C Daniel Freeman, Erik Frey, Anton Raichuk, Sertan Girgin, Igor Mordatch, and Olivier Bachem. 2021. Brax-a differentiable physics engine for large scale rigid body simulation. arXiv preprint arXiv:2106.13281 (2021)."},{"key":"e_1_3_2_1_15_1","volume-title":"International conference on machine learning. PMLR, 1587\u20131596","author":"Fujimoto Scott","year":"2018","unstructured":"Scott Fujimoto, Herke Hoof, and David Meger. 2018. Addressing function approximation error in actor-critic methods. In International conference on machine learning. PMLR, 1587\u20131596."},{"key":"e_1_3_2_1_16_1","volume-title":"International Conference on Machine Learning. PMLR, 3734\u20133744","author":"Gogianu Florin","year":"2021","unstructured":"Florin Gogianu, Tudor Berariu, Mihaela C Rosca, Claudia Clopath, Lucian Busoniu, and Razvan Pascanu. 2021. Spectral normalisation for deep reinforcement learning: an optimisation perspective. In International Conference on Machine Learning. PMLR, 3734\u20133744."},{"key":"e_1_3_2_1_17_1","volume-title":"Improved training of wasserstein gans. Advances in neural information processing systems 30","author":"Gulrajani Ishaan","year":"2017","unstructured":"Ishaan Gulrajani, Faruk Ahmed, Martin Arjovsky, Vincent Dumoulin, and Aaron C Courville. 2017. Improved training of wasserstein gans. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_18_1","volume-title":"International conference on machine learning. Pmlr","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja, Aurick Zhou, Pieter Abbeel, and Sergey Levine. 2018. Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In International conference on machine learning. Pmlr, 1861\u20131870."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_38"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"e_1_3_2_1_21_1","volume-title":"Dropout q-functions for doubly efficient reinforcement learning. arXiv preprint arXiv:2110.02034","author":"Hiraoka Takuya","year":"2021","unstructured":"Takuya Hiraoka, Takahisa Imagawa, Taisei Hashimoto, Takashi Onishi, and Yoshimasa Tsuruoka. 2021. Dropout q-functions for doubly efficient reinforcement learning. arXiv preprint arXiv:2110.02034 (2021)."},{"key":"e_1_3_2_1_22_1","volume-title":"Dissecting deep rl with high update ratios: Combatting value divergence. arXiv preprint arXiv:2403.05996","author":"Hussing Marcel","year":"2024","unstructured":"Marcel Hussing, Claas Voelcker, Igor Gilitschenski, Amir-massoud Farahmand, and Eric Eaton. 2024. Dissecting deep rl with high update ratios: Combatting value divergence. arXiv preprint arXiv:2403.05996 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift. arXiv preprint arXiv:1502.03167","author":"Ioffe Sergey","year":"2015","unstructured":"Sergey Ioffe. 2015. Batch normalization: Accelerating deep network training by reducing internal covariate shift. arXiv preprint arXiv:1502.03167 (2015)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3712256.3726394"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583131.3590388"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014504"},{"key":"e_1_3_2_1_27_1","volume-title":"Revisiting plasticity in visual reinforcement learning: Data, modules and training stages. arXiv preprint arXiv:2310.07418","author":"Ma Guozheng","year":"2023","unstructured":"Guozheng Ma, Lu Li, Sen Zhang, Zixuan Liu, Zhen Wang, Yixin Chen, Li Shen, Xueqian Wang, and Dacheng Tao. 2023. Revisiting plasticity in visual reinforcement learning: Data, modules and training stages. arXiv preprint arXiv:2310.07418 (2023)."},{"key":"e_1_3_2_1_28_1","volume-title":"Illuminating search spaces by mapping elites. arXiv preprint arXiv:1504.04909","author":"Mouret Jean-Baptiste","year":"2015","unstructured":"Jean-Baptiste Mouret and Jeff Clune. 2015. Illuminating search spaces by mapping elites. arXiv preprint arXiv:1504.04909 (2015)."},{"key":"e_1_3_2_1_29_1","volume-title":"regularized, optimistic: scaling for compute and sample efficient continuous control. Advances in neural information processing systems 37","author":"Nauman Michal","year":"2024","unstructured":"Michal Nauman, Mateusz Ostaszewski, Krzysztof Jankowski, Piotr Mi\u0142o\u015b, and Marek Cygan. 2024. Bigger, regularized, optimistic: scaling for compute and sample efficient continuous control. Advances in neural information processing systems 37 (2024), 113038\u2013113071."},{"key":"e_1_3_2_1_30_1","volume-title":"International conference on machine learning. PMLR, 16828\u201316847","author":"Nikishin Evgenii","year":"2022","unstructured":"Evgenii Nikishin, Max Schwarzer, Pierluca D'Oro, Pierre-Luc Bacon, and Aaron Courville. 2022. The primacy bias in deep reinforcement learning. In International conference on machine learning. PMLR, 16828\u201316847."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3449639.3459304"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3712256.3726475"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512290.3528845"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.3389\/frobt.2016.00040"},{"key":"e_1_3_2_1_35_1","volume-title":"SPEQ: Offline Stabilization Phases for Efficient Q-Learning in High Update-To-Data Ratio Reinforcement Learning. arXiv preprint arXiv:2501.08669","author":"Romeo Carlo","year":"2025","unstructured":"Carlo Romeo, Girolamo Macaluso, Alessandro Sestini, and Andrew D Bagdanov. 2025. SPEQ: Offline Stabilization Phases for Efficient Q-Learning in High Update-To-Data Ratio Reinforcement Learning. arXiv preprint arXiv:2501.08669 (2025)."},{"key":"e_1_3_2_1_36_1","volume-title":"Weight normalization: A simple reparameterization to accelerate training of deep neural networks. Advances in neural information processing systems 29","author":"Salimans Tim","year":"2016","unstructured":"Tim Salimans and Durk P Kingma. 2016. Weight normalization: A simple reparameterization to accelerate training of deep neural networks. Advances in neural information processing systems 29 (2016)."},{"key":"e_1_3_2_1_37_1","volume-title":"How does batch normalization help optimization? Advances in neural information processing systems 31","author":"Santurkar Shibani","year":"2018","unstructured":"Shibani Santurkar, Dimitris Tsipras, Andrew Ilyas, and Aleksander Madry. 2018. How does batch normalization help optimization? Advances in neural information processing systems 31 (2018)."},{"key":"e_1_3_2_1_38_1","volume-title":"Prioritized experience replay. arXiv preprint arXiv:1511.05952","author":"Schaul Tom","year":"2015","unstructured":"Tom Schaul, John Quan, Ioannis Antonoglou, and David Silver. 2015. Prioritized experience replay. arXiv preprint arXiv:1511.05952 (2015)."},{"key":"e_1_3_2_1_39_1","volume-title":"International Conference on Machine Learning. PMLR, 30365\u201330380","author":"Schwarzer Max","year":"2023","unstructured":"Max Schwarzer, Johan Samir Obando Ceron, Aaron Courville, Marc G Bellemare, Rishabh Agarwal, and Pablo Samuel Castro. 2023. Bigger, better, faster: Human-level atari with human-level efficiency. In International Conference on Machine Learning. PMLR, 30365\u201330380."},{"key":"e_1_3_2_1_40_1","volume-title":"D2rl: Deep dense architectures in reinforcement learning. arXiv preprint arXiv:2010.09163","author":"Sinha Samarth","year":"2020","unstructured":"Samarth Sinha, Homanga Bharadhwaj, Aravind Srinivas, and Animesh Garg. 2020. D2rl: Deep dense architectures in reinforcement learning. arXiv preprint arXiv:2010.09163 (2020)."},{"key":"e_1_3_2_1_41_1","volume-title":"Distributional policy optimization: An alternative approach for continuous control. Advances in Neural Information Processing Systems 32","author":"Tessler Chen","year":"2019","unstructured":"Chen Tessler, Guy Tennenholtz, and Shie Mannor. 2019. Distributional policy optimization: An alternative approach for continuous control. Advances in Neural Information Processing Systems 32 (2019)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2017.2735550"},{"key":"e_1_3_2_1_43_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Xue Ke","year":"2024","unstructured":"Ke Xue, Ren-Jian Wang, Pengyi Li, Dong Li, Jianye Hao, and Chao Qian. 2024. Sample-efficient quality-diversity by cooperative coevolution. In The Twelfth International Conference on Learning Representations."}],"event":{"name":"GECCO '26: Genetic and Evolutionary Computation Conference","location":"Centro Internacional de Convenciones CIC-ANDE San Jose Costa Rica","acronym":"GECCO '26","sponsor":["SIGEVO ACM Special Interest Group on Genetic and Evolutionary Computation"]},"container-title":["Proceedings of the Genetic and Evolutionary Computation Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3795095.3805197","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T17:45:03Z","timestamp":1783705503000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3795095.3805197"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,10]]},"references-count":43,"alternative-id":["10.1145\/3795095.3805197","10.1145\/3795095"],"URL":"https:\/\/doi.org\/10.1145\/3795095.3805197","relation":{},"subject":[],"published":{"date-parts":[[2026,7,10]]},"assertion":[{"value":"2026-07-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}