{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T18:16:03Z","timestamp":1783707363797,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T00:00:00Z","timestamp":1783641600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,13]]},"DOI":"10.1145\/3795095.3805144","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T17:44:29Z","timestamp":1783705469000},"page":"338-347","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["COvolve: Adversarial Co-Evolution of Large-Language-Model-Generated Policies and Environments via Two-Player Zero-Sum Game"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-4357-9533","authenticated-orcid":false,"given":"Alkis","family":"Sygkounas","sequence":"first","affiliation":[{"name":"\u00d6rebro University, \u00d6rebro, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3422-2085","authenticated-orcid":false,"given":"Rishi","family":"Hazra","sequence":"additional","affiliation":[{"name":"\u00d6rebro University, \u00d6rebro, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7649-9109","authenticated-orcid":false,"given":"Andreas","family":"Persson","sequence":"additional","affiliation":[{"name":"\u00d6rebro University, \u00d6rebro, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5834-0188","authenticated-orcid":false,"given":"Pedro","family":"Zuidberg dos Martires","sequence":"additional","affiliation":[{"name":"\u00d6rebro University, \u00d6rebro, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3122-693X","authenticated-orcid":false,"given":"Amy","family":"Loutfi","sequence":"additional","affiliation":[{"name":"\u00d6rebro University, \u00d6rebro, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,10]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2025\/1249"},{"key":"e_1_3_2_1_2_1","volume-title":"Verifiable reinforcement learning via policy extraction. Advances in neural information processing systems 31","author":"Bastani Osbert","year":"2018","unstructured":"Osbert Bastani, Yewen Pu, and Armando Solar-Lezama. 2018. Verifiable reinforcement learning via policy extraction. Advances in neural information processing systems 31 (2018)."},{"key":"e_1_3_2_1_3_1","unstructured":"Lili Chen Mihir Prabhudesai Katerina Fragkiadaki Hao Liu and Deepak Pathak. 2025. Self-Questioning Language Models. arXiv:2508.03682 [cs.LG] https:\/\/arxiv.org\/abs\/2508.03682"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3209"},{"key":"e_1_3_2_1_5_1","unstructured":"Jeff Clune. 2020. AI-GAs: AI-generating algorithms an alternate paradigm for producing general artificial intelligence. arXiv:1905.10985 [cs.AI] https:\/\/arxiv.org\/abs\/1905.10985"},{"key":"e_1_3_2_1_6_1","unstructured":"Karl Cobbe Christopher Hesse Jacob Hilton and John Schulman. 2020. Leveraging procedural generation to benchmark reinforcement learning (ICML'20). JMLR.org Article 191 9 pages."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Will Dabney Mark Rowland Marc G. Bellemare and R\u00e9mi Munos. 2017. Distributional Reinforcement Learning with Quantile Regression. arXiv:1710.10044 [cs.AI] https:\/\/arxiv.org\/abs\/1710.10044","DOI":"10.1609\/aaai.v32i1.11791"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1933"},{"key":"e_1_3_2_1_9_1","unstructured":"Michael Dennis Natasha Jaques Eugene Vinitsky Alexandre Bayen Stuart Russell Andrew Critch and Sergey Levine. 2020. Emergent complexity and zero-shot transfer via unsupervised environment design. In Advances in neural information processing systems."},{"key":"e_1_3_2_1_10_1","volume-title":"Conference on robot learning. PMLR, 1\u201316","author":"Dosovitskiy Alexey","year":"2017","unstructured":"Alexey Dosovitskiy, German Ros, Felipe Codevilla, Antonio Lopez, and Vladlen Koltun. 2017. CARLA: An open urban driving simulator. In Conference on robot learning. PMLR, 1\u201316."},{"key":"e_1_3_2_1_11_1","article-title":"DreamCoder: growing generalizable, interpretable knowledge with wake-sleep Bayesian program learning","volume":"381","author":"Ellis Kevin","year":"2023","unstructured":"Kevin Ellis, Lionel Wong, Maxwell Nye, Mathias Sable-Meyer, Luc Cary, Lore Anaya Pozo, Luke Hewitt, Armando Solar-Lezama, and Joshua B Tenenbaum. 2023. DreamCoder: growing generalizable, interpretable knowledge with wake-sleep Bayesian program learning. Philosophical Transactions of the Royal Society A 381, 2251 (2023), 20220050.","journal-title":"Philosophical Transactions of the Royal Society A"},{"key":"e_1_3_2_1_12_1","volume-title":"The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Y1XkzMJpPd","author":"Faldor Maxence","year":"2025","unstructured":"Maxence Faldor, Jenny Zhang, Antoine Cully, and Jeff Clune. 2025. OMNI-EPIC: Open-endedness via Models of human Notions of Interestingness with Environments Programmed in Code. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Y1XkzMJpPd"},{"key":"e_1_3_2_1_13_1","volume-title":"CBC: COIN-OR Branch and Cut Solver. https:\/\/github.com\/coin-or\/Cbc. Version accessed","author":"Forrest John","year":"2005","unstructured":"John Forrest and Ted Ralphs. 2005. CBC: COIN-OR Branch and Cut Solver. https:\/\/github.com\/coin-or\/Cbc. Version accessed: 2024."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10710-021-09413-9"},{"key":"e_1_3_2_1_15_1","volume-title":"Why generalization in rl is difficult: Epistemic pomdps and implicit partial observability. Advances in neural information processing systems 34","author":"Ghosh Dibya","year":"2021","unstructured":"Dibya Ghosh, Jad Rahme, Aviral Kumar, Amy Zhang, Ryan P Adams, and Sergey Levine. 2021. Why generalization in rl is difficult: Epistemic pomdps and implicit partial observability. Advances in neural information processing systems 34 (2021), 25502\u201325515."},{"key":"e_1_3_2_1_16_1","unstructured":"Tuomas Haarnoja Aurick Zhou Pieter Abbeel and Sergey Levine. 2018. Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor. arXiv:1801.01290 [cs.LG] https:\/\/arxiv.org\/abs\/1801.01290"},{"key":"e_1_3_2_1_17_1","volume-title":"The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=cJPUpL8mOw","author":"Hazra Rishi","year":"2025","unstructured":"Rishi Hazra, Alkis Sygkounas, Andreas Persson, Amy Loutfi, and Pedro Zuidberg Dos Martires. 2025. REvolve: Reward Evolution with Large Language Models using Human Feedback. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=cJPUpL8mOw"},{"key":"e_1_3_2_1_18_1","unstructured":"Chengsong Huang Wenhao Yu Xiaoyang Wang Hongming Zhang Zongxia Li Ruosen Li Jiaxin Huang Haitao Mi and Dong Yu. 2025. R-Zero: Self-Evolving Reasoning LLM from Zero Data. arXiv:2508.05004 [cs.LG] https:\/\/arxiv.org\/abs\/2508.05004"},{"key":"e_1_3_2_1_19_1","volume-title":"8th International Conference on Learning Representations.","author":"Inala Jeevana Priya","year":"2020","unstructured":"Jeevana Priya Inala, Osbert Bastani, Zenna Tavares, and Armando Solar-Lezama. 2020. Synthesizing programmatic policies that inductively generalize. In 8th International Conference on Learning Representations."},{"key":"e_1_3_2_1_20_1","volume-title":"Evolutionary robotics and the radical envelope-of-noise hypothesis. Adaptive behavior 6, 2","author":"Jakobi Nick","year":"1997","unstructured":"Nick Jakobi. 1997. Evolutionary robotics and the radical envelope-of-noise hypothesis. Adaptive behavior 6, 2 (1997), 325\u2013368."},{"key":"e_1_3_2_1_21_1","unstructured":"Minqi Jiang Edward Grefenstette and Tim Rockt\u00e4schel. 2021. Prioritized level replay. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.14174"},{"key":"e_1_3_2_1_23_1","unstructured":"Ezgi Korkmaz. 2024. A Survey Analyzing Generalization in Deep Reinforcement Learning. arXiv:2401.02349 [cs.LG] https:\/\/arxiv.org\/abs\/2401.02349"},{"key":"e_1_3_2_1_24_1","volume-title":"Advances in Neural Information Processing Systems","author":"Lanctot Marc","year":"2017","unstructured":"Marc Lanctot, Vinicius Zambaldi, Audrunas Gruslys, Angeliki Lazaridou, Karl Tuyls, Julien Perolat, David Silver, and Thore Graepel. 2017. A Unified Game-Theoretic Approach to Multiagent Reinforcement Learning. In Advances in Neural Information Processing Systems, I. Guyon, U. Von Luxburg, S. Bengio, H. Wallach, R. Fergus, S. Vishwanathan, and R. Garnett (Eds.), Vol. 30. Curran Associates, Inc. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2017\/file\/3323fe11e9595c09af38fe67567a9394-Paper.pdf"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"e_1_3_2_1_26_1","volume-title":"8th Annual Conference on Robot Learning. https:\/\/openreview.net\/forum?id=F0rWEID2gb","author":"Liang William","year":"2024","unstructured":"William Liang, Sam Wang, Hung-Ju Wang, Osbert Bastani, Dinesh Jayaraman, and Yecheng Jason Ma. 2024. Environment Curriculum Generation via Large Language Models. In 8th Annual Conference on Robot Learning. https:\/\/openreview.net\/forum?id=F0rWEID2gb"},{"key":"e_1_3_2_1_27_1","unstructured":"Zi Lin Sheng Shen Jingbo Shang Jason Weston and Yixin Nie. 2025. Learning to Solve and Verify: A Self-Play Framework for Code and Test Generation. arXiv:2502.14948 [cs.SE] https:\/\/arxiv.org\/abs\/2502.14948"},{"key":"e_1_3_2_1_28_1","volume-title":"RL-GPT: Integrating Reinforcement Learning and Code-as-policy. In The Thirty-eighth Annual Conference on Neural Information Processing Systems. https:\/\/openreview.net\/forum?id=LEzx6QRkRH","author":"Liu Shaoteng","year":"2024","unstructured":"Shaoteng Liu, Haoqi Yuan, Minda Hu, Yanwei Li, Yukang Chen, Shu Liu, Zongqing Lu, and Jiaya Jia. 2024. RL-GPT: Integrating Reinforcement Learning and Code-as-policy. In The Thirty-eighth Annual Conference on Neural Information Processing Systems. https:\/\/openreview.net\/forum?id=LEzx6QRkRH"},{"key":"e_1_3_2_1_29_1","volume-title":"The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=IEduRUO55F","author":"Ma Yecheng Jason","year":"2024","unstructured":"Yecheng Jason Ma, William Liang, Guanzhi Wang, De-An Huang, Osbert Bastani, Dinesh Jayaraman, Yuke Zhu, Linxi Fan, and Anima Anandkumar. 2024. Eureka: Human-Level Reward Design via Coding Large Language Models. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=IEduRUO55F"},{"key":"e_1_3_2_1_30_1","first-page":"25","article-title":"Pulp: a linear programming toolkit for python. The University of Auckland, Auckland","volume":"65","author":"Mitchell Stuart","year":"2011","unstructured":"Stuart Mitchell, Michael OSullivan, and Iain Dunning. 2011. Pulp: a linear programming toolkit for python. The University of Auckland, Auckland, New Zealand 65 (2011), 25.","journal-title":"New Zealand"},{"key":"e_1_3_2_1_31_1","volume-title":"Robust reinforcement learning. Neural computation 17, 2","author":"Morimoto Jun","year":"2005","unstructured":"Jun Morimoto and Kenji Doya. 2005. Robust reinforcement learning. Neural computation 17, 2 (2005), 335\u2013359."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.2307\/1969529"},{"key":"e_1_3_2_1_33_1","unstructured":"OpenAI. 2025. Introducing GPT-5.2. https:\/\/openai.com\/index\/introducing-gpt-5-2\/"},{"key":"e_1_3_2_1_34_1","volume-title":"A course in game theory","author":"Osborne Martin J","unstructured":"Martin J Osborne and Ariel Rubinstein. 1994. A course in game theory. MIT press."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989190"},{"key":"e_1_3_2_1_36_1","unstructured":"PyGame Community. 2000\u20132024. PyGame: Python Game Development. https:\/\/www.pygame.org\/. Accessed: 2025-05-09."},{"key":"e_1_3_2_1_37_1","first-page":"1","article-title":"Stable-Baselines3: Reliable Reinforcement Learning Implementations","volume":"22","author":"Raffin Antonin","year":"2021","unstructured":"Antonin Raffin, Ashley Hill, Adam Gleave, Anssi Kanervisto, Maximilian Ernestus, and Noah Dormann. 2021. Stable-Baselines3: Reliable Reinforcement Learning Implementations. Journal of Machine Learning Research 22, 268 (2021), 1\u20138. http:\/\/jmlr.org\/papers\/v22\/20-1364.html","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_38_1","first-page":"1","article-title":"Stable-baselines3: Reliable reinforcement learning implementations","volume":"22","author":"Raffin Antonin","year":"2021","unstructured":"Antonin Raffin, Ashley Hill, Adam Gleave, Anssi Kanervisto, Maximilian Ernestus, and Noah Dormann. 2021. Stable-baselines3: Reliable reinforcement learning implementations. Journal of Machine Learning Research 22, 268 (2021), 1\u20138.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_39_1","volume-title":"MAESTRO: Open-Ended Environment Design for Multi-Agent Reinforcement Learning. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=sKWlRDzPfd7","author":"Samvelyan Mikayel","year":"2023","unstructured":"Mikayel Samvelyan, Akbir Khan, Michael D Dennis, Minqi Jiang, Jack Parker-Holder, Jakob Nicolaus Foerster, Roberta Raileanu, and Tim Rockt\u00e4schel. 2023. MAESTRO: Open-Ended Environment Design for Multi-Agent Reinforcement Learning. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=sKWlRDzPfd7"},{"key":"e_1_3_2_1_40_1","unstructured":"John Schulman Filip Wolski Prafulla Dhariwal Alec Radford and Oleg Klimov. 2017. Proximal Policy Optimization Algorithms. arXiv:1707.06347 [cs.LG] https:\/\/arxiv.org\/abs\/1707.06347"},{"key":"e_1_3_2_1_41_1","volume-title":"Sutton","author":"Silver David","year":"2025","unstructured":"David Silver and Richard S. Sutton. 2025. Welcome to the Era of Experience. In Designing an Intelligence. MIT Press."},{"key":"e_1_3_2_1_42_1","volume-title":"Workshop on Language and Robotics at CoRL","author":"Singh Ishika","year":"2022","unstructured":"Ishika Singh, Valts Blukis, Arsalan Mousavian, Ankit Goyal, Danfei Xu, Jonathan Tremblay, Dieter Fox, Jesse Thomason, and Animesh Garg. 2022. ProgPrompt: Generating Situated Robot Task Plans using Large Language Models. In Workshop on Language and Robotics at CoRL 2022. https:\/\/openreview.net\/forum?id=3K4-U_5cRw"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2243"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2017.8202133"},{"key":"e_1_3_2_1_45_1","volume-title":"Learning to synthesize programs as interpretable and generalizable policies. Advances in neural information processing systems 34","author":"Trivedi Dweep","year":"2021","unstructured":"Dweep Trivedi, Jesse Zhang, Shao-Hua Sun, and Joseph J Lim. 2021. Learning to synthesize programs as interpretable and generalizable policies. Advances in neural information processing systems 34 (2021), 25146\u201325163."},{"key":"e_1_3_2_1_46_1","volume-title":"International conference on machine learning. PMLR, 5045\u20135054","author":"Verma Abhinav","year":"2018","unstructured":"Abhinav Verma, Vijayaraghavan Murali, Rishabh Singh, Pushmeet Kohli, and Swarat Chaudhuri. 2018. Programmatically interpretable reinforcement learning. In International conference on machine learning. PMLR, 5045\u20135054."},{"key":"e_1_3_2_1_47_1","unstructured":"Pablo Villalobos Anson Ho Jaime Sevilla Tamay Besiroglu Lennart Heim and Marius Hobbhahn. 2024. Will we run out of data? Limits of LLM scaling based on human-generated data. arXiv:2211.04325 [cs.LG] https:\/\/arxiv.org\/abs\/2211.04325"},{"key":"e_1_3_2_1_48_1","volume-title":"Voyager: An Open-Ended Embodied Agent with Large Language Models. Transactions on Machine Learning Research","author":"Wang Guanzhi","year":"2024","unstructured":"Guanzhi Wang, Yuqi Xie, Yunfan Jiang, Ajay Mandlekar, Chaowei Xiao, Yuke Zhu, Linxi Fan, and Anima Anandkumar. 2024. Voyager: An Open-Ended Embodied Agent with Large Language Models. Transactions on Machine Learning Research (2024). https:\/\/openreview.net\/forum?id=ehfRiF0R3a"},{"key":"e_1_3_2_1_49_1","volume-title":"The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=OI3RoHoWAN","author":"Wang Lirui","year":"2024","unstructured":"Lirui Wang, Yiyang Ling, Zhecheng Yuan, Mohit Shridhar, Chen Bao, Yuzhe Qin, Bailin Wang, Huazhe Xu, and Xiaolong Wang. 2024. GenSim: Generating Robotic Simulation Tasks via Large Language Models. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=OI3RoHoWAN"},{"key":"e_1_3_2_1_50_1","volume-title":"Stanley","author":"Wang Rui","year":"2019","unstructured":"Rui Wang, Joel Lehman, Jeff Clune, and Kenneth O. Stanley. 2019. Paired Open-Ended Trailblazer (POET): Endlessly Generating Increasingly Complex and Diverse Learning Environments and Their Solutions. arXiv:1901.01753 [cs.NE] https:\/\/arxiv.org\/abs\/1901.01753"},{"key":"e_1_3_2_1_51_1","unstructured":"Yinjie Wang Ling Yang Ye Tian Ke Shen and Mengdi Wang. 2025. Co-Evolving LLM Coder and Unit Tester via Reinforcement Learning. arXiv:2506.03136 [cs.CL] https:\/\/arxiv.org\/abs\/2506.03136"},{"key":"e_1_3_2_1_52_1","volume-title":"Jian Feng Xu, and Chee Kiong Soh","author":"Yang Yao Wen","year":"2006","unstructured":"Yao Wen Yang, Jian Feng Xu, and Chee Kiong Soh. 2006. An evolutionary programming algorithm for continuous global optimization. European journal of operational research 168, 2 (2006), 354\u2013369."}],"event":{"name":"GECCO '26: Genetic and Evolutionary Computation Conference","location":"Centro Internacional de Convenciones CIC-ANDE San Jose Costa Rica","acronym":"GECCO '26","sponsor":["SIGEVO ACM Special Interest Group on Genetic and Evolutionary Computation"]},"container-title":["Proceedings of the Genetic and Evolutionary Computation Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3795095.3805144","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T17:47:34Z","timestamp":1783705654000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3795095.3805144"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,10]]},"references-count":52,"alternative-id":["10.1145\/3795095.3805144","10.1145\/3795095"],"URL":"https:\/\/doi.org\/10.1145\/3795095.3805144","relation":{},"subject":[],"published":{"date-parts":[[2026,7,10]]},"assertion":[{"value":"2026-07-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}