{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T23:05:42Z","timestamp":1784243142061,"version":"3.55.0"},"publisher-location":"Cham","reference-count":176,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032250315","type":"print"},{"value":"9783032250322","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-25032-2_33","type":"book-chapter","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T22:09:24Z","timestamp":1784239764000},"page":"687-713","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A Research Agenda for\u00a0Usability and\u00a0Generalisation in\u00a0Reinforcement Learning"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3241-8957","authenticated-orcid":false,"given":"Dennis J. N. J.","family":"Soemers","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1902-9690","authenticated-orcid":false,"given":"Spyridon","family":"Samothrakis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7871-2495","authenticated-orcid":false,"given":"Kurt","family":"Driessens","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0125-0824","authenticated-orcid":false,"given":"Mark H. M.","family":"Winands","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,1]]},"reference":[{"key":"33_CR1","unstructured":"Abid, A., Abdalla, A., Abid, A., Khan, D., Alfozan, A., Zou, J.: Gradio: hassle-free sharing and testing of ML models in the wild. In: 2019 ICML Workshop on Human in the Loop Learning (2019)"},{"key":"33_CR2","unstructured":"Afshar, A., Li, W.: DeLF: designing learning environments with foundation models. In: AAAI 2024 Workshop on Synergy of Reinforcement Learning and Large Language Models (2024)"},{"key":"33_CR3","unstructured":"Agarwal, R., Schwarzer, M., Castro, P.S., Courville, A., Bellemare, M.G.: Deep reinforcement learning at the edge of the statistical precipice. In: Ranzato, M., Beygelzimer, A., Dauphin, Y., Liang, P., Vaughan, J.W. (eds.) Advances in Neural Information Processing Systems, vol.\u00a034, pp. 29304\u201329320. Curran Associates, Inc. (2021)"},{"key":"33_CR4","doi-asserted-by":"crossref","unstructured":"Agarwal, R., Schwarzer, M., Castro, P.S., Courville, A.C., Bellemare, M.: Reincarnating reinforcement learning: reusing prior computation to accelerate progress. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems, vol.\u00a035, pp. 28955\u201328971. Curran Associates, Inc. (2022)","DOI":"10.52202\/068431-2099"},{"key":"33_CR5","unstructured":"Ahn, M., et al.: AutoRT: embodied foundation models for large scale orchestration of robotic agents. In: IEEE ICRA 2024 Workshop on Vision-Language Models for Navigation and Manipulation (2024)"},{"key":"33_CR6","unstructured":"Andreas, J., Klein, D., Levine, S.: Modular multitask reinforcement learning with policy sketches. In: Proceedings of the 34th International Conference on Machine Learning, vol.\u00a070, pp. 166\u2013175. PMLR (2017)"},{"key":"33_CR7","unstructured":"Andrychowicz, M., et al.: What matters for on-policy deep actor-critic methods? A large-scale study. In: 2021 International Conference on Learning Representations (2021)"},{"key":"33_CR8","doi-asserted-by":"crossref","unstructured":"Aram, M., Neumann, G.: Multilayered analysis of co-development of business information systems. J. Internet Serv. Appl. 6(13) (2015)","DOI":"10.1186\/s13174-015-0030-8"},{"key":"33_CR9","doi-asserted-by":"crossref","unstructured":"Bamford, C., Huang, S., Lucas, S.: Griddly: a platform for AI research in games. In: AAAI-21 Workshop on Reinforcement Learning in Games (2021)","DOI":"10.1016\/j.simpa.2021.100066"},{"key":"33_CR10","unstructured":"Banerjee, B., Stone, P.: General game learning using knowledge transfer. In: The 20th International Joint Conference on Artificial Intelligence, pp. 672\u2013677 (2007)"},{"key":"33_CR11","unstructured":"Beck, J., et al.: A survey of meta-reinforcement learning (2023). https:\/\/arxiv.org\/abs\/2301.08028"},{"key":"33_CR12","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1038\/s41586-020-2939-8","volume":"588","author":"MG Bellemare","year":"2020","unstructured":"Bellemare, M.G., et al.: Autonomous navigation of stratospheric balloons using reinforcement learning. Nature 588, 77\u201382 (2020)","journal-title":"Nature"},{"issue":"1","key":"33_CR13","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1613\/jair.3912","volume":"47","author":"MG Bellemare","year":"2013","unstructured":"Bellemare, M.G., Naddaf, Y., Veness, J., Bowling, M.: The arcade learning environment: an evaluation platform for general agents. J. Artif. Intell. Res. 47(1), 253\u2013279 (2013)","journal-title":"J. Artif. Intell. Res."},{"key":"33_CR14","unstructured":"Benjamins, C., et al.: Contextualize me \u2013 the case for context in reinforcement learning. Trans. Mach. Learn. Res. (2023)"},{"key":"33_CR15","unstructured":"Bettini, M., Kortvelesy, R., Blumenkamp, J., Prorok, A.: VMAS: a vectorized multi-agent simulator for collective robot learning. In: Proceedings of the 16th International Symposium on Distributed Autonomous Robotic Systems, DARS 2022. Springer, Cham (2022)"},{"key":"33_CR16","unstructured":"Blattmann, A., et al.: Stable video diffusion: scaling latent video diffusion models to large datasets (2023). https:\/\/arxiv.org\/abs\/2311.15127"},{"key":"33_CR17","unstructured":"Blili-Hamelin, B., et al.: Position: stop treating \u2018AGI\u2019 as the north-star goal of AI research. In: Proceedings of the 42nd International Conference on Machine Learning (2025, to appear)"},{"key":"33_CR18","unstructured":"Bonnet, C., et al.: Jumanji: a diverse suite of scalable reinforcement learning environments in JAX. In: Proceedings of the International Conference on Learning Representations (2024)"},{"key":"33_CR19","unstructured":"Bou Ammar, H., Eaton, E., Ruvolo, P., Taylor, M.E.: Online multi-task learning for policy gradient methods. In: Xing, E.P., Jebara, T. (eds.) Proceedings of the 31st International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a032, pp. 1206\u20131214 (2014)"},{"key":"33_CR20","unstructured":"Bradbury, J., et al.: JAX: composable transformations of Python+NumPy programs (2018). http:\/\/github.com\/google\/jax"},{"key":"33_CR21","doi-asserted-by":"publisher","first-page":"661","DOI":"10.1613\/jair.3484","volume":"43","author":"SRK Branavan","year":"2012","unstructured":"Branavan, S.R.K., Silver, D., Barzilay, R.: Learning to win by reading manuals in a Monte-Carlo framework. J. Artif. Intell. Res. 43, 661\u2013704 (2012)","journal-title":"J. Artif. Intell. Res."},{"key":"33_CR22","unstructured":"Brockman, G., et al.: OpenAI gym (2016). https:\/\/arxiv.org\/abs\/1606.01540"},{"key":"33_CR23","unstructured":"Browne, C., Soemers, D.J.N.J., Piette, \u00c9., Stephenson, M., Crist, W.: Ludii language reference (2020). https:\/\/ludii.games\/downloads\/LudiiLanguageReference.pdf"},{"key":"33_CR24","unstructured":"Browne, C.B.: Automatic generation and evaluation of recombination games. Ph.D. thesis, Faculty of Information Technology, Queensland University of Technology, Queensland, Australia (2009)"},{"key":"33_CR25","unstructured":"Cobbe, K., Hesse, C., Hilton, J., Schulman, J.: Leveraging procedural generation to benchmark reinforcement learning. In: Daum\u00e9 III, H., Singh, A. (eds.) Proceedings of the 37th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0119, pp. 2048\u20132056 (2020)"},{"key":"33_CR26","unstructured":"Cobbe, K., Klimov, O., Hesse, C., Kim, T., Schulman, J.: Quantifying generalization in reinforcement learning. In: Chaudhuri, K., Salakhutdinov, R. (eds.) Proceedings of the 36th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a097, pp. 1282\u20131289. PMLR (2019)"},{"key":"33_CR27","unstructured":"Cummins, C., et al.: Compilergym: robust, performant compiler optimization environments for AI research (2021). https:\/\/arxiv.org\/abs\/2109.08267"},{"key":"33_CR28","unstructured":"Dalton, S., Frosio, I.: Accelerating reinforcement learning through GPU Atari emulation. In: Larochelle, H., Ranzato, M., Hadsell, R., Balcan, M., Lin, H. (eds.) Advances in Neural Information Processing Systems, vol.\u00a033, pp. 19773\u201319782. Curran Associates, Inc. (2020)"},{"key":"33_CR29","unstructured":"Davidson, G., Gureckis, T.M.: Toward complex and structured goals in reinforcement learning. In: Finding the Frame Workshop @ Reinforcement Learning Conference (2024)"},{"key":"33_CR30","unstructured":"Davidson, G., Todd, G., Togelius, J., Gureckis, T.M., Lake, B.M.: Goals as reward-producing programs (2024). https:\/\/arxiv.org\/abs\/2405.13242"},{"key":"33_CR31","doi-asserted-by":"publisher","first-page":"414","DOI":"10.1038\/s41586-021-04301-9","volume":"602","author":"J Degrave","year":"2022","unstructured":"Degrave, J., et al.: Magnetic control of tokamak plasmas through deep reinforcement learning. Nature 602, 414\u2013419 (2022)","journal-title":"Nature"},{"key":"33_CR32","doi-asserted-by":"crossref","unstructured":"Deisenroth, M.P., Englert, P., Peters, J., Fox, D.: Multi-task policy search for robotics. In: Proceedings of the 2014 IEEE International Conference on Robotics and Automation (ICRA), pp. 3876\u20133881 (2014)","DOI":"10.1109\/ICRA.2014.6907421"},{"key":"33_CR33","unstructured":"Dennis, M., et al.: Emergent complexity and zero-shot transfer via unsupervised environment design. In: Advances in Neural Information Processing Systems, vol.\u00a033, pp. 13049\u201313061 (2020)"},{"key":"33_CR34","doi-asserted-by":"crossref","unstructured":"Desai, A., et al.: Program synthesis using natural language. In: Proceedings of the 38th International Conference on Software Engineering, pp. 345\u2013356. Association for Computing Machinery (2016)","DOI":"10.1145\/2884781.2884786"},{"key":"33_CR35","doi-asserted-by":"crossref","unstructured":"Eimer, T., Biedenkapp, A., Reimer, M., Adriaensen, S., Hutter, F., Lindauer, M.: DACBench: a benchmark library for dynamic algorithm configuration. In: Proceedings of the Thirtieth International Joint Conference on Artificial Intelligence, pp. 1668\u20131674 (2021)","DOI":"10.24963\/ijcai.2021\/230"},{"key":"33_CR36","unstructured":"Eimer, T., Lindauer, M., Raileanu, R.: Hyperparameters in reinforcement learning and how to tune them. In: Krause, A., Brunskill, E., Cho, K., Engelhardt, B., Sabato, S., Scarlett, J. (eds.) Proceedings of the 40th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0202, pp. 9104\u20139149. PMLR (2023)"},{"key":"33_CR37","doi-asserted-by":"crossref","unstructured":"Ellis, B., et al.: SMACv2: an improved benchmark for cooperative multi-agent reinforcement learning. In: Advances in Neural Information Processing Systems (2023, accepted)","DOI":"10.52202\/075280-1634"},{"key":"33_CR38","unstructured":"Farebrother, J., Machado, M.C., Bowling, M.: Generalization and regularization in DQN (2018). https:\/\/arxiv.org\/abs\/1810.00123"},{"issue":"1","key":"33_CR39","first-page":"13","volume":"2","author":"F Fernand\u00e9z","year":"2013","unstructured":"Fernand\u00e9z, F., Veloso, M.: Learning domain structure through probabilistic policy reuse in reinforcement learning. Progress AI 2(1), 13\u201327 (2013)","journal-title":"Progress AI"},{"key":"33_CR40","unstructured":"Finn, C., Abeel, P., Levine, S.: Model-agnostic meta-learning for fast adaptation of deep networks. In: Precup, D., Teh, Y.W. (eds.) Proceedings of the 34th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a070, pp. 1126\u20131135 (2017)"},{"key":"33_CR41","unstructured":"Freeman, C.D., Frey, E., Raichuk, A., Girgin, S., Mordatch, I., Bachem, O.: Brax - a differentiable physics engine for large scale rigid body simulation (2021). http:\/\/github.com\/google\/brax"},{"key":"33_CR42","doi-asserted-by":"crossref","unstructured":"Frey, S., et al.: JAX-LOB: a GPU-accelerated limit order book simulator to unlock large scale reinforcement learning for trading. In: ICAIF 2023: Proceedings of the Fourth ACM International Conference on AI in Finance, pp. 583\u2013591 (2023)","DOI":"10.1145\/3604237.3626880"},{"key":"33_CR43","unstructured":"Gamrian, S., Goldberg, Y.: Transfer learning for related reinforcement learning tasks via image-to-image translation. In: Proceedings of the 36th International Conference on Machine Learning, pp. 2063\u20132072 (2019)"},{"key":"33_CR44","doi-asserted-by":"crossref","unstructured":"Genesereth, M., Thielscher, M.: General Game Playing. Synthesis Lectures on Artificial Intelligence and Machine Learning. Morgan & Claypool Publishers (2014)","DOI":"10.1007\/978-3-031-01569-4"},{"key":"33_CR45","unstructured":"Ghosh, D., Rahme, J., Kumar, A., Zhang, A., Adams, R.P., Levine, S.: Why generalization in RL is difficult: epistemic pomdps and implicit partial observability. In: Ranzato, M., Beygelzimer, A., Dauphin, Y., Liang, P., Vaughan, J.W. (eds.) Advances in Neural Information Processing Systems, vol.\u00a034, pp. 25502\u201325515. Curran Associates, Inc. (2021)"},{"key":"33_CR46","doi-asserted-by":"crossref","unstructured":"Glatt, R., da\u00a0Silva, F.L., Costa, A.H.R.: Towards knowledge transfer in deep reinforcement learning. In: Brazilian Conference on Intelligent Systems (BRACIS), pp. 91\u201396. IEEE (2016)","DOI":"10.1109\/BRACIS.2016.027"},{"key":"33_CR47","doi-asserted-by":"crossref","unstructured":"Glatt, R., Silva, F.L.D., da\u00a0Costa\u00a0Bianchi, R.A., Costa, A.H.R.: DECAF: deep case-based policy inference for knowledge transfer in reinforcement learning. Expert Syst. Appl. 156 (2020)","DOI":"10.1016\/j.eswa.2020.113420"},{"key":"33_CR48","doi-asserted-by":"crossref","unstructured":"Goldie, A.D., Lu, C., Jackson, M.T., Whiteson, S., Foerster, J.N.: Can learned optimization make reinforcement learning less difficult? In: AutoRL Workshop @ ICML 2024 (2024)","DOI":"10.52202\/079017-0177"},{"key":"33_CR49","doi-asserted-by":"crossref","unstructured":"Goldwaser, A., Thielscher, M.: Deep reinforcement learning for general game playing. In: The Thirty-Fourth AAAI Conference on Artificial Intelligence, pp. 1701\u20131708. AAAI Press (2020)","DOI":"10.1609\/aaai.v34i02.5533"},{"key":"33_CR50","unstructured":"Goyal, P., Niekum, S., Mooney, R.J.: PixL2R: guiding reinforcement learning using natural language by mapping pixels to rewards. In: Proceedings of the 2020 Conference on Robot Learning. PMLR, vol.\u00a0155, pp. 485\u2013497 (2021)"},{"key":"33_CR51","doi-asserted-by":"crossref","unstructured":"Gulino, C., et al.: Waymax: an accelerated, data-driven simulator for large-scale autonomous driving research. In: Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks (2023)","DOI":"10.52202\/075280-0339"},{"key":"33_CR52","unstructured":"Hartman, S., Ong, C.S., Powles, J., Kuhnert, P.: Position: we need responsible, application-driven (RAD) AI research. In: Proceedings of the 42nd International Conference on Machine Learning (2025, to appear)"},{"key":"33_CR53","doi-asserted-by":"crossref","unstructured":"Henderson, P., Islam, R., Bachman, P., Pineau, J., Precup, D., Meger, D.: Deep reinforcement learning that matters. In: Proceedings of the 32nd AAAI Conference on Artificial Intelligence, pp. 3207\u20133214. AAAI (2018)","DOI":"10.1609\/aaai.v32i1.11694"},{"issue":"12","key":"33_CR54","doi-asserted-by":"publisher","first-page":"58","DOI":"10.1145\/3467017","volume":"64","author":"S Hooker","year":"2021","unstructured":"Hooker, S.: The hardware lottery. Commun. ACM 64(12), 58\u201365 (2021)","journal-title":"Commun. ACM"},{"key":"33_CR55","volume-title":"Dynamic Programming and Markov Processes","author":"RA Howard","year":"1960","unstructured":"Howard, R.A.: Dynamic Programming and Markov Processes. The MIT Press, Cambridge (1960)"},{"key":"33_CR56","unstructured":"Irpan, A., Song, X.: The principle of unchanged optimality in reinforcement learning generalization. In: ICML 2019 Workshop on Understanding and Improving Generalization in Deep Learning (2019)"},{"key":"33_CR57","doi-asserted-by":"crossref","unstructured":"Jackson, M.T., et al.: Discovering general reinforcement learning algorithms with adversarial environment design. In: Advances in Neural Information Processing Systems, vol.\u00a036, pp. 79980\u201379998 (2023)","DOI":"10.52202\/075280-3503"},{"key":"33_CR58","doi-asserted-by":"crossref","unstructured":"Jacob, M., Devlin, S., Hofmann, K.: \u201cIt\u2019s unwieldy and it takes a lot of time\u201d\u2014challenges and opportunities for creating agents in commercial games. In: Proceedings of the Sixteenth AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment, pp. 88\u201394 (2020)","DOI":"10.1609\/aiide.v16i1.7415"},{"key":"33_CR59","unstructured":"Jordan, S.M., White, A., da\u00a0Silva, B.C., White, M., Thomas, P.S.: Position: benchmarking is limited in reinforcement learning research. In: Salakhutdinov, R., et al. (eds.) Proceedings of the 41st International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0235, pp. 22551\u201322569. PMLR (2024)"},{"key":"33_CR60","unstructured":"Jordan, S.M.: Scientific experiments in reinforcement learning (2022). https:\/\/nips.cc\/virtual\/2022\/63891, opinion talk contributed to the Deep Reinforcement Learning Workshop at NeurIPS 2022"},{"key":"33_CR61","unstructured":"Jothimurugan, K., Alur, R., Bastani, O.: A composable specification language for reinforcement learning tasks. In: Advances in Neural Information Processing Systems, vol.\u00a032, pp. 13041\u201313051 (2019)"},{"key":"33_CR62","doi-asserted-by":"crossref","unstructured":"Justesen, N., Torrado, R.R., Bontrager, P., Khalifa, A., Togelius, J., Risi, S.: Illuminating generalization in deep reinforcement learning through procedural level generation. In: NeurIPS 2018 Workshop on Deep Reinforcement Learning (2018)","DOI":"10.1109\/CIG.2018.8490422"},{"key":"33_CR63","doi-asserted-by":"crossref","unstructured":"Kaiser, \u0141., Stafiniak, \u0141.: First-order logic with counting for general game playing. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a025, pp. 791\u2013796 (2011)","DOI":"10.1609\/aaai.v25i1.7949"},{"key":"33_CR64","unstructured":"Kaufmann, T., Weng, P., Bengs, V., H\u00fcllermeier, E.: A survey of reinforcement learning from human feedback. Trans. Mach. Learn. Res. (2025)"},{"key":"33_CR65","doi-asserted-by":"crossref","unstructured":"Kharyal, C., Krishna Gottipati, S., Kumar Sinha, T., Das, S., Taylor, M.E.: GLIDE-RL: grounded language instruction through DEmonstration in RL (2024). https:\/\/arxiv.org\/abs\/2401.02991","DOI":"10.65109\/JSPM5155"},{"key":"33_CR66","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1613\/jair.1.14174","volume":"76","author":"R Kirk","year":"2023","unstructured":"Kirk, R., Zhang, A., Grefenstette, E., Rockt\u00e4schel, T.: A survey of zero-shot generalisation in deep reinforcement learning. J. Artif. Intell. Res. 76, 201\u2013264 (2023)","journal-title":"J. Artif. Intell. Res."},{"key":"33_CR67","doi-asserted-by":"crossref","unstructured":"Kowalksi, J., et al.: Efficient reasoning in regular boardgames. In: Proceedings of the 2020 IEEE Conference on Games, pp. 455\u2013462. IEEE (2020)","DOI":"10.1109\/CoG47356.2020.9231668"},{"key":"33_CR68","doi-asserted-by":"crossref","unstructured":"Kowalski, J., Maksymilian, M., Sutowicz, J., Szyku\u0142a, M.: Regular boardgames. In: Proceedings of the 33rd AAAI Conference on Artificial Intelligence, vol.\u00a033, pp. 1699\u20131706. AAAI Press (2019)","DOI":"10.1609\/aaai.v33i01.33011699"},{"key":"33_CR69","doi-asserted-by":"crossref","unstructured":"Koyamada, S., et al.: PGX: hardware-accelerated parallel game simulators for reinforcement learning. In: Advances in Neural Information Processing Systems (2023)","DOI":"10.52202\/075280-1981"},{"key":"33_CR70","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"188","DOI":"10.1007\/978-3-540-74958-5_20","volume-title":"Machine Learning: ECML 2007","author":"G Kuhlmann","year":"2007","unstructured":"Kuhlmann, G., Stone, P.: Graph-based domain mapping for transfer learning in general games. In: Kok, J.N., Koronacki, J., Mantaras, R.L., Matwin, S., Mladeni\u010d, D., Skowron, A. (eds.) ECML 2007. LNCS (LNAI), vol. 4701, pp. 188\u2013200. Springer, Heidelberg (2007). https:\/\/doi.org\/10.1007\/978-3-540-74958-5_20"},{"key":"33_CR71","unstructured":"Lange, R.T.: gymnax: a JAX-based reinforcement learning environment library (2022). http:\/\/github.com\/RobertTLange\/gymnax"},{"key":"33_CR72","doi-asserted-by":"crossref","unstructured":"Lange, S., Riedmiller, M.: Deep auto-encoder neural networks in reinforcement learning. In: Neural Networks. International Joint Conference (IJCNN 2010), pp. 1623\u20131630. IEEE (2010)","DOI":"10.1109\/IJCNN.2010.5596468"},{"key":"33_CR73","series-title":"Adaptation, Learning, and Optimization","doi-asserted-by":"publisher","first-page":"143","DOI":"10.1007\/978-3-642-27645-3_5","volume-title":"Reinforcement Learning","author":"A Lazaric","year":"2012","unstructured":"Lazaric, A.: Transfer in reinforcement learning: a framework and a survey. In: Wiering, M., van Otterlo, M. (eds.) Reinforcement Learning. Adaptation, Learning, and Optimization, vol. 12, pp. 143\u2013173. Springer, Heidelberg (2012)"},{"issue":"7553","key":"33_CR74","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1038\/nature14539","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun, Y., Bengio, Y., Hinton, G.: Deep learning. Nature 521(7553), 436\u2013444 (2015)","journal-title":"Nature"},{"key":"33_CR75","unstructured":"Lee, J.N., et al.: Supervised pretraining can learn in-context reinforcement learning (2023). https:\/\/arxiv.org\/abs\/2306.14892"},{"key":"33_CR76","unstructured":"Leivada, E., Marcus, G., G\u00fcnther, F., Murphy, E.: A sentence is worth a thousand pictures: can large language models understand human language and the world behind words? (2024). https:\/\/arxiv.org\/abs\/2308.00109"},{"issue":"40","key":"33_CR77","first-page":"1131","volume":"10","author":"H Li","year":"2009","unstructured":"Li, H., Liao, X., Carin, L.: Multi-task reinforcement learning in partially observable stochastic environments. J. Mach. Learn. Res. 10(40), 1131\u20131186 (2009)","journal-title":"J. Mach. Learn. Res."},{"key":"33_CR78","doi-asserted-by":"crossref","unstructured":"Li, X., Zhang, J., Bian, J., Tong, Y., Liu, T.Y.: A cooperative multi-agent reinforcement learning framework for resource balancing in complex logistics network. In: Agmon, N., Taylor, M.E., Veloso, E.E.M. (eds.) Proceedings of the 18th International Conference on Autonomous Agents and Multiagent Systems (AAMAS 2019), pp. 980\u2013988 (2019)","DOI":"10.65109\/BEJR9896"},{"key":"33_CR79","doi-asserted-by":"crossref","unstructured":"Lifschitz, S., Paster, K., Chan, H., Ba, J., McIlraith, S.: Steve-1: a generative model for text-to-behavior in Minecraft (2023). https:\/\/arxiv.org\/abs\/2306.00937","DOI":"10.52202\/075280-3064"},{"key":"33_CR80","doi-asserted-by":"crossref","unstructured":"Lin, S., Bercher, P.: On the expressive power of planning formalisms in conjunction with LTL. In: Proceedings of the International Conference on Automated Planning and Scheduling, vol.\u00a032, pp. 231\u2013240 (2022)","DOI":"10.1609\/icaps.v32i1.19806"},{"key":"33_CR81","unstructured":"Lindauer, M., et al.: Position: a call to action for a human-centered AutoML paradigm. In: Proceedings of the 41st International Conference on Machine Learning. PMLR, vol.\u00a0235, pp. 30566\u201330584 (2024)"},{"key":"33_CR82","doi-asserted-by":"crossref","unstructured":"Littman, M.L.: Markov games as a framework for multi-agent reinforcement learning. In: Proceedings of the Eleventh International Conference on Machine Learning, pp. 157\u2013163 (1994)","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"33_CR83","unstructured":"Love, N., Hinrichs, T., Haley, D., Schkufza, E., Genesereth, M.: General game playing: game description language specification. Technical report. LG-2006-01, Stanford Logic Group (2008)"},{"key":"33_CR84","doi-asserted-by":"crossref","unstructured":"Luketina, J., et al.: A survey of reinforcement learning informed by natural language. In: Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence, IJCAI 2019, pp. 6309\u20136317 (2019)","DOI":"10.24963\/ijcai.2019\/880"},{"key":"33_CR85","unstructured":"Luo, J., et al.: SERL: a software suite for sample-efficient robotic reinforcement learning. In: Towards Generalist Robots: Learning Paradigms for Scalable Skill Acquisition @ CoRL2023 (2023)"},{"issue":"86","key":"33_CR86","first-page":"2579","volume":"9","author":"L van der Maaten","year":"2008","unstructured":"van der Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9(86), 2579\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."},{"key":"33_CR87","doi-asserted-by":"publisher","first-page":"523","DOI":"10.1613\/jair.5699","volume":"61","author":"MC Machado","year":"2018","unstructured":"Machado, M.C., Bellemare, M.G., Talvitie, E., Veness, J., Hausknecht, M., Bowling, M.: Revisiting the arcade learning environment: evaluation protocols and open problems for general agents. J. Artif. Intell. Res. 61, 523\u2013562 (2018)","journal-title":"J. Artif. Intell. Res."},{"key":"33_CR88","doi-asserted-by":"publisher","first-page":"251","DOI":"10.1023\/A:1018020625251","volume":"22","author":"R Maclin","year":"1996","unstructured":"Maclin, R., Shavlik, J.W.: Creating advice-taking reinforcement learners. Mach. Learn. 22, 251\u2013281 (1996)","journal-title":"Mach. Learn."},{"key":"33_CR89","unstructured":"Makoviychuk, V., et al.: Isaac gym: high performance GPU based physics simulation for robot learning. In: Vanschoren, J., Yeung, S. (eds.) Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks, vol.\u00a01. Curran (2021)"},{"key":"33_CR90","unstructured":"Malik, D., Li, Y., Ravikumar, P.: When is generalizable reinforcement learning tractable? In: Ranzato, M., Beygelzimer, A., Dauphin, Y., Liang, P., Vaughan, J.W. (eds.) Advances in Neural Information Processing Systems, vol.\u00a034, pp. 8032\u20138045. Curran Associates, Inc. (2021)"},{"key":"33_CR91","unstructured":"Mannor, S., Tamar, A.: Towards deployable RL \u2013 what\u2019s broken with RL research and a potential fix (2023). https:\/\/arxiv.org\/abs\/2301.01320"},{"key":"33_CR92","doi-asserted-by":"crossref","unstructured":"Maras, M., K\u0119pa, M., Kowalski, J., Szyku\u0142a, M.: Fast and knowledge-free deep learning for general game playing (student abstract). In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 23576\u201323578 (2024)","DOI":"10.1609\/aaai.v38i21.30480"},{"key":"33_CR93","unstructured":"McDermott, D., et al.: PDDL\u2014the planning domain definition language. Technical report. CVC TR98003\/DCS TR1165, New Haven, CT: Yale Center for Computational Vision and Control (1998)"},{"key":"33_CR94","unstructured":"Mediratta, I., You, Q., Jiang, M., Raileanu, R.: A study of generalization in offline reinforcement learning. In: 2024 International Conference on Learning Representations (2024)"},{"issue":"4","key":"33_CR95","doi-asserted-by":"publisher","first-page":"316","DOI":"10.1145\/1118890.1118892","volume":"37","author":"M Mernik","year":"2005","unstructured":"Mernik, M., Heering, J., Sloane, A.M.: When and how to develop domain-specific languages. ACM Comput. Surv. 37(4), 316\u2013344 (2005)","journal-title":"ACM Comput. Surv."},{"key":"33_CR96","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1038\/s41586-021-03544-w","volume":"594","author":"A Mirhoseini","year":"2021","unstructured":"Mirhoseini, A., et al.: A graph placement methodology for fast chip design. Nature 594, 207\u2013212 (2021)","journal-title":"Nature"},{"key":"33_CR97","doi-asserted-by":"crossref","unstructured":"Misra, D., Langford, J., Artzi, Y.: Mapping instructions and visual observations to actions with reinforcement learning. In: Palmer, M., Hwa, R., Riedel, S. (eds.) Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing, pp. 1004\u20131015 (2017)","DOI":"10.18653\/v1\/D17-1106"},{"key":"33_CR98","doi-asserted-by":"crossref","unstructured":"Mittel, A., Munukutla, P.S.: Visual transfer between Atari games using competitive reinforcement learning. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW), pp. 499\u2013501 (2019)","DOI":"10.1109\/CVPRW.2019.00071"},{"key":"33_CR99","unstructured":"Mnih, V., et al.: Playing Atari with deep reinforcement learning (2013). https:\/\/arxiv.org\/abs\/1312.5602"},{"key":"33_CR100","unstructured":"Mohan, A., Benjamins, C., Wienecke, K., Dockhorn, A., Lindauer, M.: AutoRL hyperparameter landscapes. In: Faust, A., Garnett, R., White, C., Hutter, F., Gardner, J.R. (eds.) International Conference on Automated Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0224. PMLR (2023)"},{"key":"33_CR101","doi-asserted-by":"publisher","first-page":"1167","DOI":"10.1613\/jair.1.15703","volume":"79","author":"A Mohan","year":"2024","unstructured":"Mohan, A., Zhang, A., Lindauer, M.: Structure in deep reinforcement learning: a survey and open problems. J. Artif. Intell. Res. 79, 1167\u20131236 (2024)","journal-title":"J. Artif. Intell. Res."},{"key":"33_CR102","doi-asserted-by":"crossref","unstructured":"M\u00fcller-Brockhausen, M., Preuss, M., Plaat, A.: Procedural content generation: better benchmarks for transfer reinforcement learning. In: Proceedings of the 2021 IEEE Conference on Games, pp. 924\u2013931 (2021)","DOI":"10.1109\/CoG52621.2021.9619000"},{"key":"33_CR103","unstructured":"Nagabandi, A., et al.: Learning to adapt in dynamic, real-world environments through meta-reinforcement learning. In: Proceedings of the 2019 International Conference on Learning Representations (2019)"},{"key":"33_CR104","unstructured":"Nichol, A., Pfau, V., Hesse, C., Klimov, O., Schulman, J.: Gotta learn fast: a new benchmark for generalization in RL (2018). https:\/\/arxiv.org\/abs\/1804.03720"},{"key":"33_CR105","unstructured":"Obando-Ceron, J., Ara\u00fa\u2019jo, J.G.M., Courville, A., Castro, P.S.: On the consistency of hyper-parameter selection in value-based deep reinforcement learning. In: Reinforcement Learning Conference (2024, accepted)"},{"key":"33_CR106","unstructured":"Obando-Ceron, J.S., Castro, P.S.: Revisiting rainbow: promoting more insightful and inclusive deep reinforcement learning research. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning, pp. 1373\u20131383. PMLR (2021)"},{"key":"33_CR107","unstructured":"Oh, J., Singh, S., Lee, H., Kohli, P.: Zero-shot task generalization with multi-task deep reinforcement learning. In: Proceedings of the 34th International Conference on Machine Learning, pp. 2661\u20132670. PMLR (2017)"},{"key":"33_CR108","unstructured":"OpenAI: Introducing ChatGPT (2022). https:\/\/openai.com\/blog\/chatgpt. Accessed 02 Jan 2024"},{"key":"33_CR109","doi-asserted-by":"crossref","unstructured":"Oswald, J., Srinivas, K., Kokel, H., Lee, J., Katz, M., Sohrabi, S.: Large language models as planning domain generators. In: Proceedings of the International Conference on Automated Planning and Scheduling, vol.\u00a034, pp. 423\u2013431 (2024)","DOI":"10.1609\/icaps.v34i1.31502"},{"key":"33_CR110","doi-asserted-by":"crossref","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems, vol.\u00a035, pp. 27730\u201327744. Curran Associates, Inc. (2022)","DOI":"10.52202\/068431-2011"},{"key":"33_CR111","unstructured":"Parker-Holder, J., et al.: Evolving curricula with regret-based environment design. In: Proceedings of the 39th International Conference on Machine Learning. PMLR, vol.\u00a0162, pp. 17473\u201317498 (2022)"},{"key":"33_CR112","doi-asserted-by":"publisher","first-page":"517","DOI":"10.1613\/jair.1.13596","volume":"74","author":"J Parker-Holder","year":"2022","unstructured":"Parker-Holder, J., et al.: Automated reinforcement learning (autoRL): a survey and open problems. J. Artif. Intell. Res. 74, 517\u2013568 (2022)","journal-title":"J. Artif. Intell. Res."},{"key":"33_CR113","unstructured":"Patterson, A., Neumann, S., White, M., White, A.: Empirical design in reinforcement learning (2023). https:\/\/arxiv.org\/abs\/2304.01315"},{"key":"33_CR114","unstructured":"Perez-Liebana, D., Dockhorn, A., Grueso, J.H., Jeurissen, D.: The design of \u201cStratega\u201d: a general strategy games framework. In: Osborn, J.C. (ed.) Joint Proceedings of the AIIDE 2020 Workshops co-located with 16th AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment (AIIDE 2020). CEUR Workshop Proceedings (2020)"},{"key":"33_CR115","doi-asserted-by":"crossref","unstructured":"Piette, \u00c9., Soemers, D.J.N.J., Stephenson, M., Sironi, C.F., Winands, M.H.M., Browne, C.: Ludii \u2013 the ludemic general game system. In: Giacomo, G.D., et al. (eds.) Proceedings of the 24th European Conference on Artificial Intelligence (ECAI 2020). Frontiers in Artificial Intelligence and Applications, vol.\u00a0325, pp. 411\u2013418. IOS Press (2020)","DOI":"10.3233\/FAIA200120"},{"key":"33_CR116","doi-asserted-by":"crossref","unstructured":"Piette, \u00c9., Stephenson, M., Soemers, D.J.N.J., Browne, C.: General board game concepts. In: Proceedings of the 2021 IEEE Conference on Games (CoG), pp. 932\u2013939. IEEE (2021)","DOI":"10.1109\/CoG52621.2021.9618990"},{"key":"33_CR117","unstructured":"Podell, D., et al.: SDXL: improving latent diffusion models for high-resolution image synthesis (2023). https:\/\/arxiv.org\/abs\/2307.01952"},{"key":"33_CR118","unstructured":"Ponse, K., Kleuker, J.F., Moerland, T.M., Plaat, A.: Chargax: a JAX accelerated EV charging simulator. In: Reinforcement Learning Conference (2025, to appear)"},{"key":"33_CR119","unstructured":"Rakelly, K., Zhou, A., Quillen, D., Finn, C., Levine, S.: Efficient off-policy meta-reinforcement learning via probabilistic context variables. In: Chaudhuri, K., Salakhutdinov, R. (eds.) Proceedings of the 36th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a097, pp. 5331\u20135340 (2019)"},{"key":"33_CR120","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with CLIP latents (2022). https:\/\/arxiv.org\/abs\/2204.06125"},{"key":"33_CR121","doi-asserted-by":"crossref","unstructured":"Rani, S., Booth, S., Sreedharan, S.: Goals vs. rewards: a comparative study of objective specification mechanisms. In: Reinforcement Learning Conference (2025, to appear)","DOI":"10.1109\/HRI61500.2025.10973973"},{"key":"33_CR122","unstructured":"Raparthy, S.C., Hambro, E., Kirk, R., Henaff, M., Raileanu, R.: Generalization to new sequential decision making tasks with in-context learning (2023). https:\/\/arxiv.org\/abs\/2312.03801"},{"key":"33_CR123","unstructured":"Reed, S., et al.: A generalist agent. Trans. Mach. Learn. Res. (2023)"},{"key":"33_CR124","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1007\/11564096_32","volume-title":"Machine Learning: ECML 2005","author":"M Riedmiller","year":"2005","unstructured":"Riedmiller, M.: Neural fitted Q iteration \u2013 first experiences with a data efficient neural reinforcement learning method. In: Gama, J., Camacho, R., Brazdil, P.B., Jorge, A.M., Torgo, L. (eds.) ECML 2005. LNCS (LNAI), vol. 3720, pp. 317\u2013328. Springer, Heidelberg (2005). https:\/\/doi.org\/10.1007\/11564096_32"},{"key":"33_CR125","unstructured":"Rigter, M., Jiang, M., Posner, I.: Reward-free curricula for training robust world models. In: International Conference on Learning Representations (2024)"},{"key":"33_CR126","unstructured":"Rodriguez-Sanchez, R., Spiegel, B.A., Wang, J., Patel, R., Tellex, S., Konidaris, G.: RLang: a declarative language for describing partial world knowledge to reinforcement learning agents. In: Krause, A., Brunskill, E., Cho, K., Engelhardt, B., Sabato, S., Scarlett, J. (eds.) Proceedings of the 40th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0202, pp. 29161\u201329178. PMLR (2023)"},{"key":"33_CR127","unstructured":"Rolnick, D., et al.: Position: application-driven innovation in machine learning. In: Proceedings of the 41st International Conference on Machine Learning. PMLR, vol.\u00a0235, pp. 42707\u201342718 (2024)"},{"key":"33_CR128","doi-asserted-by":"publisher","first-page":"673","DOI":"10.1613\/jair.1.11304","volume":"67","author":"M Rostami","year":"2020","unstructured":"Rostami, M., Isele, D., Eaton, E.: Using task descriptions in lifelong machine learning for improved performance and zero-shot transfer. J. Artif. Intell. Res. 67, 673\u2013703 (2020)","journal-title":"J. Artif. Intell. Res."},{"key":"33_CR129","unstructured":"Rusu, A.A., et al.: Progressive neural networks (2016). https:\/\/arxiv.org\/abs\/1606.04671"},{"key":"33_CR130","unstructured":"Rutherford, A., et al.: JaxMARL: multi-agent RL environments and algorithms in JAX. https:\/\/arxiv.org\/abs\/2311.10090 (2023)"},{"key":"33_CR131","doi-asserted-by":"crossref","unstructured":"Sak\u00e7ak, B., Shell, D.A., O\u2019Kane, J.M.: Limits of specifiability for sensor-based robotic planning tasks. In: Proceedings of IEEE International Conference on Robotics and Automation (ICRA) (2025, to appear)","DOI":"10.1109\/ICRA55743.2025.11128030"},{"key":"33_CR132","unstructured":"Samvelyan, M., et al.: MAESTRO: open-ended environment design for multi-agent reinforcement learning. In: International Conference on Learning Representations (2023)"},{"key":"33_CR133","unstructured":"Samvelyan, M., et al.: Minihack the planet: a sandbox for open-ended reinforcement learning research. In: Advances in Neural Information Processing Systems (2021)"},{"key":"33_CR134","doi-asserted-by":"crossref","unstructured":"Schaul, T.: A video game description language for model-based or interactive learning. In: Proceedings of the IEEE Conference on Computational Intelligence in Games, pp. 193\u2013200. IEEE (2013)","DOI":"10.1109\/CIG.2013.6633610"},{"key":"33_CR135","unstructured":"Schaul, T., Horgan, D., Gregor, K., Silver, D.: Universal value function approximators. In: Proceedings of the 32nd International Conference on Machine Learning. JLMR: W&CP, vol.\u00a037, pp. 1312\u20131320 (2015)"},{"issue":"4","key":"33_CR136","doi-asserted-by":"publisher","first-page":"325","DOI":"10.1109\/TCIAIG.2014.2352795","volume":"6","author":"T Schaul","year":"2014","unstructured":"Schaul, T.: An extensible description language for video games. IEEE Trans. Comput. Intell. AI Games 6(4), 325\u2013331 (2014). https:\/\/doi.org\/10.1109\/TCIAIG.2014.2352795","journal-title":"IEEE Trans. Comput. Intell. AI Games"},{"key":"33_CR137","unstructured":"Schmidhuber, J.: On learning how to learn learning strategies. Technical report. FKI-198-94, Institut f\u00fcr Informatik, Technische Universit\u00e4t M\u00fcnchen (1994)"},{"key":"33_CR138","doi-asserted-by":"crossref","unstructured":"Seger, E., Ovadya, A., Siddarth, D., Garfinkel, B., Dafoe, A.: Democratising AI: multiple meanings, goals, and methods. In: Proceedings of the 2023 AAAI\/ACM Conference on AI, Ethics, and Society, pp. 715\u2013722 (2023)","DOI":"10.1145\/3600211.3604693"},{"key":"33_CR139","unstructured":"Shu, T., Xiong, C., Socher, R.: Hierarchical and interpretable skill acquisition in multi-task reinforcement learning. In: International Conference on Learning Representations (2018)"},{"key":"33_CR140","unstructured":"Silver, T., Chitnis, R.: PDDLGym: gym environments from PDDL problems. In: ICAPS Workshop on Bridging the Gap Between AI Planning and Reinforcement Learning (PRL) (2020)"},{"key":"33_CR141","unstructured":"Simkute, A., Luger, E., Evans, M., Jones, R.: \u201cIt is there, and you need it, so why do you not use it?\u201d Achieving better adoption of AI systems by domain experts, in the case study of natural science research (2024). https:\/\/arxiv.org\/abs\/2403.16895"},{"key":"33_CR142","unstructured":"Sobol, D., Wolf, L., Taigman, Y.: Visual analogies between atari games for studying transfer learning in RL (2018). https:\/\/arxiv.org\/abs\/1807.11074"},{"issue":"3","key":"33_CR143","doi-asserted-by":"publisher","first-page":"146","DOI":"10.3233\/ICG-220197","volume":"43","author":"DJNJ Soemers","year":"2022","unstructured":"Soemers, D.J.N.J., Mella, V., Browne, C., Teytaud, O.: Deep learning for general game playing with Ludii and Polygames. ICGA J. 43(3), 146\u2013161 (2022)","journal-title":"ICGA J."},{"key":"33_CR144","unstructured":"Soemers, D.J.N.J., Mella, V., Piette, \u00c9., Stephenson, M., Browne, C., Teytaud, O.: Towards a general transfer approach for policy-value networks. Trans. Mach. Learn. Res. (2023)"},{"key":"33_CR145","doi-asserted-by":"crossref","unstructured":"Soemers, D.J.N.J., Piette, \u00c9., Stephenson, M., Browne, C.: The Ludii game description language is universal. In: Proceedings of the 2024 IEEE Conference on Games, pp.\u00a01\u20138 (2024)","DOI":"10.1109\/CoG60054.2024.10645550"},{"key":"33_CR146","doi-asserted-by":"crossref","unstructured":"Soemers, D.J.N.J., Samothrakis, S., Driessens, K., Winands, M.H.M.: Environment descriptions for usability and generalisation in reinforcement learning. In: Rocha, A.P., Steels, L., van\u00a0den Herik, H.J. (eds.) Proceedings of the 17th International Conference on Agents and Artificial Intelligence, vol.\u00a03, pp. 983\u2013992 (2025)","DOI":"10.5220\/0013247300003890"},{"key":"33_CR147","unstructured":"Stability AI: Stable audio: Fast timing-conditioned latent audio diffusion (2023). https:\/\/stability.ai\/research\/stable-audio-efficient-timing-latent-diffusion. Accessed 4 Jan 2024"},{"key":"33_CR148","doi-asserted-by":"crossref","unstructured":"Stephenson, M., Soemers, D.J.N.J., Piette, \u00c9., Browne, C.: Measuring board game distance. In: Browne, C., Kishimoto, A., Schaeffer, J. (eds.) Computers and Games, CG 2022. LNCS, vol. 13865, pp. 121\u2013130. Springer, Cham (2023)","DOI":"10.1007\/978-3-031-34017-8_11"},{"key":"33_CR149","unstructured":"Stone, A., Ramirez, O., Konolige, K., Jonschkowski, R.: The distracting control suite \u2013 a challenging benchmark for reinforcement learning from pixels (2021). https:\/\/arxiv.org\/abs\/2101.02722"},{"key":"33_CR150","unstructured":"Sun, S.H., Wu, T.L., Lim, J.J.: Program guided agent. In: International Conference on Learning Representations (2020)"},{"key":"33_CR151","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction, 2nd edn. MIT Press, Cambridge (2018)","edition":"2"},{"key":"33_CR152","unstructured":"Tassa, Y., et al.: DeepMind control suite (2018)"},{"key":"33_CR153","unstructured":"Taylor, M.E., Stone, P.: Transfer learning for reinforcement learning domains: a survey. In: Mahadevan, S. (ed.) Journal of Machine Learning Research, vol.\u00a010, pp. 1633\u20131685 (2009)"},{"key":"33_CR154","unstructured":"Terry, J.K., et al.: Pettingzoo: a standard API for multi-agent reinforcement learning. In: Advances in Neural Information Processing Systems, vol.\u00a034, pp. 15032\u201315043. Curran Associates, Inc. (2021)"},{"key":"33_CR155","doi-asserted-by":"crossref","unstructured":"Tessler, C., Givony, S., Zahavy, T., Mankowitz, D., Mannor, S.: A deep hierarchical approach to lifelong learning in minecraft. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 1553\u20131561. AAAI (2017)","DOI":"10.1609\/aaai.v31i1.10744"},{"key":"33_CR156","unstructured":"Thielscher, M.: The general game playing description language is universal. In: Proceedings of the Twenty-second International Joint Conference on Artificial Intelligence, IJCAI-11, pp. 1107\u20131112 (2011)"},{"key":"33_CR157","doi-asserted-by":"crossref","unstructured":"Todd, G., Padula, A., Stephenson, M., Piette, \u00c9., Soemers, D.J.N.J., Togelius, J.: GAVEL: generating games via evolution and language models. In: Globerson, A., et al. (eds.) Advances in Neural Information Processing Systems, vol.\u00a037, pp. 110723\u2013110745. Curran Associates, Inc. (2024)","DOI":"10.52202\/079017-3515"},{"key":"33_CR158","unstructured":"Todd, G., Padula, A.G., Soemers, D.J.N.J., Togelius, J.: Ludax: a GPU-accelerated domain specific language for board games (2025, under review)"},{"key":"33_CR159","unstructured":"Trofin, M., Qian, Y., Brevdo, E., Lin, Z., Choromanski, K., Li, D.: MLGO: a machine learning guided compiler optimizations framework (2021). https:\/\/arxiv.org\/abs\/2101.04808"},{"key":"33_CR160","unstructured":"Voelcker, C., Hussing, M., Eaton, E.: Can we hop in general? A discussion of benchmark selection and design using the Hopper environment. In: Finding the Frame workshop @ Reinforcement Learning Conference (2024)"},{"key":"33_CR161","unstructured":"van\u00a0der Wal, J.: Stochastic Dynamic Programming, No.\u00a0139 in Mathematical Centre tracts, Morgan Kaufmann, Amsterdam (1981)"},{"key":"33_CR162","doi-asserted-by":"crossref","unstructured":"Whiteson, S., Tanner, B., Taylor, M.E., Stone, P.: Protecting against evaluation overfitting in empirical reinforcement learning. In: 2011 IEEE Symposium on Adaptive Dynamic Programming and Reinforcement Learning (ADPRL), pp. 120\u2013127 (2011)","DOI":"10.1109\/ADPRL.2011.5967363"},{"key":"33_CR163","doi-asserted-by":"crossref","unstructured":"Williams, E.C., Gopalan, N., Rhee, M., Tellex, S.: Learning to parse natural language to gronuded reward functions with weak supervision. In: 2018 IEEE International Conference on Robotics and Automation, pp. 4430\u20134436 (2018)","DOI":"10.1109\/ICRA.2018.8460937"},{"key":"33_CR164","doi-asserted-by":"crossref","unstructured":"Wilson, A., Fern, A., Ray, S., Tadepalli, P.: Multi-task reinforcement learning: a hierarchical Bayesian approach. In: Proceedings of the 24th International Conference on Machine Learning, pp. 1015\u20131022 (2007)","DOI":"10.1145\/1273496.1273624"},{"key":"33_CR165","unstructured":"Yang, K.: Learn dynamic-aware state embedding for transfer learning (2021). https:\/\/arxiv.org\/abs\/2101.02230"},{"key":"33_CR166","unstructured":"Yang, M., et al.: Learning interactive real-world simulators. In: International Conference on Learning Representations (2024)"},{"key":"33_CR167","unstructured":"Yoon, D., Hong, S., Lee, B.J., Kim, K.E.: Winning the L2RPN challenge: power grid management via semi-Markov afterstate actor critic. In: Proceedings of the International Conference on Learning Representations (2021)"},{"key":"33_CR168","unstructured":"Zhang, A., Ballas, N., Pineau, J.: A dissection of overfitting and generalization in continuous reinforcement learning (2020). https:\/\/arxiv.org\/abs\/1806.07937"},{"key":"33_CR169","unstructured":"Zhang, C., Vinyals, O., Munos, R., Bengio, S.: A study on overfitting in deep reinforcement learning (2018). https:\/\/arxiv.org\/abs\/1804.06893"},{"key":"33_CR170","doi-asserted-by":"crossref","unstructured":"Zhao, W., Queralta, J.P., Westerlund, T.: Sim-to-real transfer in deep reinforcement learning for robotics: a survey. In: 2020 IEEE Symposium Series on Computational Intelligence (SSCI), pp. 737\u2013744 (2020)","DOI":"10.1109\/SSCI47803.2020.9308468"},{"key":"33_CR171","unstructured":"Zhao, W.X., et al.: A survey of large language models (2023). https:\/\/arxiv.org\/abs\/2303.18223. Accessed 25 Jan 2024"},{"key":"33_CR172","unstructured":"Zhong, V., Rockt\u00e4schel, T., Grefenstette, E.: RTFM: generalising to novel environment dynamics via reading. In: International Conference on Learning Representations (2020)"},{"key":"33_CR173","unstructured":"Zhu, Z., et al.: Pearl: a production-ready reinforcement learning agent (2023). https:\/\/arxiv.org\/abs\/2312.03814"},{"issue":"11","key":"33_CR174","doi-asserted-by":"publisher","first-page":"13344","DOI":"10.1109\/TPAMI.2023.3292075","volume":"45","author":"Z Zhu","year":"2023","unstructured":"Zhu, Z., Lin, K., Jain, A.K., Zhou, J.: Transfer learning in deep reinforcement learning: a survey. IEEE Trans. Pattern Anal. Mach. Intell. 45(11), 13344\u201313362 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"33_CR175","unstructured":"Zintgraf, L.: Fast adaptation via meta reinforcement learning. Ph.D. thesis, University of Oxford, Oxford, United Kingdom (2022)"},{"key":"33_CR176","unstructured":"Zuo, M., Velez, F.P., Li, X., Littman, M.L., Bach, S.H.: Planetarium: a rigorous benchmark for translating text to structured planning languages (2024). https:\/\/arxiv.org\/abs\/2407.03321"}],"container-title":["Lecture Notes in Computer Science","Agents and Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-25032-2_33","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T22:09:40Z","timestamp":1784239780000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-25032-2_33"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032250315","9783032250322"],"references-count":176,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-25032-2_33","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"1 June 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"ICAART","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Agents and Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Porto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Portugal","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 February 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 February 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icaart2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icaart.scitevents.org\/?y=2025","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}