{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T23:18:39Z","timestamp":1743117519631,"version":"3.40.3"},"publisher-location":"Cham","reference-count":16,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031475078"},{"type":"electronic","value":"9783031475085"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-47508-5_4","type":"book-chapter","created":{"date-parts":[[2024,1,31]],"date-time":"2024-01-31T09:16:09Z","timestamp":1706692569000},"page":"41-52","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards Reinforcement Learning for Non-stationary Environments"],"prefix":"10.1007","author":[{"given":"Sebastian Gregory Dal","family":"To\u00e9","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bernard","family":"Tiddeman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Neil","family":"Mac Parthal\u00e1in","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,2,1]]},"reference":[{"key":"4_CR1","doi-asserted-by":"publisher","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D. et al.: Playing atari with deep reinforcement learning (19 Dec 2013). arXiv:1312.5602. https:\/\/doi.org\/10.48550\/arXiv.1312.5602","DOI":"10.48550\/arXiv.1312.5602"},{"key":"4_CR2","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D., Huang, A., Maddison, C., et al.: Mastering the game of Go with deep neural networks and tree search. Nature 529, 484\u2013489 (2016). https:\/\/doi.org\/10.1038\/nature16961","journal-title":"Nature"},{"key":"4_CR3","doi-asserted-by":"publisher","unstructured":"Silver, D., Hubert, T., Schrittwieser, J., et al.: Mastering chess and shogi by self-play with a general reinforcement learning algorithm (5 Dec 2017). arXiv:1712.01815. https:\/\/doi.org\/10.48550\/arXiv.1712.01815","DOI":"10.48550\/arXiv.1712.01815"},{"issue":"7782","key":"4_CR4","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals, O., Babushkin, I., Czarnecki, W., et al.: Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature 575(7782), 350\u2013354 (2019). https:\/\/doi.org\/10.1038\/s41586-019-1724-z","journal-title":"Nature"},{"issue":"7839","key":"4_CR5","doi-asserted-by":"publisher","first-page":"604","DOI":"10.1038\/s41586-020-03051-4","volume":"588","author":"J Schrittwieser","year":"2020","unstructured":"Schrittwieser, J., Antonoglou, I., Hubert, T., et al.: Mastering atari, go, chess and shogi by planning with a learned model. Nature 588(7839), 604\u2013609 (2020). https:\/\/doi.org\/10.1038\/s41586-020-03051-4","journal-title":"Nature"},{"key":"4_CR6","unstructured":"Thrun, S.: Efficient Exploration in Reinforcement Learning. Carnegie Mellon University (1992). https:\/\/citeseerx.ist.psu.edu\/viewdoc\/summary?doi=10.1.1.45.2894"},{"key":"4_CR7","unstructured":"Watkins, C.: Learning from Delayed Rewards. PhD thesis, University of Cambridge, Cambridge, England (1989). https:\/\/www.academia.edu\/3294050\/Learning_from_delayed_rewards?from=cover_page"},{"key":"4_CR8","doi-asserted-by":"publisher","unstructured":"Garnelo, M., Arulkumaran, K., Shanahan, M.: Towards deep symbolic reinforcement learning (18 Sep 2016). arXiv:1609.05518. https:\/\/doi.org\/10.48550\/arXiv.1609.05518","DOI":"10.48550\/arXiv.1609.05518"},{"key":"4_CR9","doi-asserted-by":"publisher","unstructured":"Asai, M., Kajino, H., Fukunaga, A., et al.: Classical planning in deep latent space (30 Jun 2021). arXiv:2107.00110. https:\/\/doi.org\/10.48550\/arXiv.2107.00110","DOI":"10.48550\/arXiv.2107.00110"},{"key":"4_CR10","doi-asserted-by":"publisher","unstructured":"Doersch, C.: Tutorial on variational autoencoders (19 Jun 2016). arXiv:1606.05908. https:\/\/doi.org\/10.48550\/arXiv.1606.05908","DOI":"10.48550\/arXiv.1606.05908"},{"key":"4_CR11","doi-asserted-by":"publisher","unstructured":"Asperti, A., Trentin, M.: Balancing reconstruction error and Kullback-Leibler divergence in variational autoencoders. IEEE Access 8, 199440\u2013199448 (2020). https:\/\/doi.org\/10.48550\/arXiv.2002.07514","DOI":"10.48550\/arXiv.2002.07514"},{"issue":"3","key":"4_CR12","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1007\/BF00994018","volume":"20","author":"C Cortes","year":"1995","unstructured":"Cortes, C., Vapnik, V.: Support-vector networks. Mach. Learn. 20(3), 273\u2013297 (1995). https:\/\/doi.org\/10.1007\/BF00994018","journal-title":"Mach. Learn."},{"key":"4_CR13","unstructured":"OpenAI, Gym Documentation (2022). https:\/\/www.gymlibrary.dev\/content\/api\/"},{"key":"4_CR14","unstructured":"Hill, A., Raffin, A., Ernestus, M., et al.: Stable Baselines (2018). https:\/\/stable-baselines.readthedocs.io\/en\/master\/"},{"key":"4_CR15","doi-asserted-by":"publisher","unstructured":"Wang, Z., Bapst, V., Heess, N., et al.: Sample efficient actor-critic with experience replay (3 Nov 2016). arXiv:1611.01224. https:\/\/doi.org\/10.48550\/arXiv.1611.01224","DOI":"10.48550\/arXiv.1611.01224"},{"key":"4_CR16","doi-asserted-by":"publisher","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., et al.: Proximal policy optimization algorithms (20 Jul 2017). arXiv:1707.06347. https:\/\/doi.org\/10.48550\/arXiv.1707.06347","DOI":"10.48550\/arXiv.1707.06347"}],"container-title":["Advances in Intelligent Systems and Computing","Advances in Computational Intelligence Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-47508-5_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,31]],"date-time":"2024-01-31T09:16:27Z","timestamp":1706692587000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-47508-5_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031475078","9783031475085"],"references-count":16,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-47508-5_4","relation":{},"ISSN":["2194-5357","2194-5365"],"issn-type":[{"type":"print","value":"2194-5357"},{"type":"electronic","value":"2194-5365"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"1 February 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"UKCI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"UK Workshop on Computational Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Birmingham","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 September 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 September 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ukci2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.uk-ci.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}