{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T05:00:00Z","timestamp":1762491600990,"version":"build-2065373602"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032093202","type":"print"},{"value":"9783032093219","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,8]],"date-time":"2025-11-08T00:00:00Z","timestamp":1762560000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,8]],"date-time":"2025-11-08T00:00:00Z","timestamp":1762560000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-09321-9_28","type":"book-chapter","created":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T04:58:17Z","timestamp":1762491497000},"page":"405-418","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Learning Action Strategies in\u00a0the\u00a0Wumpus World with\u00a0DQN"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7047-0070","authenticated-orcid":false,"given":"Karol","family":"Draszawka","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5029-6768","authenticated-orcid":false,"given":"Julian","family":"Szyma\u0144ski","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0791-8298","authenticated-orcid":false,"given":"David","family":"Gil","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2213-730X","authenticated-orcid":false,"given":"Maria Teresa Signes","family":"Pont","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8591-0710","authenticated-orcid":false,"given":"Higinio","family":"Mora","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,8]]},"reference":[{"key":"28_CR1","unstructured":"Akkaya, I., et\u00a0al.: Solving rubik\u2019s cube with a robot hand. arXiv preprint arXiv:1910.07113 (2019)"},{"key":"28_CR2","unstructured":"Berner, C., et\u00a0al.: Dota 2 with large scale deep reinforcement learning. arXiv preprint arXiv:1912.06680 (2019)"},{"key":"28_CR3","unstructured":"BH, R.: Q-learning in wumpus world. https:\/\/dokumen.tips\/documents\/q-learning-in-wumpus-world-bhrolenocs6601wumpuspdf-q-learning-in-wumpus-world.html. Accessed 12 Mar 2022"},{"key":"28_CR4","unstructured":"Bradshaw, J., Matthews, A.G.D.G., Ghahramani, Z.: Adversarial examples, uncertainty, and transfer testing robustness in gaussian process hybrid deep networks. arXiv preprint arXiv:1707.02476 (2017)"},{"key":"28_CR5","unstructured":"Draszawka, K.: Jak wykra\u015b\u0107 z\u0142oto smokowi? - uczenie ze wzmocnieniem w \u015bwiecie Wumpusa, pp. 90\u2013109. unknown (2021)"},{"key":"28_CR6","unstructured":"Ecoffet, A., Huizinga, J., Lehman, J., Stanley, K.O., Clune, J.: Go-explore: a new approach for hard-exploration problems. arXiv preprint arXiv:1901.10995 (2019)"},{"key":"28_CR7","unstructured":"Friesen, A.L.: A comparison of exploration\/exploitation techniques for a q-learning agent in the wumpus world (08 2009)"},{"key":"28_CR8","unstructured":"Grzes, M.: Reward shaping in episodic reinforcement learning. In: Adaptive Agents and Multi-Agent Systems (2017)"},{"key":"28_CR9","unstructured":"Guo, X., Singh, S., Lee, H., Lewis, R.L., Wang, X.: Deep learning for real-time atari game play using offline monte-carlo tree search planning. In: Advances in Neural Information Processing Systems, vol. 27 (2014)"},{"key":"28_CR10","unstructured":"Ha, D., Schmidhuber, J.: World models. arXiv preprint arXiv:1803.10122 (2018)"},{"key":"28_CR11","doi-asserted-by":"crossref","unstructured":"Haarnoja, T., Ha, S., Zhou, A., Tan, J., Tucker, G., Levine, S.: Learning to walk via deep reinforcement learning. arXiv preprint arXiv:1812.11103 (2018)","DOI":"10.15607\/RSS.2019.XV.011"},{"key":"28_CR12","unstructured":"Hafner, D., Pasukonis, J., Ba, J., Lillicrap, T.: Mastering diverse domains through world models. arXiv preprint arXiv:2301.04104 (2023)"},{"key":"28_CR13","unstructured":"Hare, J.: Dealing with sparse rewards in reinforcement learning. arXiv preprint arXiv:1910.09281 (2019)"},{"key":"28_CR14","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"28_CR15","unstructured":"Lillicrap, T.P., et al.: Continuous control with deep reinforcement learning. In: Bengio, Y., LeCun, Y. (eds.) ICLR (2016)"},{"key":"28_CR16","unstructured":"Mnih, V., et al.: Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 (2013)"},{"issue":"7540","key":"28_CR17","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015)","journal-title":"Nature"},{"key":"28_CR18","unstructured":"Pohlen, T., et\u00a0al.: Observe and look further: achieving consistent performance on atari. arXiv preprint arXiv:1805.11593 (2018)"},{"key":"28_CR19","unstructured":"Malathi, R., Venugopal, T.: The jaina logic way of wumpus world. Int. J. Eng. Sci. Invention Res. Dev. II(VI), 90\u201395 (2015)"},{"key":"28_CR20","volume-title":"Artificial Intelligence: A Modern Approach","author":"SJ Russell","year":"2003","unstructured":"Russell, S.J., Norvig, P.: Artificial Intelligence: A Modern Approach, 2nd edn. Pearson Education, London (2003)","edition":"2"},{"key":"28_CR21","unstructured":"Salimans, T., Chen, R.: Learning montezuma\u2019s revenge from a single demonstration. https:\/\/openai.com\/blog\/learning-montezumas-revenge-from-a-single-demonstration\/. Accessed 12 July 2021"},{"key":"28_CR22","unstructured":"Sardina, S., Vassos, S.: The wumpus world in indigolog: a preliminary report. In: Proceedings of the Workshop on Non-monotonic Reasoning, Action and Change at IJCAI (NRAC-05), pp. 90\u201395. Citeseer (2005)"},{"key":"28_CR23","unstructured":"Schaul, T., Quan, J., Antonoglou, I., Silver, D.: Prioritized experience replay. arXiv preprint arXiv:1511.05952 (2015)"},{"issue":"7839","key":"28_CR24","doi-asserted-by":"publisher","first-page":"604","DOI":"10.1038\/s41586-020-03051-4","volume":"588","author":"J Schrittwieser","year":"2020","unstructured":"Schrittwieser, J., et al.: Mastering Atari, go, chess and shogi by planning with a learned model. Nature 588(7839), 604\u2013609 (2020)","journal-title":"Nature"},{"key":"28_CR25","unstructured":"Shapiro, S.C., Kandefer, M.: A sneps approach to the wumpus world agent or cassie meets the wumpus. In: IJCAI-05 Workshop on Nonmonotonic Reasoning, Action, and Change (NRAC\u201905): Working Notes, pp. 96\u2013103. Citeseer (2005)"},{"issue":"6419","key":"28_CR26","doi-asserted-by":"publisher","first-page":"1140","DOI":"10.1126\/science.aar6404","volume":"362","author":"D Silver","year":"2018","unstructured":"Silver, D., et al.: A general reinforcement learning algorithm that masters chess, shogi, and go through self-play. Science 362(6419), 1140\u20131144 (2018)","journal-title":"Science"},{"key":"28_CR27","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (2018)"},{"key":"28_CR28","doi-asserted-by":"crossref","unstructured":"Trevizan, F.W., De\u00a0Barros, L.N., Da\u00a0Silva, F.S.C.: Designing logic-based robots. Inteligencia Artificial. Revista Iberoamericana de Inteligencia Artif. 10(31), 11\u201322 (2006)","DOI":"10.4114\/ia.v10i31.933"},{"issue":"7782","key":"28_CR29","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals, O., et al.: Grandmaster level in starcraft ii using multi-agent reinforcement learning. Nature 575(7782), 350\u2013354 (2019)","journal-title":"Nature"}],"container-title":["Lecture Notes in Computer Science","Computational Collective Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-09321-9_28","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T04:58:20Z","timestamp":1762491500000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-09321-9_28"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,8]]},"ISBN":["9783032093202","9783032093219"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-09321-9_28","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,8]]},"assertion":[{"value":"8 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICCCI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Computational Collective Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Ho Chi Minh City","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Vietnam","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 November 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 November 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iccci2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/iccci.pwr.edu.pl\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}