{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,27]],"date-time":"2025-10-27T05:07:15Z","timestamp":1761541635330,"version":"3.40.3"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783031144622"},{"type":"electronic","value":"9783031144639"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-14463-9_13","type":"book-chapter","created":{"date-parts":[[2022,8,10]],"date-time":"2022-08-10T12:02:48Z","timestamp":1660132968000},"page":"201-220","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["An Evaluation Study of\u00a0Intrinsic Motivation Techniques Applied to\u00a0Reinforcement Learning over\u00a0Hard Exploration Environments"],"prefix":"10.1007","author":[{"given":"Alain","family":"Andres","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Esther","family":"Villar-Rodriguez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Javier","family":"Del Ser","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,8,11]]},"reference":[{"issue":"7676","key":"13_CR1","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver, D., et al.: Mastering the game of go without human knowledge. Nature 550(7676), 354\u2013359 (2017)","journal-title":"Nature"},{"key":"13_CR2","unstructured":"Baker, B., et al.: Emergent tool use from multi-agent autocurricula. arXiv:1909.07528 (2019)"},{"issue":"1","key":"13_CR3","first-page":"1","volume":"1","author":"A Holzinger","year":"2019","unstructured":"Holzinger, A.: Introduction to machine learning & knowledge extraction (make). Mach. Learn. Knowl. Extr. 1(1), 1\u201320 (2019)","journal-title":"Mach. Learn. Knowl. Extr."},{"key":"13_CR4","unstructured":"Aubret, A., Matignon, L., Hassas, S.: A survey on intrinsic motivation in reinforcement learning. arXiv:1908.06976 (2019)"},{"key":"13_CR5","unstructured":"Ho, J., Ermon, S.: Generative adversarial imitation learning. In: Advances in Neural Information Processing Systems, vol. 29 (2016)"},{"key":"13_CR6","unstructured":"Finn, C., Levine, S., Abbeel, P.: Guided cost learning: deep inverse optimal control via policy optimization (2016)"},{"key":"13_CR7","unstructured":"Grigorescu, D.: Curiosity, intrinsic motivation and the pleasure of knowledge. J. Educ. Sci. Psychol. 10(1) (2020)"},{"key":"13_CR8","unstructured":"Raileanu, R., Rockt\u00e4schel, T.: Ride: rewarding impact-driven exploration for procedurally-generated environments. arXiv:2002.12292 (2020)"},{"key":"13_CR9","unstructured":"Badia, A.P., et al.: Never give up: learning directed exploration strategies. arXiv:2002.06038 (2020)"},{"key":"13_CR10","unstructured":"Flet-Berliac, Y., Ferret, J., Pietquin, O., Preux, P., Geist, M.: Adversarially guided actor-critic. arXiv:2102.04376 (2021)"},{"key":"13_CR11","doi-asserted-by":"crossref","unstructured":"Pathak, D., Agrawal, P., Efros, A.A., Darrell, T.: Curiosity-driven exploration by self-supervised prediction. In: International Conference on Machine Learning, pp. 2778\u20132787 (2017)","DOI":"10.1109\/CVPRW.2017.70"},{"key":"13_CR12","unstructured":"Burda, Y., Edwards, H., Storkey, A., Klimov, O.: Exploration by random network distillation. arXiv:1810.12894 (2018)"},{"key":"13_CR13","unstructured":"Andrychowicz, M., et al.: What matters in on-policy reinforcement learning? A large-scale empirical study. arXiv:2006.05990 (2020)"},{"key":"13_CR14","unstructured":"Andrychowicz, M., et al.: What matters for on-policy deep actor-critic methods? A large-scale study. In: International Conference on Learning Representations (2020)"},{"key":"13_CR15","unstructured":"Bellemare, M., Srinivasan, S., Ostrovski, G., Schaul, T., Saxton, D., Munos, R.: Unifying count-based exploration and intrinsic motivation. In: Advances in Neural Information Processing Systems, vol. 29 (2016)"},{"key":"13_CR16","unstructured":"Tang, H., et al.: # exploration: a study of count-based exploration for deep reinforcement learning. In: Advances in Neural Information Processing Systems, pp. 2753\u20132762 (2017)"},{"key":"13_CR17","doi-asserted-by":"crossref","unstructured":"Machado, M.C., Bellemare, M.G., Bowling, M.: Count-based exploration with the successor representation. In: AAAI Conference on Artificial Intelligence, vol. 34, no. 4, pp. 5125\u20135133 (2020)","DOI":"10.1609\/aaai.v34i04.5955"},{"key":"13_CR18","unstructured":"P\u00eeslar, M., Szepesvari, D., Ostrovski, G., Borsa, D., Schaul, T.: When should agents explore? arXiv:2108.11811 (2021)"},{"key":"13_CR19","unstructured":"Zhang, T., et al.: NovelD: a simple yet effective exploration criterion. In: Advances in Neural Information Processing Systems, vol. 34 (2021)"},{"issue":"2","key":"13_CR20","doi-asserted-by":"publisher","first-page":"1086","DOI":"10.1007\/s10489-020-01849-3","volume":"51","author":"N Bougie","year":"2020","unstructured":"Bougie, N., Ichise, R.: Fast and slow curiosity for high-level exploration in reinforcement learning. Appl. Intell. 51(2), 1086\u20131107 (2020). https:\/\/doi.org\/10.1007\/s10489-020-01849-3","journal-title":"Appl. Intell."},{"key":"13_CR21","unstructured":"Campero, A., Raileanu, R., K\u00fcttler, H., Tenenbaum, J.B., Rockt\u00e4schel, T., Grefenstette, E.: Learning with amigo: adversarially motivated intrinsic goals. arXiv:2006.12122 (2020)"},{"key":"13_CR22","unstructured":"Taiga, A.A., Fedus, W., Machado, M.C., Courville, A., Bellemare, M.G.: On bonus-based exploration methods in the arcade learning environment. arXiv:2109.11052 (2021)"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"Hessel, M., et al.: Rainbow: combining improvements in deep reinforcement learning. In: AAAI Conference on Artificial Intelligence (2018)","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"13_CR24","unstructured":"Burda, Y., Edwards, H., Pathak, D., Storkey, A., Darrell, T., Efros, A.A.: Large-scale study of curiosity-driven learning. In: ICLR (2019)"},{"key":"13_CR25","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv:1707.06347 (2017)"},{"key":"13_CR26","unstructured":"Orsini, M., et al.: What matters for adversarial imitation learning? In: Advances in Neural Information Processing Systems, vol. 34 (2021)"},{"key":"13_CR27","unstructured":"Jing, X., et al.: Divide and explore: multi-agent separate exploration with shared intrinsic motivations (2022)"},{"key":"13_CR28","doi-asserted-by":"crossref","unstructured":"Seurin, M., Strub, F., Preux, P., Pietquin, O.: Don\u2019t do what doesn\u2019t matter: intrinsic motivation with action usefulness. arXiv:2105.09992 (2021)","DOI":"10.24963\/ijcai.2021\/406"},{"key":"13_CR29","unstructured":"Zha, D., Ma, W., Yuan, L., Hu, X., Liu, J.: Rank the episodes: a simple approach for exploration in procedurally-generated environments. arXiv:2101.08152 (2021)"},{"key":"13_CR30","unstructured":"Espeholt, L., et al.: IMPALA: scalable distributed deep-RL with importance weighted actor-learner architectures. In: International Conference on Machine Learning, pp. 1407\u20131416 (2018)"},{"key":"13_CR31","unstructured":"Chevalier-Boisvert, M., Willems, L., Pal, S.: Minimalistic gridworld environment for OpenAI gym. http:\/\/github.com\/maximecb\/gym-minigrid (2018)"},{"key":"13_CR32","unstructured":"Schulman, J., Moritz, P., Levine, S., Jordan, M., Abbeel, P.: High-dimensional continuous control using generalized advantage estimation. arXiv:1506.02438 (2015)"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Extraction"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-14463-9_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,7]],"date-time":"2024-03-07T17:04:40Z","timestamp":1709831080000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-14463-9_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031144622","9783031144639"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-14463-9_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"11 August 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CD-MAKE","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Cross-Domain Conference for Machine Learning and Knowledge Extraction","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Vienna","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Austria","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 August 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 August 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cd-make2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/cd-make.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"45","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"23","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"51% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}