{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T14:18:34Z","timestamp":1743085114880,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":33,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819916382"},{"type":"electronic","value":"9789819916399"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-981-99-1639-9_16","type":"book-chapter","created":{"date-parts":[[2023,4,14]],"date-time":"2023-04-14T07:02:39Z","timestamp":1681455759000},"page":"189-201","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards a\u00a0Unified Benchmark for\u00a0Reinforcement Learning in\u00a0Sparse Reward Environments"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0468-9234","authenticated-orcid":false,"given":"Yongxin","family":"Kang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6117-5080","authenticated-orcid":false,"given":"Enmin","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4537-384X","authenticated-orcid":false,"given":"Yifan","family":"Zang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3840-3270","authenticated-orcid":false,"given":"Kai","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6801-0510","authenticated-orcid":false,"given":"Junliang","family":"Xing","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,4,15]]},"reference":[{"key":"16_CR1","unstructured":"Aytar, Y., Pfaff, T., Budden, D., et al.: Playing hard exploration games by watching youtube. In: NeurIPS. pp. 2935\u20132945 (2018)"},{"key":"16_CR2","doi-asserted-by":"crossref","unstructured":"Baumli, K., Warde-Farley, D., Hansen, S., et al.: Relative variational intrinsic control. In: AAAI. pp. 6732\u20136740 (2021)","DOI":"10.1609\/aaai.v35i8.16832"},{"key":"16_CR3","unstructured":"Bellemare, M., Srinivasan, S., Ostrovski, G., et al.: Unifying count-based exploration and intrinsic motivation. In: NeurIPS. pp. 1471\u20131479 (2016)"},{"issue":"1","key":"16_CR4","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1613\/jair.3912","volume":"47","author":"MG Bellemare","year":"2013","unstructured":"Bellemare, M.G., Naddaf, Y., Veness, J., et al.: The arcade learning environment: An evaluation platform for general agents. JAIR 47(1), 253\u2013279 (2013)","journal-title":"JAIR"},{"key":"16_CR5","unstructured":"Burda, Y., Edwards, H., Storkey, A., et al.: Exploration by random network distillation. In: ICLR. pp. 1\u201317 (2018)"},{"key":"16_CR6","unstructured":"Chen, Z., Lin, M.: Self-imitation learning in sparse reward settings. arXiv preprint arXiv:2010.06962 (2020)"},{"issue":"10","key":"16_CR7","doi-asserted-by":"publisher","first-page":"1742","DOI":"10.3390\/electronics9101742","volume":"9","author":"T Dai","year":"2020","unstructured":"Dai, T., Liu, H., Anthony Bharath, A.: Episodic self-imitation learning with hindsight. Electronics 9(10), 1742 (2020)","journal-title":"Electronics"},{"key":"16_CR8","unstructured":"Dhariwal, P., Hesse, C., Klimov, O., et al.: OpenAI Baselines. https:\/\/github.com\/openai\/baselines (2017)"},{"key":"16_CR9","unstructured":"Guo, Y., Choi, J., Moczulski, M., et al.: Memory based trajectory-conditioned policies for learning from sparse rewards. In: NeurIPS. pp. 4333\u20134345 (2020)"},{"key":"16_CR10","unstructured":"Guo, Y., Oh, J., Singh, S., Lee, H.: Generative adversarial self-imitation learning. arXiv preprint arXiv:1812.00950 (2018)"},{"key":"16_CR11","doi-asserted-by":"crossref","unstructured":"Hessel, M., Modayil, J., Van Hasselt, H., et al.: Rainbow: Combining improvements in deep reinforcement learning. In: AAAI. pp. 3215\u20133222 (2018)","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"16_CR12","doi-asserted-by":"crossref","unstructured":"Hester, T., Vecerik, M., Pietquin, O., et al.: Deep Q-learning from demonstrations. In: AAAI. pp. 3223\u20133230 (2018)","DOI":"10.1609\/aaai.v32i1.11757"},{"key":"16_CR13","unstructured":"Itti, L., Baldi, P.: Bayesian surprise attracts human attention. In: NeurIPS. pp. 547\u2013554 (2005)"},{"key":"16_CR14","unstructured":"Kim, H., Kim, J., Jeong, Y., et al.: EMI: Exploration with mutual information. In: ICML. pp. 3360\u20133369 (2019)"},{"key":"16_CR15","unstructured":"Leibfried, F., Pascual-D\u00edaz, S., Grau-Moya, J.: A unified bellman optimality principle combining reward maximization and empowerment. In: NeurIPS. pp. 7869\u20137880 (2019)"},{"key":"16_CR16","unstructured":"Mnih, V., Badia, A.P., Mirza, M., et al.: Asynchronous methods for deep reinforcement learning. In: ICML. pp. 1928\u20131937 (2016)"},{"issue":"7540","key":"16_CR17","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015)","journal-title":"Nature"},{"key":"16_CR18","unstructured":"Ng, A.Y., Russell, S.J., et al.: Algorithms for inverse reinforcement learning. In: ICML. pp. 663\u2013670 (2000)"},{"key":"16_CR19","unstructured":"Oh, J., Guo, Y., Singh, S., Lee, H.: Self-imitation learning. In: ICML. pp. 3875\u20133884 (2018)"},{"key":"16_CR20","unstructured":"Ostrovski, G., Bellemare, M.G., Oord, A., et al.: Count-based exploration with neural density models. In: ICML. pp. 2721\u20132730 (2017)"},{"key":"16_CR21","doi-asserted-by":"crossref","unstructured":"Pathak, D., Agrawal, P., Efros, A.A., et al.: Curiosity-driven exploration by self-supervised prediction. In: ICML. pp. 2778\u20132787 (2017)","DOI":"10.1109\/CVPRW.2017.70"},{"issue":"4","key":"16_CR22","first-page":"1","volume":"37","author":"XB Peng","year":"2018","unstructured":"Peng, X.B., Abbeel, P., Levine, S., et al.: DeepMimic: Example-guided deep reinforcement learning of physics-based character skills. TOG 37(4), 1\u201314 (2018)","journal-title":"TOG"},{"key":"16_CR23","unstructured":"Pohlen, T., Piot, B., Hester, T., et al.: Observe and look further: Achieving consistent performance on atari. arXiv preprint arXiv:1805.11593 (2018)"},{"issue":"2","key":"16_CR24","first-page":"627","volume":"1","author":"S Ross","year":"2011","unstructured":"Ross, S., Gordon, G.J., Bagnell, J.A.: A reduction of imitation learning and structured prediction to no-regret online learning. AISTATS 1(2), 627\u2013635 (2011)","journal-title":"AISTATS"},{"key":"16_CR25","unstructured":"Savinov, N., Raichuk, A., Vincent, D., et al.: Episodic curiosity through reachability. In: ICLR. pp. 1\u201320 (2019)"},{"key":"16_CR26","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., et al.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"16_CR27","unstructured":"Sekar, R., Rybkin, O., Daniilidis, K., et al.: Planning to explore via self-supervised world models. In: ICML. pp. 8583\u20138592 (2020)"},{"issue":"2","key":"16_CR28","first-page":"70","volume":"2","author":"S Singh","year":"2010","unstructured":"Singh, S., Lewis, R.L., Barto, A.G., et al.: Intrinsically motivated reinforcement learning: An evolutionary perspective. TAMD 2(2), 70\u201382 (2010)","journal-title":"TAMD"},{"issue":"8","key":"16_CR29","first-page":"1309","volume":"74","author":"AL Strehl","year":"2008","unstructured":"Strehl, A.L., Littman, M.L.: An analysis of model-based interval estimation for markov decision processes. JCSS 74(8), 1309\u20131331 (2008)","journal-title":"JCSS"},{"key":"16_CR30","unstructured":"Taiga, A.A., Fedus, W., Machado, M.C., et al.: On bonus based exploration methods in the arcade learning environment. In: ICLR. pp. 1\u201320 (2020)"},{"key":"16_CR31","unstructured":"Tang, H., Houthooft, R., Foote, D., et al.: # Exploration: A study of count-based exploration for deep reinforcement learning. In: NeurIPS. pp. 2753\u20132762 (2017)"},{"key":"16_CR32","doi-asserted-by":"crossref","unstructured":"Zhang, C., Cai, Y., Huang, L., Li, J.: Exploration by maximizing renyi entropy for reward-free rl framework. In: AAAI. pp. 10859\u201310867 (2021)","DOI":"10.1609\/aaai.v35i12.17297"},{"key":"16_CR33","unstructured":"Zhang, T., Xu, H., Wang, X., et al.: BeBold: Exploration beyond the boundary of explored regions. arXiv preprint arXiv:2012.08621 (2020)"}],"container-title":["Communications in Computer and Information Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-99-1639-9_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,4,14]],"date-time":"2023-04-14T07:27:29Z","timestamp":1681457249000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-99-1639-9_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9789819916382","9789819916399"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-981-99-1639-9_16","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"15 April 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"New Delhi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 November 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 November 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/iconip2022.apnns.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Easy Chair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"810","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"359","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"44% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.65","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"ICONIP 2022 consists of a two-volume set, LNCS & CCIS, which includes 146 and 213 papers","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}