{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T13:03:46Z","timestamp":1743080626907,"version":"3.40.3"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031703430"},{"type":"electronic","value":"9783031703447"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-70344-7_13","type":"book-chapter","created":{"date-parts":[[2024,8,29]],"date-time":"2024-08-29T08:02:43Z","timestamp":1724918563000},"page":"216-232","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Variable-Agnostic Causal Exploration for\u00a0Reinforcement Learning"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8134-3341","authenticated-orcid":false,"given":"Minh Hoang","family":"Nguyen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hung","family":"Le","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Svetha","family":"Venkatesh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,8,22]]},"reference":[{"unstructured":"Andrychowicz, M., et al.: Hindsight experience replay. In: Advances in Neural Information Processing Systems, vol. 30 (2017)","key":"13_CR1"},{"unstructured":"Bellemare, M., Srinivasan, S., Ostrovski, G., Schaul, T., Saxton, D., Munos, R.: Unifying count-based exploration and intrinsic motivation. In: Advances in Neural Information Processing Systems, vol. 29 (2016)","key":"13_CR2"},{"unstructured":"Burda, Y., Edwards, H., Storkey, A., Klimov, O.: Exploration by random network distillation. arXiv preprint arXiv:1810.12894 (2018)","key":"13_CR3"},{"unstructured":"Chevalier-Boisvert, M., et al.: Minigrid & miniworld: modular & customizable reinforcement learning environments for goal-oriented tasks. CoRR abs\/2306.13831 (2023)","key":"13_CR4"},{"unstructured":"Corcoll, O., Vicente, R.: Disentangling causal effects for hierarchical reinforcement learning. arXiv preprint arXiv:2010.01351 (2020)","key":"13_CR5"},{"unstructured":"De\u00a0Haan, P., Jayaraman, D., Levine, S.: Causal confusion in imitation learning. In: Advances in Neural Information Processing Systems, vol. 32 (2019)","key":"13_CR6"},{"unstructured":"Ding, W., Lin, H., Li, B., Zhao, D.: Generalizing goal-conditioned reinforcement learning with variational causal reasoning. In: Advances in Neural Information Processing Systems, vol. 35, pp. 26532\u201326548 (2022)","key":"13_CR7"},{"unstructured":"Hu, X., et al.: Causality-driven hierarchical structure discovery for reinforcement learning. In: Advances in Neural Information Processing Systems, vol. 35, pp. 20064\u201320076 (2022)","key":"13_CR8"},{"issue":"1","key":"13_CR9","doi-asserted-by":"publisher","first-page":"5223","DOI":"10.1038\/s41467-019-13073-w","volume":"10","author":"CC Hung","year":"2019","unstructured":"Hung, C.C., et al.: Optimizing agent behavior over long time scales by transporting value. Nat. Commun. 10(1), 5223 (2019)","journal-title":"Nat. Commun."},{"unstructured":"Ke, N.R., et al.: Learning neural causal models from unknown interventions. arXiv preprint arXiv:1910.01075 (2019)","key":"13_CR10"},{"unstructured":"de\u00a0Lazcano, R., Andreas, K., Tai, J.J., Lee, S.R., Terry, J.: Gymnasium robotics (2023). http:\/\/github.com\/Farama-Foundation\/Gymnasium-Robotics","key":"13_CR11"},{"unstructured":"Le, H., Do, K., Nguyen, D., Venkatesh, S.: Beyond surprise: improving exploration through surprise novelty. In: Proceedings of the 23rd International Conference on Autonomous Agents and Multiagent Systems, AAMAS 2024, pp. 1084\u20131092. International Foundation for Autonomous Agents and Multiagent Systems, Richland, SC (2024)","key":"13_CR12"},{"unstructured":"Levy, A., Konidaris, G., Platt, R., Saenko, K.: Learning multi-level hierarchies with hindsight. arXiv preprint arXiv:1712.00948 (2017)","key":"13_CR13"},{"unstructured":"Lillicrap, T.P., et al.: Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 (2015)","key":"13_CR14"},{"doi-asserted-by":"crossref","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015)","key":"13_CR15","DOI":"10.1038\/nature14236"},{"doi-asserted-by":"crossref","unstructured":"Pearl, J.: Causal inference in statistics: an overview (2009)","key":"13_CR16","DOI":"10.1214\/09-SS057"},{"unstructured":"Pitis, S., Chan, H., Zhao, S., Stadie, B., Ba, J.: Maximum entropy gain exploration for long horizon multi-goal reinforcement learning. In: International Conference on Machine Learning, pp. 7750\u20137761. PMLR (2020)","key":"13_CR17"},{"unstructured":"Pitis, S., Creager, E., Garg, A.: Counterfactual data augmentation using locally factored dynamics. In: Advances in Neural Information Processing Systems, vol. 33, pp. 3976\u20133990 (2020)","key":"13_CR18"},{"unstructured":"Plappert, M., et al.: Multi-goal reinforcement learning: challenging robotics environments and request for research. arXiv preprint arXiv:1802.09464 (2018)","key":"13_CR19"},{"unstructured":"Samvelyan, M., et al.: Minihack the planet: a sandbox for open-ended reinforcement learning research. In: Thirty-Fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 1) (2021). https:\/\/openreview.net\/forum?id=skFwlyefkWJ","key":"13_CR20"},{"unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)","key":"13_CR21"},{"unstructured":"Seitzer, M., Sch\u00f6lkopf, B., Martius, G.: Causal influence detection for improving efficiency in reinforcement learning. In: Advances in Neural Information Processing Systems, vol. 34, pp. 22905\u201322918 (2021)","key":"13_CR22"},{"doi-asserted-by":"crossref","unstructured":"Silver, D., et al.: Mastering the game of go without human knowledge. Nature 550(7676), 354\u2013359 (2017)","key":"13_CR23","DOI":"10.1038\/nature24270"},{"unstructured":"Sun, Z., He, B., Liu, J., Chen, X., Ma, C., Zhang, S.: Offline imitation learning with variational counterfactual reasoning. In: Advances in Neural Information Processing Systems, vol. 36 (2023)","key":"13_CR24"},{"key":"13_CR25","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (2018)"},{"unstructured":"Tang, H., et al.: # exploration: a study of count-based exploration for deep reinforcement learning. In: Advances in Neural Information Processing Systems, vol. 30 (2017)","key":"13_CR26"},{"doi-asserted-by":"publisher","unstructured":"Towers, M., et al.: Gymnasium (2023). https:\/\/doi.org\/10.5281\/zenodo.8127026. https:\/\/zenodo.org\/record\/8127025","key":"13_CR27","DOI":"10.5281\/zenodo.8127026"},{"unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)","key":"13_CR28"},{"unstructured":"Zeng, Y., Cai, R., Sun, F., Huang, L., Hao, Z.: A survey on causal reinforcement learning. arXiv preprint arXiv:2302.05209 (2023)","key":"13_CR29"},{"unstructured":"Zhang, P., Liu, F., Chen, Z., Jianye, H., Wang, J.: Deep reinforcement learning with causality-based intrinsic reward (2020)","key":"13_CR30"},{"unstructured":"Zhang, T., Guo, S., Tan, T., Hu, X., Chen, F.: Generating adjacency-constrained subgoals in hierarchical reinforcement learning. In: Advances in Neural Information Processing Systems, vol. 33, pp. 21579\u201321590 (2020)","key":"13_CR31"},{"unstructured":"Zhang, Y., et al.: Interpretable reward redistribution in reinforcement learning: a causal approach. In: Advances in Neural Information Processing Systems, vol. 36 (2024)","key":"13_CR32"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Discovery in Databases. Research Track"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-70344-7_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,29]],"date-time":"2024-08-29T08:07:20Z","timestamp":1724918840000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-70344-7_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031703430","9783031703447"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-70344-7_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"22 August 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECML PKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Joint European Conference on Machine Learning and Knowledge Discovery in Databases","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Vilnius","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lithuania","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 September 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecml2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2024.ecmlpkdd.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}