{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T07:16:51Z","timestamp":1761808611714,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":58,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,8,4]],"date-time":"2023-08-04T00:00:00Z","timestamp":1691107200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Science Foundation of China","award":["62276126"],"award-info":[{"award-number":["62276126"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["14380010"],"award-info":[{"award-number":["14380010"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022ZD0114804"],"award-info":[{"award-number":["2022ZD0114804"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,8,6]]},"DOI":"10.1145\/3580305.3599393","type":"proceedings-article","created":{"date-parts":[[2023,8,4]],"date-time":"2023-08-04T18:10:58Z","timestamp":1691172658000},"page":"2825-2837","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Internal Logical Induction for Pixel-Symbolic Reinforcement Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-4315-6976","authenticated-orcid":false,"given":"Jiacheng","family":"Xu","sequence":"first","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-4765-2397","authenticated-orcid":false,"given":"Chao","family":"Chen","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7251-4372","authenticated-orcid":false,"given":"Fuxiang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7595-3104","authenticated-orcid":false,"given":"Lei","family":"Yuan","sequence":"additional","affiliation":[{"name":"Nanjing University &amp; Polixir Technologies, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9238-4747","authenticated-orcid":false,"given":"Zongzhang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8011-3430","authenticated-orcid":false,"given":"Yang","family":"Yu","sequence":"additional","affiliation":[{"name":"Nanjing University &amp; Polixir Technologies, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,8,4]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364919887447"},{"key":"e_1_3_2_2_2_1","volume-title":"OpenAI Pieter Abbeel, and Wojciech Zaremba","author":"Andrychowicz Marcin","year":"2017","unstructured":"Marcin Andrychowicz , Filip Wolski , Alex Ray , Jonas Schneider , Rachel Fong , Peter Welinder , Bob McGrew , Josh Tobin , OpenAI Pieter Abbeel, and Wojciech Zaremba . 2017 . Hindsight Experience Replay. In Advances in Neural Information Processing Systems (NeurIPS) . 5048--5058. Marcin Andrychowicz, Filip Wolski, Alex Ray, Jonas Schneider, Rachel Fong, Peter Welinder, Bob McGrew, Josh Tobin, OpenAI Pieter Abbeel, and Wojciech Zaremba. 2017. Hindsight Experience Replay. In Advances in Neural Information Processing Systems (NeurIPS). 5048--5058."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2743240"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/371"},{"key":"e_1_3_2_2_5_1","volume-title":"Transfer in Deep Reinforcement Learning Using Successor Features and Generalised Policy Improvement. In International Conference on Machine Learning (ICML). 501--510","author":"Barreto Andre","year":"2018","unstructured":"Andre Barreto , Diana Borsa , John Quan , Tom Schaul , David Silver , Matteo Hessel , Daniel Mankowitz , Augustin Zidek , and Remi Munos . 2018 . Transfer in Deep Reinforcement Learning Using Successor Features and Generalised Policy Improvement. In International Conference on Machine Learning (ICML). 501--510 . Andre Barreto, Diana Borsa, John Quan, Tom Schaul, David Silver, Matteo Hessel, Daniel Mankowitz, Augustin Zidek, and Remi Munos. 2018. Transfer in Deep Reinforcement Learning Using Successor Features and Generalised Policy Improvement. In International Conference on Machine Learning (ICML). 501--510."},{"key":"e_1_3_2_2_6_1","unstructured":"Andre Barreto Will Dabney Remi Munos Jonathan J Hunt Tom Schaul Hado P van Hasselt and David Silver. 2017. Successor Features for Transfer in Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS). 4055--4065.  Andre Barreto Will Dabney Remi Munos Jonathan J Hunt Tom Schaul Hado P van Hasselt and David Silver. 2017. Successor Features for Transfer in Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS). 4055--4065."},{"key":"e_1_3_2_2_7_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Burda Yuri","year":"2019","unstructured":"Yuri Burda , Harrison Edwards , Amos J. Storkey , and Oleg Klimov . 2019 . Exploration by random network distillation . In International Conference on Learning Representations (ICLR). Yuri Burda, Harrison Edwards, Amos J. Storkey, and Oleg Klimov. 2019. Exploration by random network distillation. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.64"},{"key":"e_1_3_2_2_9_1","unstructured":"Nuttapong Chentanez Andrew Barto and Satinder Singh. 2004. Intrinsically Motivated Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS). 1281--1288.  Nuttapong Chentanez Andrew Barto and Satinder Singh. 2004. Intrinsically Motivated Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS). 1281--1288."},{"key":"e_1_3_2_2_10_1","volume-title":"Learning Neuro-Symbolic Relational Transition Models for Bilevel Planning. In IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS). 4166--4173","author":"Chitnis Rohan","year":"2022","unstructured":"Rohan Chitnis , Tom Silver , Joshua B. Tenenbaum , Tom\u00e1s Lozano-P\u00e9rez , and Leslie Pack Kaelbling . 2022 . Learning Neuro-Symbolic Relational Transition Models for Bilevel Planning. In IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS). 4166--4173 . Rohan Chitnis, Tom Silver, Joshua B. Tenenbaum, Tom\u00e1s Lozano-P\u00e9rez, and Leslie Pack Kaelbling. 2022. Learning Neuro-Symbolic Relational Transition Models for Bilevel Planning. In IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS). 4166--4173."},{"key":"e_1_3_2_2_11_1","volume-title":"Fast Effective Rule Induction. In International Conference on Machine Learning (ICML). 115--123","author":"Cohen William W.","year":"1995","unstructured":"William W. Cohen . 1995 . Fast Effective Rule Induction. In International Conference on Machine Learning (ICML). 115--123 . William W. Cohen. 1995. Fast Effective Rule Induction. In International Conference on Machine Learning (ICML). 115--123."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Peter Dayan. 1993. Improving Generalization for Temporal Difference Learning: The Successor Representation. In Advances in Neural Information Processing Systems (NeurIPS). 613--624.  Peter Dayan. 1993. Improving Generalization for Temporal Difference Learning: The Successor Representation. In Advances in Neural Information Processing Systems (NeurIPS). 613--624.","DOI":"10.1162\/neco.1993.5.4.613"},{"key":"e_1_3_2_2_13_1","unstructured":"Matt Deitke Eli VanderBilt Alvaro Herrasti Luca Weihs Kiana Ehsani Jordi Salvador Winson Han Eric Kolve Aniruddha Kembhavi and Roozbeh Mottaghi. 2022. ProcTHOR: Large-Scale Embodied AI Using Procedural Generation. In Advances in Neural Information Processing Systems (NeurIPS). 5982--5994.  Matt Deitke Eli VanderBilt Alvaro Herrasti Luca Weihs Kiana Ehsani Jordi Salvador Winson Han Eric Kolve Aniruddha Kembhavi and Roozbeh Mottaghi. 2022. ProcTHOR: Large-Scale Embodied AI Using Procedural Generation. In Advances in Neural Information Processing Systems (NeurIPS). 5982--5994."},{"key":"e_1_3_2_2_14_1","unstructured":"Linxi Fan Guanzhi Wang Yunfan Jiang Ajay Mandlekar Yuncong Yang Haoyi Zhu Andrew Tang De-An Huang Yuke Zhu and Anima Anandkumar. 2022. MineDojo: Building Open-Ended Embodied Agents with Internet-Scale Knowledge. In Advances in Neural Information Processing Systems (NeurIPS).  Linxi Fan Guanzhi Wang Yunfan Jiang Ajay Mandlekar Yuncong Yang Haoyi Zhu Andrew Tang De-An Huang Yuke Zhu and Anima Anandkumar. 2022. MineDojo: Building Open-Ended Embodied Agents with Internet-Scale Knowledge. In Advances in Neural Information Processing Systems (NeurIPS)."},{"volume-title":"International Conference on Autonomous Agents and Multiagent Systems (AAMAS). 720--727","author":"Fern\u00e1ndez Fernando","key":"e_1_3_2_2_15_1","unstructured":"Fernando Fern\u00e1ndez and Manuela M. Veloso . 2006. Probabilistic policy reuse in a reinforcement learning agent . In International Conference on Autonomous Agents and Multiagent Systems (AAMAS). 720--727 . Fernando Fern\u00e1ndez and Manuela M. Veloso. 2006. Probabilistic policy reuse in a reinforcement learning agent. In International Conference on Autonomous Agents and Multiagent Systems (AAMAS). 720--727."},{"key":"e_1_3_2_2_16_1","volume-title":"PathNet: Evolution Channels Gradient Descent in Super Neural Networks. arXiv preprint arXiv:1701.08734","author":"Fernando Chrisantha","year":"2017","unstructured":"Chrisantha Fernando , Dylan Banarse , Charles Blundell , Yori Zwols , David Ha , Andrei A. Rusu , Alexander Pritzel , and Daan Wierstra . 2017. PathNet: Evolution Channels Gradient Descent in Super Neural Networks. arXiv preprint arXiv:1701.08734 ( 2017 ). Chrisantha Fernando, Dylan Banarse, Charles Blundell, Yori Zwols, David Ha, Andrei A. Rusu, Alexander Pritzel, and Daan Wierstra. 2017. PathNet: Evolution Channels Gradient Descent in Super Neural Networks. arXiv preprint arXiv:1701.08734 (2017)."},{"volume-title":"Generating Accurate Rule Sets Without Global Optimization. In International Conference on Machine Learning (ICML). 144--151","author":"Frank Eibe","key":"e_1_3_2_2_17_1","unstructured":"Eibe Frank and Ian H. Witten . 1998 . Generating Accurate Rule Sets Without Global Optimization. In International Conference on Machine Learning (ICML). 144--151 . Eibe Frank and Ian H. Witten. 1998. Generating Accurate Rule Sets Without Global Optimization. In International Conference on Machine Learning (ICML). 144--151."},{"volume-title":"Foundations of Rule Learning","author":"F\u00fcrnkranz Johannes","key":"e_1_3_2_2_18_1","unstructured":"Johannes F\u00fcrnkranz , Dragan Gamberger , and Nada Lavrac . 2012. Foundations of Rule Learning . Springer . Johannes F\u00fcrnkranz, Dragan Gamberger, and Nada Lavrac. 2012. Foundations of Rule Learning. Springer."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF00962234"},{"key":"e_1_3_2_2_20_1","volume-title":"Witten","author":"Hall Mark A.","year":"2009","unstructured":"Mark A. Hall , Eibe Frank , Geoffrey Holmes , Bernhard Pfahringer , Peter Reutemann , and Ian H . Witten . 2009 . The WEKA data mining software: an update. ACM SIGKDD explorations newsletter, Vol. 11 , 1 (2009), 10--18. Mark A. Hall, Eibe Frank, Geoffrey Holmes, Bernhard Pfahringer, Peter Reutemann, and Ian H. Witten. 2009. The WEKA data mining software: an update. ACM SIGKDD explorations newsletter, Vol. 11, 1 (2009), 10--18."},{"key":"e_1_3_2_2_21_1","volume-title":"Rainbow: Combining Improvements in Deep Reinforcement Learning. In AAAI Conference on Artificial Intelligence (AAAI). 3215--3222","author":"Hessel Matteo","year":"2018","unstructured":"Matteo Hessel , Joseph Modayil , Hado van Hasselt , Tom Schaul , Georg Ostrovski , Will Dabney , Dan Horgan , Bilal Piot , Mohammad Gheshlaghi Azar , and David Silver . 2018 . Rainbow: Combining Improvements in Deep Reinforcement Learning. In AAAI Conference on Artificial Intelligence (AAAI). 3215--3222 . Matteo Hessel, Joseph Modayil, Hado van Hasselt, Tom Schaul, Georg Ostrovski, Will Dabney, Dan Horgan, Bilal Piot, Mohammad Gheshlaghi Azar, and David Silver. 2018. Rainbow: Combining Improvements in Deep Reinforcement Learning. In AAAI Conference on Artificial Intelligence (AAAI). 3215--3222."},{"key":"e_1_3_2_2_22_1","volume-title":"Distilling the Knowledge in a Neural Network. arXiv preprint arXiv:1503.02531","author":"Hinton Geoffrey E.","year":"2015","unstructured":"Geoffrey E. Hinton , Oriol Vinyals , and Jeffrey Dean . 2015. Distilling the Knowledge in a Neural Network. arXiv preprint arXiv:1503.02531 ( 2015 ). Geoffrey E. Hinton, Oriol Vinyals, and Jeffrey Dean. 2015. Distilling the Knowledge in a Neural Network. arXiv preprint arXiv:1503.02531 (2015)."},{"key":"e_1_3_2_2_23_1","volume-title":"Filip De Turck, and Pieter Abbeel","author":"Houthooft Rein","year":"2016","unstructured":"Rein Houthooft , Xi Chen , Yan Duan , John Schulman , Filip De Turck, and Pieter Abbeel . 2016 . VIME : Variational Information Maximizing Exploration. In Advances in Neural Information Processing Systems (NeurIPS) . 1109--1117. Rein Houthooft, Xi Chen, Yan Duan, John Schulman, Filip De Turck, and Pieter Abbeel. 2016. VIME: Variational Information Maximizing Exploration. In Advances in Neural Information Processing Systems (NeurIPS). 1109--1117."},{"key":"e_1_3_2_2_24_1","volume-title":"Generalizable Episodic Memory for Deep Reinforcement Learning. In International Conference on Machine Learning (ICML). 4380--4390","author":"Hu Hao","year":"2021","unstructured":"Hao Hu , Jianing Ye , Guangxiang Zhu , Zhizhou Ren , and Chongjie Zhang . 2021 . Generalizable Episodic Memory for Deep Reinforcement Learning. In International Conference on Machine Learning (ICML). 4380--4390 . Hao Hu, Jianing Ye, Guangxiang Zhu, Zhizhou Ren, and Chongjie Zhang. 2021. Generalizable Episodic Memory for Deep Reinforcement Learning. In International Conference on Machine Learning (ICML). 4380--4390."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622737.1622748"},{"key":"e_1_3_2_2_26_1","volume-title":"Multi-View Dreaming: Multi-View World Model with Contrastive Learning. arXiv preprint arXiv:2203.11024","author":"Kinose Akira","year":"2022","unstructured":"Akira Kinose , Masashi Okada , Ryo Okumura , and Tadahiro Taniguchi . 2022. Multi-View Dreaming: Multi-View World Model with Contrastive Learning. arXiv preprint arXiv:2203.11024 ( 2022 ). Akira Kinose, Masashi Okada, Ryo Okumura, and Tadahiro Taniguchi. 2022. Multi-View Dreaming: Multi-View World Model with Contrastive Learning. arXiv preprint arXiv:2203.11024 (2022)."},{"key":"e_1_3_2_2_27_1","volume-title":"Contrastive Representation Learning: A Framework and Review. IEEE Aerospace Conference","volume":"8","author":"Le-Khac Phuc H.","year":"2020","unstructured":"Phuc H. Le-Khac , Graham Healy , and Alan F. Smeaton . 2020 . Contrastive Representation Learning: A Framework and Review. IEEE Aerospace Conference , Vol. 8 ( 2020 ), 193907--193934. Phuc H. Le-Khac, Graham Healy, and Alan F. Smeaton. 2020. Contrastive Representation Learning: A Framework and Review. IEEE Aerospace Conference, Vol. 8 (2020), 193907--193934."},{"key":"e_1_3_2_2_28_1","volume-title":"SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning. In International Conference on Machine Learning (ICML). 6131--6141","author":"Lee Kimin","year":"2021","unstructured":"Kimin Lee , Michael Laskin , Aravind Srinivas , and Pieter Abbeel . 2021 . SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning. In International Conference on Machine Learning (ICML). 6131--6141 . Kimin Lee, Michael Laskin, Aravind Srinivas, and Pieter Abbeel. 2021. SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning. In International Conference on Machine Learning (ICML). 6131--6141."},{"key":"e_1_3_2_2_29_1","unstructured":"Minne Li Lisheng Wu Jun WANG and Haitham Bou Ammar. 2019. Multi-View Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS). 1418--1429.  Minne Li Lisheng Wu Jun WANG and Haitham Bou Ammar. 2019. Multi-View Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS). 1418--1429."},{"key":"e_1_3_2_2_30_1","volume-title":"SDRL: Interpretable and Data-Efficient Deep Reinforcement Learning Leveraging Symbolic Planning. In AAAI Conference on Artificial Intelligence (AAAI). 2970--2977","author":"Lyu Daoming","year":"2019","unstructured":"Daoming Lyu , Fangkai Yang , Bo Liu , and Steven Gustafson . 2019 . SDRL: Interpretable and Data-Efficient Deep Reinforcement Learning Leveraging Symbolic Planning. In AAAI Conference on Artificial Intelligence (AAAI). 2970--2977 . Daoming Lyu, Fangkai Yang, Bo Liu, and Steven Gustafson. 2019. SDRL: Interpretable and Data-Efficient Deep Reinforcement Learning Leveraging Symbolic Planning. In AAAI Conference on Artificial Intelligence (AAAI). 2970--2977."},{"key":"e_1_3_2_2_31_1","volume-title":"Nature","volume":"518","author":"Mnih Volodymyr","year":"2015","unstructured":"Volodymyr Mnih , Koray Kavukcuoglu , David Silver , Andrei A. Rusu , Joel Veness , Marc G. Bellemare , Alex Graves , Martin A. Riedmiller , Andreas Fidjeland , Georg Ostrovski , Stig Petersen , Charles Beattie , Amir Sadik , Ioannis Antonoglou , Helen King , Dharshan Kumaran , Daan Wierstra , Shane Legg , and Demis Hassabis . 2015 . Human-level Control Through Deep Reinforcement Learning . Nature , Vol. 518 , 7540 (2015), 529--533. Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Andrei A. Rusu, Joel Veness, Marc G. Bellemare, Alex Graves, Martin A. Riedmiller, Andreas Fidjeland, Georg Ostrovski, Stig Petersen, Charles Beattie, Amir Sadik, Ioannis Antonoglou, Helen King, Dharshan Kumaran, Daan Wierstra, Shane Legg, and Demis Hassabis. 2015. Human-level Control Through Deep Reinforcement Learning. Nature, Vol. 518, 7540 (2015), 529--533."},{"key":"e_1_3_2_2_32_1","volume-title":"Self-Imitation Learning. In International Conference on Machine Learning (ICML). 3878--3887","author":"Oh Junhyuk","year":"2018","unstructured":"Junhyuk Oh , Yijie Guo , Satinder Singh , and Honglak Lee . 2018 . Self-Imitation Learning. In International Conference on Machine Learning (ICML). 3878--3887 . Junhyuk Oh, Yijie Guo, Satinder Singh, and Honglak Lee. 2018. Self-Imitation Learning. In International Conference on Machine Learning (ICML). 3878--3887."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2019.2941685"},{"key":"e_1_3_2_2_34_1","volume-title":"Solving Rubik's Cube with a Robot Hand. arXiv preprint arXiv:1910.07113","author":"Ilge Akkaya AI","year":"2019","unstructured":"Open AI , Ilge Akkaya , Marcin Andrychowicz , Maciek Chociej , Mateusz Litwin , Bob McGrew , Arthur Petron , Alex Paino , Matthias Plappert , Glenn Powell , Raphael Ribas , Jonas Schneider , Nikolas Tezak , Jerry Tworek , Peter Welinder , Lilian Weng , Qiming Yuan , Wojciech Zaremba , and Lei Zhang . 2019. Solving Rubik's Cube with a Robot Hand. arXiv preprint arXiv:1910.07113 ( 2019 ). OpenAI, Ilge Akkaya, Marcin Andrychowicz, Maciek Chociej, Mateusz Litwin, Bob McGrew, Arthur Petron, Alex Paino, Matthias Plappert, Glenn Powell, Raphael Ribas, Jonas Schneider, Nikolas Tezak, Jerry Tworek, Peter Welinder, Lilian Weng, Qiming Yuan, Wojciech Zaremba, and Lei Zhang. 2019. Solving Rubik's Cube with a Robot Hand. arXiv preprint arXiv:1910.07113 (2019)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01183"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453160"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF00117105"},{"key":"e_1_3_2_2_39_1","first-page":"1","article-title":"Stable-Baselines3: Reliable Reinforcement Learning Implementations","volume":"22","author":"Raffin Antonin","year":"2021","unstructured":"Antonin Raffin , Ashley Hill , Adam Gleave , Anssi Kanervisto , Maximilian Ernestus , and Noah Dormann . 2021 . Stable-Baselines3: Reliable Reinforcement Learning Implementations . Journal of Machine Learning Research , Vol. 22 , 268 (2021), 1 -- 8 . Antonin Raffin, Ashley Hill, Adam Gleave, Anssi Kanervisto, Maximilian Ernestus, and Noah Dormann. 2021. Stable-Baselines3: Reliable Reinforcement Learning Implementations. Journal of Machine Learning Research, Vol. 22, 268 (2021), 1--8.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_2_40_1","volume-title":"Caglar Gulcehre, Guillaume Desjardins, James Kirkpatrick, Razvan Pascanu, Volodymyr Mnih, Koray Kavukcuoglu, and Raia Hadsell.","author":"Rusu Andrei A","year":"2015","unstructured":"Andrei A Rusu , Sergio Gomez Colmenarejo , Caglar Gulcehre, Guillaume Desjardins, James Kirkpatrick, Razvan Pascanu, Volodymyr Mnih, Koray Kavukcuoglu, and Raia Hadsell. 2015 . Policy distillation. arXiv preprint arXiv:1511.06295 (2015). Andrei A Rusu, Sergio Gomez Colmenarejo, Caglar Gulcehre, Guillaume Desjardins, James Kirkpatrick, Razvan Pascanu, Volodymyr Mnih, Koray Kavukcuoglu, and Raia Hadsell. 2015. Policy distillation. arXiv preprint arXiv:1511.06295 (2015)."},{"key":"e_1_3_2_2_41_1","volume-title":"Progressive Neural Networks. arXiv preprint arXiv:1606.04671","author":"Rusu Andrei A.","year":"2016","unstructured":"Andrei A. Rusu , Neil C. Rabinowitz , Guillaume Desjardins , Hubert Soyer , James Kirkpatrick , Koray Kavukcuoglu , Razvan Pascanu , and Raia Hadsell . 2016. Progressive Neural Networks. arXiv preprint arXiv:1606.04671 ( 2016 ). Andrei A. Rusu, Neil C. Rabinowitz, Guillaume Desjardins, Hubert Soyer, James Kirkpatrick, Koray Kavukcuoglu, Razvan Pascanu, and Raia Hadsell. 2016. Progressive Neural Networks. arXiv preprint arXiv:1606.04671 (2016)."},{"key":"e_1_3_2_2_42_1","volume-title":"Universal Value Function Approximators. In International Conference on Machine Learning (ICML). 1312--1320","author":"Schaul Tom","year":"2015","unstructured":"Tom Schaul , Daniel Horgan , Karol Gregor , and David Silver . 2015 . Universal Value Function Approximators. In International Conference on Machine Learning (ICML). 1312--1320 . Tom Schaul, Daniel Horgan, Karol Gregor, and David Silver. 2015. Universal Value Function Approximators. In International Conference on Machine Learning (ICML). 1312--1320."},{"key":"e_1_3_2_2_43_1","volume-title":"Prioritized Experience Replay. In International Conference on Learning Representations (ICLR).","author":"Schaul Tom","year":"2016","unstructured":"Tom Schaul , John Quan , Ioannis Antonoglou , and David Silver . 2016 . Prioritized Experience Replay. In International Conference on Learning Representations (ICLR). Tom Schaul, John Quan, Ioannis Antonoglou, and David Silver. 2016. Prioritized Experience Replay. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_44_1","volume-title":"Mastering the Game of Go with Deep Neural Networks and Tree Search. Nature","author":"Silver David","year":"2016","unstructured":"David Silver , Aja Huang , Chris J. Maddison , Arthur Guez , Laurent Sifre , George van den Driessche , Julian Schrittwieser , Ioannis Antonoglou , Vedavyas Panneershelvam , Marc Lanctot , Sander Dieleman , Dominik Grewe , John Nham , Nal Kalchbrenner , Ilya Sutskever , Timothy P. Lillicrap , Madeleine Leach , Koray Kavukcuoglu , Thore Graepel , and Demis Hassabis . 2016. Mastering the Game of Go with Deep Neural Networks and Tree Search. Nature , Vol. 529 , 7587 ( 2016 ), 484--489. David Silver, Aja Huang, Chris J. Maddison, Arthur Guez, Laurent Sifre, George van den Driessche, Julian Schrittwieser, Ioannis Antonoglou, Vedavyas Panneershelvam, Marc Lanctot, Sander Dieleman, Dominik Grewe, John Nham, Nal Kalchbrenner, Ilya Sutskever, Timothy P. Lillicrap, Madeleine Leach, Koray Kavukcuoglu, Thore Graepel, and Demis Hassabis. 2016. Mastering the Game of Go with Deep Neural Networks and Tree Search. Nature, Vol. 529, 7587 (2016), 484--489."},{"key":"e_1_3_2_2_45_1","unstructured":"Robert L Solso M Kimberly MacLin and Otto H MacLin. 2005. Cognitive Psychology. Pearson Education New Zealand.  Robert L Solso M Kimberly MacLin and Otto H MacLin. 2005. Cognitive Psychology. Pearson Education New Zealand."},{"key":"e_1_3_2_2_46_1","volume-title":"The Distracting Control Suite - A Challenging Benchmark for Reinforcement Learning from Pixels. arXiv preprint arXiv:2101.02722","author":"Stone Austin","year":"2021","unstructured":"Austin Stone , Oscar Ramirez , Kurt Konolige , and Rico Jonschkowski . 2021. The Distracting Control Suite - A Challenging Benchmark for Reinforcement Learning from Pixels. arXiv preprint arXiv:2101.02722 ( 2021 ). Austin Stone, Oscar Ramirez, Kurt Konolige, and Rico Jonschkowski. 2021. The Distracting Control Suite - A Challenging Benchmark for Reinforcement Learning from Pixels. arXiv preprint arXiv:2101.02722 (2021)."},{"volume-title":"On Bonus Based Exploration Methods in The Arcade Learning Environment. In International Conference on Learning Representations (ICLR).","author":"Ta\u00efga Adrien Ali","key":"e_1_3_2_2_47_1","unstructured":"Adrien Ali Ta\u00efga , William Fedus , Marlos C. Machado , Aaron C. Courville , and Marc G. Bellemare . 2020 . On Bonus Based Exploration Methods in The Arcade Learning Environment. In International Conference on Learning Representations (ICLR). Adrien Ali Ta\u00efga, William Fedus, Marlos C. Machado, Aaron C. Courville, and Marc G. Bellemare. 2020. On Bonus Based Exploration Methods in The Arcade Learning Environment. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_48_1","unstructured":"Yunhao Tang. 2020. Self-Imitation Learning via Generalized Lower Bound Q-learning. In Advances in Neural Information Processing Systems (NeurIPS). 13964--13975.  Yunhao Tang. 2020. Self-Imitation Learning via Generalized Lower Bound Q-learning. In Advances in Neural Information Processing Systems (NeurIPS). 13964--13975."},{"key":"e_1_3_2_2_49_1","volume-title":"Survey on Deep Multi-modal Data Analytics: Collaboration, Rivalry and Fusion. arXiv preprint arXiv:2006.08159","author":"Wang Yang","year":"2020","unstructured":"Yang Wang . 2020. Survey on Deep Multi-modal Data Analytics: Collaboration, Rivalry and Fusion. arXiv preprint arXiv:2006.08159 ( 2020 ). Yang Wang. 2020. Survey on Deep Multi-modal Data Analytics: Collaboration, Rivalry and Fusion. arXiv preprint arXiv:2006.08159 (2020)."},{"key":"e_1_3_2_2_50_1","volume-title":"Generalized Linear Rule Models. In International Conference on Machine Learning (ICML). 6687--6696","author":"Wei Dennis","year":"2019","unstructured":"Dennis Wei , Sanjeeb Dash , Tian Gao , and Oktay Gunluk . 2019 . Generalized Linear Rule Models. In International Conference on Machine Learning (ICML). 6687--6696 . Dennis Wei, Sanjeeb Dash, Tian Gao, and Oktay Gunluk. 2019. Generalized Linear Rule Models. In International Conference on Machine Learning (ICML). 6687--6696."},{"key":"e_1_3_2_2_51_1","volume-title":"Deep Embedded Complementary and Interactive Information for Multi-View Classification. In AAAI Conference on Artificial Intelligence (AAAI). 6494--6501","author":"Xu Jinglin","year":"2020","unstructured":"Jinglin Xu , Wenbin Li , Xinwang Liu , Dingwen Zhang , Ji Liu , and Junwei Han . 2020 . Deep Embedded Complementary and Interactive Information for Multi-View Classification. In AAAI Conference on Artificial Intelligence (AAAI). 6494--6501 . Jinglin Xu, Wenbin Li, Xinwang Liu, Dingwen Zhang, Ji Liu, and Junwei Han. 2020. Deep Embedded Complementary and Interactive Information for Multi-View Classification. In AAAI Conference on Artificial Intelligence (AAAI). 6494--6501."},{"key":"e_1_3_2_2_52_1","unstructured":"Deheng Ye Guibin Chen Wen Zhang Sheng Chen Bo Yuan Bo Liu Jia Chen Zhao Liu Fuhao Qiu Hongsheng Yu Yinyuting Yin Bei Shi Liang Wang Tengfei Shi Qiang Fu Wei Yang Lanxiao Huang and Wei Liu. 2020. Towards Playing Full MOBA Games with Deep Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS). 621--632.  Deheng Ye Guibin Chen Wen Zhang Sheng Chen Bo Yuan Bo Liu Jia Chen Zhao Liu Fuhao Qiu Hongsheng Yu Yinyuting Yin Bei Shi Liang Wang Tengfei Shi Qiang Fu Wei Yang Lanxiao Huang and Wei Liu. 2020. Towards Playing Full MOBA Games with Deep Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS). 621--632."},{"key":"e_1_3_2_2_53_1","volume-title":"Knowledge Transfer for Deep Reinforcement Learning with Hierarchical Experience Replay. In AAAI Conference on Artificial Intelligence (AAAI). 1640--1646","author":"Yin Haiyan","year":"2017","unstructured":"Haiyan Yin and Sinno Jialin Pan . 2017 . Knowledge Transfer for Deep Reinforcement Learning with Hierarchical Experience Replay. In AAAI Conference on Artificial Intelligence (AAAI). 1640--1646 . Haiyan Yin and Sinno Jialin Pan. 2017. Knowledge Transfer for Deep Reinforcement Learning with Hierarchical Experience Replay. In AAAI Conference on Artificial Intelligence (AAAI). 1640--1646."},{"key":"e_1_3_2_2_54_1","volume-title":"Towards Sample Efficient Reinforcement Learning. In International Joint Conference on Artificial Intelligence (IJCAI). 5739--5743","author":"Yu Yang","year":"2018","unstructured":"Yang Yu . 2018 . Towards Sample Efficient Reinforcement Learning. In International Joint Conference on Artificial Intelligence (IJCAI). 5739--5743 . Yang Yu. 2018. Towards Sample Efficient Reinforcement Learning. In International Joint Conference on Artificial Intelligence (IJCAI). 5739--5743."},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2983149"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-018-9801-4"},{"key":"e_1_3_2_2_57_1","volume-title":"Transfer Learning in Deep Reinforcement Learning: A Survey. arXiv preprint arXiv:2009.07888","author":"Zhu Zhuangdi","year":"2020","unstructured":"Zhuangdi Zhu , Kaixiang Lin , and Jiayu Zhou . 2020. Transfer Learning in Deep Reinforcement Learning: A Survey. arXiv preprint arXiv:2009.07888 ( 2020 ). Zhuangdi Zhu, Kaixiang Lin, and Jiayu Zhou. 2020. Transfer Learning in Deep Reinforcement Learning: A Survey. arXiv preprint arXiv:2009.07888 (2020)."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2021.3134702"}],"event":{"name":"KDD '23: The 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"],"location":"Long Beach CA USA","acronym":"KDD '23"},"container-title":["Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3580305.3599393","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3580305.3599393","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:37:36Z","timestamp":1750178256000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3580305.3599393"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,4]]},"references-count":58,"alternative-id":["10.1145\/3580305.3599393","10.1145\/3580305"],"URL":"https:\/\/doi.org\/10.1145\/3580305.3599393","relation":{},"subject":[],"published":{"date-parts":[[2023,8,4]]},"assertion":[{"value":"2023-08-04","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}