{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T17:12:21Z","timestamp":1761930741514,"version":"build-2065373602"},"reference-count":28,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,5,1]],"date-time":"2024-05-01T00:00:00Z","timestamp":1714521600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,1]],"date-time":"2024-05-01T00:00:00Z","timestamp":1714521600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,5,1]]},"DOI":"10.1109\/iwcit62550.2024.10552959","type":"proceedings-article","created":{"date-parts":[[2024,6,13]],"date-time":"2024-06-13T13:40:07Z","timestamp":1718286007000},"page":"1-6","source":"Crossref","is-referenced-by-count":0,"title":["Deep ExRL: Experience-Driven Deep Reinforcement Learning in Control Problems"],"prefix":"10.1109","author":[{"given":"Ali","family":"Ghandi","sequence":"first","affiliation":[{"name":"Sharif University of Technology,EE,Tehran,Iran"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Saeed Bagheri","family":"Shouraki","sequence":"additional","affiliation":[{"name":"Sharif University of Technology,EE,Tehran,Iran"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mahyar","family":"Riazati","sequence":"additional","affiliation":[{"name":"University of Tehran,ECE,Tehran,Iran"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2023.104399"},{"journal-title":"Successor Features for Transfer in Reinforcement Learning","year":"2016","author":"Barreto","key":"ref2"},{"journal-title":"Transfer in Deep Reinforcement Learning Using Successor Features and Generalised Policy Improvement","year":"2019","author":"Barreto","key":"ref3"},{"journal-title":"A Survey of Meta-Reinforcement Learning","year":"2023","author":"Beck","key":"ref4"},{"journal-title":"OpenAI Gym","year":"2016","author":"Brockman","key":"ref5"},{"journal-title":"Imagined Value Gradients: Model-Based Policy Optimization with Transferable Latent Dynamics Models","year":"2019","author":"Byravan","key":"ref6"},{"volume-title":"Deep Reinforcement Learning: Fundamentals, Research and Applications","year":"2020","key":"ref7"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/0167-8655(85)90023-6"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-74048-3_4"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/S0959-440X(96)80056-X"},{"journal-title":"Model-Agnostic Meta-Learning for Fast Adaptation of Deep Networks","year":"2017","author":"Finn","key":"ref11"},{"journal-title":"Learning Invariant Feature Spaces to Transfer Skills with Reinforcement Learning","year":"2017","author":"Gupta","key":"ref12"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v29i1.9628"},{"journal-title":"Domain Adaptive Imitation Learning","year":"2019","author":"Kim","key":"ref14"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2020.3048361"},{"journal-title":"Asynchronous Methods for Deep Reinforcement Learning","year":"2016","author":"Mnih","key":"ref16"},{"journal-title":"Skill-based Meta-Reinforcement Learning","year":"2022","author":"Nam","key":"ref17"},{"journal-title":"Deep Exploration via Randomized Value Functions","year":"2017","author":"Osband","key":"ref18"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/s11431-020-1647-3"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/s13253-021-00483-x"},{"journal-title":"Proximal Policy Optimization Algorithms","year":"2017","author":"Schulman","key":"ref21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45622-8_16"},{"article-title":"Model transfer for Markov decision tasks via parameter matching","volume-title":"Proceedings of the 25th Workshop of the UK Planning and Scheduling Special Interest Group (PlanSIG 2006)","author":"Sunmola","key":"ref23"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/tnn.1998.712192"},{"journal-title":"Distral: Robust Multitask Reinforcement Learning","year":"2017","author":"Whye Teh","key":"ref25"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.petrol.2022.110868"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9412011"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3292075"}],"event":{"name":"2024 12th Iran Workshop on Communication and Information Theory (IWCIT)","start":{"date-parts":[[2024,5,1]]},"location":"Tehran, Iran, Islamic Republic of","end":{"date-parts":[[2024,5,2]]}},"container-title":["2024 12th Iran Workshop on Communication and Information Theory (IWCIT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10552865\/10552958\/10552959.pdf?arnumber=10552959","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T17:07:21Z","timestamp":1761930441000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10552959\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,1]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/iwcit62550.2024.10552959","relation":{},"subject":[],"published":{"date-parts":[[2024,5,1]]}}}