{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:25:39Z","timestamp":1740122739106,"version":"3.37.3"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,12,24]],"date-time":"2024-12-24T00:00:00Z","timestamp":1734998400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,24]],"date-time":"2024-12-24T00:00:00Z","timestamp":1734998400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100011033","name":"Agencia Estatal de Investigaci\u00f3n","doi-asserted-by":"publisher","award":["PID2020-119367RB-I00","PID2023-153341OB-I00"],"award-info":[{"award-number":["PID2020-119367RB-I00","PID2023-153341OB-I00"]}],"id":[{"id":"10.13039\/501100011033","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,2]]},"DOI":"10.1007\/s10489-024-06190-7","type":"journal-article","created":{"date-parts":[[2024,12,24]],"date-time":"2024-12-24T05:13:03Z","timestamp":1735017183000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Learning state-action correspondence across reinforcement learning control tasks via partially paired trajectories"],"prefix":"10.1007","volume":"55","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5638-5240","authenticated-orcid":false,"given":"Javier","family":"Garc\u00eda","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"I\u00f1aki","family":"Ra\u00f1\u00f3","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"J. Miguel","family":"Bur\u00e9s","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xos\u00e9 R.","family":"Fdez-Vidal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Roberto","family":"Iglesias","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,24]]},"reference":[{"key":"6190_CR1","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2011","unstructured":"Sutton RS, Barto AG (2011) Reinforcement Learning: An Introduction. MIT Press, Cambridge, MA"},{"key":"6190_CR2","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Graves A, Antonoglou I, Wierstra D, Riedmiller M (2023) Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602"},{"key":"6190_CR3","doi-asserted-by":"crossref","unstructured":"Silver D, Huang A, Maddison CJ, Guez A, Sifre L, Van Den\u00a0Driessche G, Schrittwieser J, Antonoglou I, Panneershelvam V, Lanctot M et al (2016) Mastering the game of go with deep neural networks and tree search. Nat 529(7587):484","DOI":"10.1038\/nature16961"},{"key":"6190_CR4","unstructured":"Sinha S, Mandlekar A, Garg A (2022) S4rl: Surprisingly simple self-supervision for offline reinforcement learning in robotics. In: Conference on Robot Learning, PMLR. pp 907\u2013917"},{"key":"6190_CR5","doi-asserted-by":"crossref","unstructured":"Taylor ME, Stone P (2009) Transfer learning for reinforcement learning domains: A survey. J Mach Learn Res 10(7)","DOI":"10.1007\/978-3-642-01882-4"},{"key":"6190_CR6","doi-asserted-by":"crossref","unstructured":"Lazaric A (2012) Transfer in reinforcement learning: a framework and a survey. Reinforcement Learning: State of the Art, 143\u2013173","DOI":"10.1007\/978-3-642-27645-3_5"},{"key":"6190_CR7","doi-asserted-by":"crossref","unstructured":"Fern\u00e1ndez F, Veloso M (2006) Probabilistic policy reuse in a reinforcement learning agent. In: Proceedings of the 5th International Joint Conference on Autonomous Agents and Multiagent Systems (AAMAS\u201906)","DOI":"10.1145\/1160633.1160762"},{"key":"6190_CR8","unstructured":"Zhang Q, Xiao T, Efros AA, Pinto L, Wang X (2020) Learning cross-domain correspondence for control with dynamics cycle-consistency. arXiv preprint arXiv:2012.09811"},{"key":"6190_CR9","unstructured":"You H, Yang T, Zheng Y, Hao J, E\u00a0Taylor M (2022) Cross-domain adaptive transfer reinforcement-learning based on state-action correspondence. In: Uncertainty in Artificial Intelligence, PMLR, pp 2299\u20132309"},{"key":"6190_CR10","unstructured":"Gupta A, Devin C, Liu Y, Abbeel P, Levine S (2017) Learning invariant feature spaces to transfer skills with reinforcement learning. arXiv preprint arXiv:1703.02949"},{"key":"6190_CR11","unstructured":"Taylor ME, Kuhlmann G, Stone P (2008) Autonomous transfer for reinforcement learning. In: Proceedings of the 7th International Joint Conference on Autonomous Agents and Multiagent Systems, ACM, pp 283\u2013290"},{"issue":"11","key":"6190_CR12","doi-asserted-by":"publisher","first-page":"4217","DOI":"10.1007\/s10994-022-06242-4","volume":"111","author":"J Garc\u00eda","year":"2022","unstructured":"Garc\u00eda J, Vis\u00fas \u00c1, Fern\u00e1ndez F (2022) A taxonomy for similarity metrics between markov decision processes. Mach Learn 111(11):4217\u20134247","journal-title":"Mach Learn"},{"key":"6190_CR13","unstructured":"Wan M, Gangwani T, Peng J (2020) Mutual information based knowledge transfer under state-action dimension mismatch. arXiv preprint arXiv:2006.07041"},{"issue":"1","key":"6190_CR14","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1007\/s13748-012-0026-6","volume":"2","author":"F Fern\u00e1ndez","year":"2013","unstructured":"Fern\u00e1ndez F, Veloso M (2013) Learning domain structure through probabilistic policy reuse in reinforcement learning. Prog Artif Intell 2(1):13\u201327","journal-title":"Prog Artif Intell"},{"key":"6190_CR15","unstructured":"Gamrian S, Goldberg Y (2019) Transfer learning for related reinforcement learning tasks via image-to-image translation. In: International Conference on Machine Learning, PMLR, pp 2063\u20132072"},{"key":"6190_CR16","unstructured":"Watkins C (1989) Learning from delayed rewards. PhD thesis, King\u2019s College, Cambridge, UK"},{"issue":"5","key":"6190_CR17","doi-asserted-by":"publisher","first-page":"1636","DOI":"10.1287\/opre.2022.2396","volume":"71","author":"SR Sinclair","year":"2023","unstructured":"Sinclair SR, Banerjee S, Yu CL (2023) Adaptive discretization in online reinforcement learning. Oper Res 71(5):1636\u20131652","journal-title":"Oper Res"},{"key":"6190_CR18","unstructured":"Reinforcement Learning (2014) State-of-the-Art. In: Wiering M, Van\u00a0Otterlo M (eds) Adaptation, Learning, and Optimization, vol 12. Springer, Berlin, Germany"},{"issue":"10","key":"6190_CR19","first-page":"4100","volume":"32","author":"F Zhuang","year":"2021","unstructured":"Zhuang F, Qi Z, Duan K, Xi D, Zhu Y, Zhu H, Xiong H, He Q (2021) A comprehensive survey on transfer learning. IEEE Trans Neural Netw Learn Syst 32(10):4100\u20134122","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"6190_CR20","doi-asserted-by":"crossref","unstructured":"Fern\u00e1ndez D, Fern\u00e1ndez F, Garc\u00eda J (2021) Probabilistic multi-knowledge transfer in reinforcement learning. In: 2021 20th IEEE International Conference on Machine Learning and Applications (ICMLA), IEEE, pp 471\u2013476","DOI":"10.1109\/ICMLA52953.2021.00079"},{"key":"6190_CR21","doi-asserted-by":"crossref","unstructured":"Torrey L, Walker T, Shavlik J, Maclin R (2005) Using advice to transfer knowledge acquired in one reinforcement learning task to another. In: Machine Learning: ECML 2005: 16th European Conference on Machine Learning. Proceedings 16, Springer, Porto, Portugal, 3-7 Oct 2005. pp 412\u2013424","DOI":"10.1007\/11564096_40"},{"issue":"1","key":"6190_CR22","first-page":"2125","volume":"8","author":"ME Taylor","year":"2007","unstructured":"Taylor ME, Stone P, Liu Y (2007) Transfer learning via inter-task mappings for temporal difference learning. J Mach Learn Res 8(1):2125\u20132167","journal-title":"J Mach Learn Res"},{"issue":"7","key":"6190_CR23","doi-asserted-by":"publisher","first-page":"866","DOI":"10.1016\/j.robot.2010.03.007","volume":"58","author":"F Fern\u00e1ndez","year":"2010","unstructured":"Fern\u00e1ndez F, Garc\u00eda J, Veloso M (2010) Probabilistic policy reuse for inter-task transfer learning. Robot Auton Syst 58(7):866\u2013871","journal-title":"Robot Auton Syst"},{"key":"6190_CR24","doi-asserted-by":"crossref","unstructured":"Ammar HB, Taylor ME (2012) Reinforcement learning transfer via common subspaces. In: Adaptive and Learning Agents: International Workshop, ALA 2011, Held at AAMAS 2011, Taipei, Taiwan, May 2, 2011, Revised Selected Papers, Springer, pp 21\u201336","DOI":"10.1007\/978-3-642-28499-1_2"},{"key":"6190_CR25","doi-asserted-by":"crossref","unstructured":"Sun, Y., Yin, X., Huang, F.: Temple: Learning template of transitions for sample efficient multi-task rl. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 9765\u20139773 (2021)","DOI":"10.1609\/aaai.v35i11.17174"},{"key":"6190_CR26","unstructured":"Sun Y, Zheng R, Wang X, Cohen A, Huang F (2022) Transfer rl across observation feature spaces via model-based regularization. arXiv preprint arXiv:2201.00248"},{"key":"6190_CR27","unstructured":"Chen Y, Chen Y, Hu Z, Yang T, Fan C, Yu Y, Hao J (2019) Learning action-transferable policy with action embedding. arXiv preprint arXiv:1909.02291"},{"key":"6190_CR28","unstructured":"Raiman J, Zhang S, Dennison C (2019) Neural network surgery with sets. arXiv preprint arXiv:1912.06719"},{"key":"6190_CR29","unstructured":"Buljan M, Canal O, Taschin F (2021) Neural Network Surgery in Deep Reinforcement Learning. Accessed 10 Dec 2024. https:\/\/campusai.github.io\/pdf\/nn-surgery-report.pdf"},{"key":"6190_CR30","doi-asserted-by":"crossref","unstructured":"Sermanet P, Lynch C, Chebotar Y, Hsu J, Jang E, Schaal S, Levine S, Brain G (2018) Time-contrastive networks: Self-supervised learning from video. In: 2018 IEEE International Conference on Robotics and Automation (ICRA), IEEE, pp 1134\u20131141","DOI":"10.1109\/ICRA.2018.8462891"},{"issue":"4","key":"6190_CR31","doi-asserted-by":"publisher","first-page":"160","DOI":"10.1145\/122344.122377","volume":"2","author":"RS Sutton","year":"1991","unstructured":"Sutton RS (1991) Dyna, an integrated architecture for learning, planning, and reacting. ACM Sigart Bull 2(4):160\u2013163","journal-title":"ACM Sigart Bull"},{"key":"6190_CR32","doi-asserted-by":"crossref","unstructured":"Wu G, Fang W, Wang J, Ge P, Cao J, Ping Y, Gou P (2022) Dyna-ppo reinforcement learning with gaussian process for the continuous action decision-making in autonomous driving. Appl Intell 1\u201315","DOI":"10.1007\/s10489-022-04354-x"},{"key":"6190_CR33","unstructured":"Brockman G, Cheung V, Pettersson L, Schneider J, Schulman J, Tang J, Zaremba W (2016) OpenAI Gym"},{"issue":"7540","key":"6190_CR34","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nat 518(7540):529","journal-title":"Nat"},{"key":"6190_CR35","unstructured":"Lillicrap TP, Hunt JJ, Pritzel A, Heess N, Erez T, Tassa Y, Silver D, Wierstra D (2015) Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971"},{"key":"6190_CR36","unstructured":"Barnett SA (2018) Convergence problems with generative adversarial networks (gans). arXiv preprint arXiv:1806.11382"},{"key":"6190_CR37","doi-asserted-by":"crossref","unstructured":"Da Silva FL, Costa AHR (2019) A survey on transfer learning for multiagent reinforcement learning systems. J Artif Intell Res 64:645\u2013703","DOI":"10.1613\/jair.1.11396"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-06190-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-06190-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-06190-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,30]],"date-time":"2025-01-30T16:05:06Z","timestamp":1738253106000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-06190-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,24]]},"references-count":37,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,2]]}},"alternative-id":["6190"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-06190-7","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2024,12,24]]},"assertion":[{"value":"11 December 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 December 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"On behalf of all authors, the corresponding author states there is no conflict of interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}}],"article-number":"219"}}