{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T05:44:04Z","timestamp":1757310244827,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":34,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819722617"},{"type":"electronic","value":"9789819722594"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-981-97-2259-4_17","type":"book-chapter","created":{"date-parts":[[2024,4,24]],"date-time":"2024-04-24T09:02:31Z","timestamp":1713949351000},"page":"223-234","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Meta-Reinforcement Learning Algorithm Based on\u00a0Reward and\u00a0Dynamic Inference"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-9195-7052","authenticated-orcid":false,"given":"Jinhao","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3008-1887","authenticated-orcid":false,"given":"Chunhong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8874-5466","authenticated-orcid":false,"given":"Zheng","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,4,25]]},"reference":[{"key":"17_CR1","doi-asserted-by":"publisher","unstructured":"Bellemare, M.G., et al.: Autonomous navigation of stratospheric balloons using reinforcement learning. Nature 588(7836), 77\u201382. https:\/\/doi.org\/10.1038\/s41586-020-2939-8. https:\/\/www.nature.com\/articles\/s41586-020-2939-8","DOI":"10.1038\/s41586-020-2939-8"},{"key":"17_CR2","doi-asserted-by":"publisher","unstructured":"Miki, T., Lee, J., Hwangbo, J., Wellhausen, L., Koltun, V., Hutter, M.: Learning robust perceptive locomotion for quadrupedal robots in the wild. Sci. Robot. 7(62), eabk2822. https:\/\/doi.org\/10.1126\/scirobotics.abk2822. https:\/\/www.science.org\/doi\/full\/10.1126\/scirobotics.abk2822","DOI":"10.1126\/scirobotics.abk2822"},{"key":"17_CR3","doi-asserted-by":"publisher","first-page":"e253","DOI":"10.1017\/S0140525X16001837","volume":"40","author":"BM Lake","year":"2017","unstructured":"Lake, B.M., Ullman, T.D., Tenenbaum, J.B., Gershman, S.J.: Building machines that learn and think like people. Behav. Brain Sci. 40, e253 (2017). https:\/\/doi.org\/10.1017\/S0140525X16001837","journal-title":"Behav. Brain Sci."},{"key":"17_CR4","unstructured":"Peng, M., Zhu, B., Jiao, J.: Linear representation meta-reinforcement learning for instant adaptation. arXiv arXiv:2101.04750v1 (2021)"},{"key":"17_CR5","doi-asserted-by":"publisher","unstructured":"Beck, J., et al.: A survey of meta-reinforcement learning. arXiv arXiv:2301.08028 (2023). https:\/\/doi.org\/10.48550\/arXiv.2301.08028","DOI":"10.48550\/arXiv.2301.08028"},{"key":"17_CR6","doi-asserted-by":"publisher","unstructured":"Imagawa, T., Hiraoka, T., Tsuruoka, Y.: Off-policy meta-reinforcement learning with belief-based task inference. IEEE Access 10, 49494\u201349507. https:\/\/doi.org\/10.1109\/ACCESS.2022.3170582. https:\/\/ieeexplore.ieee.org\/abstract\/document\/9763505","DOI":"10.1109\/ACCESS.2022.3170582"},{"key":"17_CR7","unstructured":"Wang, J.X., et al.: Learning to reinforcement learn. arXiv arXiv:1611.05763 (2017)"},{"key":"17_CR8","unstructured":"Melo, L.C.: Transformers are meta-reinforcement learners. arXiv arXiv:2206.06614 (2022)"},{"key":"17_CR9","unstructured":"Rakelly, K., Zhou, A., Quillen, D., Finn, C., Levine, S.: Efficient off-policy meta-reinforcement learning via probabilistic context variables, p.\u00a010 (2019)"},{"key":"17_CR10","doi-asserted-by":"publisher","unstructured":"Jiang, P., Song, S., Huang, G.: Exploration with task information for meta reinforcement learning. IEEE Trans. Neural Netw. Learn. Syst. 34(8), 4033\u20134046 (2023). https:\/\/doi.org\/10.1109\/TNNLS.2021.3121432. https:\/\/ieeexplore.ieee.org\/document\/9604770\/","DOI":"10.1109\/TNNLS.2021.3121432"},{"key":"17_CR11","unstructured":"Humplik, J., Galashov, A., Hasenclever, L., Ortega, P.A., Teh, Y.W., Heess, N.: Meta reinforcement learning as task inference. arXiv arXiv:1905.06424 (2019)"},{"key":"17_CR12","unstructured":"Han, X., Wu, F.: Meta reinforcement learning with successor feature based context. arXiv arXiv:2207.14723 (2022)"},{"key":"17_CR13","unstructured":"Gupta, A., Mendonca, R., Liu, Y., Abbeel, P., Levine, S.: Meta-reinforcement learning of structured exploration strategies. In: Advances in Neural Information Processing Systems, vol.\u00a031. Curran Associates, Inc. (2018). https:\/\/proceedings.neurips.cc\/paper\/2018\/hash\/4de754248c196c85ee4fbdcee89179bd-Abstract.html"},{"key":"17_CR14","unstructured":"Stadie, B.C., et al.: Some considerations on learning to explore via meta-reinforcement learning. arXiv arXiv:1803.01118 (2018)"},{"key":"17_CR15","doi-asserted-by":"publisher","unstructured":"Rothfuss, J., Lee, D., Clavera, I., Asfour, T., Abbeel, P.: ProMP: proximal meta-policy search (2018). https:\/\/doi.org\/10.48550\/arXiv.1810.06784. http:\/\/arxiv.org\/abs\/1810.06784","DOI":"10.48550\/arXiv.1810.06784"},{"key":"17_CR16","unstructured":"Zintgraf, L., Shiarli, K., Kurin, V., Hofmann, K., Whiteson, S.: Fast context adaptation via meta-learning. In: Proceedings of the 36th International Conference on Machine Learning, pp. 7693\u20137702. PMLR (2018). ISSN 2640-3498. https:\/\/proceedings.mlr.press\/v97\/zintgraf19a.html"},{"key":"17_CR17","unstructured":"Vuorio, R., Beck, J., Farquhar, G., Foerster, J., Whiteson, S.: No dice: an investigation of the bias- variance tradeoff in meta-gradients (2022)"},{"key":"17_CR18","unstructured":"Mendonca, R., Gupta, A., Kralev, R., Abbeel, P., Levine, S., Finn, C.: Guided meta-policy search. In: Advances in Neural Information Processing Systems, vol.\u00a032. Curran Associates, Inc. (2019). https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/d324a0cc02881779dcda44a675fdcaaa-Abstract.html"},{"key":"17_CR19","unstructured":"Finn, C., Abbeel, P., Levine, S.: Model-agnostic meta-learning for fast adaptation of deep networks, p.\u00a010 (2017)"},{"key":"17_CR20","unstructured":"Korshunova, I., Degrave, J., Dambre, J., Gretton, A., Husz\u00e1r, F.: Exchangeable models in meta reinforcement learning (2020)"},{"key":"17_CR21","doi-asserted-by":"publisher","unstructured":"Raileanu, R., Goldstein, M., Szlam, A., Fergus, R.: Fast adaptation via policy-dynamics value functions (2020). https:\/\/doi.org\/10.48550\/arXiv.2007.02879. http:\/\/arxiv.org\/abs\/2007.02879","DOI":"10.48550\/arXiv.2007.02879"},{"key":"17_CR22","doi-asserted-by":"publisher","unstructured":"He, J.Z.Y., Raghunathan, A., Brown, D.S., Erickson, Z., Dragan, A.D.: Learning representations that enable generalization in assistive tasks (2022). https:\/\/doi.org\/10.48550\/arXiv.2212.03175. https:\/\/arxiv.org\/abs\/2212.03175v1","DOI":"10.48550\/arXiv.2212.03175"},{"key":"17_CR23","doi-asserted-by":"publisher","unstructured":"Beck, J., Jackson, M.T., Vuorio, R., Whiteson, S.: Hypernetworks in meta-reinforcement learning (2022). https:\/\/doi.org\/10.48550\/arXiv.2210.11348. https:\/\/arxiv.org\/abs\/2210.11348v1","DOI":"10.48550\/arXiv.2210.11348"},{"key":"17_CR24","unstructured":"Duan, Y., Schulman, J., Chen, X., Bartlett, P.L., Sutskever, I., Abbeel, P.: RL$$^{2}$$: fast reinforcement learning via slow reinforcement learning. arXiv arXiv:1611.02779 (2017)"},{"key":"17_CR25","doi-asserted-by":"publisher","unstructured":"Greenberg, I., Mannor, S., Chechik, G., Meirom, E.: Train hard, fight easy: robust meta reinforcement learning (2023). https:\/\/doi.org\/10.48550\/arXiv.2301.11147. http:\/\/arxiv.org\/abs\/2301.11147","DOI":"10.48550\/arXiv.2301.11147"},{"key":"17_CR26","doi-asserted-by":"publisher","unstructured":"Kaelbling, L.P., Littman, M.L., Cassandra, A.R.: Planning and acting in partially observable stochastic domains. Artif. Intell. 101(1), 99\u2013134 (1998). https:\/\/doi.org\/10.1016\/S0004-3702(98)00023-X. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S000437029800023X","DOI":"10.1016\/S0004-3702(98)00023-X"},{"key":"17_CR27","doi-asserted-by":"publisher","unstructured":"Zintgraf, L., et al.: VariBAD: a very good method for Bayes-adaptive deep RL via meta-learning (2020). https:\/\/doi.org\/10.48550\/arXiv.1910.08348. https:\/\/arxiv.org\/abs\/1910.08348v2","DOI":"10.48550\/arXiv.1910.08348"},{"key":"17_CR28","unstructured":"Yu, T., et al.: Meta-world: a benchmark and evaluation for multi-task and meta reinforcement learning, p.\u00a017 (2021)"},{"key":"17_CR29","unstructured":"Yang, R., Xu, H., Wu, Y., Wang, X.: Multi-task reinforcement learning with soft modularization. arXiv arXiv:2003.13661 (2020)"},{"key":"17_CR30","unstructured":"Li, L., Huang, Y., Chen, M., Luo, S., Luo, D., Huang, J.: Provably improved context-based offline meta-RL with attention and contrastive learning, p.\u00a021 (2021)"},{"key":"17_CR31","doi-asserted-by":"publisher","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes (2022). https:\/\/doi.org\/10.48550\/arXiv.1312.6114. http:\/\/arxiv.org\/abs\/1312.6114","DOI":"10.48550\/arXiv.1312.6114"},{"key":"17_CR32","doi-asserted-by":"publisher","unstructured":"Alemi, A.A., Fischer, I., Dillon, J.V., Murphy, K.: Deep variational information bottleneck (2019). https:\/\/doi.org\/10.48550\/arXiv.1612.00410. http:\/\/arxiv.org\/abs\/1612.00410","DOI":"10.48550\/arXiv.1612.00410"},{"key":"17_CR33","doi-asserted-by":"publisher","unstructured":"Haarnoja, T., Zhou, A., Abbeel, P., Levine, S.: Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor (2018). https:\/\/doi.org\/10.48550\/arXiv.1801.01290. http:\/\/arxiv.org\/abs\/1801.01290","DOI":"10.48550\/arXiv.1801.01290"},{"key":"17_CR34","doi-asserted-by":"publisher","unstructured":"Todorov, E., Erez, T., Tassa, Y.: MuJoCo: a physics engine for model-based control. In: 2012 IEEE\/RSJ International Conference on Intelligent Robots and Systems, pp. 5026\u20135033 (2012). ISSN 2153-0866. https:\/\/doi.org\/10.1109\/IROS.2012.6386109. https:\/\/ieeexplore.ieee.org\/abstract\/document\/6386109","DOI":"10.1109\/IROS.2012.6386109"}],"container-title":["Lecture Notes in Computer Science","Advances in Knowledge Discovery and Data Mining"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-2259-4_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,4,24]],"date-time":"2024-04-24T23:18:33Z","timestamp":1714000713000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-2259-4_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9789819722617","9789819722594"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-2259-4_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"25 April 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PAKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pacific-Asia Conference on Knowledge Discovery and Data Mining","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Taipei","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Taiwan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 May 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 May 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"pakdd2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/pakdd2024.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}