{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T12:54:47Z","timestamp":1769086487673,"version":"3.49.0"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030640958","type":"print"},{"value":"9783030640965","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-64096-5_3","type":"book-chapter","created":{"date-parts":[[2020,11,25]],"date-time":"2020-11-25T00:30:27Z","timestamp":1606264227000},"page":"29-39","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["MGHRL: Meta Goal-Generation for Hierarchical Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Haotian","family":"Fu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongyao","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianye","family":"Hao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wulong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chen","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,11,25]]},"reference":[{"key":"3_CR1","unstructured":"Andrychowicz, M., et al.: Hindsight experience replay. In: Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, pp. 5048\u20135058 (2017). http:\/\/papers.nips.cc\/paper\/7090-hindsight-experience-replay"},{"key":"3_CR2","unstructured":"Bacon, P., Harb, J., Precup, D.: The option-critic architecture. In: Proceedings of the Thirty-First AAAI Conference on Artificial Intelligence, pp. 1726\u20131734 (2017). http:\/\/aaai.org\/ocs\/index.php\/AAAI\/AAAI17\/paper\/view\/14858"},{"key":"3_CR3","doi-asserted-by":"publisher","unstructured":"Barto, A.G., Mahadevan, S.: Recent advances in hierarchical reinforcement learning. Discrete Event Dyn. Syst. 13(1\u20132), 41\u201377 (2003). https:\/\/doi.org\/10.1023\/A:1022140919877","DOI":"10.1023\/A:1022140919877"},{"key":"3_CR4","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Bengio, S., Cloutier, J.: Learning a synaptic learning rule. In: IJCNN-91-Seattle International Joint Conference on Neural Networks II, vol. 2, p. 969 (1991)","DOI":"10.1109\/IJCNN.1991.155621"},{"key":"3_CR5","unstructured":"Bengio, Y., LeCun, Y. (eds.): 4th International Conference on Learning Representations, ICLR 2016 (2016). https:\/\/iclr.cc\/archive\/www\/doku.php%3Fid=iclr2016:accepted-main.html"},{"key":"3_CR6","unstructured":"Dayan, P., Hinton, G.E.: Feudal reinforcement learning. In: Advances in Neural Information Processing Systems 5, [NIPS Conference], pp. 271\u2013278 (1992). http:\/\/papers.nips.cc\/paper\/714-feudal-reinforcement-learning"},{"key":"3_CR7","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1613\/jair.639","volume":"13","author":"TG Dietterich","year":"2000","unstructured":"Dietterich, T.G.: Hierarchical reinforcement learning with the MAXQ value function decomposition. J. Artif. Intell. Res. 13, 227\u2013303 (2000). https:\/\/doi.org\/10.1613\/jair.639","journal-title":"J. Artif. Intell. Res."},{"key":"3_CR8","unstructured":"Duan, Y., Schulman, J., Chen, X., Bartlett, P.L., Sutskever, I., Abbeel, P.: RL$$^{2}$$: fast reinforcement learning via slow reinforcement learning. CoRR abs\/1611.02779 (2016). http:\/\/arxiv.org\/abs\/1611.02779"},{"key":"3_CR9","unstructured":"Finn, C., Abbeel, P., Levine, S.: Model-agnostic meta-learning for fast adaptation of deep networks. In: Proceedings of the 34th International Conference on Machine Learning, ICML 2017, pp. 1126\u20131135 (2017). http:\/\/proceedings.mlr.press\/v70\/finn17a.html"},{"key":"3_CR10","unstructured":"Frans, K., Ho, J., Chen, X., Abbeel, P., Schulman, J.: Meta learning shared hierarchies. In: 6th International Conference on Learning Representations, ICLR 2018 (2018)"},{"key":"3_CR11","unstructured":"Haarnoja, T., Zhou, A., Abbeel, P., Levine, S.: Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: Proceedings of the 35th International Conference on Machine Learning, ICML 2018, pp. 1856\u20131865 (2018). http:\/\/proceedings.mlr.press\/v80\/haarnoja18b.html"},{"key":"3_CR12","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes. In: 2nd International Conference on Learning Representations, ICLR 2014 (2014). http:\/\/arxiv.org\/abs\/1312.6114"},{"key":"3_CR13","unstructured":"Levine, S., Finn, C., Darrell, T., Abbeel, P.: End-to-end training of deep visuomotor policies. J. Mach. Learn. Res. 17, 39:1\u201339:40 (2016). http:\/\/jmlr.org\/papers\/v17\/15-522.html"},{"key":"3_CR14","unstructured":"Levy, A., Konidaris, G., Platt Jr, R., Saenko, K.: Learning multi-level hierarchies with hindsight. In: 7th International Conference on Learning Representations, ICLR 2019 (2019). https:\/\/openreview.net\/forum?id=ryzECoAcY7"},{"key":"3_CR15","unstructured":"Mishra, N., Rohaninejad, M., Chen, X., Abbeel, P.: A simple neural attentive meta-learner. In: 6th International Conference on Learning Representations, ICLR 2018 (2018). https:\/\/openreview.net\/forum?id=B1DmUzWAW"},{"issue":"7540","key":"3_CR16","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015). https:\/\/doi.org\/10.1038\/nature14236","journal-title":"Nature"},{"key":"3_CR17","unstructured":"Nachum, O., Gu, S., Lee, H., Levine, S.: Data-efficient hierarchical reinforcement learning. In: Advances in Neural Information Processing Systems 31: Annual Conference on Neural Information Processing Systems 2018, NeurIPS 2018, pp. 3307\u20133317 (2018). http:\/\/papers.nips.cc\/paper\/7591-data-efficient-hierarchical-reinforcement-learning"},{"key":"3_CR18","unstructured":"Parr, R., Russell, S.J.: Reinforcement learning with hierarchies of machines. In: Advances in Neural Information Processing Systems 10, [NIPS Conference], pp. 1043\u20131049 (1997). http:\/\/papers.nips.cc\/paper\/1384-reinforcement-learning-with-hierarchies-of-machines"},{"key":"3_CR19","unstructured":"Plappert, M., et al.: Multi-goal reinforcement learning: challenging robotics environments and request for research. CoRR abs\/1802.09464 (2018). http:\/\/arxiv.org\/abs\/1802.09464"},{"key":"3_CR20","unstructured":"Rakelly, K., Zhou, A., Finn, C., Levine, S., Quillen, D.: Efficient off-policy meta-reinforcement learning via probabilistic context variables. In: Proceedings of the 36th International Conference on Machine Learning, ICML 2019, pp. 5331\u20135340 (2019). http:\/\/proceedings.mlr.press\/v97\/rakelly19a.html"},{"key":"3_CR21","unstructured":"Ravi, S., Larochelle, H.: Optimization as a model for few-shot learning. In: ICLR (2017)"},{"key":"3_CR22","unstructured":"Rothfuss, J., Lee, D., Clavera, I., Asfour, T., Abbeel, P.: ProMP: proximal meta-policy search. In: 7th International Conference on Learning Representations, ICLR 2019 (2019). https:\/\/openreview.net\/forum?id=SkxXCi0qFX"},{"key":"3_CR23","unstructured":"Santoro, A., Bartunov, S., Botvinick, M., Wierstra, D., Lillicrap, T.P.: Meta-learning with memory-augmented neural networks. In: Proceedings of the 33nd International Conference on Machine Learning, ICML 2016, pp. 1842\u20131850 (2016). http:\/\/proceedings.mlr.press\/v48\/santoro16.html"},{"key":"3_CR24","unstructured":"Schmidhuber, J.: Evolutionary principles in self-referential learning (1987)"},{"key":"3_CR25","unstructured":"Stadie, B.C., et al.: Some considerations on learning to explore via meta-reinforcement learning. CoRR abs\/1803.01118 (2018). http:\/\/arxiv.org\/abs\/1803.01118"},{"key":"3_CR26","doi-asserted-by":"publisher","unstructured":"Sutton, R.S., Precup, D., Singh, S.P.: Between MDPs and semi-MDPs: a framework for temporal abstraction in reinforcement learning. Artif. Intell. 112(1\u20132), 181\u2013211 (1999). https:\/\/doi.org\/10.1016\/S0004-3702(99)00052-1","DOI":"10.1016\/S0004-3702(99)00052-1"},{"key":"3_CR27","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-5529-2","volume-title":"Learning to Learn","author":"S Thrun","year":"1998","unstructured":"Thrun, S., Pratt, L.Y.: Learning to Learn. Springer, Boston (1998). https:\/\/doi.org\/10.1007\/978-1-4615-5529-2"},{"key":"3_CR28","doi-asserted-by":"publisher","unstructured":"Todorov, E., Erez, T., Tassa, Y.: MuJoCo: a physics engine for model-based control. In: 2012 IEEE\/RSJ International Conference on Intelligent Robots and Systems, IROS 2012, pp. 5026\u20135033 (2012). https:\/\/doi.org\/10.1109\/IROS.2012.6386109","DOI":"10.1109\/IROS.2012.6386109"},{"key":"3_CR29","unstructured":"Vezhnevets, A.S., et al.: Feudal networks for hierarchical reinforcement learning. In: Proceedings of the 34th International Conference on Machine Learning, ICML 2017, pp. 3540\u20133549 (2017). http:\/\/proceedings.mlr.press\/v70\/vezhnevets17a.html"},{"key":"3_CR30","unstructured":"Vinyals, O., Blundell, C., Lillicrap, T., Kavukcuoglu, K., Wierstra, D.: Matching networks for one shot learning. In: Advances in Neural Information Processing Systems 29: Annual Conference on Neural Information Processing Systems 2016, pp. 3630\u20133638 (2016). http:\/\/papers.nips.cc\/paper\/6385-matching-networks-for-one-shot-learning"},{"key":"3_CR31","unstructured":"Wang, J.X., et al.: Learning to reinforcement learn. CoRR abs\/1611.05763 (2016). http:\/\/arxiv.org\/abs\/1611.05763"},{"key":"3_CR32","unstructured":"Xu, T., Liu, Q., Zhao, L., Peng, J.: Learning to explore via meta-policy gradient. In: Proceedings of the 35th International Conference on Machine Learning, ICML 2018, pp. 5459\u20135468 (2018). http:\/\/proceedings.mlr.press\/v80\/xu18d.html"}],"container-title":["Lecture Notes in Computer Science","Distributed Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-64096-5_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,18]],"date-time":"2024-08-18T00:02:30Z","timestamp":1723939350000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-64096-5_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030640958","9783030640965"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-64096-5_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"25 November 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"DAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Distributed Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Nanjing","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 October 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"dai22020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.adai.ai\/dai\/2020\/2020.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"22","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"9","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"41% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.77","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1.4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Due to the Corona pandemic this event was held virtually.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}