{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T09:26:45Z","timestamp":1743067605054,"version":"3.40.3"},"publisher-location":"Cham","reference-count":52,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031730030"},{"type":"electronic","value":"9783031730047"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73004-7_13","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T17:04:43Z","timestamp":1730394283000},"page":"214-230","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Reinforcement Learning via\u00a0Auxiliary Task Distillation"],"prefix":"10.1007","author":[{"given":"Abhinav Narayan","family":"Harish","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Larry","family":"Heck","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Josiah P.","family":"Hanna","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zsolt","family":"Kira","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andrew","family":"Szot","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"13_CR1","unstructured":"Mnih, V., et al.: Playing atari with deep reinforcement learning. arXiv preprintarXiv:1312.5602 (2013)"},{"key":"13_CR2","unstructured":"Berner, C., et al.: Dota 2 with large scale deep reinforcement learning. arXiv preprintarXiv:1912.06680 (2019)"},{"issue":"7676","key":"13_CR3","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver, D., et al.: Mastering the game of go without human knowledge. Nature 550(7676), 354\u2013359 (2017)","journal-title":"Nature"},{"key":"13_CR4","unstructured":"Shah, P., Hakkani-T\u00fcr, D., Heck, L.: Interactive reinforcement learning for task-oriented dialogue management. In: NIPS 2016 Deep Learning for Action and Interaction Workshop, vol. 11 (2016)"},{"key":"13_CR5","unstructured":"Liu, B., Tur, G., Hakkani-Tur, D., Shah, P., Heck, L.: End-to-end optimization of task-oriented dialogue model with deep reinforcement learning. arXiv preprintarXiv:1711.10712 (2017)"},{"key":"13_CR6","unstructured":"Nakano, R., et\u00a0al.: Webgpt: browser-assisted question-answering with human feedback. arXiv preprintarXiv:2112.09332 (2021)"},{"key":"13_CR7","unstructured":"Achiam, J., et al.: Gpt-4 technical report. arXiv preprintarXiv:2303.08774 (2023)"},{"key":"13_CR8","unstructured":"Touvron, H., et al.: Llama: open and efficient foundation language models. arXiv preprintarXiv:2302.13971 (2023)"},{"key":"13_CR9","unstructured":"Akkaya, I., et\u00a0al.: Solving rubik\u2019s cube with a robot hand. arXiv preprintarXiv:1910.07113 (2019)"},{"key":"13_CR10","unstructured":"Qi, H., Kumar, A., Calandra, R., Ma, Y., Malik, J.: In-hand object rotation via rapid motor adaptation. In: Conference on Robot Learning, pp. 1722\u20131732. PMLR (2023)"},{"key":"13_CR11","unstructured":"Kalashnikov, D., et al.: Scalable deep reinforcement learning for vision-based robotic manipulation. In: 2nd Annual Conference on Robot Learning, CoRL 2018, Z\u00fcrich, Switzerland, 29-31 October 2018, Proceedings, vol. 87, pp. 651\u2013673. Proceedings of Machine Learning Research. PMLR (2018). http:\/\/proceedings.mlr.press\/v87\/kalashnikov18a.html"},{"key":"13_CR12","unstructured":"Batra, D., et al.: Rearrangement: a challenge for embodied AI. arXiv preprintarXiv:2011.01975 (2020)"},{"key":"13_CR13","unstructured":"Harutyunyan, A., et al.: Hindsight credit assignment. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"13_CR14","unstructured":"Ni, T., Ma, M., Eysenbach, B., Bacon, P-L.: When do transformers shine in RL? decoupling memory from credit assignment. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"13_CR15","doi-asserted-by":"crossref","unstructured":"Berges, V-P., et al.: Galactic: scaling end-to-end reinforcement learning for rearrangement at 100k steps-per-second. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13767\u201313777 (2023)","DOI":"10.1109\/CVPR52729.2023.01323"},{"key":"13_CR16","doi-asserted-by":"crossref","unstructured":"Huang, X., Batra, D., Rai, A., Szot, A.: Skill transformer: a monolithic policy for mobile manipulation. arXiv preprintarXiv:2308.09873 (2023)","DOI":"10.1109\/ICCV51070.2023.00996"},{"key":"13_CR17","unstructured":"Gu, J., Chaplot, D.S., Su, H., Malik, J.: Multi-skill mobile manipulation for object rearrangement. arXiv preprintarXiv:2209.02778 (2022)"},{"issue":"1","key":"13_CR18","first-page":"7382","volume":"21","author":"S Narvekar","year":"2020","unstructured":"Narvekar, S., Peng, B., Leonetti, M., Sinapov, J., Taylor, M.E., Stone, P.: Curriculum learning for reinforcement learning domains: a framework and survey. J. Mach. Learn. Res. 21(1), 7382\u20137431 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"13_CR19","unstructured":"Dennis, M., et al.: Emergent complexity and zero-shot transfer via unsupervised environment design. In: Advance in Neural Information Processing System, vol. 33, pp. 13049\u201313061 (2020)"},{"key":"13_CR20","unstructured":"Azad, A.S., et al.: Clutr: curriculum learning via unsupervised task representation learning. In: International Conference on Machine Learning, pp. 1361\u20131395. PMLR (2023)"},{"key":"13_CR21","unstructured":"Fang, K., Migimatsu, T., Mandlekar, A., Fei-Fei, L., Bohg, J.: Active task randomization: Learning visuomotor skills for sequential manipulation by proposing feasible and novel tasks. arXiv preprintarXiv:2211.06134 (2022)"},{"key":"13_CR22","unstructured":"Szot, A., et al.: Habitat 2.0: training home assistants to rearrange their habitat. In: Advances in Neural Information Processing Systems, vol. 34 (2021)"},{"issue":"1\u20132","key":"13_CR23","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"RS Sutton","year":"1999","unstructured":"Sutton, R.S., Precup, D., Singh, S.: Between MDPs and semi-MDPs: a framework for temporal abstraction in reinforcement learning. Artif. Intell. 112(1\u20132), 181\u2013211 (1999)","journal-title":"Artif. Intell."},{"key":"13_CR24","doi-asserted-by":"crossref","unstructured":"Bacon, P-L., Harb, J., Precup, D.: The option-critic architecture. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 31 (2017)","DOI":"10.1609\/aaai.v31i1.10916"},{"key":"13_CR25","unstructured":"Zhang, S., Whiteson, S.: DAC: the double actor-critic architecture for learning options. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"13_CR26","unstructured":"Xia, F., Li, C., Mart\u00edn-Mart\u00edn, R., Litany, O., Toshev, A., Savarese, S.: Relmogen: leveraging motion generation in reinforcement learning for mobile manipulation. arXiv preprintarXiv:2008.07792 (2020)"},{"key":"13_CR27","unstructured":"Karkus, P., et al.: Beyond tabula-rasa: a modular reinforcement learning approach for physically embedded 3d sokoban. arXiv preprintarXiv:2010.01298 (2020)"},{"key":"13_CR28","unstructured":"Dalal, M., Pathak, D., Salakhutdinov, R.R.: Accelerating robotic reinforcement learning via parameterized action primitives. In: Advances in Neural Information Processing Systems, vol. 34, pp. 21847\u201321859 (2021)"},{"key":"13_CR29","unstructured":"Hafner, D., Lee, K.-H., Fischer, I., Abbeel, P.: Deep hierarchical planning from pixels. IN: Advance in Neural Information Processing System, vol. 35, pp. 26091\u201326104 (2022)"},{"key":"13_CR30","unstructured":"Vezzani, G., et al.: Skills: adaptive skill sequencing for efficient temporally-extended exploration. arXiv preprintarXiv:2211.13743 (2022)"},{"key":"13_CR31","unstructured":"Chen, Y., Wang, C., Fei-Fei, L., Karen Liu, C.: Sequential dexterity: Chaining dexterous policies for long-horizon manipulation. arXiv preprintarXiv:2309.00987 (2023)"},{"key":"13_CR32","unstructured":"Utkarsh\u00a0Aashu Mishra, Shangjie Xue, Yongxin Chen, and Danfei Xu. Generative skill chaining: Long-horizon skill planning with diffusion models. In Conference on Robot Learning, pages 2905\u20132925. PMLR, 2023"},{"key":"13_CR33","unstructured":"Lee, Y., Lim, J.J., Anandkumar, A., Zhu, Y.: Adversarial skill chaining for long-horizon robot manipulation via terminal state regularization. arXiv preprintarXiv:2111.07999 (2021)"},{"key":"13_CR34","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprintarXiv:1707.06347 (2017)"},{"key":"13_CR35","unstructured":"Rudin, N., Hoeller, D., Reist, P., Hutter, M.: Learning to walk in minutes using massively parallel deep reinforcement learning. In: Conference on Robot Learning, pp. 91\u2013100. PMLR (2022)"},{"key":"13_CR36","unstructured":"Agarwal, A., Kumar, A., Malik, J., Pathak, D.: Legged locomotion in challenging terrains using egocentric vision. In: Conference on Robot Learning, pp. 403\u2013415. PMLR (2023)"},{"key":"13_CR37","unstructured":"Fu, Z., Cheng, X., Pathak, D.: Deep whole-body control: learning a unified policy for manipulation and locomotion. In: Conference on Robot Learning, pp. 138\u2013149. PMLR (2023)"},{"key":"13_CR38","doi-asserted-by":"crossref","unstructured":"Kumar, A., Zipeng, F., Pathak, D., Malik, J.: Rapid motor adaptation for legged robots. RSS, Rma (2021)","DOI":"10.15607\/RSS.2021.XVII.011"},{"key":"13_CR39","unstructured":"Radosavovic, I., Xiao, T., Zhang, B., Darrell, T., Malik, J., Sreenath, K.: Learning humanoid locomotion with transformers. arXiv preprintarXiv:2303.03381 (2023)"},{"key":"13_CR40","doi-asserted-by":"crossref","unstructured":"Katara, P., Xian, Z., Fragkiadaki, K.: Gen2sim: scaling up robot learning in simulation with generative models. arXiv preprintarXiv:2310.18308 (2023)","DOI":"10.1109\/ICRA57147.2024.10610566"},{"key":"13_CR41","unstructured":"Ye, J., Batra, D., Wijmans, E., Das, A.: Auxiliary tasks speed up learning point goal navigation. In: Conference on Robot Learning, pp. 498\u2013516. PMLR (2021)"},{"key":"13_CR42","doi-asserted-by":"crossref","unstructured":"Ye, J., Batra, D., Das, A., Wijmans, E.: Auxiliary tasks and exploration enable objectgoal navigation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16117\u201316126 (2021)","DOI":"10.1109\/ICCV48922.2021.01581"},{"key":"13_CR43","unstructured":"Gregor, K., Jimenez Rezende, D., Besse, F., Wu, Y., Merzic, H., van den Oord, A.: Shaping belief states with generative environment models for RL. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"13_CR44","doi-asserted-by":"crossref","unstructured":"Kuo, C.W., Ma, C.Y., Hoffman, J., Kira, Z.: Structure-encoding auxiliary tasks for improved visual representation in vision-and-language navigation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1104\u20131113 (2023)","DOI":"10.1109\/WACV56688.2023.00116"},{"key":"13_CR45","unstructured":"Jia, Z., Li, X., Ling, Z., Liu, S., Wu, Y., Su, H.: Improving policy optimization with generalist-specialist learning. In: International Conference on Machine Learning, pp. 10104\u201310119. PMLR (2022)"},{"key":"13_CR46","unstructured":"Baker, B., et al.: Emergent tool use from multi-agent autocurricula. arXiv preprintarXiv:1909.07528 (2019)"},{"key":"13_CR47","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"13_CR48","doi-asserted-by":"crossref","unstructured":"Hessel, M., Soyer, H., Espeholt, L., Czarnecki, W., Schmitt, S., Van Hasselt, H.: Multi-task deep reinforcement learning with popart. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 3796\u20133803 (2019)","DOI":"10.1609\/aaai.v33i01.33013796"},{"key":"13_CR49","unstructured":"Fetch robotics. Fetch (2020). http:\/\/fetchrobotics.com\/"},{"key":"13_CR50","unstructured":"Szot, A., et al.: Habitat rearrangement challenge (2022). https:\/\/aihabitat.org\/challenge\/2022_rearrange"},{"issue":"3\u20134","key":"13_CR51","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1016\/0004-3702(71)90010-5","volume":"2","author":"RE Fikes","year":"1971","unstructured":"Fikes, R.E., Nilsson, N.J.: STRIPS: a new approach to the application of theorem proving to problem solving. Artif. Intell. 2(3\u20134), 189\u2013208 (1971)","journal-title":"Artif. Intell."},{"key":"13_CR52","unstructured":"Wijmans, E., et al.: DD-PPO: learning near-perfect pointgoal navigators from 2.5 billion frames. arXiv preprintarXiv:1911.00357 (2019)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73004-7_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T17:07:14Z","timestamp":1730394434000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73004-7_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9783031730030","9783031730047"],"references-count":52,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73004-7_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}