{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T02:33:37Z","timestamp":1730255617666,"version":"3.28.0"},"reference-count":41,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,5,13]]},"DOI":"10.1109\/icra57147.2024.10610861","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T17:51:05Z","timestamp":1723139465000},"page":"10821-10828","source":"Crossref","is-referenced-by-count":1,"title":["HyperPPO: A scalable method for finding small policies for robotic control"],"prefix":"10.1109","author":[{"given":"Shashank","family":"Hegde","sequence":"first","affiliation":[{"name":"University of Southern California"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhehui","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Southern California"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gaurav S.","family":"Sukhatme","sequence":"additional","affiliation":[{"name":"University of Southern California"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Learning to walk via deep reinforcement learning","author":"Haarnoja","year":"2019","journal-title":"Robotics: Science and Systems"},{"key":"ref2","first-page":"1291","article-title":"Visual-locomotion: Learning to walk on complex terrains with vision","volume-title":"Proceedings of the 5th Conference on Robot Learning","volume":"164","author":"Yu"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/MMAR.2017.8046794"},{"article-title":"Neural architecture search: Insights from 1000 papers","year":"2023","author":"White","key":"ref4"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160791"},{"article-title":"Quadswarm: A modular multi-quadrotor simulator for deep reinforcement learning with direct thrust control","year":"2023","author":"Huang","key":"ref6"},{"article-title":"Proximal policy optimization algorithms","year":"2017","author":"Schulman","key":"ref7"},{"article-title":"What matters in on-policy reinforcement learning? a large-scale empirical study","volume-title":"ICLR 2021-Ninth International Conference on Learning Representations","author":"Andrychowicz","key":"ref8"},{"key":"ref9","article-title":"The 37 implementation details of proximal policy optimization","volume-title":"ICLR Blog Track","author":"Huang","year":"2022"},{"key":"ref10","first-page":"22409","article-title":"Envpool: A highly parallel reinforcement learning environment execution engine","volume":"35","author":"Weng","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Brax-a differentiable physics engine for large scale rigid body simulation","volume-title":"Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 1)","author":"Freeman","key":"ref11"},{"article-title":"Isaac gym: High performance gpu based physics simulation for robot learning","volume-title":"Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2)","author":"Makoviychuk","key":"ref12"},{"article-title":"Learning to walk in minutes using massively parallel deep reinforcement learning","volume-title":"5th Annual Conference on Robot Learning","author":"Rudin","key":"ref13"},{"article-title":"Neural architecture search with reinforcement learning","volume-title":"International Conference on Learning Representations","author":"Zoph","key":"ref14"},{"article-title":"Darts: Differentiable architecture search","volume-title":"International Conference on Learning Representations","author":"Liu","key":"ref15"},{"key":"ref16","first-page":"20\/1","article-title":"Differentiable architecture search for reinforcement learning","volume-title":"Proceedings of the First International Conference on Automated Machine Learning","volume":"188","author":"Miao"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561998"},{"article-title":"Reinforcement learning with chromatic networks for compact architecture search","year":"2019","author":"Song","key":"ref18"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-80126-7_42"},{"article-title":"Smash: One-shot model architecture search through hypernetworks","volume-title":"6th International Conference on Learning Representations 2018","author":"Brock","key":"ref20"},{"key":"ref21","article-title":"Hypernetworks","volume-title":"CoRR","author":"Ha","year":"2016"},{"article-title":"Graph hypernetworks for neural architecture search","volume-title":"2019, publisher Copyright: \u00a9 7th International Conference on Learning Representations, ICLR 2019. All Rights Reserved.; 7th International Conference on Learning Representations, ICLR 2019","author":"Zhang","key":"ref22"},{"key":"ref23","first-page":"29 433","article-title":"Parameter prediction for unseen deep architectures","volume":"34","author":"Knyazev","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Hyperdynamics: Meta-learning object and agent dynamics with hypernetworks","volume-title":"International Conference on Learning Representations","author":"Xian","key":"ref24"},{"article-title":"Continual learning with hypernetworks","volume-title":"International Conference on Learning Representations","author":"von Oswald","key":"ref25"},{"article-title":"Hyperdecision transformer for efficient online policy adaptation","year":"2023","author":"Xu","key":"ref26"},{"key":"ref27","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"International conference on machine learning","author":"Haarnoja"},{"key":"ref28","article-title":"Hindsight experience replay","volume":"30","author":"Andrychowicz","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref29","article-title":"Continuous control with deep reinforcement learning","author":"Lillicrap","year":"2016","journal-title":"ICLR (Poster)"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2017.2720851"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8967695"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2019.2930489"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/IROS51168.2021.9636053"},{"key":"ref34","first-page":"576","article-title":"Decentralized control of quadrotor swarms with end-to-end deep reinforcement learning","volume-title":"Conference on Robot Learning","author":"Batra"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/icra57147.2024.10611499"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.abg5810"},{"key":"ref37","article-title":"Openai gym","volume-title":"CoRR","author":"Brockman","year":"2016"},{"key":"ref38","first-page":"7652","article-title":"Sample factory: Egocentric 3d control from pixels at 100000 FPS with asynchronous reinforcement learning","volume-title":"Proceedings of the 37th International Conference on Machine Learning, ICML 2020, 13-18 July 2020, Virtual Event","volume":"119","author":"Petrenko"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"article-title":"Proximal policy gradient arborescence for quality diversity reinforcement learning","year":"2023","author":"Batra","key":"ref40"},{"article-title":"Generating behaviorally diverse policies with latent diffusion models","year":"2023","author":"Hegde","key":"ref41"}],"event":{"name":"2024 IEEE International Conference on Robotics and Automation (ICRA)","start":{"date-parts":[[2024,5,13]]},"location":"Yokohama, Japan","end":{"date-parts":[[2024,5,17]]}},"container-title":["2024 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10609961\/10609862\/10610861.pdf?arnumber=10610861","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,10]],"date-time":"2024-08-10T05:22:28Z","timestamp":1723267348000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10610861\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,13]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/icra57147.2024.10610861","relation":{},"subject":[],"published":{"date-parts":[[2024,5,13]]}}}