{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T14:35:03Z","timestamp":1785767703648,"version":"3.56.0"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2024,8,27]],"date-time":"2024-08-27T00:00:00Z","timestamp":1724716800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,8,27]],"date-time":"2024-08-27T00:00:00Z","timestamp":1724716800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100000006","name":"Office of Naval Research","doi-asserted-by":"publisher","award":["N00014-21-1-4010"],"award-info":[{"award-number":["N00014-21-1-4010"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CMMI-2037101"],"award-info":[{"award-number":["CMMI-2037101"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CMMI-2037101"],"award-info":[{"award-number":["CMMI-2037101"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Robot"],"published-print":{"date-parts":[[2024,10]]},"DOI":"10.1007\/s10514-024-10170-8","type":"journal-article","created":{"date-parts":[[2024,8,27]],"date-time":"2024-08-27T21:02:16Z","timestamp":1724792536000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["R $$\\times $$ R: Rapid eXploration for Reinforcement learning via sampling-based reset distributions and imitation pre-training"],"prefix":"10.1007","volume":"48","author":[{"given":"Gagan","family":"Khandate","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tristan L.","family":"Saidi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siqi","family":"Shang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Eric T.","family":"Chang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Seth","family":"Dennis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Johnson","family":"Adams","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Matei","family":"Ciocarlie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,8,27]]},"reference":[{"key":"10170_CR1","unstructured":"Agarwal, A., Kakade, S.\u00a0M., Lee, J.\u00a0D., & Mahajan, G. (2020). Optimality and approximation with policy gradient methods in markov decision processes. In Jacob, A. & Shivani A. (eds.), Proceedings of thirty third conference on learning theory, vol. 125 of Proceedings of machine learning research, pp. 64\u201366. PMLR."},{"key":"10170_CR2","unstructured":"Akkaya, I., Andrychowicz, M., Chociej, M., Litwin, M., McGrew, B., Petron, A., & Zhang, L. (2019). Solving rubik\u2019s cube with a robot hand"},{"key":"10170_CR3","doi-asserted-by":"crossref","unstructured":"Allshire, A., Mittal, M., Lodaya, V., Makoviychuk, V., Makoviichuk, D., Widmaier, F., W\u00fcthrich, M., Bauer, S., Handa, A., & Garg, A. (2021). Transferring dexterous manipulation from GPU simulation to a remote real-world TriFinger.","DOI":"10.1109\/IROS47612.2022.9981458"},{"key":"10170_CR4","unstructured":"Amin, S., Gomrokchi, M., Satija, H., van Hoof, H., & Precup, D. (2021). A survey of exploration methods in reinforcement learning. arXiv:2109.00157"},{"issue":"1","key":"10170_CR5","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1177\/0278364919887447","volume":"39","author":"OM Andrychowicz","year":"2020","unstructured":"Andrychowicz, O. M., Baker, B., Chociej, M., Jozefowicz, R., McGrew, B., Pachocki, J., Petron, A., Plappert, M., Powell, G., Ray, A., & Schneider, J. (2020). Learning dexterous in-hand manipulation. The International Journal of Robotics Research, 39(1), 3\u201320.","journal-title":"The International Journal of Robotics Research"},{"key":"10170_CR6","doi-asserted-by":"crossref","unstructured":"Bhatt, A., Sieler, A., Puhlmann, S., & Brock, O. (2022). Surprisingly robust in-hand manipulation: An empirical study.","DOI":"10.15607\/RSS.2021.XVII.089"},{"key":"10170_CR7","unstructured":"Chen, T., Tippur, M., Wu, S., Kumar, V., Adelson, E. & Agrawal, P. (2022). Visual dexterity: In-hand dexterous manipulation from depth."},{"key":"10170_CR8","unstructured":"Chen, T., Xu, J., & Agrawal, P. (2021). A system for general in-hand object re-orientation"},{"issue":"4","key":"10170_CR9","doi-asserted-by":"publisher","first-page":"4298","DOI":"10.1109\/LRA.2019.2931199","volume":"4","author":"HTL Chiang","year":"2019","unstructured":"Chiang, H. T. L., Hsu, J., Fiser, M., Tapia, L., & Faust, A. (2019). RL-RRT: Kinodynamic motion planning via learning reachability estimators from RL policies. IEEE Robotics and Automation Letters, 4(4), 4298\u20134305.","journal-title":"IEEE Robotics and Automation Letters"},{"key":"10170_CR10","unstructured":"Duan, Y., Chen, X., Houthooft, R., Schulman, J., & Abbeel, P. (2016). Benchmarking deep reinforcement learning for continuous control."},{"key":"10170_CR11","unstructured":"Ecoffet, A., Huizinga, J., Lehman, J., Stanley, K. O., & Clune, J. (2019). Go-explore: a new approach for hard-exploration problems."},{"issue":"7847","key":"10170_CR12","doi-asserted-by":"publisher","first-page":"580","DOI":"10.1038\/s41586-020-03157-9","volume":"590","author":"A Ecoffet","year":"2021","unstructured":"Ecoffet, A., Huizinga, J., Lehman, J., Stanley, K. O., & Clune, J. (2021). First return, then explore. Nature, 590(7847), 580\u2013586.","journal-title":"Nature"},{"issue":"4","key":"10170_CR13","doi-asserted-by":"publisher","first-page":"1115","DOI":"10.1109\/TRO.2020.2975428","volume":"36","author":"A Francis","year":"2020","unstructured":"Francis, A., Faust, A., Chiang, H. T. L., Hsu, J., Kew, J. C., Fiser, M., & Lee, T. W. E. (2020). Long-range indoor navigation with PRM-RL. IEEE Transactions on Robotics, 36(4), 1115\u20131134.","journal-title":"IEEE Transactions on Robotics"},{"key":"10170_CR14","unstructured":"Ha, H., Xu, J., & Song, S. (2020). Learning a decentralized multi-arm motion planner."},{"key":"10170_CR15","unstructured":"Haarnoja, T., Zhou, A., & Abbeel, P. (2018). Soft actor-critic: Off-Policy maximum entropy deep reinforcement learning with a stochastic actor."},{"key":"10170_CR16","doi-asserted-by":"crossref","unstructured":"Han, L, & Trinkle, J\u00a0C. (1998). Dextrous manipulation by rolling and finger gaiting. In Proceedings 1998 IEEE international conference on robotics and automation (Cat. No.98CH36146), vol.\u00a01, pp. 730\u2013735.","DOI":"10.1109\/ROBOT.1998.677060"},{"key":"10170_CR17","doi-asserted-by":"crossref","unstructured":"Handa, A., Allshire, A., Makoviychuk, V., Petrenko, A., Singh, R., Liu, J., Makoviichuk, D., Van Wyk, K., Zhurkevich, A., Sundaralingam, B. & Narang, Y. (2022). DeXtreme: Transfer of agile in-hand manipulation from simulation to reality.","DOI":"10.1109\/ICRA48891.2023.10160216"},{"key":"10170_CR18","unstructured":"Hansen, N., Lin, Y., Hao, S., Wang, X., Kumar, V. & Aravind, R. (2022). MoDem: Accelerating visual model-based reinforcement learning with demonstrations."},{"key":"10170_CR19","doi-asserted-by":"crossref","unstructured":"Hu, H., Mirchandani, S., & Sadigh, D. (2023). Imitation bootstrapped reinforcement learning.","DOI":"10.15607\/RSS.2024.XX.056"},{"key":"10170_CR20","doi-asserted-by":"crossref","unstructured":"Jurgenson, T., & Tamar, A. (2019). Harnessing reinforcement learning for neural motion planning.","DOI":"10.15607\/RSS.2019.XV.026"},{"key":"10170_CR21","doi-asserted-by":"crossref","unstructured":"Karaman, S., & Frazzoli, E. (2010). Optimal kinodynamic motion planning using incremental sampling-based methods. In 49th IEEE conference on decision and control (CDC), pp. 7681\u20137687.","DOI":"10.1109\/CDC.2010.5717430"},{"issue":"1","key":"10170_CR22","doi-asserted-by":"publisher","first-page":"166","DOI":"10.1109\/70.660866","volume":"14","author":"LE Kavraki","year":"1998","unstructured":"Kavraki, L. E., Kolountzakis, M. N., & Latombe, J.-C. (1998). Analysis of probabilistic roadmaps for path planning. IEEE Transactions on Robotics and Automation, 14(1), 166\u2013171.","journal-title":"IEEE Transactions on Robotics and Automation"},{"issue":"4","key":"10170_CR23","doi-asserted-by":"publisher","first-page":"566","DOI":"10.1109\/70.508439","volume":"12","author":"LE Kavraki","year":"1996","unstructured":"Kavraki, L. E., Svestka, P., Latombe, J.-C., & Overmars, M. H. (1996). Probabilistic roadmaps for path planning in high-dimensional configuration spaces. IEEE Transactions on Robotics and Automation, 12(4), 566\u2013580.","journal-title":"IEEE Transactions on Robotics and Automation"},{"key":"10170_CR24","doi-asserted-by":"crossref","unstructured":"Khandate, G., Haas-Heger, M., Ciocarlie, M. (2022). On the feasibility of learning finger-gaiting in-hand manipulation with intrinsic sensing. In 2022 International Conference on Robotics and Automation (ICRA), pp. 2752\u20132758","DOI":"10.1109\/ICRA46639.2022.9812212"},{"key":"10170_CR25","doi-asserted-by":"crossref","unstructured":"Khandate, S., Shang, S., Chang, E. T., Saidi, T. L., Liu, Y., Dennis, S.\u00a0M., Adams, J., & Ciocarlie, M. (2023). Sampling-based exploration for reinforcement learning of dexterous manipulation.","DOI":"10.15607\/RSS.2023.XIX.020"},{"key":"10170_CR26","doi-asserted-by":"crossref","unstructured":"King, J. E., Cognetti, M., & Srinivasa, S. S. (2016). Rearrangement planning using object-centric and robot-centric action spaces. In 2016 IEEE international conference on robotics and automation (ICRA). IEEE","DOI":"10.1109\/ICRA.2016.7487583"},{"key":"10170_CR27","unstructured":"LaValle, S. (1998). Rapidly-exploring random trees : A new tool for path planning. The annual research report."},{"key":"10170_CR28","doi-asserted-by":"crossref","unstructured":"Leveroni, S., & Salisbury, K. (1996). Reorienting objects with a robot hand using grasp gaits. In Robotics research, pp 39\u201351. Springer London.","DOI":"10.1007\/978-1-4471-1021-7_5"},{"issue":"3","key":"10170_CR29","doi-asserted-by":"publisher","first-page":"4496","DOI":"10.1109\/LRA.2021.3067847","volume":"6","author":"L Li","year":"2021","unstructured":"Li, L., Miao, Y., Qureshi, A. H., & Yip, M. C. (2021). MPC-MPNet: Model-predictive motion planning networks for fast, near-optimal planning under kinodynamic constraints. IEEE Robotics and Automation Letters, 6(3), 4496\u20134503.","journal-title":"IEEE Robotics and Automation Letters"},{"key":"10170_CR30","doi-asserted-by":"crossref","unstructured":"Ma, R. R., & Dollar, A. M. (2011). On dexterity and dexterous manipulation. In 2011 15th International conference on advanced robotics (ICAR), pp. 1\u20137.","DOI":"10.1109\/ICAR.2011.6088576"},{"key":"10170_CR31","unstructured":"Makoviychuk, V., Wawrzyniak, L., Guo, Y., Michelle, L., Storey, K., Macklin, M., Hoeller, D., Rudin, N., Allshire, A., Handa, A., & Gavriel S. (2021). Isaac gym: High performance GPU-based physics simulation for robot learning."},{"key":"10170_CR32","unstructured":"Morere, P., Francis, G., Blau, T., & Ramos, F. (2020). Reinforcement learning with probabilistically complete exploration."},{"key":"10170_CR33","doi-asserted-by":"crossref","unstructured":"Morgan, A.S., Nandha, D., Chalvatzaki, G., D\u2019Eramo, C., Dollar, A.\u00a0M., & Peters, J. (2021). Model predictive actor-critic: Accelerating robot skill acquisition with deep reinforcement learning.","DOI":"10.1109\/ICRA48506.2021.9561298"},{"issue":"2","key":"10170_CR34","doi-asserted-by":"publisher","first-page":"4821","DOI":"10.1109\/LRA.2022.3145961","volume":"7","author":"AS Morgan","year":"2022","unstructured":"Morgan, A. S., Hang, K., Wen, B., Bekris, K., & Dollar, A. M. (2022). Complex in-hand manipulation via compliance-enabled finger gaiting and multi-modal planning. IEEE Robotics and Automation Letters, 7(2), 4821\u20134828.","journal-title":"IEEE Robotics and Automation Letters"},{"key":"10170_CR35","doi-asserted-by":"crossref","unstructured":"Nair, A., McGrew, B., Andrychowicz, M., Zaremba, W., & Abbeel, P. (2017). Overcoming exploration in reinforcement learning with demonstrations.","DOI":"10.1109\/ICRA.2018.8463162"},{"key":"10170_CR36","doi-asserted-by":"crossref","unstructured":"Pathak, D., Agrawal, P., Efros, A. A., & Darrell, T. (2017). Curiosity-driven exploration by self-supervised prediction.","DOI":"10.1109\/CVPRW.2017.70"},{"issue":"5","key":"10170_CR37","doi-asserted-by":"publisher","first-page":"2416","DOI":"10.1109\/TMECH.2020.2975578","volume":"25","author":"P Piacenza","year":"2020","unstructured":"Piacenza, P., Behrman, K., Schifferer, B., Kymissis, I., & Ciocarlie, M. (2020). A sensorized multicurved robot finger with Data-Driven touch sensing via overlapping light signals. IEEE\/ASME Transactions on Mechatronics, 25(5), 2416\u20132427.","journal-title":"IEEE\/ASME Transactions on Mechatronics"},{"key":"10170_CR38","doi-asserted-by":"crossref","unstructured":"Pinto, L., Andrychowicz, M., Welinder, P., Zaremba, W., & Abbeel, P. (2017). Asymmetric actor critic for image based robot learning.","DOI":"10.15607\/RSS.2018.XIV.008"},{"key":"10170_CR39","unstructured":"Pinto, L., Mandalika, A., Hou, B., & Srinivasa, S. (2018). Sample-efficient learning of nonprehensile manipulation policies via physics-based informed state distributions."},{"key":"10170_CR40","doi-asserted-by":"crossref","unstructured":"Pitz, J., R\u00f6stel, L., Sievers, L., & B\u00e4uml, B. (2023). Dextrous tactile in-hand manipulation using a modular reinforcement learning architecture.","DOI":"10.1109\/ICRA48891.2023.10160756"},{"key":"10170_CR41","unstructured":"Plappert, M., Houthooft, R., Dhariwal, P., Sidor, S., Chen, R. Y., Chen, X., Asfour, T., Abbeel, P., & Andrychowicz, M. (2017). Parameter space noise for exploration."},{"key":"10170_CR42","unstructured":"Qi, H., Kumar, A., Calandra, R., Ma, Y., & Malik, J. (2022). In-hand object rotation via rapid motor adaptation."},{"key":"10170_CR43","unstructured":"Qi, H., Yi, B., Suresh, S., Lambeta, M., Ma, Y, Calandra, R., & Malik, J. (2023). General in-hand object rotation with vision and touch. CoRL. arXiv:2309.09979"},{"key":"10170_CR44","doi-asserted-by":"crossref","unstructured":"R\u00f6stel, L., Pitz, J., Sievers, L., & B\u00e4uml, B. (2023). Estimator-coupled reinforcement learning for robust purely tactile in-hand manipulation.","DOI":"10.1109\/Humanoids57100.2023.10375194"},{"key":"10170_CR45","doi-asserted-by":"crossref","unstructured":"Schramm, L., & Boularias, A. (2022). Learning-guided exploration for efficient sampling-based motion planning in high dimensions. In 2022 International conference on robotics and automation (ICRA). IEEE.","DOI":"10.1109\/ICRA46639.2022.9812184"},{"key":"10170_CR46","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., & Klimov, O. (2017). Proximal policy optimization algorithms."},{"key":"10170_CR47","doi-asserted-by":"crossref","unstructured":"Sievers, L., Pitz, J., & B\u00e4uml, B. (2022). Learning purely tactile In-Hand manipulation with a Torque-Controlled hand. In 2022 International conference on robotics and automation (ICRA), pp. 2745\u20132751.","DOI":"10.1109\/ICRA46639.2022.9812093"},{"key":"10170_CR48","unstructured":"Tavakoli, A., Levdik, V., Islam, R., Smith, C.\u00a0M. & Kormushev, P. (2018). Exploring restart distributions."},{"key":"10170_CR49","doi-asserted-by":"crossref","unstructured":"Webb, DJ. & van\u00a0den Berg, J. (2013). Kinodynamic RRT*: Asymptotically optimal motion planning for robots with linear dynamics. In 2013 IEEE international conference on robotics and automation, pp 5054\u20135061.","DOI":"10.1109\/ICRA.2013.6631299"},{"key":"10170_CR50","doi-asserted-by":"crossref","unstructured":"Xu, J., Koo, TJ. & Li, Z. (2007). Finger gaits planning for multifingered manipulation. In 2007 IEEE\/RSJ international conference on intelligent robots and systems, pp. 2932\u20132937.","DOI":"10.1109\/IROS.2007.4399189"},{"key":"10170_CR51","doi-asserted-by":"crossref","unstructured":"Yashima, M, Shiina, Y, & Yamaguchi, H. (2003). Randomized manipulation planning for a multi-fingered hand by switching contact modes. In 2003 IEEE international conference on robotics and automation (Cat. No. 03CH37422), vol.\u00a02, pp. 2689\u20132694.","DOI":"10.1109\/ROBOT.2003.1241999"},{"key":"10170_CR52","doi-asserted-by":"crossref","unstructured":"Yin, ZH., Huang, B., Qin, Y., Chen, Q. & Wang, X. (2023). Rotating without seeing: Towards in-hand dexterity through touch.","DOI":"10.15607\/RSS.2023.XIX.036"},{"key":"10170_CR53","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Che, H., Qin, Y., Huang, B., Yin, ZH., Lee, KW., Yi, W., Lim, SC. & Wang, X. (2023). Robot synesthesia: In-hand manipulation with visuotactile sensing.","DOI":"10.1109\/ICRA57147.2024.10610532"},{"key":"10170_CR54","unstructured":"Zhuang, Z., Fu, Z., Wang, J., Atkeson, C., Schwertfeger, S., Finn, C. & Zhao, H. (2023). Robot parkour learning."}],"container-title":["Autonomous Robots"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10514-024-10170-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10514-024-10170-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10514-024-10170-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,15]],"date-time":"2024-10-15T15:01:34Z","timestamp":1729004494000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10514-024-10170-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,27]]},"references-count":54,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2024,10]]}},"alternative-id":["10170"],"URL":"https:\/\/doi.org\/10.1007\/s10514-024-10170-8","relation":{},"ISSN":["0929-5593","1573-7527"],"issn-type":[{"value":"0929-5593","type":"print"},{"value":"1573-7527","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8,27]]},"assertion":[{"value":"16 January 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 July 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 August 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose. The authors have no Conflict of interest to declare that are relevant to the content of this article. All authors certify that they have no affiliations with or involvement in any organization or entity with any financial interest or non-financial interest in the subject matter or materials discussed in this manuscript.The authors have no financial or proprietary interests in any material discussed in this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"17"}}