{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T06:53:56Z","timestamp":1781592836980,"version":"3.54.5"},"reference-count":68,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T00:00:00Z","timestamp":1778198400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T00:00:00Z","timestamp":1778198400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/100000006","name":"Office of Naval Research","doi-asserted-by":"publisher","award":["N00014-24-1-2550"],"award-info":[{"award-number":["N00014-24-1-2550"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000183","name":"Army Research Office","doi-asserted-by":"publisher","award":["W911NF-17-2-0181"],"award-info":[{"award-number":["W911NF-17-2-0181"]}],"id":[{"id":"10.13039\/100000183","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000185","name":"Defense Advanced Research Projects Agency","doi-asserted-by":"publisher","award":["HR00112520004"],"award-info":[{"award-number":["HR00112520004"]}],"id":[{"id":"10.13039\/100000185","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Science Foundation","award":["FRR2145283"],"award-info":[{"award-number":["FRR2145283"]}]},{"DOI":"10.13039\/100002186","name":"Lockheed Martin","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100002186","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Good Systems"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Robot"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s10514-026-10253-8","type":"journal-article","created":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T01:53:03Z","timestamp":1778205183000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Vision-based manipulation from single human video with open-world object graphs"],"prefix":"10.1007","volume":"50","author":[{"given":"Yifeng","family":"Zhu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Arisrei","family":"Lim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peter","family":"Stone","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuke","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,8]]},"reference":[{"key":"10253_CR1","doi-asserted-by":"crossref","unstructured":"Bharadhwaj, H., Gupta, A., Kumar, V. & Tulsiani, S. (2023). Towards generalizable zero-shot manipulation via translating human interaction plans. arXiv preprint arXiv:2312.00775","DOI":"10.1109\/ICRA57147.2024.10610288"},{"key":"10253_CR2","doi-asserted-by":"crossref","unstructured":"Bahl, S., Gupta, A. & Pathak, D. (2022). Human-to-robot imitation in the wild. arXiv preprint arXiv:2207.09450","DOI":"10.15607\/RSS.2022.XVIII.026"},{"key":"10253_CR3","unstructured":"Bharadhwaj, H., Gupta, A., Tulsiani, S. & Kumar, V. (2023). Zero-shot robot manipulation from passive human videos. arXiv preprint arXiv:2302.02011"},{"key":"10253_CR4","doi-asserted-by":"crossref","unstructured":"Chen, A.S., Nair, S. & Finn, C. (2021). Learning generalizable robotic reward functions from\" in-the-wild\" human videos. arXiv preprint arXiv:2103.16817","DOI":"10.15607\/RSS.2021.XVII.012"},{"key":"10253_CR5","doi-asserted-by":"crossref","unstructured":"Cheng, H.K., Oh, S.W., Price, B., Lee, J.-Y. & Schwing, A. (2023). Putting the object back into video object segmentation. arXiv preprint arXiv:2310.12982","DOI":"10.1109\/CVPR52733.2024.00304"},{"key":"10253_CR6","doi-asserted-by":"crossref","unstructured":"Choi, S., Zhou, Q.-Y. & Koltun, V. (2015). Robust reconstruction of indoor scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5556\u20135565.","DOI":"10.1109\/CVPR.2015.7299195"},{"key":"10253_CR7","doi-asserted-by":"crossref","unstructured":"Devin, C., Abbeel, P., Darrell, T. & Levine, S. (2018). Deep object-centric representations for generalizable robot learning. In: 2018 IEEE International Conference on Robotics and Automation (ICRA), pp. 7111\u20137118 . IEEE.","DOI":"10.1109\/ICRA.2018.8461196"},{"key":"10253_CR8","unstructured":"Duan, Y., Andrychowicz, M., Stadie, B., Jonathan\u00a0Ho, O., Schneider, J., Sutskever, I., Abbeel, P. & Zaremba, W. (2017). One-shot imitation learning. Advances in neural information processing systems 30."},{"key":"10253_CR9","unstructured":"Deepmind, G. (2024). State-of-the-art video and image generation with Veo 2 and Imagen 3. Google Labs blog. https:\/\/blog.google\/technology\/google-labs\/video-image-generation-update-december-2024"},{"key":"10253_CR10","unstructured":"Di\u00a0Palo, N. & Johns, E. (2022). Learning multi-stage tasks with one demonstration via self-replay. In: Conference on Robot Learning. PMLR. pp. 1180\u20131189."},{"key":"10253_CR11","first-page":"21847","volume":"34","author":"M Dalal","year":"2021","unstructured":"Dalal, M., Pathak, D., & Salakhutdinov, R. R. (2021). Accelerating robotic reinforcement learning via parameterized action primitives. Advances in Neural Information Processing Systems, 34, 21847\u201321859.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10253_CR12","doi-asserted-by":"crossref","unstructured":"Heppert, N., Argus, M., Welschehold, T., Brox, T. & Valada, A. (2024). Ditto: Demonstration imitation by trajectory transformation. arXiv preprint arXiv:2403.15203.","DOI":"10.1109\/IROS58592.2024.10801982"},{"key":"10253_CR13","doi-asserted-by":"crossref","unstructured":"Huang, Y., Conkey, A. & Hermans, T. (2023). Planning for multi-object manipulation with graph neural network relational classifiers. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 1822\u20131829. IEEE.","DOI":"10.1109\/ICRA48891.2023.10161204"},{"key":"10253_CR14","unstructured":"Haldar, S., Mathur, V., Yarats, D. & Pinto, L. (2023). Watch and match: Supercharging imitation with regularized optimal transport. In: Conference on Robot Learning, pp. 32\u201343. PMLR."},{"key":"10253_CR15","doi-asserted-by":"crossref","unstructured":"Haldar, S., Pari, J., Rai, A. & Pinto, L. (2023). Teach a robot to fish: Versatile imitation from one minute of demonstrations. arXiv preprint arXiv:2303.01497.","DOI":"10.15607\/RSS.2023.XIX.009"},{"key":"10253_CR16","doi-asserted-by":"crossref","unstructured":"Joseph, K., Khan, S., Khan, F.S. & Balasubramanian, V.N. (2021). Towards open world object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5830\u20135840 .","DOI":"10.1109\/CVPR46437.2021.00577"},{"key":"10253_CR17","doi-asserted-by":"crossref","unstructured":"Johns, E. (2021). Coarse-to-fine imitation learning: Robot manipulation from a single demonstration. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 4613\u20134619. IEEE.","DOI":"10.1109\/ICRA48506.2021.9560942"},{"issue":"500","key":"10253_CR18","doi-asserted-by":"publisher","first-page":"1590","DOI":"10.1080\/01621459.2012.737745","volume":"107","author":"R Killick","year":"2012","unstructured":"Killick, R., Fearnhead, P., & Eckley, I. A. (2012). Optimal detection of changepoints with a linear computational cost. Journal of the American Statistical Association, 107(500), 1590\u20131598.","journal-title":"Journal of the American Statistical Association"},{"key":"10253_CR19","unstructured":"Ko, P.-C., Mao, J., Du, Y., Sun, S.-H. & Tenenbaum, J.B. (2023). Learning to act from actionless videos through dense correspondences. arXiv preprint arXiv:2310.08576"},{"key":"10253_CR20","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.-Y., et al. (2023). Segment anything. arXiv preprint arXiv:2304.02643","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"10253_CR21","doi-asserted-by":"crossref","unstructured":"Ke, B., Narnhofer, D., Huang, S., Ke, L., Peters, T., Fragkiadaki, K., Obukhov, A. & Schindler, K. (2025). Video Depth without Video Models. https:\/\/arxiv.org\/abs\/2411.19189.","DOI":"10.1109\/CVPR52734.2025.00678"},{"key":"10253_CR22","doi-asserted-by":"crossref","unstructured":"Karaev, N., Rocco, I., Graham, B., Neverova, N., Vedaldi, A. & Rupprecht, C. (2023). Cotracker: It is better to track together. arXiv preprint arXiv:2307.07635.","DOI":"10.1007\/978-3-031-73033-7_2"},{"key":"10253_CR23","doi-asserted-by":"crossref","unstructured":"Karnan, H., Torabi, F., Warnell, G. & Stone, P. (2022). Adversarial imitation learning from video using a state observer. In: 2022 International Conference on Robotics and Automation (ICRA), pp. 2452\u20132458. IEEE.","DOI":"10.1109\/ICRA46639.2022.9811570"},{"key":"10253_CR24","unstructured":"Kumar, S., Zamora, J., Hansen, N., Jangir, R. & Wang, X. (2023). Graph inverse reinforcement learning from diverse videos. In: Conference on Robot Learning, pp. 55\u201366. PMLR."},{"key":"10253_CR25","doi-asserted-by":"crossref","unstructured":"Liu, Y., Gupta, A., Abbeel, P. & Levine, S. (2018). Imitation from observation: Learning to imitate behaviors from raw video via context translation. In: 2018 IEEE International Conference on Robotics and Automation (ICRA), pp. 1118\u20131125. IEEE.","DOI":"10.1109\/ICRA.2018.8462901"},{"key":"10253_CR26","doi-asserted-by":"crossref","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Li, C., Yang, J., Su, H., Zhu, J., et al. (2023). Grounding dino: Marrying dino with grounded pre-training for open-set object detection. arXiv preprint arXiv:2303.05499.","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"10253_CR27","unstructured":"Li, J., Zhu, Y., Xie, Y., Jiang, Z., Seo, M., Pavlakos, G. & Zhu, Y. (2024). Okami: Teaching humanoid robots manipulation skills through single video imitation. arXiv preprint arXiv:2410.11792"},{"issue":"2","key":"10253_CR28","doi-asserted-by":"publisher","first-page":"844","DOI":"10.1109\/LRA.2020.2965875","volume":"5","author":"T Migimatsu","year":"2020","unstructured":"Migimatsu, T., & Bohg, J. (2020). Object-centric task and motion planning in dynamic environments. IEEE Robotics and Automation Letters, 5(2), 844\u2013851.","journal-title":"IEEE Robotics and Automation Letters"},{"key":"10253_CR29","doi-asserted-by":"crossref","unstructured":"Mo, K., Guerrero, P., Yi, L., Su, H., Wonka, P., Mitra, N. & Guibas, L.J. (2019). Structurenet: Hierarchical graph networks for 3d shape generation. arXiv preprint arXiv:1908.00575.","DOI":"10.1145\/3355089.3356527"},{"key":"10253_CR30","unstructured":"Ma, Y.J., Sodhani, S., Jayaraman, D., Bastani, O., Kumar, V. & Zhang, A. (2022). Vip: Towards universal visual reward and representation via value-implicit pre-training. arXiv preprint arXiv:2210.00030"},{"key":"10253_CR31","doi-asserted-by":"crossref","unstructured":"Mandlekar, A., Xu, D., Mart\u00edn-Mart\u00edn, R., Savarese, S. & Fei-Fei, L. (2020). Learning to generalize across long-horizon tasks from human demonstrations. arXiv preprint arXiv:2003.06085","DOI":"10.15607\/RSS.2020.XVI.061"},{"key":"10253_CR32","doi-asserted-by":"crossref","unstructured":"Nasiriany, S., Liu, H. & Zhu, Y. (2022). Augmenting reinforcement learning with behavior primitives for diverse manipulation tasks. In: 2022 International Conference on Robotics and Automation (ICRA), pp. 7477\u20137484. IEEE.","DOI":"10.1109\/ICRA46639.2022.9812140"},{"key":"10253_CR33","unstructured":"Nair, S., Rajeswaran, A., Kumar, V., Finn, C. & Gupta, A. (2022). R3m: A universal visual representation for robot manipulation. arXiv preprint arXiv:2203.12601 ."},{"key":"10253_CR34","unstructured":"Oquab, M., Darcet, T., Moutakanni, T., Vo, H., Szafraniec, M., Khalidov, V., Fernandez, P., Haziza, D., Massa, F., El-Nouby, A., et al. (2023). Dinov2: Learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 ."},{"key":"10253_CR35","unstructured":"Park, S., Bharadhwaj, H. & Tulsiani, S. (2025). Demodiffusion: One-shot human imitation using pre-trained diffusion policy. arXiv preprint arXiv:2506.20668"},{"key":"10253_CR36","unstructured":"Patel, S., Mohan, S., Mai, H., Jain, U., Lazebnik, S. & Li, Y. (2025). Robotic manipulation by imitating generated videos without physical demonstrations. arXiv preprint arXiv:2507.00990"},{"key":"10253_CR37","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Shan, D., Radosavovic, I., Kanazawa, A., Fouhey, D. & Malik, J. (2023). Reconstructing hands in 3d with transformers. arXiv preprint arXiv:2312.05251","DOI":"10.1109\/CVPR52733.2024.00938"},{"issue":"4","key":"10253_CR38","doi-asserted-by":"publisher","first-page":"6262","DOI":"10.1109\/LRA.2020.3010750","volume":"5","author":"BS Pavse","year":"2020","unstructured":"Pavse, B. S., Torabi, F., Hanna, J., Warnell, G., & Stone, P. (2020). Ridm: Reinforced inverse dynamics modeling for learning from a single observed demonstration. IEEE Robotics and Automation Letters, 5(4), 6262\u20136269.","journal-title":"IEEE Robotics and Automation Letters"},{"key":"10253_CR39","doi-asserted-by":"crossref","unstructured":"Qureshi, A.H., Mousavian, A., Paxton, C., Yip, M.C. & Fox, D. (2021). Nerp: Neural rearrangement planning for unknown objects. arXiv preprint arXiv:2106.01352.","DOI":"10.15607\/RSS.2021.XVII.072"},{"key":"10253_CR40","doi-asserted-by":"crossref","unstructured":"Rusu, R.B., Blodow, N. & Beetz, M. (2009). Fast point feature histograms (fpfh) for 3d registration. In: 2009 IEEE International Conference on Robotics and Automation, pp. 3212\u20133217. IEEE.","DOI":"10.1109\/ROBOT.2009.5152473"},{"key":"10253_CR41","doi-asserted-by":"crossref","unstructured":"Smith, L., Dhawan, N., Zhang, M., Abbeel, P. & Levine, S. (2019). Avid: Learning multi-stage tasks via pixel-level translation of human videos. arXiv preprint arXiv:1912.04443 .","DOI":"10.15607\/RSS.2020.XVI.024"},{"key":"10253_CR42","unstructured":"Sharma, P., Pathak, D. & Gupta, A. (2019). Third-person visual imitation learning via decoupled hierarchical controller. Advances in Neural Information Processing Systems 32."},{"key":"10253_CR43","doi-asserted-by":"crossref","unstructured":"Shi, J., Qian, J., Ma, Y.J. & Jayaraman, D. (2024). Plug-and-play object-centric representations from \u201cwhat\u201d and \u201cwhere\u201d foundation models","DOI":"10.1109\/ICRA57147.2024.10610695"},{"key":"10253_CR44","unstructured":"Stone, A., Xiao, T., Lu, Y., Gopalakrishnan, K., Lee, K.-H., Vuong, Q., Wohlhart, P., Zitkovich, B., Xia, F., Finn, C., et al. (2023). Open-world object manipulation using pre-trained vision-language models. arXiv preprint arXiv:2303.00905"},{"key":"10253_CR45","doi-asserted-by":"crossref","unstructured":"Teed, Z. & Deng, J. (2021). Tangent space backpropagation for 3d transformation groups. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10338\u201310347.","DOI":"10.1109\/CVPR46437.2021.01020"},{"key":"10253_CR46","unstructured":"Torabi, F. (2021). Imitation learning from observation. PhD thesis, University of Texas at Austin. PhD Thesis."},{"key":"10253_CR47","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2019.107299","volume":"167","author":"C Truong","year":"2020","unstructured":"Truong, C., Oudre, L., & Vayatis, N. (2020). Selective review of offline change point detection methods. Signal Processing, 167, Article 107299.","journal-title":"Signal Processing"},{"key":"10253_CR48","unstructured":"Tremblay, J., To, T., Sundaralingam, B., Xiang, Y., Fox, D. & Birchfield, S. (2018). Deep object pose estimation for semantic robotic grasping of household objects. arXiv preprint arXiv:1809.10790."},{"key":"10253_CR49","doi-asserted-by":"crossref","unstructured":"Tyree, S., Tremblay, J., To, T., Cheng, J., Mosier, T., Smith, J. & Birchfield, S. (2022). 6-dof pose estimation of household objects for robotic manipulation: An accessible dataset and benchmark. arXiv preprint arXiv:2203.05701","DOI":"10.1109\/IROS47612.2022.9981838"},{"key":"10253_CR50","doi-asserted-by":"crossref","unstructured":"Torabi, F., Warnell, G. & Stone, P. (2018). Behavioral cloning from observation. In: Proceedings of the 27th International Joint Conference on Artificial Intelligence, pp. 4950\u20134957.","DOI":"10.24963\/ijcai.2018\/687"},{"key":"10253_CR51","unstructured":"Torabi, F., Warnell, G. & Stone, P. (2019). Generative adversarial imitation from observation. In: Imitation, Intent, and Interaction (I3) Workshop at ICML 2019 ."},{"key":"10253_CR52","doi-asserted-by":"crossref","unstructured":"Torabi, F., Warnell, G. & Stone, P. (2019). Imitation learning from video by leveraging proprioception. In: Proceedings of the 28th International Joint Conference on Artificial Intelligence, pp. 3585\u20133591.","DOI":"10.24963\/ijcai.2019\/497"},{"key":"10253_CR53","doi-asserted-by":"crossref","unstructured":"Vecerik, M., Doersch, C., Yang, Y., Davchev, T., Aytar, Y., Zhou, G., Hadsell, R., Agapito, L. & Scholz, J. (2023). Robotap: Tracking arbitrary points for few-shot visual imitation. arXiv preprint arXiv:2308.15975.","DOI":"10.1109\/ICRA57147.2024.10611409"},{"key":"10253_CR54","doi-asserted-by":"crossref","unstructured":"Valassakis, E., Papagiannis, G., Di\u00a0Palo, N. & Johns, E. (2022). Demonstrate once, imitate immediately (dome): Learning visual servoing for one-shot imitation learning. In: 2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 8614\u20138621 IEEE.","DOI":"10.1109\/IROS47612.2022.9981982"},{"key":"10253_CR55","doi-asserted-by":"crossref","unstructured":"Wang, D., Devin, C., Cai, Q.-Z., Yu, F. & Darrell, T. (2019). Deep object-centric policies for autonomous driving. In: 2019 International Conference on Robotics and Automation (ICRA), pp. 8853\u20138859. IEEE.","DOI":"10.1109\/ICRA.2019.8794224"},{"key":"10253_CR56","unstructured":"Wang, C., Fan, L., Sun, J., Zhang, R., Fei-Fei, L., Xu, D., Zhu, Y. & Anandkumar, A. (2023). Mimicplay: Long-horizon imitation learning by watching human play. arXiv preprint arXiv:2302.12422"},{"key":"10253_CR57","doi-asserted-by":"crossref","unstructured":"Wen, B., Lian, W., Bekris, K., & Schaal, S. (2022). You only demonstrate once: Category-level manipulation from single visual demonstration. arXiv preprint arXiv:2201.12716","DOI":"10.15607\/RSS.2022.XVIII.044"},{"key":"10253_CR58","doi-asserted-by":"crossref","unstructured":"Wen, C., Lin, X., So, J., Chen, K., Dou, Q., Gao, Y. & Abbeel, P. (2023). Any-point trajectory modeling for policy learning. arXiv preprint arXiv:2401.00025.","DOI":"10.15607\/RSS.2024.XX.092"},{"key":"10253_CR59","doi-asserted-by":"crossref","unstructured":"Wen, B., Yang, W., Kautz, J. & Birchfield, S. (2024). Foundationpose: Unified 6d pose estimation and tracking of novel objects. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17868\u201317879 .","DOI":"10.1109\/CVPR52733.2024.01692"},{"key":"10253_CR60","unstructured":"Xu, J., Cheng, W., Gao, Y., Wang, X., Gao, S. & Shan, Y. (2024). Instantmesh: Efficient 3d mesh generation from a single image with sparse-view large reconstruction models. arXiv preprint arXiv:2404.07191."},{"key":"10253_CR61","doi-asserted-by":"crossref","unstructured":"Xiong, H., Li, Q., Chen, Y.-C., Bharadhwaj, H., Sinha, S. & Garg, A. (2021). Learning by watching: Physical imitation of manipulation skills from human videos. In: 2021 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 7827\u20137834 IEEE","DOI":"10.1109\/IROS51168.2021.9636080"},{"key":"10253_CR62","unstructured":"Xu, M., Xu, Z., Chi, C., Veloso, M. & Song, S. (2023). Xskill: Cross embodiment skill discovery. In: Conference on Robot Learning, pp. 3536\u20133555 PMLR."},{"key":"10253_CR63","unstructured":"Zhu, Y., Joshi, A., Stone, P. & Zhu, Y. (2022). Viola: Imitation learning for vision-based manipulation with object proposal priors. arXiv preprint arXiv:2210.11339"},{"key":"10253_CR64","unstructured":"Zhu, Y., Jiang, Z., Stone, P. & Zhu, Y. (2023). Learning generalizable manipulation policies with object-centric 3d representations. In: 7th Annual Conference on Robot Learning."},{"key":"10253_CR65","unstructured":"Zhang, R., Lee, S., Hwang, M., Hiranaka, A., Wang, C., Ai, W., Tan, J.J.R., Gupta, S., Hao, Y., Levine, G., et al. (2023). Noir: Neural signal operated intelligent robots for everyday activities. arXiv preprint arXiv:2311.01454."},{"key":"10253_CR66","unstructured":"Zhou, Q.-Y., Park, J. & Koltun, V. (2018). Open3d: A modern library for 3d data processing. arXiv preprint arXiv:1801.09847"},{"key":"10253_CR67","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Tremblay, J., Birchfield, S. & Zhu, Y. (2021). Hierarchical planning for long-horizon manipulation with geometric and symbolic scene graphs. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 6541\u20136548 IEEE.","DOI":"10.1109\/ICRA48506.2021.9561548"},{"key":"10253_CR68","unstructured":"Zhu, Y., Wong, J., Mandlekar, A., Mart\u00edn-Mart\u00edn, R., Joshi, A., Nasiriany, S., Zhu, Y. & Lin, K. (2020). robosuite: A modular simulation framework and benchmark for robot learning. In: arXiv Preprint arXiv:2009.12293."}],"container-title":["Autonomous Robots"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10514-026-10253-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10514-026-10253-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10514-026-10253-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T06:20:14Z","timestamp":1781590814000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10514-026-10253-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,8]]},"references-count":68,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["10253"],"URL":"https:\/\/doi.org\/10.1007\/s10514-026-10253-8","relation":{},"ISSN":["0929-5593","1573-7527"],"issn-type":[{"value":"0929-5593","type":"print"},{"value":"1573-7527","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,8]]},"assertion":[{"value":"6 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 February 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 March 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 May 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Peter Stone serves as the Chief Scientist of Sony AI and receives financial compensation for that role. The terms of this arrangement have been reviewed and approved by the University of Texas at Austin in accordance with its policy on objectivity in research.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"27"}}