{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T21:26:40Z","timestamp":1783632400748,"version":"3.55.0"},"reference-count":83,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T00:00:00Z","timestamp":1782950400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100001537","name":"University of Auckland","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001537","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004543","name":"China Scholarship Council","doi-asserted-by":"publisher","award":["202306290029"],"award-info":[{"award-number":["202306290029"]}],"id":[{"id":"10.13039\/501100004543","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Robotics and Computer-Integrated Manufacturing"],"published-print":{"date-parts":[[2027,2]]},"DOI":"10.1016\/j.rcim.2026.103371","type":"journal-article","created":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T16:45:19Z","timestamp":1783442719000},"page":"103371","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Video2Knowledge: Extracting temporally consistent task knowledge from monocular video for robot skill learning"],"prefix":"10.1016","volume":"103","author":[{"given":"Jinyi","family":"Huang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yinwang","family":"Ren","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenqing","family":"Wan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xun","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8513-7302","authenticated-orcid":false,"given":"Jan","family":"Polzer","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.rcim.2026.103371_b1","doi-asserted-by":"crossref","first-page":"259","DOI":"10.1016\/j.jmsy.2025.12.001","article-title":"Human-centric manufacturing: Re-thinking, Re-justifying, and Re-envisioning","volume":"84","author":"Xu","year":"2026","journal-title":"J. Manuf. Syst."},{"key":"10.1016\/j.rcim.2026.103371_b2","doi-asserted-by":"crossref","DOI":"10.1016\/j.rcim.2022.102510","article-title":"Proactive human\u2013robot collaboration: Mutual-cognitive, predictable, and self-organising perspectives","volume":"81","author":"Li","year":"2023","journal-title":"Robot. Comput.-Integr. Manuf."},{"key":"10.1016\/j.rcim.2026.103371_b3","doi-asserted-by":"crossref","first-page":"349","DOI":"10.1016\/j.jmsy.2024.02.010","article-title":"Unlocking the power of industrial artificial intelligence towards Industry 5.0: Insights, pathways, and challenges","volume":"73","author":"Leng","year":"2024","journal-title":"J. Manuf. Syst."},{"key":"10.1016\/j.rcim.2026.103371_b4","doi-asserted-by":"crossref","first-page":"29","DOI":"10.1016\/j.jmsy.2025.08.018","article-title":"Towards instructional collaborative robots: From video-based learning to feedback-adapted instruction","volume":"83","author":"Huang","year":"2025","journal-title":"J. Manuf. Syst."},{"key":"10.1016\/j.rcim.2026.103371_b5","doi-asserted-by":"crossref","DOI":"10.1016\/j.rcim.2025.103148","article-title":"From perception to precision: Vision-based mobile robotic manipulation for assembly screwdriving","volume":"98","author":"Stefanov","year":"2026","journal-title":"Robot. Comput.-Integr. Manuf."},{"key":"10.1016\/j.rcim.2026.103371_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.rcim.2025.103064","article-title":"Empowering natural human\u2013robot collaboration through multimodal language models and spatial intelligence: Pathways and perspectives","volume":"97","author":"Wu","year":"2026","journal-title":"Robot. Comput.-Integr. Manuf."},{"key":"10.1016\/j.rcim.2026.103371_b7","series-title":"Conference on Robot Learning","first-page":"2679","article-title":"OpenVLA: An open-source vision-language-action model","author":"Kim","year":"2025"},{"key":"10.1016\/j.rcim.2026.103371_b8","series-title":"European Conference on Computer Vision","first-page":"127","article-title":"Adaptive computationally efficient network for monocular 3d hand pose estimation","author":"Fan","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b9","series-title":"European Conference on Computer Vision","first-page":"548","article-title":"InterHand2.6M: A dataset and baseline for 3d interacting hand pose estimation from a single rgb image","author":"Moon","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b10","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"6202","article-title":"Slowfast networks for video recognition","author":"Feichtenhofer","year":"2019"},{"key":"10.1016\/j.rcim.2026.103371_b11","series-title":"European Conference on Computer Vision","first-page":"769","article-title":"Collaborative learning of gesture recognition and 3D hand pose estimation with multi-order feature analysis","author":"Yang","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b12","series-title":"2020 IEEE International Conference on Robotics and Automation","first-page":"6210","article-title":"Robust, occlusion-aware pose estimation for objects grasped by adaptive hands","author":"Wen","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b13","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"3560","article-title":"DualPoseNet: Category-level 6d object pose and size estimation using dual pose network with refined learning of pose consistency","author":"Lin","year":"2021"},{"key":"10.1016\/j.rcim.2026.103371_b14","series-title":"2025 International Conference on 3D Vision","first-page":"1427","article-title":"InterTrack: Tracking human object interaction without object templates","author":"Xie","year":"2025"},{"key":"10.1016\/j.rcim.2026.103371_b15","doi-asserted-by":"crossref","DOI":"10.1016\/j.rcim.2023.102686","article-title":"Human-robot shared assembly taxonomy: A step toward seamless human-robot knowledge transfer","volume":"86","author":"Lee","year":"2024","journal-title":"Robot. Comput.-Integr. Manuf."},{"key":"10.1016\/j.rcim.2026.103371_b16","doi-asserted-by":"crossref","first-page":"67069","DOI":"10.52202\/075280-2930","article-title":"HA-ViD: A human assembly video dataset for comprehensive assembly knowledge understanding","volume":"36","author":"Zheng","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.rcim.2026.103371_b17","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"21096","article-title":"Assembly101: A large-scale multi-view video dataset for understanding procedural activities","author":"Sener","year":"2022"},{"key":"10.1016\/j.rcim.2026.103371_b18","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10843","article-title":"3D hand shape and pose from images in the wild","author":"Boukhayma","year":"2019"},{"issue":"6","key":"10.1016\/j.rcim.2026.103371_b19","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3130800.3130883","article-title":"Embodied hands: modeling and capturing hands and bodies together","volume":"36","author":"Romero","year":"2017","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.rcim.2026.103371_b20","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"2354","article-title":"End-to-end hand mesh recovery from a monocular rgb image","author":"Zhang","year":"2019"},{"key":"10.1016\/j.rcim.2026.103371_b21","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"4990","article-title":"Weakly-supervised mesh-convolutional hand reconstruction in the wild","author":"Kulon","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b22","series-title":"European Conference on Computer Vision","first-page":"211","article-title":"Weakly supervised 3d hand pose estimation via biomechanical constraints","author":"Spurr","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b23","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1496","article-title":"HandOccNet: Occlusion-robust 3d hand mesh estimation network","author":"Park","year":"2022"},{"key":"10.1016\/j.rcim.2026.103371_b24","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"17028","article-title":"Bringing inputs to shared domains for 3D interacting hands recovery in the wild","author":"Moon","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b25","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"203","article-title":"X3D: Expanding architectures for efficient video recognition","author":"Feichtenhofer","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b26","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"4511","article-title":"H+O: Unified egocentric recognition of 3d hand-object poses and interactions","author":"Tekin","year":"2019"},{"key":"10.1016\/j.rcim.2026.103371_b27","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"10138","article-title":"H2O: Two hands manipulating objects for first person interaction recognition","author":"Kwon","year":"2021"},{"key":"10.1016\/j.rcim.2026.103371_b28","doi-asserted-by":"crossref","DOI":"10.1016\/j.rcim.2025.102976","article-title":"A human-robot collaborative assembly framework with quality checking based on real-time dual-hand action segmentation","volume":"94","author":"Zheng","year":"2025","journal-title":"Robot. Comput.-Integr. Manuf."},{"key":"10.1016\/j.rcim.2026.103371_b29","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"4769","article-title":"Transformer-based unified recognition of two hands manipulating objects","author":"Cho","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b30","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"12999","article-title":"AssemblyHands: Towards egocentric activity understanding via 3d hand pose estimation","author":"Ohkawa","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b31","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"7083","article-title":"TSM: Temporal shift module for efficient video understanding","author":"Lin","year":"2019"},{"key":"10.1016\/j.rcim.2026.103371_b32","series-title":"Conference on Robot Learning","first-page":"715","article-title":"MegaPose: 6D pose estimation of novel objects via render & compare","author":"Labb\u00e9","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b33","series-title":"European Conference on Computer Vision","first-page":"163","article-title":"FoundPose: Unseen object pose estimation with foundation features","author":"\u00d6rnek","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b34","article-title":"DINOv2: Learning robust visual features without supervision","author":"Oquab","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.rcim.2026.103371_b35","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"13916","article-title":"Multi-path learning for object pose estimation across domains","author":"Sundermeyer","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b36","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6835","article-title":"OSOP: A multi-stage one shot object pose estimation framework","author":"Shugurov","year":"2022"},{"key":"10.1016\/j.rcim.2026.103371_b37","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"9903","article-title":"GigaPose: Fast and robust novel object pose estimation via one correspondence","author":"Nguyen","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b38","doi-asserted-by":"crossref","first-page":"1251","DOI":"10.1109\/TCSVT.2024.3482439","article-title":"ZeroPose: CAD-prompted zero-shot object 6d pose estimation in cluttered scenes","volume":"35","author":"Chen","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.rcim.2026.103371_b39","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6825","article-title":"OnePose: One-shot object pose estimation without CAD models","author":"Sun","year":"2022"},{"key":"10.1016\/j.rcim.2026.103371_b40","doi-asserted-by":"crossref","first-page":"35103","DOI":"10.52202\/068431-2544","article-title":"OnePose++: Keypoint-free one-shot object pose estimation without CAD models","volume":"35","author":"He","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.rcim.2026.103371_b41","series-title":"European Conference on Computer Vision","first-page":"298","article-title":"Gen6D: Generalizable model-free 6-dof object pose estimation from rgb images","author":"Liu","year":"2022"},{"key":"10.1016\/j.rcim.2026.103371_b42","series-title":"Conference on Robot Learning","first-page":"1060","article-title":"PoET: Pose estimation transformer for single-view, multi-object 6D pose estimation","author":"Jantos","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b43","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"17868","article-title":"FoundationPose: Unified 6d pose estimation and tracking of novel objects","author":"Wen","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b44","series-title":"Conference on Robot Learning","first-page":"168","article-title":"One view, many worlds: Single-image to 3D object meets generative domain randomization for one-shot 6D pose estimation","author":"Geng","year":"2025"},{"key":"10.1016\/j.rcim.2026.103371_b45","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"25050","article-title":"Hi3DGen: High-fidelity 3d geometry generation from images via normal bridging","author":"Ye","year":"2025"},{"key":"10.1016\/j.rcim.2026.103371_b46","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"21469","article-title":"Structured 3d latents for scalable and versatile 3d generation","author":"Xiang","year":"2025"},{"key":"10.1016\/j.rcim.2026.103371_b47","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"4896","article-title":"VP3D: Unleashing 2d visual prompt for text-to-3d generation","author":"Chen","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b48","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"1237","article-title":"IT3D: Improved text-to-3d generation with explicit view synthesis","volume":"Vol. 38","author":"Chen","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b49","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"7320","article-title":"Cycle3D: High-quality and consistent image-to-3d generation via generation-reconstruction cycle","volume":"Vol. 39","author":"Tang","year":"2025"},{"key":"10.1016\/j.rcim.2026.103371_b50","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"12179","article-title":"Vision transformers for dense prediction","author":"Ranftl","year":"2021"},{"issue":"3","key":"10.1016\/j.rcim.2026.103371_b51","doi-asserted-by":"crossref","first-page":"5405","DOI":"10.1109\/LRA.2021.3067308","article-title":"Three-filters-to-normal: An accurate and ultrafast surface normal estimator","volume":"6","author":"Fan","year":"2021","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.rcim.2026.103371_b52","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"12991","article-title":"Planar surface reconstruction from sparse views","author":"Jin","year":"2021"},{"key":"10.1016\/j.rcim.2026.103371_b53","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"55","article-title":"Total3DUnderstanding: Joint layout, object pose and mesh reconstruction for indoor scenes from a single image","author":"Nie","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b54","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10218","article-title":"Joint reconstruction of 3d human and object via contact-based refinement transformer","author":"Nam","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b55","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1911","article-title":"G-HOP: Generative hand-object prior for interaction reconstruction and grasp synthesis","author":"Ye","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b56","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10003","article-title":"Template free reconstruction of human-object interaction with procedural interaction generation","author":"Xie","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b57","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8834","article-title":"NeuralDome: A neural modeling pipeline on multi-view human-object interactions","author":"Zhang","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b58","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"595","article-title":"Instant-NVR: Instant neural volumetric rendering for human-object interactions from monocular rgbd stream","author":"Jiang","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b59","series-title":"European Conference on Computer Vision","first-page":"125","article-title":"CHORE: Contact, human and object reconstruction from a single rgb image","author":"Xie","year":"2022"},{"key":"10.1016\/j.rcim.2026.103371_b60","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"571","article-title":"Leveraging photometric consistency over time for sparsely supervised hand-object reconstruction","author":"Hasson","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b61","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"19717","article-title":"Diffusion-guided reconstruction of everyday hand-object interaction clips","author":"Ye","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b62","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"494","article-title":"HOLD: Category-agnostic 3d reconstruction of interacting hands and objects from video","author":"Fan","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b63","series-title":"2024 International Conference on 3D Vision","first-page":"1006","article-title":"Interaction replica: Tracking human\u2013object interaction and scene changes from human motion","author":"Guzov","year":"2024"},{"key":"10.1016\/j.rcim.2026.103371_b64","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"4757","article-title":"Visibility aware human-object interaction tracking from single rgb camera","author":"Xie","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b65","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18995","article-title":"Ego4D: Around the world in 3,000 hours of egocentric video","author":"Grauman","year":"2022"},{"key":"10.1016\/j.rcim.2026.103371_b66","series-title":"Robotics: Science and Systems","article-title":"Robotic telekinesis: Learning a robotic hand imitator by watching humans on youtube","author":"Sivakumar","year":"2022"},{"key":"10.1016\/j.rcim.2026.103371_b67","series-title":"International Conference on Machine Learning","first-page":"23301","article-title":"LIV: Language-image representations and rewards for robotic control","author":"Ma","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b68","series-title":"Conference on Robot Learning","first-page":"416","article-title":"Real-world robot learning with masked visual pre-training","author":"Radosavovic","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b69","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"1749","article-title":"FrankMocap: A monocular 3d whole-body pose estimation system via regression and integration","author":"Rong","year":"2021"},{"issue":"4","key":"10.1016\/j.rcim.2026.103371_b70","doi-asserted-by":"crossref","first-page":"513","DOI":"10.1177\/02783649241227559","article-title":"Learning dexterity from human hand motion in internet videos","volume":"43","author":"Shaw","year":"2024","journal-title":"Int. J. Robot. Res."},{"key":"10.1016\/j.rcim.2026.103371_b71","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"9869","article-title":"Understanding human hands in contact at internet scale","author":"Shan","year":"2020"},{"key":"10.1016\/j.rcim.2026.103371_b72","series-title":"2024 IEEE International Conference on Robotics and Automation","first-page":"6904","article-title":"Towards generalizable zero-shot manipulation via translating human interaction plans","author":"Bharadhwaj","year":"2024"},{"issue":"30","key":"10.1016\/j.rcim.2026.103371_b73","first-page":"1","article-title":"A review of robot learning for manipulation: Challenges, representations, and algorithms","volume":"22","author":"Kroemer","year":"2021","journal-title":"J. Mach. Learn. Res. JMLR"},{"key":"10.1016\/j.rcim.2026.103371_b74","series-title":"2022 International Conference on Robotics and Automation","first-page":"8591","article-title":"Learning sensorimotor primitives of sequential manipulation tasks from visual demonstrations","author":"Liang","year":"2022"},{"issue":"6\u20137","key":"10.1016\/j.rcim.2026.103371_b75","doi-asserted-by":"crossref","first-page":"866","DOI":"10.1177\/02783649211004615","article-title":"Learning compositional models of robot skills for task and motion planning","volume":"40","author":"Wang","year":"2021","journal-title":"Int. J. Robot. Res."},{"issue":"37","key":"10.1016\/j.rcim.2026.103371_b76","doi-asserted-by":"crossref","first-page":"eaay4663","DOI":"10.1126\/scirobotics.aay4663","article-title":"A tale of two explanations: Enhancing human trust by explaining robot behavior","volume":"4","author":"Edmonds","year":"2019","journal-title":"Sci. Robot."},{"key":"10.1016\/j.rcim.2026.103371_b77","series-title":"2021 IEEE International Conference on Robotics and Automation","first-page":"4679","article-title":"Learning multimodal contact-rich skills from demonstrations without reward engineering","author":"Balakuntala","year":"2021"},{"key":"10.1016\/j.rcim.2026.103371_b78","series-title":"Conference on Robot Learning","first-page":"201","article-title":"MimicPlay: Long-horizon imitation learning by watching human play","author":"Wang","year":"2023"},{"key":"10.1016\/j.rcim.2026.103371_b79","unstructured":"A. Levy, G. Konidaris, R. Platt, K. Saenko, Learning multi-level hierarchies with hindsight, in: Proceedings of International Conference on Learning Representations, 2019."},{"key":"10.1016\/j.rcim.2026.103371_b80","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"12242","article-title":"WiLoR: End-to-end 3d hand localization and reconstruction in-the-wild","author":"Potamias","year":"2025"},{"key":"10.1016\/j.rcim.2026.103371_b81","doi-asserted-by":"crossref","unstructured":"K. Li, Y. Wang, Y. He, Y. Li, Y. Wang, L. Wang, Y. Qiao, UniFormerV2: Unlocking the potential of image vits for video understanding, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 1632\u20131643, http:\/\/dx.doi.org\/10.1109\/iccv51070.2023.00157.","DOI":"10.1109\/ICCV51070.2023.00157"},{"key":"10.1016\/j.rcim.2026.103371_b82","doi-asserted-by":"crossref","unstructured":"L. Wang, B. Huang, Z. Zhao, Z. Tong, Y. He, Y. Wang, Y. Wang, Y. Qiao, VideoMAE V2: Scaling video masked autoencoders with dual masking, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 14549\u201314560, http:\/\/dx.doi.org\/10.1109\/cvpr52729.2023.01398.","DOI":"10.1109\/CVPR52729.2023.01398"},{"key":"10.1016\/j.rcim.2026.103371_b83","series-title":"2015 IEEE-RAS 15th International Conference on Humanoid Robots","first-page":"928","article-title":"TRAC-IK: An open-source library for improved solving of generic inverse kinematics","author":"Beeson","year":"2015"}],"container-title":["Robotics and Computer-Integrated Manufacturing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0736584526001936?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0736584526001936?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T21:11:30Z","timestamp":1783631490000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0736584526001936"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,2]]},"references-count":83,"alternative-id":["S0736584526001936"],"URL":"https:\/\/doi.org\/10.1016\/j.rcim.2026.103371","relation":{"is-supplemented-by":[{"id-type":"uri","id":"https:\/\/github.com\/jhua528\/video2knowledge-processed-data","asserted-by":"subject"}]},"ISSN":["0736-5845"],"issn-type":[{"value":"0736-5845","type":"print"}],"subject":[],"published":{"date-parts":[[2027,2]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Video2Knowledge: Extracting temporally consistent task knowledge from monocular video for robot skill learning","name":"articletitle","label":"Article Title"},{"value":"Robotics and Computer-Integrated Manufacturing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.rcim.2026.103371","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"103371"}}