{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T00:18:55Z","timestamp":1787012335580,"version":"3.56.0"},"reference-count":35,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,10,24]],"date-time":"2020-10-24T00:00:00Z","timestamp":1603497600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,10,24]],"date-time":"2020-10-24T00:00:00Z","timestamp":1603497600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,10,24]],"date-time":"2020-10-24T00:00:00Z","timestamp":1603497600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,10,24]]},"DOI":"10.1109\/iros45743.2020.9340905","type":"proceedings-article","created":{"date-parts":[[2021,3,15]],"date-time":"2021-03-15T10:49:56Z","timestamp":1615805396000},"page":"8366-8372","source":"Crossref","is-referenced-by-count":11,"title":["Understanding Contexts Inside Robot and Human Manipulation Tasks through Vision-Language Model and Ontology System in Video Streams"],"prefix":"10.1109","author":[{"given":"Chen","family":"Jiang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Masood","family":"Dehghan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Martin","family":"Jagersand","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.515"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1016\/j.artmed.2017.07.002"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/2757001.2757003"},{"key":"ref30","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"2014"},{"key":"ref35","article-title":"Microsoft coco captions: Data collection and evaluation server","author":"chen","year":"2015"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.334"},{"key":"ref10","article-title":"Visual commonsense for scene understanding using perception, semantic parsing and reasoning","author":"aditya","year":"2015","journal-title":"AAAI Spring Symposium Series"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989535"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2017.12.004"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8460964"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/873"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/HUMANOIDS.2014.7041483"},{"key":"ref16","first-page":"6435","article-title":"A multitask convolutional neural network for autonomous robotic grasping in object stacking scenes","author":"zhang","year":"2018","journal-title":"International Conference on Intelligent Robots and Systems (IROS)"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/COASE.2019.8843293"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794104"},{"key":"ref19","first-page":"8839","article-title":"Towards blended reactive planning and acting using behavior trees","author":"colledanchise","year":"2016","journal-title":"International Conference on Robotics and Automation (ICRA)"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794254"},{"key":"ref4","article-title":"Robot learning and execution of collaborative manipulation plans from youtube videos","author":"zhang","year":"2019"},{"key":"ref27","first-page":"619","article-title":"In the eye of beholder: Joint learning of gaze and actions in first person video","author":"li","year":"2018","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8460857"},{"key":"ref6","first-page":"229","article-title":"Attention is all we need: Nailing down object-centric attention for egocentric activity recognition","author":"sudhakaran","year":"2018","journal-title":"British Machine Vision Conference (BMVC)"},{"key":"ref29","first-page":"2048","article-title":"Show, attend and tell: Neural image caption generation with visual attention","author":"xu","year":"2015","journal-title":"International Conference on Machine Learning"},{"key":"ref5","article-title":"V2cnet: A deep learning framework to translate videos to commands for robotic manipulation","author":"nguyen","year":"2019"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00539"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2019.2901707"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8968278"},{"key":"ref9","article-title":"Sequence to sequence learning with neural networks","author":"sutskever","year":"2014","journal-title":"NIPS"},{"key":"ref1","doi-asserted-by":"crossref","DOI":"10.1609\/aaai.v29i1.9671","article-title":"Robot learning manipulation action plans by&#x201D; watching&#x201D; unconstrained videos from the world wide web","author":"yang","year":"2015","journal-title":"AAAI"},{"key":"ref20","first-page":"3774","article-title":"Interactively picking real-world objects with unconstrained spoken language instructions","author":"hatori","year":"2017","journal-title":"International Conference on Robotics and Automation (ICRA)"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794036"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2018.XIV.028"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8967621"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794441"},{"key":"ref26","first-page":"720","article-title":"Scaling egocentric vision: The epic-kitchens dataset","author":"damen","year":"2018","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref25","article-title":"On the effectiveness of task granularity for transfer learning","author":"mahdisoltani","year":"2018"}],"event":{"name":"2020 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","location":"Las Vegas, NV, USA","start":{"date-parts":[[2020,10,24]]},"end":{"date-parts":[[2021,1,24]]}},"container-title":["2020 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9340668\/9340635\/09340905.pdf?arnumber=9340905","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,21]],"date-time":"2022-12-21T09:52:40Z","timestamp":1671616360000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9340905\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10,24]]},"references-count":35,"URL":"https:\/\/doi.org\/10.1109\/iros45743.2020.9340905","relation":{},"subject":[],"published":{"date-parts":[[2020,10,24]]}}}