{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T16:29:17Z","timestamp":1783096157915,"version":"3.54.6"},"reference-count":102,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21A20488"],"award-info":[{"award-number":["U21A20488"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62072323"],"award-info":[{"award-number":["62072323"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhejiang Lab Open Research Project","award":["K2022NB0AB04"],"award-info":[{"award-number":["K2022NB0AB04"]}]},{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["22511105902"],"award-info":[{"award-number":["22511105902"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Postdoctoral Fellowship Program of CPSF","award":["GZC20232292"],"award-info":[{"award-number":["GZC20232292"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Knowl. Data Eng."],"published-print":{"date-parts":[[2024,11]]},"DOI":"10.1109\/tkde.2024.3399746","type":"journal-article","created":{"date-parts":[[2024,5,16]],"date-time":"2024-05-16T17:33:56Z","timestamp":1715880836000},"page":"6962-6976","source":"Crossref","is-referenced-by-count":45,"title":["Scene-Driven Multimodal Knowledge Graph Construction for Embodied AI"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8146-2236","authenticated-orcid":false,"given":"Yaoxian","family":"Song","sequence":"first","affiliation":[{"name":"Shanghai Key Laboratory of Data Science, School of Computer Science, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2290-0944","authenticated-orcid":false,"given":"Penglei","family":"Sun","sequence":"additional","affiliation":[{"name":"Shanghai Key Laboratory of Data Science, School of Computer Science, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8246-7056","authenticated-orcid":false,"given":"Haoyu","family":"Liu","sequence":"additional","affiliation":[{"name":"Research Center for Intelligent Robotics, Zhejiang Lab, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2355-288X","authenticated-orcid":false,"given":"Zhixu","family":"Li","sequence":"additional","affiliation":[{"name":"Shanghai Key Laboratory of Data Science, School of Computer Science, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0828-7486","authenticated-orcid":false,"given":"Wei","family":"Song","sequence":"additional","affiliation":[{"name":"Research Center for Intelligent Robotics, Zhejiang Lab, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8403-9591","authenticated-orcid":false,"given":"Yanghua","family":"Xiao","sequence":"additional","affiliation":[{"name":"Shanghai Key Laboratory of Data Science, School of Computer Science, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6343-1455","authenticated-orcid":false,"given":"Xiaofang","family":"Zhou","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science, Technology, Hong Kong, SAR, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.tics.2003.09.006"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.array.2021.100057"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/nrn2277"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.newideapsych.2021.100916"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/219717.219748"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1023\/B:BTTJ.0000047600.45421.6d"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/2629489"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3224228"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE55515.2023.00229"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2022.104294"},{"key":"ref13","article-title":"PP-YOLOE: An evolved version of YOLO","author":"Xu","year":"2022"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.5555\/3045118.3045336"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.2002.1021896"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298990"},{"key":"ref17","first-page":"1","article-title":"SQA3D: Situated question answering in 3D scenes","volume-title":"Proc. 11th Int. Conf. Learn. Representations","author":"Ma"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/access.2024.3387941"},{"key":"ref20","article-title":"ChatGPT: Optimizing language models for dialogue","year":"2023"},{"key":"ref21","article-title":"GPT-4 technical report","author":"Achiam","year":"2023"},{"key":"ref22","first-page":"8469","article-title":"PALM-E: An embodied multimodal language model","author":"Driess","year":"2023","journal-title":"Proc. Int. Conf. Mach. Learn."},{"key":"ref23","article-title":"ChatGPT is a knowledgeable but inexperienced solver: An investigation of commonsense problem in large language models","author":"Bian","year":"2023"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3150080"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/2213836.2213891"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i7.16792"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/2556195.2556245"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1111\/j.1745-6622.1995.tb00283.x"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1038\/s41928-020-0422-z"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.308"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482470"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/n19-1306"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1177\/02783649922066501"},{"key":"ref35","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-540-30301-5","volume-title":"Springer Handbook of Robotics","volume":"200","author":"Siciliano","year":"2008"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/4527.001.0001"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2019.2893920"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3220625"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE55515.2023.00379"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00387"},{"key":"ref41","article-title":"Retrospectives on the embodied ai workshop","author":"Deitke","year":"2022"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref43","first-page":"1691","article-title":"Language grounding with 3D objects","volume-title":"Proc. Conf. Robot Learn.","author":"Thomason"},{"key":"ref44","first-page":"13","article-title":"VilBERT: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lu"},{"key":"ref45","article-title":"LegoFormer: Transformers for block-by-block multi-view 3D reconstruction","author":"Yagubbayli","year":"2021"},{"key":"ref46","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023","journal-title":"Proc. Int. Conf. Mach. Learn."},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1007\/s10514-023-10099-4"},{"key":"ref48","first-page":"1","article-title":"How much is a triple?","volume-title":"Proc. IEEE Int. Semantic Web Conf.","author":"Paulheim"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1145\/219717.219745"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1145\/1376616.1376746"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.3233\/SW-140134"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1145\/3191513"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00308"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00690"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01000"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-short.7"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01045"},{"key":"ref58","volume-title":"Surfaces and Essences: Analogy as the Fuel and Fire of Thinking","author":"Hofstadter","year":"2013"},{"key":"ref59","article-title":"Robobrain: Large-scale knowledge engine for robots","author":"Saxena","year":"2014"},{"key":"ref60","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brown"},{"key":"ref61","first-page":"466","article-title":"ObjectFolder: A dataset of objects with implicit visual, auditory, and tactile representations","volume-title":"Proc. Conf. Robot Learn.","author":"Gao"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01034"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2020.XVI.079"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00576"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2019.2931042"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10605-2_27"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00108"},{"key":"ref68","article-title":"Learning 6-DoF fine-grained grasp detection based on part affordance grounding","author":"Song","year":"2023"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013027"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/584"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/MRA.2011.941632"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1007\/s13218-015-0364-1"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref74","article-title":"RoBERTa: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"issue":"1","key":"ref75","first-page":"5485","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.703"},{"issue":"8","key":"ref77","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog."},{"key":"ref78","article-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref79","article-title":"Stanford alpaca: An instruction-following llama model","author":"Taori","year":"2023"},{"key":"ref80","article-title":"Huggingchat","year":"2023"},{"key":"ref81","article-title":"LLaMA 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2017.2785279"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.511"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.330"},{"key":"ref86","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.303"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1086"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01441"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.332"},{"key":"ref92","first-page":"8821","article-title":"Zero-shot text-to-image generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ramesh"},{"key":"ref93","article-title":"Hierarchical text-conditional image generation with clip latents","author":"Ramesh","year":"2022"},{"key":"ref94","article-title":"Point-E: A system for generating 3D point clouds from complex prompts","author":"Nichol","year":"2022"},{"key":"ref95","article-title":"mplug-owl: Modularization empowers large language models with multimodality","author":"Ye","year":"2023"},{"key":"ref96","article-title":"MiniGPT-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","year":"2023","journal-title":"Proc. 12th Int. Conf. Learn. Representations"},{"key":"ref97","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.019"},{"key":"ref99","first-page":"1","article-title":"R3M: A universal visual representation for robot manipulation","volume-title":"Proc. 6th Annu. Conf. Robot Learn.","author":"Nair"},{"key":"ref100","article-title":"Where are we in the search for an artificial visual cortex for embodied intelligence?","volume":"36","author":"Majumdar","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"240","key":"ref101","first-page":"1","article-title":"PALM: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2023","journal-title":"J. Mach. Learn. Res."},{"key":"ref102","first-page":"7480","article-title":"Scaling vision transformers to 22 billion parameters","author":"Dehghani","year":"2023","journal-title":"Proc. Int. Conf. Mach. Learn."}],"container-title":["IEEE Transactions on Knowledge and Data Engineering"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/69\/10709365\/10531671.pdf?arnumber=10531671","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,9]],"date-time":"2024-10-09T05:41:20Z","timestamp":1728452480000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10531671\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11]]},"references-count":102,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/tkde.2024.3399746","relation":{},"ISSN":["1041-4347","1558-2191","2326-3865"],"issn-type":[{"value":"1041-4347","type":"print"},{"value":"1558-2191","type":"electronic"},{"value":"2326-3865","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11]]}}}