{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,19]],"date-time":"2025-12-19T18:58:40Z","timestamp":1766170720317,"version":"3.48.0"},"reference-count":34,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iros60139.2025.11247209","type":"proceedings-article","created":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T18:54:45Z","timestamp":1764269685000},"page":"20479-20486","source":"Crossref","is-referenced-by-count":0,"title":["Sensing Differently: Unifying Vision, Language, Posture and Tactile in Robotic Perception"],"prefix":"10.1109","author":[{"given":"Yanmin","family":"Zhou","sequence":"first","affiliation":[{"name":"Tongji University,College of Electronics and Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yiyang","family":"Jin","sequence":"additional","affiliation":[{"name":"Tongji University,College of Electronics and Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rong","family":"Jiang","sequence":"additional","affiliation":[{"name":"Tongji University,College of Electronics and Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Li","sequence":"additional","affiliation":[{"name":"Tongji University,College of Electronics and Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongrui","family":"Sang","sequence":"additional","affiliation":[{"name":"Shanghai Maritime University,School of Logistics Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuo","family":"Jiang","sequence":"additional","affiliation":[{"name":"Tongji University,College of Electronics and Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhipeng","family":"Wang","sequence":"additional","affiliation":[{"name":"Tongji University,College of Electronics and Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"He","sequence":"additional","affiliation":[{"name":"Tongji University,College of Electronics and Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/bs.acdb.2016.12.002"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1037\/h0060252"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.2514\/6.2024-2419"},{"article-title":"Mat: Multi-fingered adaptive tactile grasping via deep reinforcement learning","year":"2019","author":"Wu","key":"ref4"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/IROS51168.2021.9636259"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.025"},{"article-title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control","year":"2023","author":"Brohan","key":"ref7"},{"article-title":"Open x-embodiment: Robotic learning datasets and rt-x models","year":"2023","author":"Padalkar","key":"ref8"},{"article-title":"A touch, vision, and language dataset for multimodal alignment","year":"2024","author":"Fu","key":"ref9"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02488"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2024.XX.066"},{"article-title":"A survey of large language models","year":"2023","author":"Zhao","key":"ref12"},{"article-title":"Gpt-4 technical report","year":"2023","author":"Achiam","key":"ref13"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/s11023-020-09548-1"},{"article-title":"A label is worth a thousand images in dataset distillation","year":"2024","author":"Qin","key":"ref15"},{"key":"ref16","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.65286\/icic.v20i2.91720"},{"article-title":"An early evaluation of gpt-4v (ision)","year":"2023","author":"Wu","key":"ref18"},{"article-title":"Imu2clip: Multimodal contrastive learning for imu motion sensors from egocentric videos and text","year":"2022","author":"Moon","key":"ref19"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747631"},{"article-title":"Lora: Low-rank adaptation of large language models","year":"2021","author":"Hu","key":"ref21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8460495"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196787"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.3390\/s17122762"},{"article-title":"Cogvlm: Visual expert for pretrained language models","year":"2023","author":"Wang","key":"ref25"},{"article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","year":"2020","author":"Dosovitskiy","key":"ref26"},{"article-title":"Omni-scale cnns: a simple and effective kernel size configuration for time series classification","year":"2020","author":"Tang","key":"ref27"},{"article-title":"Goldbach conjecture","year":"2002","author":"Weisstein","key":"ref28"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.018"},{"article-title":"The feeling of success: Does touch sensing help predict grasp outcomes?","year":"2017","author":"Calandra","key":"ref30"},{"article-title":"Touch and go: Learning from human-collected vision and touch","year":"2022","author":"Yang","key":"ref31"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9981562"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00276"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01100"}],"event":{"name":"2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","start":{"date-parts":[[2025,10,19]]},"location":"Hangzhou, China","end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11245651\/11245652\/11247209.pdf?arnumber=11247209","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,19]],"date-time":"2025-12-19T18:55:03Z","timestamp":1766170503000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11247209\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":34,"URL":"https:\/\/doi.org\/10.1109\/iros60139.2025.11247209","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}