{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T20:24:32Z","timestamp":1776889472153,"version":"3.51.2"},"reference-count":62,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T00:00:00Z","timestamp":1685318400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T00:00:00Z","timestamp":1685318400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61972404,12071478"],"award-info":[{"award-number":["61972404,12071478"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004260","name":"Public Computing Cloud, Renmin University of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004260","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,5,29]]},"DOI":"10.1109\/icra48891.2023.10160658","type":"proceedings-article","created":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T17:20:56Z","timestamp":1688491256000},"page":"7234-7242","source":"Crossref","is-referenced-by-count":6,"title":["ViPFormer: Efficient Vision-and-Pointcloud Transformer for Unsupervised Pointcloud Understanding"],"prefix":"10.1109","author":[{"given":"Hongyu","family":"Sun","sequence":"first","affiliation":[{"name":"School of Information, Renmin University of China,Department of Computer Science,Beijing,China,100872"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongcai","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Information, Renmin University of China,Department of Computer Science,Beijing,China,100872"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xudong","family":"Cai","sequence":"additional","affiliation":[{"name":"School of Information, Renmin University of China,Department of Computer Science,Beijing,China,100872"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuewei","family":"Bai","sequence":"additional","affiliation":[{"name":"School of Information, Renmin University of China,Department of Computer Science,Beijing,China,100872"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Deying","family":"Li","sequence":"additional","affiliation":[{"name":"School of Information, Renmin University of China,Department of Computer Science,Beijing,China,100872"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","article-title":"Self-supervised deep learning on point clouds by reconstructing space","volume":"32","author":"sauder","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref57","article-title":"Decoupled weight decay regularization","author":"loshchilov","year":"0","journal-title":"International Conference on Learning Representations 2019"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20086-1_35"},{"key":"ref56","article-title":"Disn: Deep implicit surface network for high-quality single-view 3d reconstruction","volume":"32","author":"xu","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01227"},{"key":"ref59","article-title":"3d shapenets: A deep representation for volumetric shapes","author":"wu","year":"2015","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00964"},{"key":"ref58","author":"katsura","year":"2021","journal-title":"Pytorch cosineannealing with warmup restarts"},{"key":"ref53","article-title":"Perceiver IO: A general architecture for structured inputs & outputs","author":"jaegle","year":"0","journal-title":"International Conference on Learning Representations 2022"},{"key":"ref52","first-page":"4651","article-title":"Perceiver: General perception with iterative attention","volume":"139","author":"jaegle","year":"2021","journal-title":"Proceedings of the 38th International Conference on Machine Learning"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01871"},{"key":"ref55","article-title":"Shapenet: An information-rich 3d model repository","author":"chang","year":"2015","journal-title":"CoRR"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00029"},{"key":"ref54","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","volume":"119","author":"chen","year":"2020","journal-title":"Proceedings of the 37th International Conference on Machine Learning"},{"key":"ref17","first-page":"40","article-title":"Learning representations and generative models for 3D point clouds","volume":"80","author":"achlioptas","year":"2018","journal-title":"Proceedings of the 35th International Conference on Machine Learning"},{"key":"ref16","article-title":"Learning a probabilistic latent space of object shapes via 3d generative-adversarial modeling","volume":"29","author":"wu","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref19","doi-asserted-by":"crossref","first-page":"574","DOI":"10.1007\/978-3-030-58580-8_34","article-title":"Pointcontrast: Unsupervised pre-training for 3d point cloud understanding","author":"xie","year":"2020","journal-title":"Computer Vision - ECCV 2020"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018376"},{"key":"ref51","author":"guo","year":"2020","journal-title":"Pct Point cloud transformer"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00294"},{"key":"ref46","first-page":"213","article-title":"End-to-end object detection with transformers","author":"carion","year":"2020","journal-title":"Compu Vis - ECCV 2020 - 16th Eur Conf"},{"key":"ref45","article-title":"Rethinking network design and local geometry in point cloud: A simple residual MLP framework","author":"ma","year":"0","journal-title":"International Conference on Learning Representations 2022"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00315"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00290"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00738"},{"key":"ref41","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3142440"},{"key":"ref43","first-page":"7212","article-title":"Self-supervised few-shot learning on point clouds","volume":"33","author":"sharma","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00274"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00937"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00472"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01180"},{"key":"ref4","first-page":"16259","article-title":"Point trans-former","author":"zhao","year":"2021","journal-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01112"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00798"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01644"},{"key":"ref40","author":"radford","year":"2019","journal-title":"Language Models are Unsupervised Multitask Learners"},{"key":"ref35","article-title":"Pointnet++: Deep hierarchical feature learning on point sets in a metric space","volume":"30","author":"qi","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref34","doi-asserted-by":"crossref","first-page":"234","DOI":"10.1007\/978-3-319-24574-4_28","article-title":"U-net: Convolutional networks for biomedical image segmentation","author":"ronneberger","year":"2015","journal-title":"Medical Image Com-puting and Computer-Assisted Intervention - MICCAI 2015"},{"key":"ref37","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"dosovitskiy","year":"2021","journal-title":"International Conference on Learning Representations"},{"key":"ref36","article-title":"Attention is all you need","volume":"30","author":"vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref31","article-title":"Pointnet: Deep learning on point sets for 3d classification and segmentation","author":"qi","year":"2017","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CASE49997.2022.9926589"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/3326362"},{"key":"ref2","article-title":"Rethinking network design and local geometry in point cloud: A simple residual MLP framework","author":"ma","year":"0","journal-title":"International Conference on Learning Representations 2022"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00095"},{"key":"ref39","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2019","journal-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics Human Language Technologies Volume 1 (Long and Short Papers)"},{"key":"ref38","article-title":"BEit: BERT pre-training of image transformers","author":"bao","year":"2022","journal-title":"International Conference on Learning Representations"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref23","first-page":"8489","article-title":"Contrastive bound-ary learning for point cloud segmentation","author":"tang","year":"2022","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01101"},{"key":"ref25","first-page":"5583","article-title":"Vilt: Vision-and-language transformer without convolution or region supervision","volume":"139","author":"kim","year":"2021","journal-title":"Proceedings of the 38th International Conference on Machine Learning"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01009"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00967"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01533"},{"key":"ref28","doi-asserted-by":"crossref","first-page":"639","DOI":"10.1007\/978-3-030-01231-1_39","article-title":"Audio-Visual Scene Analysis with Self-Supervised Multisensory Features","volume":"11210","author":"owens","year":"2018","journal-title":"Computer Vision - ECCV 2018"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58598-3_10"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01229"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00167"},{"key":"ref62","first-page":"2579","article-title":"Visualizing data using t-sne","volume":"9","author":"van der maaten","year":"2008","journal-title":"Journal of Machine Learning Research"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1145\/2980179.2980238"}],"event":{"name":"2023 IEEE International Conference on Robotics and Automation (ICRA)","location":"London, United Kingdom","start":{"date-parts":[[2023,5,29]]},"end":{"date-parts":[[2023,6,2]]}},"container-title":["2023 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10160211\/10160212\/10160658.pdf?arnumber=10160658","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,24]],"date-time":"2023-07-24T17:36:03Z","timestamp":1690220163000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10160658\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,29]]},"references-count":62,"URL":"https:\/\/doi.org\/10.1109\/icra48891.2023.10160658","relation":{},"subject":[],"published":{"date-parts":[[2023,5,29]]}}}