{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T10:21:44Z","timestamp":1777890104446,"version":"3.51.4"},"reference-count":77,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00592","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"6273-6283","source":"Crossref","is-referenced-by-count":0,"title":["Mamba-3VL: Taming State Space Model for 3D Vision Language Learning"],"prefix":"10.1109","author":[{"given":"Yuan","family":"Wang","sequence":"first","affiliation":[{"name":"Tsinghua University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxin","family":"Chen","sequence":"additional","affiliation":[{"name":"Tencent PCG,ARC Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongang","family":"Qi","sequence":"additional","affiliation":[{"name":"Tencent PCG,ARC Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lijun","family":"Liu","sequence":"additional","affiliation":[{"name":"UCAS"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jile","family":"Jiao","sequence":"additional","affiliation":[{"name":"Tsinghua University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuetao","family":"Feng","sequence":"additional","affiliation":[{"name":"Tsinghua University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yujia","family":"Liang","sequence":"additional","affiliation":[{"name":"HUST"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ying","family":"Shan","sequence":"additional","affiliation":[{"name":"Tencent PCG,ARC Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhipeng","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, SJTU"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_25"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01854"},{"key":"ref3","author":"Brohan","year":"2023","journal-title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01597"},{"key":"ref5","author":"Cheang","year":"2024","journal-title":"Gr-2: A generative video-language-action model with web-scale knowledge for robot manipulation"},{"key":"ref6","first-page":"202","article-title":"Scanrefer: 3d object localization in rgb-d scans using natural language","volume-title":"European Conference on Computer Vision","author":"Zhenyu Chen","year":"2020"},{"key":"ref7","first-page":"487","article-title":"D 3 net: A unified speaker-listener architecture for 3d dense captioning and visual grounding","volume-title":"European Conference on Computer Vision","author":"Zhenyu Chen","year":"2022"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1492"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01070"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00321"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01660"},{"issue":"3","key":"ref13","volume":"2","author":"Chiang","year":"2023","journal-title":"Vicuna: An open-source chatbot impressing gpt-4 with 90 chatgpt quality"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01058"},{"key":"ref15","author":"Daniel","year":"2022","journal-title":"Hungry hungry hippos: Towards language modeling with state space models"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_31"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.2307\/jj.36032637.26"},{"key":"ref18","author":"Gu","year":"2021","journal-title":"Efficiently modeling long sequences with structured state spaces"},{"key":"ref19","first-page":"572","article-title":"Combining recurrent, convolutional, and continuous-time models with linear state space layers","volume":"34","author":"Gu","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00548"},{"key":"ref21","author":"Huang","year":"2023","journal-title":"An embodied generalist agent in 3d world"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.01508"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73033-7_10"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_24"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72673-6_16"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_31"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72649-1_21"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01057"},{"key":"ref29","author":"Jin Kim","year":"2024","journal-title":"Openvla: An open-source vision-language-action model"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"ref31","author":"Li","year":"2024","journal-title":"With spatial intelligence, ai will understand the real world, 2024"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.01116"},{"key":"ref33","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref34","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified visionlanguage understanding and generation","volume-title":"International Conference on Machine Learning","author":"Li","year":"2022"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73347-5_14"},{"key":"ref36","author":"Liang","year":"2024","journal-title":"Pointmamba: A simple state space model for point cloud analysis"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.52202\/079017-4473"},{"key":"ref38","author":"Loshchilov","year":"2017","journal-title":"Decoupled weight decay regularization"},{"key":"ref39","first-page":"21297","article-title":"Soft: Softmax-free transformer with linear complexity","volume":"34","author":"Lu","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref40","first-page":"1610","article-title":"Ovir-3d: Open-vocabulary 3d instance retrieval without training on 3d data","volume-title":"Conference on Robot Learning","author":"Lu","year":"2023"},{"key":"ref41","author":"Jun","year":"2024","journal-title":"U-mamba: Enhancing longrange dependency for biomedical image segmentation"},{"key":"ref42","author":"Ma","year":"2022","journal-title":"Sqa3d: Situated question answering in 3d scenes"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0031"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00385"},{"key":"ref45","author":"O\u2019Neill","year":"2023","journal-title":"Open x-embodiment: Robotic learning datasets and rt-x models"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00085"},{"key":"ref47","article-title":"Pointnet++: Deep hierarchical feature learning on point sets in a metric space","volume":"30","author":"Ruizhongtai Qi","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref48","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International Conference on Machine Learning","author":"Radford","year":"2021"},{"issue":"140","key":"ref49","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"Journal of Machine Learning Research"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00511"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_8"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160590"},{"key":"ref53","first-page":"894","article-title":"Cliport: What and where pathways for robotic manipulation","volume-title":"Conference on Robot Learning","author":"Shridhar","year":"2022"},{"key":"ref54","author":"Takmaz","year":"2023","journal-title":"Openmask3d: Open-vocabulary 3d instance segmentation"},{"key":"ref55","author":"Wang","year":"2020","journal-title":"Linformer: Self-attention with linear complexity"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1145\/3326362"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.00703"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-02116-5"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01320"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i8.32875"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01843"},{"key":"ref62","author":"Yang","year":"2023","journal-title":"Learning interactive real-world simulators"},{"key":"ref63","author":"Yang","year":"2023","journal-title":"Sam3d: Segment anything in 3d scenes"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00837"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2590"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i10.33098"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01397"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73232-4_15"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/iccv48922.2021.00292"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01293"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i7.28597"},{"key":"ref72","author":"Zhu","year":"2024","journal-title":"Vision mamba: Efficient visual representation learning with bidirectional state space model"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00272"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72784-9_11"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01451"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0868"},{"key":"ref77","first-page":"40164","article-title":"Interfacing foundation models\u2019 embeddings","volume":"37","author":"Zou","year":"2025","journal-title":"Advances in Neural Information Processing Systems"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11444484.pdf?arnumber=11444484","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T05:29:43Z","timestamp":1777613383000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11444484\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":77,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00592","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}