{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:02:40Z","timestamp":1784268160927,"version":"3.55.0"},"reference-count":86,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China (NSFC)","doi-asserted-by":"publisher","award":["62206172,62432008"],"award-info":[{"award-number":["62206172,62432008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00480","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"1-12","source":"Crossref","is-referenced-by-count":4,"title":["Detect Anything 3D in the Wild"],"prefix":"10.1109","author":[{"given":"Hanxue","family":"Zhang","sequence":"first","affiliation":[{"name":"OpenDriveLab at Shanghai AI Laboratory"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haoran","family":"Jiang","sequence":"additional","affiliation":[{"name":"OpenDriveLab at Shanghai AI Laboratory"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingsong","family":"Yao","sequence":"additional","affiliation":[{"name":"Stanford University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanan","family":"Sun","sequence":"additional","affiliation":[{"name":"OpenDriveLab at Shanghai AI Laboratory"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Renrui","family":"Zhang","sequence":"additional","affiliation":[{"name":"CUHK MMLab"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Zhao","sequence":"additional","affiliation":[{"name":"Tsinghua University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongyang","family":"Li","sequence":"additional","affiliation":[{"name":"OpenDriveLab at Shanghai AI Laboratory"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongzi","family":"Zhu","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zetong","family":"Yang","sequence":"additional","affiliation":[{"name":"OpenDriveLab at Shanghai AI Laboratory"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","author":"Achiam","year":"2023","journal-title":"Gpt-4 technical report"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00773"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/677"},{"key":"ref4","article-title":"Arkitscenes: A diverse real-world dataset for 3d indoor scene understanding using mobile rgb-d data","author":"Baruch","year":"2021","journal-title":"NeurIPS Datasets"},{"key":"ref5","author":"Bi","year":"2024","journal-title":"Deepseek 1lm: Scaling opensource language models with longtermism"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2007.363129"},{"key":"ref7","author":"Bochkovskii","year":"2024","journal-title":"Depth pro: Sharp monocular metric depth in less than a second"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01264"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01417"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3435937"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02496"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.236"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.691"},{"key":"ref16","article-title":"Vision transformer adapter for dense predictions","author":"Chen","year":"2023","journal-title":"ICLR"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.73"},{"key":"ref19","article-title":"Depth map prediction from a single image using a multi-scale deep network","author":"Eigen","year":"2014","journal-title":"NeurIPS"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01146"},{"key":"ref21","author":"G\u00e4hlert","year":"2020","journal-title":"Cityscapes 3d: Dataset and benchmark for 9 dof vehicle detection"},{"key":"ref22","article-title":"Magicdrive: Street view generation with diverse 3d geometry control","author":"Gao","year":"2023","journal-title":"ICLR"},{"key":"ref23","author":"Gao","year":"2024","journal-title":"Magicdrive3d: Controllable 3d generation for any-view rendering in street scenes"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913491297"},{"key":"ref25","author":"Geyer","year":"2020","journal-title":"A2d2: Audi autonomous driving dataset"},{"key":"ref26","author":"Guo","year":"2024","journal-title":"Deepseek-coder: When the large language model meets programming-the rise of code intelligence"},{"key":"ref27","author":"Guo","year":"2023","journal-title":"Point-bind & point-llm: Aligning point cloud with multi-modality for 3d understanding, generation, and instruction following"},{"key":"ref28","author":"Guo","year":"2024","journal-title":"Sam2point: Segment any 3d as videos in zero-shot and promptable manners"},{"key":"ref29","author":"Guo","year":"2025","journal-title":"Can we generate images with cot? let\u2019s verify and reinforce image generation step by step"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i3.32355"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01712"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00927"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-11015-4_25"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i4.32459"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00111"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73229-4_27"},{"key":"ref39","article-title":"Bevformer:\\\\ learning bird\u2019s-eye-view representation from lidar-camera via spatiotemporal transformers","author":"Li","year":"2024","journal-title":"IEEE TPAMI"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01567"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0757"},{"key":"ref42","author":"Lin","year":"2022","journal-title":"Sparse4d: Multi-view 3d object detection with sparse spatial-temporal fusion"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3300061.3300116"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00506"},{"key":"ref46","author":"Loshchilov","year":"2016","journal-title":"Sgdr: Stochastic gradient descent with warm restarts"},{"key":"ref47","author":"Loshchilov","year":"2017","journal-title":"Decoupled weight decay regularization"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2023.3346386"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01790-1"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00013"},{"key":"ref51","article-title":"Dinov2: Learning robust visual features without supervision","author":"Oquab","year":"2024","journal-title":"TMLR"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ISMAR.2008.4637336"},{"key":"ref53","article-title":"Pytorch: An imperative style, high-performance deep learning library","author":"Paszke","year":"2019","journal-title":"NeurIPS"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00963"},{"key":"ref55","author":"Qi","year":"2025","journal-title":"Gpt4scene: Understand 3d scenes from videos with vision-language models"},{"key":"ref56","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021","journal-title":"ICML"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01073"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00133"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1023\/A:1014573219977"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72943-0_15"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00252"},{"key":"ref63","author":"Touvron","year":"2023","journal-title":"Llama: Open and efficient foundation language models"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.596"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00775"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00107"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01868"},{"key":"ref68","article-title":"Uni3detr: Unified 3d detection transformer","author":"Wang","year":"2023","journal-title":"NeurIPS"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72970-6_5"},{"key":"ref70","article-title":"Argoverse 2: Next generation datasets for self-driving perception and forecasting","author":"Wilson","year":"2023","journal-title":"NeurIPS Datasets"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00099"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01604"},{"key":"ref73","article-title":"Scenecraft: Layout-guided 3d scene generation","author":"Yang","year":"2025","journal-title":"NeurIPS"},{"key":"ref74","author":"Yao","year":"2024","journal-title":"Open vocabulary monocular 3d object detection"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1145\/3730841"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00830"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00391"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00836"},{"key":"ref79","article-title":"Personalize segment anything model with one shot","author":"Zhang","year":"2023","journal-title":"ICLR"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00840"},{"key":"ref81","article-title":"Llama-adapter: Efficient fine-tuning of large language models with zeroinitialized attention","author":"Zhang","year":"2024","journal-title":"ICLR"},{"key":"ref82","author":"Zhang","year":"2024","journal-title":"Mavis: Mathematical visual instruction tuning with an automatic data engine"},{"key":"ref83","author":"Zhu","year":"2023","journal-title":"Ponderv2: Pave the way for 3 d foundation model with a universal pre-training paradigm"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2014.6907430"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72784-9_11"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/3dv66043.2025.00122"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11444544.pdf?arnumber=11444544","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T05:18:13Z","timestamp":1777612693000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11444544\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":86,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00480","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}