{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T20:47:56Z","timestamp":1757623676323,"version":"3.44.0"},"reference-count":86,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T00:00:00Z","timestamp":1748044800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T00:00:00Z","timestamp":1748044800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s11263-025-02464-w","type":"journal-article","created":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T08:48:24Z","timestamp":1748076504000},"page":"6074-6087","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["MIM4D: Masked Modeling with Multi-View Video for Autonomous Driving Representation Learning"],"prefix":"10.1007","volume":"133","author":[{"given":"Jialv","family":"Zou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bencheng","family":"Liao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qian","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenyu","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6732-7823","authenticated-orcid":false,"given":"Xinggang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,24]]},"reference":[{"key":"2464_CR1","doi-asserted-by":"crossref","unstructured":"Park, D., Ambrus, R., Guizilini, V., Li, J., & Gaidon, A. (2021). Is pseudo-lidar needed for monocular 3d object detection? In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3142\u20133152.","DOI":"10.1109\/ICCV48922.2021.00313"},{"issue":"1","key":"2464_CR2","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3503250","volume":"65","author":"B Mildenhall","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P. P., Tancik, M., Barron, J. T., Ramamoorthi, R., & Ng, R. (2021). Nerf: Representing scenes as neural radiance fields for view synthesis. Communications of the ACM, 65(1), 99\u2013106.","journal-title":"Communications of the ACM"},{"key":"2464_CR3","unstructured":"Zhang, C., Yan, J., Wei, Y., Li, J., Liu, L., Tang, Y., Duan, Y., & Lu, J. (2023). Occnerf: Self-supervised multi-camera occupancy prediction with neural radiance fields. arXiv preprint arXiv:2312.09243."},{"key":"2464_CR4","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., & Lo, W.-Y., et al. (2023). Segment anything. arXiv preprint arXiv:2304.02643.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"2464_CR5","doi-asserted-by":"crossref","unstructured":"Yang, H., Zhang, S., Huang, D., Wu, X., Zhu, H., He, T., Tang, S., Zhao, H., Qiu, Q., & Lin, B., et al. (2023). Unipad: A universal pre-training paradigm for autonomous driving. arXiv preprint arXiv:2310.08370.","DOI":"10.1109\/CVPR52733.2024.01443"},{"key":"2464_CR6","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., & Girshick, R. (2022). Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16000\u201316009.","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"2464_CR7","doi-asserted-by":"crossref","unstructured":"Yang, Z., Chen, L., Sun, Y., & Li, H. (2024). Visual point cloud forecasting enables scalable autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14673\u201314684.","DOI":"10.1109\/CVPR52733.2024.01390"},{"key":"2464_CR8","unstructured":"Guo, M., Zhang, Z., He, Y., Wang, K., & Jing, L. (2024) End-to-end autonomous driving without costly modularization and 3d manual annotation. arXiv preprint arXiv:2406.17680."},{"key":"2464_CR9","doi-asserted-by":"crossref","unstructured":"Hu, Y., Yang, J., Chen, L., Li, K., Sima, C., Zhu, X., Chai, S., Du, S., Lin, T., & Wang, W. (2023). Planning-oriented autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17853\u201317862.","DOI":"10.1109\/CVPR52729.2023.01712"},{"key":"2464_CR10","doi-asserted-by":"crossref","unstructured":"Jiang, B., Chen, S., Xu, Q., Liao, B., Chen, J., Zhou, H., Zhang, Q., Liu, W., Huang, C., & Wang, X. (2023). Vad: Vectorized scene representation for efficient autonomous driving. arXiv preprint arXiv:2303.12077.","DOI":"10.1109\/ICCV51070.2023.00766"},{"key":"2464_CR11","unstructured":"Chen, S., Jiang, B., Gao, H., Liao, B., Xu, Q., Zhang, Q., Huang, C., Liu, W., & Wang, X. (2024). Vadv2: End-to-end vectorized autonomous driving via probabilistic planning. arXiv preprint arXiv:2402.13243."},{"key":"2464_CR12","doi-asserted-by":"crossref","unstructured":"Caesar, H., Bankiti, V., Lang, A.H., Vora, S., Liong, V.E., Xu, Q., Krishnan, A., Pan, Y., Baldan, G., & Beijbom, O. (2020). nuscenes: A multimodal dataset for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11621\u201311631.","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"2464_CR13","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. Ieee.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"2464_CR14","doi-asserted-by":"crossref","unstructured":"Zhou, B., & Kr\u00e4henb\u00fchl, P. (2022). Cross-view transformers for real-time map-view semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13760\u201313769.","DOI":"10.1109\/CVPR52688.2022.01339"},{"key":"2464_CR15","doi-asserted-by":"crossref","unstructured":"Liu, Y., Wang, T., Zhang, X., & Sun, J. (2022) Petr: Position embedding transformation for multi-view 3d object detection. arXiv preprint arXiv:2203.05625.","DOI":"10.1007\/978-3-031-19812-0_31"},{"key":"2464_CR16","unstructured":"Huang, J., & Huang, G. (2022). Bevdet4d: Exploit temporal cues in multi-camera 3d object detection. arXiv preprint arXiv:2203.17054."},{"key":"2464_CR17","unstructured":"Lin, X., Pei, Z., Lin, T., Huang, L., & Su, Z. (2023). Sparse4d v3: Advancing end-to-end 3d detection and tracking. arXiv preprint arXiv:2311.11722."},{"key":"2464_CR18","unstructured":"Liao, B., Chen, S., Wang, X., Cheng, T., Zhang, Q., Liu, W., & Huang, C. (2022). Maptr: Structured modeling and learning for online vectorized hd map construction. arXiv preprint arXiv:2208.14437."},{"key":"2464_CR19","unstructured":"Chen, T., Kornblith, S., Norouzi, M., & Hinton, G. (2020) A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607. PMLR."},{"key":"2464_CR20","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., & Girshick, R. (2020) Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9729\u20139738.","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"2464_CR21","unstructured":"Chen, X., Fan, H., Girshick, R., & He, K. (2020). Improved baselines with momentum contrastive learning. arXiv preprint arXiv:2003.04297."},{"key":"2464_CR22","doi-asserted-by":"crossref","unstructured":"Fan, H., Xiong, B., Mangalam, K., Li, Y., Yan, Z., Malik, J., & Feichtenhofer, C.(2021). Multiscale vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6824\u20136835.","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"2464_CR23","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., & Joulin, A. (2021). Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660.","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"2464_CR24","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., & Clark, J. (2021). Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR."},{"key":"2464_CR25","unstructured":"Fang, Y., Dong, L., Bao, H., Wang, X., & Wei, F. (2023).Corrupted image modeling for self-supervised visual pre-training. In: The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=09hVcSDkea."},{"key":"2464_CR26","unstructured":"Bao, H., Dong, L., Piao, S., & Wei, F. (2022). BEiT: BERT pre-training of image transformers. In: International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=p-BhZSz59o4."},{"key":"2464_CR27","doi-asserted-by":"crossref","unstructured":"Fang, Y., Wang, W., Xie, B., Sun, Q., Wu, L., Wang, X., Huang, T., Wang, X., & Cao, Y. (2023). Eva: Exploring the limits of masked visual representation learning at scale. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19358\u201319369.","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"2464_CR28","doi-asserted-by":"crossref","unstructured":"Xie, Z., Zhang, Z., Cao, Y., Lin, Y., Bao, J., Yao, Z., Dai, Q., & Hu, H. (2022). Simmim: A simple framework for masked image modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9653\u20139663.","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"2464_CR29","doi-asserted-by":"crossref","unstructured":"Li, Y., Fan, H., Hu, R., Feichtenhofer, C., & He, K. (2023). Scaling language-image pre-training via masking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23390\u201323400.","DOI":"10.1109\/CVPR52729.2023.02240"},{"key":"2464_CR30","unstructured":"Tian, K., Jiang, Y., diao, Lin, C., Wang, L., & Yuan, Z. (2023). Designing BERT for convolutional networks: Sparse and hierarchical masked modeling. In: The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=NRxydtWup1S."},{"key":"2464_CR31","doi-asserted-by":"crossref","unstructured":"Wang, L., Huang, B., Zhao, Z., Tong, Z., He, Y., Wang, Y., Wang, Y., & Qiao, Y. (2023). Videomae v2: Scaling video masked autoencoders with dual masking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14549\u201314560.","DOI":"10.1109\/CVPR52729.2023.01398"},{"key":"2464_CR32","first-page":"10078","volume":"35","author":"Z Tong","year":"2022","unstructured":"Tong, Z., Song, Y., Wang, J., & Wang, L. (2022). Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems, 35, 10078\u201310093.","journal-title":"Advances in neural information processing systems"},{"key":"2464_CR33","first-page":"35946","volume":"35","author":"C Feichtenhofer","year":"2022","unstructured":"Feichtenhofer, C., Li, Y., & He, K. (2022). Masked autoencoders as spatiotemporal learners. Advances in neural information processing systems, 35, 35946\u201335958.","journal-title":"Advances in neural information processing systems"},{"issue":"1","key":"2464_CR34","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3503250","volume":"65","author":"B Mildenhall","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P. P., Tancik, M., Barron, J. T., Ramamoorthi, R., & Ng, R. (2021). Nerf: Representing scenes as neural radiance fields for view synthesis. Communications of the ACM, 65(1), 99\u2013106.","journal-title":"Communications of the ACM"},{"key":"2464_CR35","doi-asserted-by":"crossref","unstructured":"Sun, C., Sun, M., & Chen, H.-T. (2022). Direct voxel grid optimization: Super-fast convergence for radiance fields reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5459\u20135469.","DOI":"10.1109\/CVPR52688.2022.00538"},{"issue":"4","key":"2464_CR36","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3528223.3530127","volume":"41","author":"T M\u00fcller","year":"2022","unstructured":"M\u00fcller, T., Evans, A., Schied, C., & Keller, A. (2022). Instant neural graphics primitives with a multiresolution hash encoding. ACM Transactions on Graphics (ToG), 41(4), 1\u201315.","journal-title":"ACM Transactions on Graphics (ToG)"},{"key":"2464_CR37","doi-asserted-by":"crossref","unstructured":"Barron, J.T., Mildenhall, B., Tancik, M., Hedman, P., Martin-Brualla, R., & Srinivasan, P.P. (2021). Mip-nerf: A multiscale representation for anti-aliasing neural radiance fields. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5855\u20135864.","DOI":"10.1109\/ICCV48922.2021.00580"},{"key":"2464_CR38","unstructured":"Wang, P., Liu, L., Liu, Y., Theobalt, C., Komura, T., & Wang, W. (2021). Neus: Learning neural implicit surfaces by volume rendering for multi-view reconstruction. arXiv preprint arXiv:2106.10689."},{"key":"2464_CR39","doi-asserted-by":"crossref","unstructured":"Fang, J., Yi, T., Wang, X., Xie, L., Zhang, X., Liu, W., Nie\u00dfner, M., & Tian, Q. (2022). Fast dynamic radiance fields with time-aware neural voxels. In: SIGGRAPH Asia 2022 Conference Papers, pp. 1\u20139.","DOI":"10.1145\/3550469.3555383"},{"key":"2464_CR40","doi-asserted-by":"crossref","unstructured":"Huang, C., Hou, Y., Ye, W., Huang, D., Huang, X., Lin, B., Cai, D., & Ouyang, W. (2024). Nerf-det++: Incorporating semantic cues and perspective-aware depth supervision for indoor multi-view 3d detection. arXiv preprint arXiv:2402.14464.","DOI":"10.1109\/TIP.2025.3560240"},{"key":"2464_CR41","doi-asserted-by":"crossref","unstructured":"Xu, C., Wu, B., Hou, J., Tsai, S., Li, R., Wang, J., Zhan, W., He, Z., Vajda, P., & Keutzer, K. (2023). Nerf-det: Learning geometry-aware volumetric representation for multi-view 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 23320\u201323330.","DOI":"10.1109\/ICCV51070.2023.02131"},{"key":"2464_CR42","doi-asserted-by":"crossref","unstructured":"Xu, J., Peng, L., Cheng, H., Li, H., Qian, W., Li, K., Wang, W., Cai, D. (2023). Mononerd: Nerf-like representations for monocular 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6814\u20136824.","DOI":"10.1109\/ICCV51070.2023.00627"},{"key":"2464_CR43","doi-asserted-by":"crossref","unstructured":"Chen, Y., Nie\u00dfner, M., & Dai, A. (2022). 4dcontrast: Contrastive learning with dynamic correspondences for 3d scene understanding. In: European Conference on Computer Vision, pp. 543\u2013560. Springer.","DOI":"10.1007\/978-3-031-19824-3_32"},{"key":"2464_CR44","doi-asserted-by":"crossref","unstructured":"Li, L., Heizmann, M. (2022). A closer look at invariances in self-supervised pre-training for 3d vision. In: European Conference on Computer Vision, pp. 656\u2013673. Springer.","DOI":"10.1007\/978-3-031-20056-4_38"},{"key":"2464_CR45","doi-asserted-by":"crossref","unstructured":"Liang, H., Jiang, C., Feng, D., Chen, X., Xu, H., Liang, X., Zhang, W., Li, Z., & Van\u00a0Gool, L. (2021). Exploring geometry-aware contrast and clustering harmonization for self-supervised 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3293\u20133302.","DOI":"10.1109\/ICCV48922.2021.00328"},{"key":"2464_CR46","doi-asserted-by":"crossref","unstructured":"Liu, H., Cai, M., & Lee, Y.J. (2022). Masked discrimination for self-supervised learning on point clouds. In: European Conference on Computer Vision, pp. 657\u2013675. Springer.","DOI":"10.1007\/978-3-031-20086-1_38"},{"key":"2464_CR47","doi-asserted-by":"crossref","unstructured":"Pang, Y., Wang, W., Tay, F.E., Liu, W., Tian, Y., & Yuan, L. (2022). Masked autoencoders for point cloud self-supervised learning. In: European Conference on Computer Vision, pp. 604\u2013621. Springer.","DOI":"10.1007\/978-3-031-20086-1_35"},{"key":"2464_CR48","doi-asserted-by":"crossref","unstructured":"Tian, X., Ran, H., Wang, Y., & Zhao, H. (2023). Geomae: Masked geometric target prediction for self-supervised point cloud pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13570\u201313580.","DOI":"10.1109\/CVPR52729.2023.01304"},{"key":"2464_CR49","doi-asserted-by":"crossref","unstructured":"Huang, D., Peng, S., He, T., Yang, H., Zhou, X., & Ouyang, W. (2023). Ponder: Point cloud pre-training via neural rendering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16089\u201316098.","DOI":"10.1109\/ICCV51070.2023.01474"},{"key":"2464_CR50","doi-asserted-by":"crossref","unstructured":"Liu, J., Wang, T., Liu, B., Zhang, Q., Liu, Y., & Li, H. (2023). Geomim: Towards better 3d knowledge transfer via masked image modeling for multi-view 3d understanding. 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), 17793\u201317803.","DOI":"10.1109\/ICCV51070.2023.01635"},{"key":"2464_CR51","unstructured":"Liu, B., Wang, M., Foroosh, H., Tappen, M., & Pensky, M. (2015). Sparse convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 806\u2013814."},{"key":"2464_CR52","unstructured":"Tian, K., Jiang, Y., Diao, Q., Lin, C., Wang, L., & Yuan, Z. (2023). Designing bert for convolutional networks: Sparse and hierarchical masked modeling. arXiv preprint arXiv:2301.03580."},{"key":"2464_CR53","doi-asserted-by":"crossref","unstructured":"Philion, J., & Fidler, S. (2020). Lift, splat, shoot: Encoding images from arbitrary camera rigs by implicitly unprojecting to 3d. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XIV 16, pp. 194\u2013210. Springer.","DOI":"10.1007\/978-3-030-58568-6_12"},{"key":"2464_CR54","doi-asserted-by":"crossref","unstructured":"Zong, Z., Jiang, D., Song, G., Xue, Z., Su, J., Li, H., & Liu, Y. (2023). Temporal enhanced training of multi-view 3d object detector via historical object prediction. arXiv preprint arXiv:2304.00967.","DOI":"10.1109\/ICCV51070.2023.00350"},{"key":"2464_CR55","doi-asserted-by":"crossref","unstructured":"Wang, S., Liu, Y., Wang, T., Li, Y., & Zhang, X. (2023). Exploring object-centric temporal modeling for efficient multi-view 3d object detection. arXiv preprint arXiv:2303.11926.","DOI":"10.1109\/ICCV51070.2023.00335"},{"key":"2464_CR56","unstructured":"Yu, Z., Shu, C., Deng, J., Lu, K., Liu, Z., Yu, J., Yang, D., Li, H., & Chen, Y. (2023). Flashocc: Fast and memory-efficient occupancy prediction via channel-to-height plugin. arXiv preprint arXiv:2311.12058."},{"key":"2464_CR57","unstructured":"Park, J., Xu, C., Yang, S., Keutzer, K., Kitani, K., Tomizuka, M., & Zhan, W. (2022). Time will tell: New outlooks and a baseline for temporal multi-view 3d object detection. arXiv preprint arXiv:2210.02443."},{"key":"2464_CR58","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., & Dai, J. (2020). Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159."},{"key":"2464_CR59","doi-asserted-by":"crossref","unstructured":"Huang, Y., Zheng, W., Zhang, B., Zhou, J., Lu, J. (2023). Selfocc: Self-supervised vision-based 3d occupancy prediction. arXiv preprint arXiv:2311.12754.","DOI":"10.1109\/CVPR52733.2024.01885"},{"key":"2464_CR60","doi-asserted-by":"crossref","unstructured":"Sun, C., Sun, M., Chen, H.-T. (2022). Direct voxel grid optimization: Super-fast convergence for radiance fields reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5459\u20135469.","DOI":"10.1109\/CVPR52688.2022.00538"},{"key":"2464_CR61","unstructured":"Wang, P., Liu, L., Liu, Y., Theobalt, C., Komura, T., & Wang, W. (2021). Neus: Learning neural implicit surfaces by volume rendering for multi-view reconstruction. arXiv preprint arXiv:2106.10689."},{"key":"2464_CR62","doi-asserted-by":"crossref","unstructured":"Oechsle, M., Peng, S., & Geiger, A. (2021). Unisurf: Unifying neural implicit surfaces and radiance fields for multi-view reconstruction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5589\u20135599.","DOI":"10.1109\/ICCV48922.2021.00554"},{"key":"2464_CR63","unstructured":"Liu, Y., Yuan, T., Wang, Y., Wang, Y., & Zhao, H. (2023). Vectormapnet: End-to-end vectorized hd map learning. In: International Conference on Machine Learning, pp. 22352\u201322369. PMLR."},{"key":"2464_CR64","doi-asserted-by":"crossref","unstructured":"Hu, S., Chen, L., Wu, P., Li, H., Yan, J., & Tao, D. (2022). St-p3: End-to-end vision-based autonomous driving via spatial-temporal feature learning. In: European Conference on Computer Vision, pp. 533\u2013549. Springer.","DOI":"10.1007\/978-3-031-19839-7_31"},{"key":"2464_CR65","unstructured":"Zhu, B., Jiang, Z., Zhou, X., Li, Z., & Yu, G. (2019) Class-balanced grouping and sampling for point cloud 3d object detection. arXiv preprint arXiv:1908.09492."},{"key":"2464_CR66","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.-Y., Feichtenhofer, C., Darrell, T., &Xie, S. (2022). A convnet for the 2020s. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11976\u201311986.","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"2464_CR67","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2464_CR68","first-page":"18442","volume":"35","author":"Y Li","year":"2022","unstructured":"Li, Y., Chen, Y., Qi, X., Li, Z., Sun, J., & Jia, J. (2022). Unifying voxel-based representation with transformer for 3d object detection. Advances in Neural Information Processing Systems, 35, 18442\u201318455.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2464_CR69","first-page":"10078","volume":"35","author":"Z Tong","year":"2022","unstructured":"Tong, Z., Song, Y., Wang, J., & Wang, L. (2022). Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems, 35, 10078\u201310093.","journal-title":"Advances in neural information processing systems"},{"key":"2464_CR70","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., & Girshick, R. (2017). Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969.","DOI":"10.1109\/ICCV.2017.322"},{"key":"2464_CR71","doi-asserted-by":"crossref","unstructured":"Wang, T., Zhu, X., Pang, J., & Lin, D. (2021). Fcos3d: Fully convolutional one-stage monocular 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 913\u2013922.","DOI":"10.1109\/ICCVW54120.2021.00107"},{"key":"2464_CR72","doi-asserted-by":"crossref","unstructured":"Roddick, T., & Cipolla, R. (2020). Predicting semantic map representations from images using pyramid occupancy networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11138\u201311147.","DOI":"10.1109\/CVPR42600.2020.01115"},{"issue":"3","key":"2464_CR73","doi-asserted-by":"publisher","first-page":"4867","DOI":"10.1109\/LRA.2020.3004325","volume":"5","author":"B Pan","year":"2020","unstructured":"Pan, B., Sun, J., Leung, H. Y. T., Andonian, A., & Zhou, B. (2020). Cross-view semantic segmentation for sensing surroundings. IEEE Robotics and Automation Letters, 5(3), 4867\u20134873.","journal-title":"IEEE Robotics and Automation Letters"},{"key":"2464_CR74","doi-asserted-by":"crossref","unstructured":"Saha, A., Mendez, O., Russell, C., & Bowden, R. (2021). Enabling spatio-temporal aggregation in birds-eye-view vehicle estimation. In: 2021 Ieee International Conference on Robotics and Automation (icra), pp. 5133\u20135139. IEEE.","DOI":"10.1109\/ICRA48506.2021.9561169"},{"key":"2464_CR75","doi-asserted-by":"crossref","unstructured":"Hu, A., Murez, Z., Mohan, N., Dudas, S., Hawke, J., Badrinarayanan, V., Cipolla, R., & Kendall, A. (2021). Fiery: Future instance prediction in bird\u2019s-eye view from surround monocular cameras. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15273\u201315282.","DOI":"10.1109\/ICCV48922.2021.01499"},{"key":"2464_CR76","unstructured":"Huang, J., Huang, G., Zhu, Z., Ye, Y., Du, D. (2021). Bevdet: High-performance multi-camera 3d object detection in bird-eye-view. arXiv preprint arXiv:2112.11790."},{"key":"2464_CR77","doi-asserted-by":"crossref","unstructured":"Zhang, H., Zhou, W., Zhu, Y., Yan, X., Gao, J., Bai, D., Cai, Y., Liu, B., Cui, S., & Li, Z. (2024). Visionpad: A vision-centric pre-training paradigm for autonomous driving. arXiv preprint arXiv:2411.14716.","DOI":"10.1109\/CVPR52734.2025.01600"},{"key":"2464_CR78","doi-asserted-by":"crossref","unstructured":"Yang, C., Chen, Y., Tian, H., Tao, C., Zhu, X., Zhang, Z., Huang, G., Li, H., Qiao, Y., & Lu, L. (2023). Bevformer v2: Adapting modern image backbones to bird\u2019s-eye-view recognition via perspective supervision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17830\u201317839.","DOI":"10.1109\/CVPR52729.2023.01710"},{"key":"2464_CR79","doi-asserted-by":"crossref","unstructured":"Han, C., Sun, J., Ge, Z., Yang, J., Dong, R., Zhou, H., Mao, W., Peng, Y., & Zhang, X. (2023). Exploring recurrent long-term temporal fusion for multi-view 3d perception. arXiv preprint arXiv:2303.05970.","DOI":"10.1109\/LRA.2024.3401172"},{"key":"2464_CR80","unstructured":"Lin, X., Lin, T., Pei, Z., Huang, L., & Su, Z. (2023). Sparse4D v2: Recurrent Temporal Fusion with Sparse Model."},{"issue":"4","key":"2464_CR81","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3592433","volume":"42","author":"B Kerbl","year":"2023","unstructured":"Kerbl, B., Kopanas, G., Leimk\u00fchler, T., & Drettakis, G. (2023). 3d gaussian splatting for real-time radiance field rendering. ACM Trans. Graph., 42(4), 139\u20131.","journal-title":"ACM Trans. Graph."},{"key":"2464_CR82","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, W., Li, H., Xie, E., Sima, C., Lu, T., Qiao, Y., & Dai, J. (2022). Bevformer: Learning bird\u2019s-eye-view representation from multi-camera images via spatiotemporal transformers. In: European Conference on Computer Vision, pp. 1\u201318. Springer.","DOI":"10.1007\/978-3-031-20077-9_1"},{"key":"2464_CR83","unstructured":"Chen, S., Cheng, T., Wang, X., Meng, W., Zhang, Q., & Liu, W. (2022). Efficient and robust 2d-to-bev representation learning via geometry-guided kernel transformer. arXiv preprint arXiv:2206.04584."},{"key":"2464_CR84","doi-asserted-by":"crossref","unstructured":"Li, Q., Wang, Y., Wang, Y., & Zhao, H. (2022). Hdmapnet: An online hd map construction and evaluation framework. In: 2022 International Conference on Robotics and Automation (ICRA), pp. 4628\u20134634. IEEE.","DOI":"10.1109\/ICRA46639.2022.9812383"},{"key":"2464_CR85","unstructured":"Tan, M., & Le, Q. (2019). Efficientnet: Rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, pp. 6105\u20136114. PMLR."},{"key":"2464_CR86","doi-asserted-by":"crossref","unstructured":"Lang, A.H., Vora, S., Caesar, H., Zhou, L., Yang, J., & Beijbom, O. (2019). Pointpillars: Fast encoders for object detection from point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12697\u201312705.","DOI":"10.1109\/CVPR.2019.01298"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02464-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02464-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02464-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,9]],"date-time":"2025-09-09T07:58:46Z","timestamp":1757404726000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02464-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,24]]},"references-count":86,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["2464"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02464-w","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"type":"print","value":"0920-5691"},{"type":"electronic","value":"1573-1405"}],"subject":[],"published":{"date-parts":[[2025,5,24]]},"assertion":[{"value":"21 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 April 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All authors declare no conflicts of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}},{"value":"The toolkit and experimental results are publicly available at .","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Code availability"}}]}}