{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T06:04:30Z","timestamp":1785305070021,"version":"3.55.0"},"reference-count":60,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Zhejiang Provincial Natural Science Foundation of China","award":["LD24F020016"],"award-info":[{"award-number":["LD24F020016"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s11263-026-02934-9","type":"journal-article","created":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T12:45:57Z","timestamp":1784033157000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["LightOcc: Lightweight Spatial Embedding for Efficient Vision-based 3D Occupancy Prediction"],"prefix":"10.1007","volume":"134","author":[{"given":"Jinqing","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenrui","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5181-6451","authenticated-orcid":false,"given":"Qingjie","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunhong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,14]]},"reference":[{"key":"2934_CR1","doi-asserted-by":"crossref","unstructured":"Behley, J., Garbade, M., Milioto, A., Quenzel, J., Behnke, S., Stachniss, C., & Gall, J. (2019). Semantickitti: A dataset for semantic scene understanding of lidar sequences. Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 9297\u20139307)","DOI":"10.1109\/ICCV.2019.00939"},{"key":"2934_CR2","doi-asserted-by":"crossref","unstructured":"Berman, M., Triki, A. R., & Blaschko, M. B. (2018). The lov\u00e1sz-softmax loss: A tractable surrogate for the optimization of the intersection-over-union measure in neural networks. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (pp. 4413\u20134421)","DOI":"10.1109\/CVPR.2018.00464"},{"key":"2934_CR3","doi-asserted-by":"crossref","unstructured":"Caesar, H., Bankiti, V., Lang, A. H., Vora, S., Liong, V. E., Xu, Q., Krishnan, A., Pan, Y., Baldan, G., & Beijbom, O. (2020). nuscenes: A multimodal dataset for autonomous driving. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 11621\u201311631)","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"2934_CR4","doi-asserted-by":"crossref","unstructured":"Cao, A.-Q., Charette, D., & R. (2022). Monoscene: Monocular 3d semantic scene completion. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 3991\u20134001)","DOI":"10.1109\/CVPR52688.2022.00396"},{"key":"2934_CR5","doi-asserted-by":"crossref","unstructured":"Duan, Z., Dang, C., Hu, X., An, P., Ding, J., Zhan, J., Xu, Y., & Ma, J. (2025). Sdgocc: Semantic and depth-guided bird\u2019s-eye view transformation for 3d multimodal occupancy prediction. Proceedings of the Computer Vision and Pattern Recognition Conference (pp. 6751\u20136760)","DOI":"10.1109\/CVPR52734.2025.00633"},{"key":"2934_CR6","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., & Urtasun, R. (2012). Are we ready for autonomous driving? the kitti vision benchmark suite. IEEE Conference on Computer Vision and Pattern Recognition (pp. 3354\u20133361). IEEE.","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"2934_CR7","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (pp. 770\u2013778)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2934_CR8","doi-asserted-by":"crossref","unstructured":"Hou, J., Li, X., Guan, W., Zhang, G., Feng, D., Du, Y., Xue, X., & Pu, J. (2024). Fastocc: Accelerating 3d occupancy prediction by fusing the 2d bird\u2019s-eye view and perspective view arXiv:2403.02710.","DOI":"10.1109\/ICRA57147.2024.10610625"},{"key":"2934_CR9","unstructured":"Huang, J.,& Huang, G. (2022). Bevdet4d: Exploit temporal cues in multi-camera 3d object detection. arXiv:2203.17054"},{"key":"2934_CR10","unstructured":"Huang, J.,& Huang, G. (2022). Bevpoolv2: A cutting-edge implementation of bevdet toward deployment. arXiv:2211.17111"},{"key":"2934_CR11","unstructured":"Huang, J., Huang, G., Zhu, Z., Ye, Y., & Du, D. (2021). Bevdet: High-performance multi-camera 3d object detection in bird-eye-view. arXiv:2112.11790"},{"key":"2934_CR12","doi-asserted-by":"crossref","unstructured":"Huang, Y., Thammatadatrakoon, A., Zheng, W., Zhang, Y., Du, D., & Lu, J. (2025). Gaussianformer-2: Probabilistic gaussian superposition for efficient 3d occupancy prediction. Proceedings of the Computer Vision and Pattern Recognition Conference (pp. 27477\u201327486)","DOI":"10.1109\/CVPR52734.2025.02559"},{"key":"2934_CR13","doi-asserted-by":"crossref","unstructured":"Huang, Y., Zheng, W., Zhang, Y., Zhou, J., & Lu, J. (2023). Tri-perspective view for vision-based 3d semantic occupancy prediction. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 9223\u20139232)","DOI":"10.1109\/CVPR52729.2023.00890"},{"key":"2934_CR14","doi-asserted-by":"crossref","unstructured":"Huang, Y., Zheng, W., Zhang, B., Zhou, J., & Lu, J. (2024). Selfocc: Self-supervised vision-based 3d occupancy prediction. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 19946\u201319956)","DOI":"10.1109\/CVPR52733.2024.01885"},{"key":"2934_CR15","doi-asserted-by":"crossref","unstructured":"Huang, Y., Zheng, W., Zhang, Y., Zhou, J., & Lu, J. (2024). Gaussianformer: Scene as gaussians for vision-based 3d semantic occupancy prediction. European Conference on Computer Vision (pp. 376\u2013393). Springer.","DOI":"10.1007\/978-3-031-73383-3_22"},{"key":"2934_CR16","doi-asserted-by":"crossref","unstructured":"Jiang, Y., Zhang, L., Miao, Z., Zhu, X., Gao, J., & Hu, W.,& Jiang, Y.-G. (2023). Polarformer: Multi-camera 3d object detection with polar transformer. Proceedings of the AAAI Conference on Artificial Intelligence,37, 1042\u20131050.","DOI":"10.1609\/aaai.v37i1.25185"},{"key":"2934_CR17","doi-asserted-by":"crossref","unstructured":"Jiang, H., Cheng, T., Gao, N., Zhang, H., Lin, T., Liu, W., W., & X. (2024). Symphonize 3d semantic scene completion with contextual instance queries. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 20258\u201320267)","DOI":"10.1109\/CVPR52733.2024.01915"},{"key":"2934_CR18","doi-asserted-by":"crossref","unstructured":"Jiang, Z., Zhang, J., Zhang, Y., Liu, Q., Hu, Z., Wang, B.,& Wang, Y. (2025). Fsd-bev: Foreground self-distillation for multi-view 3d object detection. In: European Conference on Computer Vision, pp. 110\u2013126 . Springer","DOI":"10.1007\/978-3-031-73242-3_7"},{"key":"2934_CR19","doi-asserted-by":"crossref","unstructured":"Li, Y., Bao, H., Ge, Z., Yang, J., & Sun, J.,& Li, Z. (2023). Bevstereo: Enhancing depth estimation in multi-view 3d object detection with temporal stereo. Proceedings of the AAAI Conference on Artificial Intelligence,37, 1486\u20131494.","DOI":"10.1609\/aaai.v37i2.25234"},{"key":"2934_CR20","doi-asserted-by":"crossref","unstructured":"Li, Y., Ge, Z., Yu, G., Yang, J., Wang, Z., Shi, Y., & Sun, J.,& Li, Z. (2023). Bevdepth: Acquisition of reliable depth for multi-view 3d object detection. Proceedings of the AAAI Conference on Artificial Intelligence,37, 1477\u20131485.","DOI":"10.1609\/aaai.v37i2.25233"},{"key":"2934_CR21","doi-asserted-by":"crossref","unstructured":"Li, H., Hou, Y., Xing, X., Ma, Y., Sun, X., & Zhang, Y. (2025). Occmamba: Semantic occupancy prediction with state space models. Proceedings of the Computer Vision and Pattern Recognition Conference (pp. 11949\u201311959)","DOI":"10.1109\/CVPR52734.2025.01116"},{"key":"2934_CR22","doi-asserted-by":"crossref","unstructured":"Li, Z., Lan, S., Alvarez, J. M., & Wu, Z. (2024). Bevnext: Reviving dense bev frameworks for 3d object detection. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 20113\u201320123)","DOI":"10.1109\/CVPR52733.2024.01901"},{"key":"2934_CR23","doi-asserted-by":"crossref","unstructured":"Liu, H., Chen, Y., Wang, H., Yang, Z., Li, T., Zeng, J., Chen, L., Li, H., & Wang, L. (2024). Fully sparse 3d occupancy prediction. In: European Conference on Computer Vision, pp. 54\u201371 . Springer","DOI":"10.1007\/978-3-031-72698-9_4"},{"key":"2934_CR24","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S.,& Guo, B. (2021). Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2934_CR25","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, W., Li, H., Xie, E., Sima, C., Lu, T., Qiao, Y.,& Dai, J. (2022). Bevformer: Learning bird\u2019s-eye-view representation from multi-camera images via spatiotemporal transformers. In: European Conference on Computer Vision, pp. 1\u201318 Springer","DOI":"10.1007\/978-3-031-20077-9_1"},{"key":"2934_CR26","unstructured":"Li, Z., Yu, Z., Austin, D., Fang, M., Lan, S., Kautz, J., & Alvarez, J. M. (2023). Fb-occ: 3d occupancy prediction based on forward-backward view transformation arXiv:2307.01492."},{"key":"2934_CR27","doi-asserted-by":"crossref","unstructured":"Li, Y., Yu, Z., Choy, C., Xiao, C., Alvarez, J. M., Fidler, S., Feng, C., & Anandkumar, A. (2023). Voxformer: Sparse voxel transformer for camera-based 3d semantic scene completion. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 9087\u20139098)","DOI":"10.1109\/CVPR52729.2023.00877"},{"key":"2934_CR28","doi-asserted-by":"crossref","unstructured":"Li, H., Zhang, H., Zeng, Z., Liu, S., Li, F., Ren, T.,& Zhang, L. (2023). Dfa3d: 3d deformable attention for 2d-to-3d feature lifting. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6684\u20136693","DOI":"10.1109\/ICCV51070.2023.00615"},{"key":"2934_CR29","unstructured":"Mao, J., Shi, S., Wang, X.,& Li, H. (2022). 3d object detection for autonomous driving: A review and new outlooks. arXiv:2206.09474 1, 1"},{"key":"2934_CR30","doi-asserted-by":"crossref","unstructured":"Ma, Q., Tan, X., Qu, Y., Ma, L., Zhang, Z., & Xie, Y. (2024). Cotr: Compact occupancy transformer for vision-based 3d occupancy prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19936\u201319945","DOI":"10.1109\/CVPR52733.2024.01884"},{"key":"2934_CR31","doi-asserted-by":"crossref","unstructured":"Ma, Y., Wang, T., Bai, X., Yang, H., Hou, Y., Wang, Y., Qiao, Y., Yang, R.,& Zhu, X. (2024). Vision-centric bev perception: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence","DOI":"10.1109\/TPAMI.2024.3449912"},{"key":"2934_CR32","doi-asserted-by":"crossref","unstructured":"Murez, Z., Van As, T., Bartolozzi, J., Sinha, A., Badrinarayanan, V., & Rabinovich, A. (2020). Atlas: End-to-end 3d scene reconstruction from posed images. European Conference on Computer Vision (pp. 414\u2013431). Springer.","DOI":"10.1007\/978-3-030-58571-6_25"},{"key":"2934_CR33","doi-asserted-by":"crossref","unstructured":"Oh, G., Kim, S., Ko, H., Chi, H.-g., Kim, J., Lee, D., Ji, D., Choi, S., Jang, S.,& Kim, S. (2025). 3d occupancy prediction with low-resolution queries via prototype-aware view transformation. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 17134\u201317144","DOI":"10.1109\/CVPR52734.2025.01597"},{"key":"2934_CR34","doi-asserted-by":"crossref","unstructured":"Pan, M., Liu, J., Zhang, R., Huang, P., Li, X., Xie, H., Wang, B., Liu, L.,& Zhang, S. (2024). Renderocc: Vision-centric 3d occupancy prediction with 2d rendering supervision. In: 2024 IEEE International Conference on Robotics and Automation (ICRA), pp. 12404\u201312411 . IEEE","DOI":"10.1109\/ICRA57147.2024.10611537"},{"key":"2934_CR35","doi-asserted-by":"crossref","unstructured":"Philion, J., & Fidler, S. (2020). Lift, splat, shoot: Encoding images from arbitrary camera rigs by implicitly unprojecting to 3d. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XIV 16, pp. 194\u2013210 . Springer","DOI":"10.1007\/978-3-030-58568-6_12"},{"key":"2934_CR36","doi-asserted-by":"crossref","unstructured":"Qian, R., Lai, X., & Li, X. (2022). 3d object detection for autonomous driving: A survey. Pattern Recognition,130, Article 108796.","DOI":"10.1016\/j.patcog.2022.108796"},{"key":"2934_CR37","doi-asserted-by":"crossref","unstructured":"Shamsafar, F., Woerz, S., Rahim, R., & Zell, A. (2022). Mobilestereonet: Towards lightweight deep networks for stereo matching. Proceedings of the Ieee\/cvf Winter Conference on Applications of Computer Vision (pp. 2417\u20132426)","DOI":"10.1109\/WACV51458.2022.00075"},{"key":"2934_CR38","doi-asserted-by":"crossref","unstructured":"Shi, Y., Cheng, T., Zhang, Q., Liu, W., W., & X. (2024). Occupancy as set of points arXiv:2407.04049.","DOI":"10.1007\/978-3-031-73030-6_5"},{"key":"2934_CR39","unstructured":"Shi, Y., Jiang, K., Li, J., Qian, Z., Wen, J., Yang, M., Wang, K.,& Yang, D. (2023). Grid-centric traffic scenario perception for autonomous driving: A comprehensive review. arXiv:2303.01212"},{"key":"2934_CR40","unstructured":"Silva, S., Wannigama, S. B., Ragel, R., & Jayatilaka, G. (2024). S2tpvformer: Spatio-temporal tri-perspective view for temporally coherent 3d semantic occupancy prediction arXiv:2401.13785."},{"key":"2934_CR41","doi-asserted-by":"crossref","unstructured":"Tian, X., Jiang, T., Yun, L., Mao, Y., Yang, H., Wang, Y., Wang, Y., & Zhao, H. (2024). Occ3d: A large-scale 3d occupancy prediction benchmark for autonomous driving. Advances in Neural Information Processing Systems 36","DOI":"10.52202\/075280-2809"},{"key":"2934_CR42","doi-asserted-by":"crossref","unstructured":"Wang, Y., Chen, Y., Liao, X., Fan, L., & Zhang, Z. (2024). Panoocc: Unified occupancy representation for camera-based 3d panoptic segmentation. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 17158\u201317168)","DOI":"10.1109\/CVPR52733.2024.01624"},{"key":"2934_CR43","doi-asserted-by":"publisher","first-page":"119861","DOI":"10.52202\/079017-3809","volume":"37","author":"J Wang","year":"2024","unstructured":"Wang, J., Liu, Z., Meng, Q., Yan, L., Wang, K., Yang, J., Liu, W., Hou, Q., & Cheng, M.-M. (2024). Opus: occupancy prediction using a sparse set. Advances in Neural Information Processing Systems, 37, 119861\u2013119885.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2934_CR44","doi-asserted-by":"crossref","unstructured":"Wang, G., Wang, Z., Tang, P., Zheng, J., Ren, X., Feng, B.,& Ma, C. (2024). Occgen: Generative multi-modal 3d occupancy prediction for autonomous driving. In: European Conference on Computer Vision, pp. 95\u2013112 . Springer","DOI":"10.1007\/978-3-031-72661-3_6"},{"issue":"7","key":"2934_CR45","doi-asserted-by":"publisher","first-page":"3781","DOI":"10.1109\/TIV.2023.3264658","volume":"8","author":"L Wang","year":"2023","unstructured":"Wang, L., Zhang, X., Song, Z., Bi, J., Zhang, G., Wei, H., Tang, L., Yang, L., Li, J., & Jia, C. (2023). Multi-modal 3d object detection in autonomous driving: A survey and taxonomy. IEEE Transactions on Intelligent Vehicles, 8(7), 3781\u20133798.","journal-title":"IEEE Transactions on Intelligent Vehicles"},{"key":"2934_CR46","unstructured":"Wang, K., Zhu, J., Ren, M., Liu, Z., Li, S., Zhang, Z., Zhang, C., Wu, X., Zhan, Q., Liu, Q., et\u00a0al. (2024). A survey on data synthesis and augmentation for large language models. arXiv:2410.12896"},{"key":"2934_CR47","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhu, Z., Xu, W., Zhang, Y., Wei, Y., Chi, X., Ye, Y., Du, D., Lu, J.,& Wang, X. (2023). Openoccupancy: A large scale benchmark for surrounding semantic occupancy perception. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 17850\u201317859","DOI":"10.1109\/ICCV51070.2023.01636"},{"key":"2934_CR48","doi-asserted-by":"crossref","unstructured":"Wei, Y., Zhao, L., Zheng, W., Zhu, Z., Zhou, J., & Lu, J. (2023). Surroundocc: Multi-camera 3d occupancy prediction for autonomous driving. Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 21729\u201321740)","DOI":"10.1109\/ICCV51070.2023.01986"},{"key":"2934_CR49","doi-asserted-by":"crossref","unstructured":"Wu, Y., Yan, Z., Wang, Z., Li, X., Hui, L., & Yang, J. (2024). Deep height decoupling for precise vision-based 3d occupancy prediction. arXiv:2409.07972","DOI":"10.1109\/ICRA55743.2025.11128708"},{"key":"2934_CR50","doi-asserted-by":"crossref","unstructured":"Xu, H., Chen, J., Meng, S., Wang, Y., & Chau, L.-P. (2025). A survey on occupancy perception for autonomous driving: The information fusion perspective. Information Fusion,114, Article 102671.","DOI":"10.1016\/j.inffus.2024.102671"},{"key":"2934_CR51","doi-asserted-by":"crossref","unstructured":"Yang, C., Chen, Y., Tian, H., Tao, C., Zhu, X., Zhang, Z., Huang, G., Li, H., Qiao, Y., & Lu, L. (2023). Bevformer v2: Adapting modern image backbones to bird\u2019s-eye-view recognition via perspective supervision. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 17830\u201317839)","DOI":"10.1109\/CVPR52729.2023.01710"},{"key":"2934_CR52","doi-asserted-by":"crossref","unstructured":"Ye, Z., Jiang, T., Xu, C., Li, Y., & Zhao, H. (2024). Cvt-occ: Cost volume temporal fusion for 3d occupancy prediction. European Conference on Computer Vision (pp. 381\u2013397). Springer.","DOI":"10.1007\/978-3-031-73464-9_23"},{"key":"2934_CR53","unstructured":"Yu, Z., Shu, C., Deng, J., Lu, K., Liu, Z., Yu, J., Yang, D., Li, H.,& Chen, Y. (2023). Flashocc: Fast and memory-efficient occupancy prediction via channel-to-height plugin. arXiv:2311.12058"},{"key":"2934_CR54","doi-asserted-by":"publisher","first-page":"1531","DOI":"10.52202\/079017-0049","volume":"37","author":"Z Yu","year":"2024","unstructured":"Yu, Z., Zhang, R., Ying, J., Yu, J., Hu, X., Luo, L., Cao, S.-Y., & Shen, H.-L. (2024). Context and geometry aware voxel transformer for semantic scene completion. Advances in Neural Information Processing Systems, 37, 1531\u20131555.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2934_CR55","doi-asserted-by":"crossref","unstructured":"Zhang, J., Zhang, Y., Liu, Q.,& Wang, Y. (2023). Sa-bev: Generating semantic-aware bird\u2019s-eye-view feature for multi-view 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3348\u20133357","DOI":"10.1109\/ICCV51070.2023.00310"},{"key":"2934_CR56","doi-asserted-by":"crossref","unstructured":"Zhang, J., Zhang, Y., Qi, Y., Fu, Z., Liu, Q.,& Wang, Y. (2024). Geobev: Learning geometric bev representation for multi-view 3d object detection. arXiv:2409.01816","DOI":"10.1609\/aaai.v39i9.33080"},{"key":"2934_CR57","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Zhang, J., Wang, Z., Xu, J.,& Huang, D. (2024). Vision-based 3d occupancy prediction in autonomous driving: a review and outlook. arXiv:2405.02595","DOI":"10.1007\/s11704-024-40443-5"},{"key":"2934_CR58","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Zhu, Z., & Du, D. (2023). Occformer: Dual-path transformer for vision-based 3d semantic occupancy prediction. Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 9433\u20139443)","DOI":"10.1109\/ICCV51070.2023.00865"},{"key":"2934_CR59","doi-asserted-by":"crossref","unstructured":"Zhao, L., Xu, X., Wang, Z., Zhang, Y., Zhang, B., Zheng, W., Du, D., Zhou, J.,& Lu, J. (2024). Lowrankocc: tensor decomposition and low-rank recovery for vision-based 3d semantic occupancy prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9806\u20139815","DOI":"10.1109\/CVPR52733.2024.00936"},{"key":"2934_CR60","doi-asserted-by":"crossref","unstructured":"Zuo, S., Zheng, W., Huang, Y., Zhou, J., & Lu, J. (2025). Gaussianworld: Gaussian world model for streaming 3d occupancy prediction. Proceedings of the Computer Vision and Pattern Recognition Conference (pp. 6772\u20136781)","DOI":"10.1109\/CVPR52734.2025.00635"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02934-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02934-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02934-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T05:49:55Z","timestamp":1785304195000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02934-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":60,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["2934"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02934-9","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"17 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 June 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"347"}}