{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T07:37:17Z","timestamp":1773128237080,"version":"3.50.1"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,2,8]],"date-time":"2026-02-08T00:00:00Z","timestamp":1770508800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,8]],"date-time":"2026-02-08T00:00:00Z","timestamp":1770508800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Machine Vision and Applications"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s00138-026-01793-1","type":"journal-article","created":{"date-parts":[[2026,2,8]],"date-time":"2026-02-08T16:51:04Z","timestamp":1770569464000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["LS-Occ:light specific-target-focus vision-based 3D occupancy prediction with adaptive combined head"],"prefix":"10.1007","volume":"37","author":[{"given":"Shuo","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xu","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zihang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,8]]},"reference":[{"key":"1793_CR1","unstructured":"Huang, J., Huang, G., Zhu, Z., Ye, Y., Du, D.: Bevdet: High-performance multi-camera 3d object detection in bird-eye-view. arXiv preprint (2021) arXiv:2307.01492"},{"key":"1793_CR2","doi-asserted-by":"publisher","unstructured":"Li, Y., Ge, Z., Yu, G., Yang, J., Wang, Z., Shi, Y., Sun, J., Li, Z.: Bevdepth: Acquisition of reliable depth for multi-view 3d object detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 1477\u20131485 (2023). https:\/\/doi.org\/10.1609\/aaai.v37i2.25233","DOI":"10.1609\/aaai.v37i2.25233"},{"key":"1793_CR3","doi-asserted-by":"publisher","unstructured":"Li, Y., Bao, H., Ge, Z., Yang, J., Sun, J., Li, Z.: Bevstereo: Enhancing depth estimation in multi-view 3d object detection with temporal stereo. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 1486\u20131494 (2023). https:\/\/doi.org\/10.1609\/aaai.v37i2.25234","DOI":"10.1609\/aaai.v37i2.25234"},{"key":"1793_CR4","doi-asserted-by":"publisher","unstructured":"Zhang, J., Zhang, Y., Liu, Q., Wang, Y.: Sa-bev: Generating semantic-aware bird\u2019s-eye-view feature for multi-view 3d object detection. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 3325\u20133334 (2023). https:\/\/doi.org\/10.1109\/ICCV51070.2023.00310","DOI":"10.1109\/ICCV51070.2023.00310"},{"key":"1793_CR5","doi-asserted-by":"publisher","unstructured":"Yang, C., Chen, Y., Tian, H., Tao, C., Zhu, X., Zhang, Z., Huang, G., Li, H., Qiao, Y., Lu, L., Zhou, J., Dai, J.: Bevformer v2: Adapting modern image backbones to bird\u2019s-eye-view recognition via perspective supervision. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 17830\u201317839 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.01710","DOI":"10.1109\/CVPR52729.2023.01710"},{"issue":"3","key":"1793_CR6","doi-asserted-by":"publisher","first-page":"2020","DOI":"10.1109\/TPAMI.2024.3515454","volume":"47","author":"Z Li","year":"2025","unstructured":"Li, Z., Wang, W., Li, H., Xie, E., Sima, C., Lu, T., Yu, Q., Dai, J.: Bevformer: learning bird\u2019s-eye-view representation from lidar-camera via spatiotemporal transformers. IEEE Trans. Pattern Anal. Mach. Intell. 47(3), 2020\u20132036 (2025). https:\/\/doi.org\/10.1109\/TPAMI.2024.3515454","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1793_CR7","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.108796","volume":"130","author":"R Qian","year":"2022","unstructured":"Qian, R., Lai, X., Li, X.: 3d object detection for autonomous driving: a survey. Pattern Recogn. 130, 108796 (2022). https:\/\/doi.org\/10.1016\/j.patcog.2022.108796","journal-title":"Pattern Recogn."},{"key":"1793_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102671","volume":"114","author":"H Xu","year":"2025","unstructured":"Xu, H., Chen, J., Meng, S., Wang, Y., Chau, L.-P.: A survey on occupancy perception for autonomous driving: the information fusion perspective. Inf. Fusion 114, 102671 (2025). https:\/\/doi.org\/10.1016\/j.inffus.2024.102671","journal-title":"Inf. Fusion"},{"key":"1793_CR9","unstructured":"Li, Z., Yu, Z., Austin, D., Fang, M., Lan, S., Kautz, J., Alvarez, J.M.: FB-OCC: 3D occupancy prediction based on forward-backward view transformation. arXiv e-prints, 2307\u201301492 (2023) arXiv:2307.01492 [cs.CV]"},{"key":"1793_CR10","doi-asserted-by":"publisher","unstructured":"Wang, Y., Chen, Y., Liao, X., Fan, L., Zhang, Z.: Panoocc: Unified occupancy representation for camera-based 3d panoptic segmentation. In: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 17158\u201317168 (2024). https:\/\/doi.org\/10.1109\/CVPR52733.2024.01624","DOI":"10.1109\/CVPR52733.2024.01624"},{"key":"1793_CR11","unstructured":"Yu, Z., Shu, C., Deng, J., Lu, K., Liu, Z., Yu, J., Yang, D., Li, H., Chen, Y.: FlashOcc: fast and memory-efficient occupancy prediction via channel-to-height plugin. arXiv e-prints, 2311\u201312058 (2023) arXiv:2311.12058 [cs.CV]"},{"key":"1793_CR12","doi-asserted-by":"publisher","unstructured":"Tang, P., Wang, Z., Wang, G., Zheng, J., Ren, X., Feng, B., Ma, C.: Sparseocc: Rethinking sparse latent representation for vision-based semantic occupancy prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15035\u201315044. Springer, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-72698-9_4","DOI":"10.1007\/978-3-031-72698-9_4"},{"key":"1793_CR13","doi-asserted-by":"publisher","unstructured":"Ma, J., Zhao, Z., Yi, X., Chen, J., Hong, L., Chi, E.H.: Modeling task relationships in multi-task learning with multi-gate mixture-of-experts. In: Proceedings of the 24th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 1930\u20131939. Association for Computing Machinery, New York, NY, USA (2018). https:\/\/doi.org\/10.1145\/3219819.3220007","DOI":"10.1145\/3219819.3220007"},{"key":"1793_CR14","doi-asserted-by":"publisher","unstructured":"Cipolla, R., Gal, Y., Kendall, A.: Multi-task learning using uncertainty to weigh losses for scene geometry and semantics. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7482\u20137491 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00781","DOI":"10.1109\/CVPR.2018.00781"},{"key":"1793_CR15","doi-asserted-by":"publisher","unstructured":"Tang, H., Liu, J., Zhao, M., Gong, X.: Progressive layered extraction (ple): A novel multi-task learning (mtl) model for personalized recommendations. In: Proceedings of the 14th ACM Conference on Recommender Systems, pp. 269\u2013278. Association for Computing Machinery, New York, NY, USA (2020). https:\/\/doi.org\/10.1145\/3383313.3412236","DOI":"10.1145\/3383313.3412236"},{"key":"1793_CR16","unstructured":"Wang, Y., Guizilini, V.C., Zhang, T., Wang, Y., Zhao, H., Solomon, J.: Detr3d: 3d object detection from multi-view images via 3d-to-2d queries. In: Conference on Robot Learning, pp. 180\u2013191 (2022). PMLR"},{"key":"1793_CR17","doi-asserted-by":"publisher","unstructured":"Wei, Y., Zhao, L., Zheng, W., Zhu, Z., Zhou, J., Lu, J.: Surroundocc: Multi-camera 3d occupancy prediction for autonomous driving. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 21672\u201321683 (2023). https:\/\/doi.org\/10.1109\/ICCV51070.2023.01986","DOI":"10.1109\/ICCV51070.2023.01986"},{"key":"1793_CR18","unstructured":"Hong, Y., Liu, Q., Cheng, H., Ma, D., Dai, H., Wang, Y., Cao, G., Ding, Y.: UniVision: a unified framework for vision-centric 3D perception. arXiv e-prints, 2401\u201306994 (2024) arXiv:2401.06994 [cs.CV]"},{"key":"1793_CR19","doi-asserted-by":"publisher","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"1793_CR20","doi-asserted-by":"publisher","unstructured":"Lin, T.-Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 936\u2013944 (2017). https:\/\/doi.org\/10.1109\/CVPR.2017.106","DOI":"10.1109\/CVPR.2017.106"},{"key":"1793_CR21","doi-asserted-by":"publisher","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 9992\u201310002 (2021). https:\/\/doi.org\/10.1109\/ICCV48922.2021.00986","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"1793_CR22","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint (2020) arXiv:2010.11929"},{"key":"1793_CR23","doi-asserted-by":"publisher","unstructured":"Ghiasi, G., Lin, T.-Y., Le, Q.V.: Nas-fpn: Learning scalable feature pyramid architecture for object detection. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7029\u20137038 (2019). https:\/\/doi.org\/10.1109\/CVPR.2019.00720","DOI":"10.1109\/CVPR.2019.00720"},{"key":"1793_CR24","doi-asserted-by":"publisher","unstructured":"Tan, M., Pang, R., Le, Q.V.: Efficientdet: scalable and efficient object detection. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10778\u201310787 (2020). https:\/\/doi.org\/10.1109\/CVPR42600.2020.01079","DOI":"10.1109\/CVPR42600.2020.01079"},{"key":"1793_CR25","doi-asserted-by":"publisher","unstructured":"Qiao, S., Chen, L.-C., Yuille, A.: Detectors: detecting objects with recursive feature pyramid and switchable atrous convolution. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10208\u201310219 (2021). https:\/\/doi.org\/10.1109\/CVPR46437.2021.01008","DOI":"10.1109\/CVPR46437.2021.01008"},{"key":"1793_CR26","doi-asserted-by":"publisher","unstructured":"Yang, G., Lei, J., Tian, H., Feng, Z., Liang, R.: Asymptotic feature pyramid network for labeling pixels and regions 34, 7820\u20137829 (2024). https:\/\/doi.org\/10.1109\/TCSVT.2024.3376773","DOI":"10.1109\/TCSVT.2024.3376773"},{"key":"1793_CR27","doi-asserted-by":"publisher","unstructured":"Hou, J., Li, X., Guan, W., Zhang, G., Feng, D., Du, Y., Xue, X., Pu, J.: Fastocc: accelerating 3d occupancy prediction by fusing the 2d bird\u2019s-eye view and perspective view. In: 2024 IEEE International Conference on Robotics and Automation (ICRA), pp. 16425\u201316431 (2024). https:\/\/doi.org\/10.1109\/ICRA57147.2024.10610625","DOI":"10.1109\/ICRA57147.2024.10610625"},{"key":"1793_CR28","doi-asserted-by":"crossref","unstructured":"Zhou, B., Kr\u00e4henb\u00fchl, P.: Cross-view transformers for real-time map-view semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13760\u201313769 (2022)","DOI":"10.1109\/CVPR52688.2022.01339"},{"key":"1793_CR29","doi-asserted-by":"crossref","unstructured":"Philion, J., Fidler, S.: Lift, splat, shoot: Encoding images from arbitrary camera rigs by implicitly unprojecting to 3d. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) Computer Vision - ECCV 2020, pp. 194\u2013210. Springer, Cham (2020)","DOI":"10.1007\/978-3-030-58568-6_12"},{"key":"1793_CR30","unstructured":"Huang, J., Huang, G.: Bevpoolv2: A cutting-edge implementation of bevdet toward deployment. arXiv preprint (2022) arXiv:2409.22111"},{"key":"1793_CR31","doi-asserted-by":"publisher","unstructured":"Cao, A.-Q., Charette, R.: Monoscene: monocular 3d semantic scene completion. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3981\u20133991 (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.00396","DOI":"10.1109\/CVPR52688.2022.00396"},{"key":"1793_CR32","doi-asserted-by":"crossref","unstructured":"Caesar, H., Bankiti, V., Lang, A.H., Vora, S., Liong, V.E., Xu, Q., Krishnan, A., Pan, Y., Baldan, G., Beijbom, O.: nuscenes: a multimodal dataset for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11621\u201311631 (2020)","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"1793_CR33","first-page":"64318","volume":"36","author":"X Tian","year":"2023","unstructured":"Tian, X., Jiang, T., Yun, L., Mao, Y., Yang, H., Wang, Y., Wang, Y., Zhao, H.: Occ3d: A large-scale 3d occupancy prediction benchmark for autonomous driving. Adv. Neural. Inf. Process. Syst. 36, 64318\u201364330 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1793_CR34","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint (2017) arXiv:1711.05101"},{"key":"1793_CR35","doi-asserted-by":"publisher","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.-Y., Doll\u00e1r, P., Girshick, R.: Segment anything, 3992\u20134003 (2023) https:\/\/doi.org\/10.1109\/ICCV51070.2023.00371","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"1793_CR36","doi-asserted-by":"crossref","unstructured":"Wu, Y., Yan, Z., Wang, Z., Li, X., Hui, L., Yang, J.: Deep height decoupling for precise vision-based 3d occupancy prediction. arXiv preprint (2024) arXiv:2409.07972","DOI":"10.1109\/ICRA55743.2025.11128708"},{"key":"1793_CR37","doi-asserted-by":"publisher","unstructured":"Huang, Y., Zheng, W., Zhang, Y., Zhou, J., Lu, J.: Tri-perspective view for vision-based 3d semantic occupancy prediction. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 9223\u20139232 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.00890","DOI":"10.1109\/CVPR52729.2023.00890"},{"key":"1793_CR38","doi-asserted-by":"publisher","unstructured":"Zhang, Y., Zhu, Z., Du, D.: Occformer: Dual-path transformer for vision-based 3d semantic occupancy prediction. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 9399\u20139409 (2023). https:\/\/doi.org\/10.1109\/ICCV51070.2023.00865","DOI":"10.1109\/ICCV51070.2023.00865"},{"key":"1793_CR39","doi-asserted-by":"publisher","unstructured":"Wang, X., Zhu, Z., Xu, W., Zhang, Y., Wei, Y., Chi, X., Ye, Y., Du, D., Lu, J., Wang, X.: Openoccupancy: a large scale benchmark for surrounding semantic occupancy perception. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 17804\u201317813 (2023). https:\/\/doi.org\/10.1109\/ICCV51070.2023.01636","DOI":"10.1109\/ICCV51070.2023.01636"},{"key":"1793_CR40","doi-asserted-by":"publisher","unstructured":"Berman, M., Triki, A.R., Blaschko, M.B.: The lovasz-softmax loss: a tractable surrogate for the optimization of the intersection-over-union measure in neural networks. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4413\u20134421 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00464","DOI":"10.1109\/CVPR.2018.00464"}],"container-title":["Machine Vision and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-026-01793-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00138-026-01793-1","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-026-01793-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,9]],"date-time":"2026-03-09T18:12:54Z","timestamp":1773079974000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00138-026-01793-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,8]]},"references-count":40,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["1793"],"URL":"https:\/\/doi.org\/10.1007\/s00138-026-01793-1","relation":{},"ISSN":["0932-8092","1432-1769"],"issn-type":[{"value":"0932-8092","type":"print"},{"value":"1432-1769","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,8]]},"assertion":[{"value":"16 August 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 January 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 January 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 February 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"32"}}