{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:24:50Z","timestamp":1767324290688,"version":"3.48.0"},"publisher-location":"Cham","reference-count":76,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032128393","type":"print"},{"value":"9783032128409","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-12840-9_30","type":"book-chapter","created":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:21:40Z","timestamp":1767324100000},"page":"471-487","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MT-Occ: Single-View 3D Occupancy Prediction via\u00a0Multi-task Distillation"],"prefix":"10.1007","author":[{"given":"Zhi","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rahaf","family":"Aljundi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daniel Olmeda","family":"Reino","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bernt","family":"Schiele","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,2]]},"reference":[{"key":"30_CR1","doi-asserted-by":"publisher","unstructured":"Bachmann, R., Mizrahi, D., Atanov, A., Zamir, A.: MultiMAE: multi-modal multi-task masked autoencoders. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022. ECCV 2022. LNCS, vol. 13697, pp. 348\u2013367. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19836-6_20","DOI":"10.1007\/978-3-031-19836-6_20"},{"key":"30_CR2","doi-asserted-by":"crossref","unstructured":"Bhattacharjee, D., Zhang, T., S\u00fcsstrunk, S., Salzmann, M.: Mult: an end-to-end multitask learning transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12031\u201312041 (2022)","DOI":"10.1109\/CVPR52688.2022.01172"},{"key":"30_CR3","unstructured":"Boeder, S., Gigengack, F., Risse, B.: Gaussianflowocc: sparse and weakly supervised occupancy estimation using gaussian splatting and temporal flow. arXiv preprint arXiv:2502.17288 (2025)"},{"key":"30_CR4","doi-asserted-by":"crossref","unstructured":"Caesar, H., et al.: nuScenes: a multimodal dataset for autonomous driving. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"30_CR5","doi-asserted-by":"crossref","unstructured":"Cao, A.Q., Dai, A., de\u00a0Charette, R.: Pasco: urban 3d panoptic scene completion with uncertainty awareness. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14554\u201314564 (2024)","DOI":"10.1109\/CVPR52733.2024.01379"},{"key":"30_CR6","doi-asserted-by":"crossref","unstructured":"Cao, A.Q., De\u00a0Charette, R.: Monoscene: monocular 3d semantic scene completion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3991\u20134001 (2022)","DOI":"10.1109\/CVPR52688.2022.00396"},{"key":"30_CR7","unstructured":"Chambon, L., Zablocki, E., Boulch, A., Chen, M., Cord, M.: Gaussrender: learning 3d occupancy with gaussian rendering. arXiv preprint arXiv:2502.05040 (2025)"},{"key":"30_CR8","doi-asserted-by":"crossref","unstructured":"Cheng, B., et al.: Panoptic-deeplab: a simple, strong, and fast baseline for bottom-up panoptic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12475\u201312485 (2020)","DOI":"10.1109\/CVPR42600.2020.01249"},{"key":"30_CR9","doi-asserted-by":"crossref","unstructured":"Cheng, R., Agia, C., Ren, Y., Li, X., Bingbing, L.: S3cnet: a sparse semantic scene completion network for lidar point clouds. In: Conference on Robot Learning, pp. 2148\u20132161. PMLR (2021)","DOI":"10.1109\/ICRA48506.2021.9561305"},{"key":"30_CR10","doi-asserted-by":"crossref","unstructured":"Chou, Z.T., Huang, S.Y., Liu, I., Wang, Y.C.F., et\u00a0al.: Gsnerf: generalizable semantic neural radiance fields with enhanced 3d scene understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20806\u201320815 (2024)","DOI":"10.1109\/CVPR52733.2024.01966"},{"key":"30_CR11","doi-asserted-by":"crossref","unstructured":"Cordts, M., et al.: The cityscapes dataset for semantic urban scene understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3213\u20133223 (2016)","DOI":"10.1109\/CVPR.2016.350"},{"key":"30_CR12","unstructured":"Eigen, D., Puhrsch, C., Fergus, R.: Depth map prediction from a single image using a multi-scale deep network. Adv. Neural Inf. Process. Syst. 27 (2014)"},{"key":"30_CR13","doi-asserted-by":"publisher","unstructured":"Feng, Z., Yang, L., Jing, L., Wang, H., Tian, Y., Li, B.: Disentangling object motion and occlusion for unsupervised multi-frame monocular depth. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022. ECCV 2022. LNCS, vol. 13692, pp. 228\u2013244. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19824-3_14","DOI":"10.1007\/978-3-031-19824-3_14"},{"key":"30_CR14","first-page":"27503","volume":"34","author":"C Fifty","year":"2021","unstructured":"Fifty, C., Amid, E., Zhao, Z., Yu, T., Anil, R., Finn, C.: Efficiently identifying task groupings for multi-task learning. Adv. Neural Inf. Process. Syst. 34, 27503\u201327516 (2021)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"30_CR15","doi-asserted-by":"crossref","unstructured":"Fu, X., et al.: Panoptic nerf: 3d-to-2d label transfer for panoptic urban scene segmentation. In: 2022 International Conference on 3D Vision (3DV), pp. 1\u201311. IEEE (2022)","DOI":"10.1109\/3DV57658.2022.00042"},{"key":"30_CR16","unstructured":"Gan, W., Liu, F., Xu, H., Mo, N., Yokoya, N.: Gaussianocc: fully self-supervised and efficient 3d occupancy estimation with gaussian splatting. arXiv preprint arXiv:2408.11447 (2024)"},{"key":"30_CR17","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac\u00a0Aodha, O., Firman, M., Brostow, G.J.: Digging into self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3828\u20133838 (2019)","DOI":"10.1109\/ICCV.2019.00393"},{"key":"30_CR18","doi-asserted-by":"crossref","unstructured":"Hayler, A., Wimbauer, F., Muhle, D., Rupprecht, C., Cremers, D.: S4c: self-supervised semantic scene completion with neural fields. In: International Conference on 3D Vision (3DV) (2024)","DOI":"10.1109\/3DV62453.2024.00133"},{"key":"30_CR19","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"30_CR20","doi-asserted-by":"crossref","unstructured":"Huang, Y., Thammatadatrakoon, A., Zheng, W., Zhang, Y., Du, D., Lu, J.: Probabilistic gaussian superposition for efficient 3d occupancy prediction. arXiv preprint arXiv:2412.04384 (2024)","DOI":"10.1109\/CVPR52734.2025.02559"},{"key":"30_CR21","doi-asserted-by":"crossref","unstructured":"Huang, Y., Zheng, W., Zhang, B., Zhou, J., Lu, J.: Selfocc: self-supervised vision-based 3D occupancy prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19946\u201319956 (2024)","DOI":"10.1109\/CVPR52733.2024.01885"},{"key":"30_CR22","doi-asserted-by":"crossref","unstructured":"Huang, Y., Zheng, W., Zhang, Y., Zhou, J., Lu, J.: Tri-perspective view for vision-based 3D semantic occupancy prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9223\u20139232 (2023)","DOI":"10.1109\/CVPR52729.2023.00890"},{"key":"30_CR23","doi-asserted-by":"publisher","unstructured":"Huang, Y., Zheng, W., Zhang, Y., Zhou, J., Lu, J.: GaussianFormer: scene as gaussians for vision-based 3D semantic occupancy prediction. In: Leonardis, A., Ricci, E., Roth, S., Russakovsky, O., Sattler, T., Varol, G. (eds.) Computer Vision \u2013 ECCV 2024. ECCV 2024. LNCS, vol. 15085, pp. 376\u2013393. Springer, Cham (2025). https:\/\/doi.org\/10.1007\/978-3-031-73383-3_22","DOI":"10.1007\/978-3-031-73383-3_22"},{"key":"30_CR24","doi-asserted-by":"crossref","unstructured":"Kerbl, B., Kopanas, G., Leimk\u00fchler, T., Drettakis, G.: 3D gaussian splatting for real-time radiance field rendering. ACM Trans. Graph. 42(4), 139\u20131 (2023)","DOI":"10.1145\/3592433"},{"key":"30_CR25","doi-asserted-by":"crossref","unstructured":"Kerr, J., Kim, C.M., Goldberg, K., Kanazawa, A., Tancik, M.: Lerf: language embedded radiance fields. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19729\u201319739 (2023)","DOI":"10.1109\/ICCV51070.2023.01807"},{"key":"30_CR26","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. In: 3rd International Conference on Learning Representations (ICLR) (2015). https:\/\/arxiv.org\/abs\/1412.6980"},{"key":"30_CR27","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4015\u20134026 (2023)"},{"key":"30_CR28","doi-asserted-by":"crossref","unstructured":"Kundu, A., et al.: Panoptic neural fields: a semantic object-aware neural scene representation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12871\u201312881 (2022)","DOI":"10.1109\/CVPR52688.2022.01253"},{"key":"30_CR29","doi-asserted-by":"crossref","unstructured":"Li, J., Han, K., Wang, P., Liu, Y., Yuan, X.: Anisotropic convolutional networks for 3d semantic scene completion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3351\u20133359 (2020)","DOI":"10.1109\/CVPR42600.2020.00341"},{"key":"30_CR30","doi-asserted-by":"crossref","unstructured":"Li, J., et al.: Rgbd based dimensional decomposition residual network for 3D semantic scene completion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7693\u20137702 (2019)","DOI":"10.1109\/CVPR.2019.00788"},{"issue":"1","key":"30_CR31","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1109\/LRA.2019.2953639","volume":"5","author":"J Li","year":"2019","unstructured":"Li, J., Liu, Y., Yuan, X., Zhao, C., Siegwart, R., Reid, I., Cadena, C.: Depth based semantic scene completion with position importance aware loss. IEEE Robot. Autom. Lett. 5(1), 219\u2013226 (2019)","journal-title":"IEEE Robot. Autom. Lett."},{"key":"30_CR32","doi-asserted-by":"crossref","unstructured":"Li, R., Fischer, T., Segu, M., Pollefeys, M., Van\u00a0Gool, L., Tombari, F.: Know your neighbors: improving single-view reconstruction via spatial vision-language reasoning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9848\u20139858 (2024)","DOI":"10.1109\/CVPR52733.2024.00940"},{"key":"30_CR33","doi-asserted-by":"crossref","unstructured":"Li, Y., et\u00a0al.: Sscbench: a large-scale 3d semantic scene completion benchmark for autonomous driving. In: 2024 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 13333\u201313340. IEEE (2024)","DOI":"10.1109\/IROS58592.2024.10802143"},{"key":"30_CR34","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Voxformer: sparse voxel transformer for camera-based 3D semantic scene completion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9087\u20139098 (2023)","DOI":"10.1109\/CVPR52729.2023.00877"},{"key":"30_CR35","doi-asserted-by":"publisher","unstructured":"Li, Z., et al.: BEVFormer: learning bird\u2019s-eye-view representation from multi-camera images via spatiotemporal transformers. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022. ECCV 2022. LNCS, vol. 13669, pp. 1\u201318. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20077-9_1","DOI":"10.1007\/978-3-031-20077-9_1"},{"key":"30_CR36","doi-asserted-by":"crossref","unstructured":"Li, Z., Yu, Z., Wang, W., Anandkumar, A., Lu, T., Alvarez, J.M.: Fb-bev: bev representation from forward-backward view transformations. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6919\u20136928 (2023)","DOI":"10.1109\/ICCV51070.2023.00637"},{"key":"30_CR37","doi-asserted-by":"crossref","unstructured":"Liao, Y., Xie, J., Geiger, A.: KITTI-360: a novel dataset and benchmarks for urban scene understanding in 2d and 3d. Pattern Anal. Mach. Intell. (PAMI) (2022)","DOI":"10.1109\/TPAMI.2022.3179507"},{"key":"30_CR38","first-page":"18878","volume":"34","author":"B Liu","year":"2021","unstructured":"Liu, B., Liu, X., Jin, X., Stone, P., Liu, Q.: Conflict-averse gradient descent for multi-task learning. Adv. Neural Inf. Process. Syst. 34, 18878\u201318890 (2021)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"30_CR39","unstructured":"Liu, S., et al.: See and think: disentangling semantic scene completion. Adv. Neural Inf. Process. Syst. 31 (2018)"},{"key":"30_CR40","doi-asserted-by":"crossref","unstructured":"Lopes, I., Vu, T.H., de\u00a0Charette, R.: Cross-task attention mechanism for dense multi-task learning. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2329\u20132338 (2023)","DOI":"10.1109\/WACV56688.2023.00236"},{"issue":"1","key":"30_CR41","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3503250","volume":"65","author":"B Mildenhall","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P.P., Tancik, M., Barron, J.T., Ramamoorthi, R., Ng, R.: Nerf: representing scenes as neural radiance fields for view synthesis. Commun. ACM 65(1), 99\u2013106 (2021)","journal-title":"Commun. ACM"},{"key":"30_CR42","unstructured":"Mobileye: Mobileye ces 2020 presentation (2020). https:\/\/youtu.be\/HPWGFzqd7pI. Accessed 06 Mar 2025"},{"key":"30_CR43","unstructured":"Oquab, M., et al.: DINOv2: learning robust visual features without supervision. In: Proceedings of the International Conference on Learning Representations (ICLR) (2025). https:\/\/arxiv.org\/abs\/2304.07193"},{"key":"30_CR44","doi-asserted-by":"crossref","unstructured":"Pan, M., et al.: Renderocc: vision-centric 3d occupancy prediction with 2d rendering supervision. In: 2024 IEEE International Conference on Robotics and Automation (ICRA), pp. 12404\u201312411. IEEE (2024)","DOI":"10.1109\/ICRA57147.2024.10611537"},{"key":"30_CR45","doi-asserted-by":"crossref","unstructured":"Peng, S., Genova, K., Jiang, C., Tagliasacchi, A., Pollefeys, M., Funkhouser, T., et\u00a0al.: Openscene: 3d scene understanding with open vocabularies. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 815\u2013824 (2023)","DOI":"10.1109\/CVPR52729.2023.00085"},{"key":"30_CR46","doi-asserted-by":"crossref","unstructured":"Prinzler, M., Hilliges, O., Thies, J.: Diner: depth-aware image-based neural radiance fields. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12449\u201312459 (2023)","DOI":"10.1109\/CVPR52729.2023.01198"},{"key":"30_CR47","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"issue":"10","key":"30_CR48","doi-asserted-by":"publisher","first-page":"7205","DOI":"10.1109\/TPAMI.2021.3095302","volume":"44","author":"CB Rist","year":"2021","unstructured":"Rist, C.B., Emmerichs, D., Enzweiler, M., Gavrila, D.M.: Semantic scene completion using local deep implicit functions on lidar data. IEEE Trans. Pattern Anal. Mach. Intell. 44(10), 7205\u20137218 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"30_CR49","doi-asserted-by":"crossref","unstructured":"Roldao, L., de\u00a0Charette, R., Verroust-Blondet, A.: LMSCNet: lightweight multiscale 3D semantic completion. In: 2020 International Conference on 3D Vision (3DV), pp. 111\u2013119. IEEE (2020)","DOI":"10.1109\/3DV50981.2020.00021"},{"issue":"8","key":"30_CR50","doi-asserted-by":"publisher","first-page":"1978","DOI":"10.1007\/s11263-021-01504-5","volume":"130","author":"L Roldao","year":"2022","unstructured":"Roldao, L., De Charette, R., Verroust-Blondet, A.: 3D semantic scene completion: a survey. Int. J. Comput. Vis. 130(8), 1978\u20132005 (2022)","journal-title":"Int. J. Comput. Vis."},{"key":"30_CR51","doi-asserted-by":"publisher","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation. In: Navab, N., Hornegger, J., Wells, W., Frangi, A. (eds.) Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2015. MICCAI 2015. LNCS, vol. 9351, pp. 234\u2013241. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"30_CR52","unstructured":"Shen, Z., Zhang, M., Zhao, H., Yi, S., Li, H.: Efficient attention: attention with linear complexities. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 3531\u20133539 (2021)"},{"key":"30_CR53","doi-asserted-by":"crossref","unstructured":"Song, S., Yu, F., Zeng, A., Chang, A.X., Savva, M., Funkhouser, T.: Semantic scene completion from a single depth image. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1746\u20131754 (2017)","DOI":"10.1109\/CVPR.2017.28"},{"key":"30_CR54","unstructured":"Tesla: Tesla ai day 2021 (2021). https:\/\/www.youtube.com\/watch?v=j0z4FweCy4M. Accessed 6 Mar 2025"},{"key":"30_CR55","doi-asserted-by":"crossref","unstructured":"Tong, W., et\u00a0al.: Scene as occupancy. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8406\u20138415 (2023)","DOI":"10.1109\/ICCV51070.2023.00772"},{"key":"30_CR56","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"527","DOI":"10.1007\/978-3-030-58548-8_31","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Vandenhende","year":"2020","unstructured":"Vandenhende, S., Georgoulis, S., Van Gool, L.: MTI-Net: multi-scale task interaction networks for multi-task learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12349, pp. 527\u2013543. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58548-8_31"},{"key":"30_CR57","doi-asserted-by":"crossref","unstructured":"Wang, X., et al.: Openoccupancy: a large scale benchmark for surrounding semantic occupancy perception. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 17850\u201317859 (2023)","DOI":"10.1109\/ICCV51070.2023.01636"},{"issue":"4","key":"30_CR58","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z., Bovik, A.C., Sheikh, H.R., Simoncelli, E.P.: Image quality assessment: from error visibility to structural similarity. IEEE Trans. Image Process. 13(4), 600\u2013612 (2004)","journal-title":"IEEE Trans. Image Process."},{"key":"30_CR59","doi-asserted-by":"crossref","unstructured":"Wei, Y., Zhao, L., Zheng, W., Zhu, Z., Zhou, J., Lu, J.: Surroundocc: multi-camera 3d occupancy prediction for autonomous driving. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 21729\u201321740 (2023)","DOI":"10.1109\/ICCV51070.2023.01986"},{"key":"30_CR60","doi-asserted-by":"crossref","unstructured":"Wimbauer, F., Yang, N., Rupprecht, C., Cremers, D.: Behind the scenes: density fields for single view reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9076\u20139086 (2023)","DOI":"10.1109\/CVPR52729.2023.00876"},{"key":"30_CR61","doi-asserted-by":"crossref","unstructured":"Wu, S.C., Tateno, K., Navab, N., Tombari, F.: Scfusion: real-time incremental scene reconstruction with semantic completion. In: 2020 International Conference on 3D Vision (3DV), pp. 801\u2013810. IEEE (2020)","DOI":"10.1109\/3DV50981.2020.00090"},{"key":"30_CR62","doi-asserted-by":"crossref","unstructured":"Wu, Y., Lee, J.Y., Zou, C., Wang, S., Hoiem, D.: Monopatchnerf: improving neural radiance fields with patch-based monocular guidance. arXiv preprint arXiv:2404.08252 (2024)","DOI":"10.1109\/3DV66043.2025.00049"},{"key":"30_CR63","doi-asserted-by":"crossref","unstructured":"Xu, D., Ouyang, W., Wang, X., Sebe, N.: Pad-net: multi-tasks guided prediction-and-distillation network for simultaneous depth estimation and scene parsing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 675\u2013684 (2018)","DOI":"10.1109\/CVPR.2018.00077"},{"key":"30_CR64","doi-asserted-by":"publisher","unstructured":"Xu, X., Zhao, H., Vineet, V., Lim, S.N., Torralba, A.: Mtformer: multi-task learning via transformer and cross-task reasoning. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022. ECCV 2022. LNCS, vol. 13687, pp. 304\u2013321. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19812-0_18","DOI":"10.1007\/978-3-031-19812-0_18"},{"key":"30_CR65","doi-asserted-by":"crossref","unstructured":"Yan, X., et al.: Sparse single sweep lidar point cloud segmentation via learning contextual shape priors from scene completion. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035, pp. 3101\u20133109 (2021)","DOI":"10.1609\/aaai.v35i4.16419"},{"key":"30_CR66","doi-asserted-by":"crossref","unstructured":"Yang, H., et al.: Contranerf: generalizable neural radiance fields for synthetic-to-real novel view synthesis via contrastive learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16508\u201316517 (2023)","DOI":"10.1109\/CVPR52729.2023.01584"},{"key":"30_CR67","doi-asserted-by":"crossref","unstructured":"Yang, L., Kang, B., Huang, Z., Xu, X., Feng, J., Zhao, H.: Depth anything: unleashing the power of large-scale unlabeled data. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2024)","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"30_CR68","unstructured":"Yang, L., et al.: Depth anything v2. arXiv preprint arXiv:2406.09414 (2024)"},{"key":"30_CR69","doi-asserted-by":"crossref","unstructured":"Yao, J., et al.: Ndc-scene: boost monocular 3d semantic scene completion in normalized device coordinates space. In: Proceedings of IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 9421\u20139431. IEEE Computer Society (2023)","DOI":"10.1109\/ICCV51070.2023.00867"},{"key":"30_CR70","doi-asserted-by":"crossref","unstructured":"Yu, A., Ye, V., Tancik, M., Kanazawa, A.: pixelNeRF: neural radiance fields from one or few images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4578\u20134587 (2021)","DOI":"10.1109\/CVPR46437.2021.00455"},{"key":"30_CR71","first-page":"25018","volume":"35","author":"Z Yu","year":"2022","unstructured":"Yu, Z., Peng, S., Niemeyer, M., Sattler, T., Geiger, A.: Monosdf: exploring monocular geometric cues for neural implicit surface reconstruction. Adv. Neural Inf. Process. Syst. 35, 25018\u201325032 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"30_CR72","unstructured":"Zhang, C., et al.: Occnerf: self-supervised multi-camera occupancy prediction with neural radiance fields. arXiv preprint arXiv:2312.09243 (2023)"},{"key":"30_CR73","doi-asserted-by":"crossref","unstructured":"Zhang, J., Dong, R., Ma, K.: Clip-fo3d: learning free open-world 3D scene representations from 2d dense clip. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2048\u20132059 (2023)","DOI":"10.1109\/ICCVW60793.2023.00219"},{"key":"30_CR74","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Zhu, Z., Du, D.: Occformer: dual-path transformer for vision-based 3d semantic occupancy prediction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9433\u20139443 (2023)","DOI":"10.1109\/ICCV51070.2023.00865"},{"key":"30_CR75","doi-asserted-by":"crossref","unstructured":"Zhi, S., Laidlow, T., Leutenegger, S., Davison, A.J.: In-place scene labelling and understanding with implicit scene representation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15838\u201315847 (2021)","DOI":"10.1109\/ICCV48922.2021.01554"},{"key":"30_CR76","unstructured":"Zhou, X., Wang, J., Wang, Y., Wei, Y., Dong, N., Yang, M.H.: Occgs: zero-shot 3d occupancy reconstruction with semantic and geometric-aware gaussian splatting. arXiv preprint arXiv:2502.04981 (2025)"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-12840-9_30","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:21:47Z","timestamp":1767324107000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-12840-9_30"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032128393","9783032128409"],"references-count":76,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-12840-9_30","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 January 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"DAGM GCPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"DAGM German Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Freiburg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"47","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"dagm2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.dagm-gcpr.de\/year\/2025","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}