{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T16:59:48Z","timestamp":1777568388030,"version":"3.51.4"},"publisher-location":"Cham","reference-count":51,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031734038","type":"print"},{"value":"9783031734045","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T00:00:00Z","timestamp":1730246400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T00:00:00Z","timestamp":1730246400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73404-5_3","type":"book-chapter","created":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T16:03:13Z","timestamp":1730217793000},"page":"38-54","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Monocular Occupancy Prediction for\u00a0Scalable Indoor Scenes"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-9249-2726","authenticated-orcid":false,"given":"Hongxiao","family":"Yu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6360-1431","authenticated-orcid":false,"given":"Yuqi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9555-1897","authenticated-orcid":false,"given":"Yuntao","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2648-3875","authenticated-orcid":false,"given":"Zhaoxiang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,30]]},"reference":[{"key":"3_CR1","doi-asserted-by":"crossref","unstructured":"Arshad, M.S., Beksi, W.J.: LIST: learning implicitly from spatial transformers for single-view 3D reconstruction. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00855"},{"key":"3_CR2","doi-asserted-by":"publisher","first-page":"355","DOI":"10.1162\/pres.1997.6.4.355","volume":"6","author":"RT Azuma","year":"1997","unstructured":"Azuma, R.T.: A survey of augmented reality. Presence Teleop. Virt. Environ. 6, 355\u2013385 (1997)","journal-title":"Presence Teleop. Virt. Environ."},{"key":"3_CR3","unstructured":"Bhat, S.F., Birkl, R., Wofk, D., Wonka, P., M\u00fcller, M.: Zoedepth: zero-shot transfer by combining relative and metric depth. arXiv preprint arXiv:2302.12288 (2023)"},{"key":"3_CR4","unstructured":"Birkl, R., Wofk, D., M\u00fcller, M.: MiDaS V3. 1\u2013a model zoo for robust monocular relative depth estimation. arXiv preprint arXiv:2307.14460 (2023)"},{"key":"3_CR5","doi-asserted-by":"crossref","unstructured":"Cao, A.Q., de\u00a0Charette, R.: MonoScene: monocular 3D semantic scene completion. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00396"},{"key":"3_CR6","doi-asserted-by":"crossref","unstructured":"Chen, X., Lin, K.Y., Qian, C., Zeng, G., Li, H.: 3D sketch-aware semantic scene completion via semi-supervised structure prior. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00425"},{"key":"3_CR7","unstructured":"Dahnert, M., Hou, J., Nie\u00dfner, M., Dai, A.: Panoptic 3D scene reconstruction from a single RGB image. In: NeurIPS (2021)"},{"key":"3_CR8","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A.X., Savva, M., Halber, M., Funkhouser, T., Nie\u00dfner, M.: ScanNet: richly-annotated 3D reconstructions of indoor scenes. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.261"},{"key":"3_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"51","DOI":"10.1007\/978-3-030-58542-6_4","volume-title":"Computer Vision \u2013 ECCV 2020","author":"M Denninger","year":"2020","unstructured":"Denninger, M., Triebel, R.: 3D scene reconstruction from a single viewport. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12367, pp. 51\u201367. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58542-6_4"},{"key":"3_CR10","doi-asserted-by":"publisher","first-page":"237","DOI":"10.1109\/34.982903","volume":"24","author":"GN DeSouza","year":"2002","unstructured":"DeSouza, G.N., Kak, A.C.: Vision for mobile robot navigation: a survey. TPAMI 24, 237\u2013267 (2002)","journal-title":"TPAMI"},{"key":"3_CR11","doi-asserted-by":"crossref","unstructured":"Fan, H., Su, H., Guibas, L.J.: A point set generation network for 3D object reconstruction from a single image. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.264"},{"key":"3_CR12","unstructured":"Fieraru, M., Zanfir, M., Oneata, E., Popa, A.I., Olaru, V., Sminchisescu, C.: Reconstructing three-dimensional models of interacting humans. arXiv preprint arXiv:2308.01854 (2023)"},{"key":"3_CR13","doi-asserted-by":"crossref","unstructured":"Goel, S., Pavlakos, G., Rajasegaran, J., Kanazawa, A., Malik, J.: Humans in 4D: reconstructing and tracking humans with transformers. arXiv preprint arXiv:2305.20091 (2023)","DOI":"10.1109\/ICCV51070.2023.01358"},{"key":"3_CR14","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-023-3242-2","volume":"18","author":"H Guan","year":"2024","unstructured":"Guan, H., Song, C., Zhang, Z.: GRAMO: geometric resampling augmentation for monocular 3D object detection. Front. Comput. Sci. 18, 185706 (2024)","journal-title":"Front. Comput. Sci."},{"key":"3_CR15","unstructured":"Huang, J., Huang, G., Zhu, Z., Ye, Y., Du, D.: BEVDet: high-performance multi-camera 3D object detection in bird-eye-view. arXiv preprint arXiv:2112.11790 (2021)"},{"key":"3_CR16","doi-asserted-by":"crossref","unstructured":"Huang, Y., Zheng, W., Zhang, B., Zhou, J., Lu, J.: SelfOcc: self-supervised vision-based 3D occupancy prediction. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.01885"},{"key":"3_CR17","doi-asserted-by":"crossref","unstructured":"Huang, Y., Zheng, W., Zhang, Y., Zhou, J., Lu, J.: Tri-perspective view for vision-based 3D semantic occupancy prediction. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00890"},{"key":"3_CR18","doi-asserted-by":"crossref","unstructured":"Li, J., Han, K., Wang, P., Liu, Y., Yuan, X.: Anisotropic convolutional networks for 3D semantic scene completion. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00341"},{"key":"3_CR19","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: VoxFormer: sparse voxel transformer for camera-based 3D semantic scene completion. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00877"},{"key":"3_CR20","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: BEVDepth: acquisition of reliable depth for multi-view 3D object detection. In: AAAI (2023)","DOI":"10.1609\/aaai.v37i2.25233"},{"key":"3_CR21","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-031-20077-9_1","volume-title":"ECCV 2022","author":"Z Li","year":"2022","unstructured":"Li, Z., et al.: BEVFormer: learning bird\u2019s-eye-view representation from multi-camera images via spatiotemporal transformers. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13669, pp. 1\u201318. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20077-9_1"},{"key":"3_CR22","unstructured":"Li, Z., et al.: FB-OCC: 3D occupancy prediction based on forward-backward view transformation. arXiv preprint arXiv:2307.01492 (2023)"},{"key":"3_CR23","unstructured":"Liu, S., et al.: See and think: disentangling semantic scene completion. In: NeurIPS (2018)"},{"key":"3_CR24","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"414","DOI":"10.1007\/978-3-030-58571-6_25","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Z Murez","year":"2020","unstructured":"Murez, Z., van As, T., Bartolozzi, J., Sinha, A., Badrinarayanan, V., Rabinovich, A.: Atlas: end-to-end 3D scene reconstruction from posed images. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12352, pp. 414\u2013431. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58571-6_25"},{"key":"3_CR25","doi-asserted-by":"crossref","unstructured":"Pan, M., et al.: RenderOcc: vision-centric 3D occupancy prediction with 2D rendering supervision. arXiv preprint arXiv:2309.09502 (2023)","DOI":"10.1109\/ICRA57147.2024.10611537"},{"key":"3_CR26","doi-asserted-by":"crossref","unstructured":"Park, J.J., Florence, P., Straub, J., Newcombe, R., Lovegrove, S.: DeepSDF: learning continuous signed distance functions for shape representation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00025"},{"key":"3_CR27","doi-asserted-by":"publisher","first-page":"1623","DOI":"10.1109\/TPAMI.2020.3019967","volume":"44","author":"R Ranftl","year":"2020","unstructured":"Ranftl, R., Lasinger, K., Hafner, D., Schindler, K., Koltun, V.: Towards robust monocular depth estimation: mixing datasets for zero-shot cross-dataset transfer. TPAMI 44, 1623\u20131637 (2020)","journal-title":"TPAMI"},{"key":"3_CR28","doi-asserted-by":"crossref","unstructured":"Roldao, L., de\u00a0Charette, R., Verroust-Blondet, A.: LMSCNet: lightweight multiscale 3D semantic completion. In: 3DV (2020)","DOI":"10.1109\/3DV50981.2020.00021"},{"key":"3_CR29","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"746","DOI":"10.1007\/978-3-642-33715-4_54","volume-title":"Computer Vision \u2013 ECCV 2012","author":"N Silberman","year":"2012","unstructured":"Silberman, N., Hoiem, D., Kohli, P., Fergus, R.: Indoor segmentation and support inference from RGBD images. In: Fitzgibbon, A., Lazebnik, S., Perona, P., Sato, Y., Schmid, C. (eds.) ECCV 2012. LNCS, vol. 7576, pp. 746\u2013760. Springer, Heidelberg (2012). https:\/\/doi.org\/10.1007\/978-3-642-33715-4_54"},{"key":"3_CR30","doi-asserted-by":"crossref","unstructured":"Song, S., Yu, F., Zeng, A., Chang, A.X., Savva, M., Funkhouser, T.: Semantic scene completion from a single depth image. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.28"},{"key":"3_CR31","unstructured":"Tan, M., Le, Q.: EfficientNet: rethinking model scaling for convolutional neural networks. In: ICML (2019)"},{"key":"3_CR32","doi-asserted-by":"crossref","unstructured":"Tang, Y., Dorn, S., Savani, C.: Center3D: center-based monocular 3D object detection with joint depth understanding. In: DAGM GCPR (2020)","DOI":"10.1007\/978-3-030-71278-5_21"},{"key":"3_CR33","unstructured":"Tian, X., et al.: OCC3D: a large-scale 3D occupancy prediction benchmark for autonomous driving. In: NeurIPS (2023)"},{"key":"3_CR34","doi-asserted-by":"crossref","unstructured":"Tong, W., et\u00a0al.: Scene as occupancy. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00772"},{"key":"3_CR35","doi-asserted-by":"crossref","unstructured":"Wang, S., Liu, Y., Wang, T., Li, Y., Zhang, X.: Exploring object-centric temporal modeling for efficient multi-view 3D object detection. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00335"},{"key":"3_CR36","doi-asserted-by":"crossref","unstructured":"Wang, X., et al.: OpenOccupancy: a large scale benchmark for surrounding semantic occupancy perception. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01636"},{"key":"3_CR37","unstructured":"Wang, Y., Guizilini, V.C., Zhang, T., Wang, Y., Zhao, H., Solomon, J.: DETR3D: 3D object detection from multi-view images via 3D-to-2D queries. In: CoRL (2022)"},{"key":"3_CR38","doi-asserted-by":"crossref","unstructured":"Wang, Y., Chen, Y., Liao, X., Fan, L., Zhang, Z.: PanoOcc: unified occupancy representation for camera-based 3d panoptic segmentation. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.01624"},{"key":"3_CR39","doi-asserted-by":"crossref","unstructured":"Wang, Y., Chen, Y., Zhang, Z.: FrustumFormer: adaptive instance-aware resampling for multi-view 3D detection. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00493"},{"key":"3_CR40","doi-asserted-by":"crossref","unstructured":"Wei, Y., Zhao, L., Zheng, W., Zhu, Z., Zhou, J., Lu, J.: SurroundOcc: multi-camera 3D occupancy prediction for autonomous driving. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01986"},{"key":"3_CR41","doi-asserted-by":"crossref","unstructured":"Wu, Q., Wang, K., Li, K., Zheng, J., Cai, J.: ObjectSDF++: improved object-compositional neural implicit surfaces. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01989"},{"key":"3_CR42","doi-asserted-by":"crossref","unstructured":"Wu, S.C., Tateno, K., Navab, N., Tombari, F.: SCFusion: real-time incremental scene reconstruction with semantic completion. In: 3DV (2020)","DOI":"10.1109\/3DV50981.2020.00090"},{"key":"3_CR43","doi-asserted-by":"crossref","unstructured":"Yang, C., et\u00a0al.: BEVFormer V2: adapting modern image backbones to bird\u2019s-eye-view recognition via perspective supervision. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01710"},{"key":"3_CR44","doi-asserted-by":"crossref","unstructured":"Yang, L., Kang, B., Huang, Z., Xu, X., Feng, J., Zhao, H.: Depth anything: unleashing the power of large-scale unlabeled data. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"3_CR45","doi-asserted-by":"crossref","unstructured":"Yao, J., et al.: NDC-scene: boost monocular 3D semantic scene completion in normalized device coordinates space. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00867"},{"key":"3_CR46","unstructured":"Yu, Z., et al.: FlashOcc: fast and memory-efficient occupancy prediction via channel-to-height plugin. arXiv preprint arXiv:2311.12058 (2023)"},{"key":"3_CR47","doi-asserted-by":"publisher","first-page":"58443","DOI":"10.1109\/ACCESS.2020.2983149","volume":"8","author":"E Yurtsever","year":"2020","unstructured":"Yurtsever, E., Lambert, J., Carballo, A., Takeda, K.: A survey of autonomous driving: common practices and emerging technologies. IEEE Access 8, 58443\u201358469 (2020)","journal-title":"IEEE Access"},{"key":"3_CR48","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"749","DOI":"10.1007\/978-3-030-01258-8_45","volume-title":"Computer Vision \u2013 ECCV 2018","author":"J Zhang","year":"2018","unstructured":"Zhang, J., Zhao, H., Yao, A., Chen, Y., Zhang, L., Liao, H.: Efficient semantic scene completion network with spatial group convolution. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11216, pp. 749\u2013765. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01258-8_45"},{"key":"3_CR49","doi-asserted-by":"crossref","unstructured":"Zhang, X., Bi, S., Sunkavalli, K., Su, H., Xu, Z.: NeRFusion: fusing radiance fields for large-scale scene reconstruction. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00537"},{"key":"3_CR50","doi-asserted-by":"crossref","unstructured":"Zheng, Z., Yu, T., Wei, Y., Dai, Q., Liu, Y.: DeepHuman: 3D human reconstruction from a single image. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00783"},{"key":"3_CR51","doi-asserted-by":"crossref","unstructured":"Zhong, M., Zeng, G.: Semantic point completion network for 3D semantic scene completion. In: ECAI (2020)","DOI":"10.3233\/FAIA200424"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73404-5_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,24]],"date-time":"2025-04-24T19:45:34Z","timestamp":1745523934000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73404-5_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,30]]},"ISBN":["9783031734038","9783031734045"],"references-count":51,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73404-5_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,30]]},"assertion":[{"value":"30 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}