{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,10]],"date-time":"2025-04-10T10:10:06Z","timestamp":1744279806835,"version":"3.40.4"},"publisher-location":"Singapore","reference-count":48,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819636785"},{"type":"electronic","value":"9789819636792"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-3679-2_13","type":"book-chapter","created":{"date-parts":[[2025,4,8]],"date-time":"2025-04-08T21:11:47Z","timestamp":1744146707000},"page":"195-210","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["SIE-DepthNet: Semantic-Guided Monocular Depth Estimation for\u00a0Dynamic Environment"],"prefix":"10.1007","author":[{"given":"Zilong","family":"Song","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sijia","family":"Dai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuai","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aimin","family":"Hao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shoulong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,30]]},"reference":[{"key":"13_CR1","doi-asserted-by":"crossref","unstructured":"Bian, J.W., Zhan, H., Reid, I.: NVSS: high-quality novel view selfie synthesis. In: 2021 International Conference on 3D Vision (3DV), pp. 1085\u20131094 (2021)","DOI":"10.1109\/3DV53792.2021.00116"},{"key":"13_CR2","doi-asserted-by":"crossref","unstructured":"Bian, J.W., Zhan, H., Wang, N., Chin, T.J., Shen, C., Reid, I.: Auto-rectify network for unsupervised indoor depth estimation. IEEE Trans. Pattern Anal. Mach. Intell. 9802\u20139813 (2022)","DOI":"10.1109\/TPAMI.2021.3136220"},{"key":"13_CR3","unstructured":"Bian, J.W., et al.: Unsupervised scale-consistent depth learning from video. In: IJCV 2021 (2021)"},{"key":"13_CR4","doi-asserted-by":"crossref","unstructured":"Bolya, D., Fu, C., Dai, X., Zhang, P., Hoffman, J.: Hydra attention: efficient attention with many heads. In: Computer Vision - ECCV 2022 Workshops - Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part VII. Springer (2022)","DOI":"10.1007\/978-3-031-25082-8_3"},{"key":"13_CR5","doi-asserted-by":"crossref","unstructured":"Casser, V., Pirk, S., Mahjourian, R., Angelova, A.: Depth prediction without the sensors: leveraging structure for unsupervised learning from monocular videos. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 8001\u20138008 (2019)","DOI":"10.1609\/aaai.v33i01.33018001"},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Chen, P.Y., Liu, A.H., Liu, Y.C., Wang, Y.C.F.: Towards scene understanding: unsupervised monocular depth estimation with semantic-aware representation. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00273"},{"key":"13_CR7","unstructured":"Chen, W., Zhao, F., Yang, D., Deng, J.: Single-image depth perception in the wild. Cornell University - arXiv, Cornell University - arXiv (2016)"},{"key":"13_CR8","unstructured":"Choi, J., Jung, D., Lee, D.H., Kim, C.: Safenet: self-supervised monocular depth estimation with semantic-aware feature extraction. Cornell University - arXiv, Cornell University - arXiv (2020)"},{"key":"13_CR9","unstructured":"Contributors, M.: MMSegmentation: Openmmlab semantic segmentation toolbox and benchmark (2020). https:\/\/github.com\/open-mmlab\/mmsegmentation"},{"key":"13_CR10","doi-asserted-by":"crossref","unstructured":"Cordts, M., et al.: The cityscapes dataset for semantic urban scene understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.350"},{"key":"13_CR11","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"13_CR12","doi-asserted-by":"crossref","unstructured":"Dickson, A., Knott, A., Zollmann, S.: Benchmarking monocular depth estimation models for VR content creation from a user perspective. In: 2021 36th International Conference on Image and Vision Computing New Zealand (IVCNZ), pp.\u00a01\u20136 (2021)","DOI":"10.1109\/IVCNZ54163.2021.9653344"},{"key":"13_CR13","unstructured":"Eigen, D., Puhrsch, C., Fergus, R.: Depth map prediction from a single image using a multi-scale deep network. Cornell University - arXiv, Cornell University - arXiv (2014)"},{"key":"13_CR14","doi-asserted-by":"crossref","unstructured":"Feng, Z., Yang, L., Jing, L., Wang, H., Tian, Y., Li, B.: Disentangling object motion and occlusion for unsupervised multi-frame monocular depth. In: Computer Vision - ECCV 2022 - 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part XXXII, pp. 228\u2013244. Springer (2022)","DOI":"10.1007\/978-3-031-19824-3_14"},{"key":"13_CR15","doi-asserted-by":"crossref","unstructured":"Garg, R., B.G., V.K., Carneiro, G., Reid, I.: Unsupervised CNN for single view depth estimation: geometry to the rescue, pp. 740\u2013756 (2016)","DOI":"10.1007\/978-3-319-46484-8_45"},{"key":"13_CR16","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., Stiller, C., Urtasun, R.: Vision meets robotics: the kitti dataset. Int. J. Robot. Res. 1231\u20131237 (2013)","DOI":"10.1177\/0278364913491297"},{"key":"13_CR17","doi-asserted-by":"crossref","unstructured":"Godard, C., Aodha, O.M., Brostow, G.J.: Unsupervised monocular depth estimation with left-right consistency. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.699"},{"key":"13_CR18","doi-asserted-by":"crossref","unstructured":"Godard, C., Aodha, O.M., Firman, M., Brostow, G.: Digging into self-supervised monocular depth estimation. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00393"},{"key":"13_CR19","doi-asserted-by":"crossref","unstructured":"Guizilini, V., Ambrus, R., Pillai, S., Raventos, A., Gaidon, A.: 3D packing for self-supervised monocular depth estimation. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00256"},{"key":"13_CR20","doi-asserted-by":"crossref","unstructured":"Guizilini, V., Hou, R., Li, J., Ambrus, R., Gaidon, A.: Semantically-guided representation learning for self-supervised monocular depth. In: International Conference on Learning Representations (2020)","DOI":"10.1109\/CVPR42600.2020.00256"},{"key":"13_CR21","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"13_CR22","doi-asserted-by":"crossref","unstructured":"Ji, P., Li, R., Bhanu, B., Xu, Y.: Monoindoor: towards good practice of self-supervised monocular depth estimation for indoor environments. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV) (2021)","DOI":"10.1109\/ICCV48922.2021.01255"},{"key":"13_CR23","unstructured":"Kingma, D., Ba, J.: Adam: a method for stochastic optimization. arXiv: Learning,arXiv: Learning (2014)"},{"key":"13_CR24","doi-asserted-by":"crossref","unstructured":"Klingner, M., Term\u00f6hlen, J.A., Mikolajczyk, J., Fingscheidt, T.: Self-supervised monocular depth estimation: solving the dynamic object problem by semantic guidance, pp. 582\u2013600 (2020)","DOI":"10.1007\/978-3-030-58565-5_35"},{"key":"13_CR25","doi-asserted-by":"crossref","unstructured":"Li, R., He, X., Zhu, Y., Li, X., Sun, J., Zhang, Y.: Enhancing self-supervised monocular depth estimation via incorporating robust constraints. In: Proceedings of the 28th ACM International Conference on Multimedia, vol.\u00a01, pp. 3108\u20133117 (2020)","DOI":"10.1145\/3394171.3413706"},{"key":"13_CR26","doi-asserted-by":"crossref","unstructured":"Newcombe, R.A., et al.: Kinectfusion: real-time dense surface mapping and tracking. In: 2011 10th IEEE International Symposium on Mixed and Augmented Reality (2011)","DOI":"10.1109\/ISMAR.2011.6162880"},{"key":"13_CR27","unstructured":"Paszke, A., et al.: Automatic differentiation in pytorch (2017)"},{"key":"13_CR28","unstructured":"Ramirez, P., Poggi, M., Tosi, F., Mattoccia, S., Stefano, L.: Geometry meets semantics for semi-supervised monocular depth estimation. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (2018)"},{"key":"13_CR29","doi-asserted-by":"crossref","unstructured":"Ranjan, A., et al.: Competitive collaboration: joint unsupervised learning of depth, camera motion, optical flow and motion segmentation. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.01252"},{"key":"13_CR30","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation, pp. 234\u2013241 (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"13_CR31","doi-asserted-by":"crossref","unstructured":"Shu, C., Yu, K., Duan, Z., Yang, K.: Feature-metric loss for self-supervised learning of depth and egomotion, pp. 572\u2013588 (2020)","DOI":"10.1007\/978-3-030-58529-7_34"},{"key":"13_CR32","doi-asserted-by":"crossref","unstructured":"Silberman, N., Hoiem, D., Kohli, P., Fergus, R.: Indoor segmentation and support inference from RGBD images, pp. 746\u2013760 (2012)","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"13_CR33","doi-asserted-by":"crossref","unstructured":"Sturm, J., Engelhard, N., Endres, F., Burgard, W., Cremers, D.: A benchmark for the evaluation of RGB-D slam systems. In: 2012 IEEE\/RSJ International Conference on Intelligent Robots and Systems (2012)","DOI":"10.1109\/IROS.2012.6385773"},{"key":"13_CR34","doi-asserted-by":"crossref","unstructured":"Su, H., Jampani, V., Sun, D., Gallo, O., Learned-Miller, E., Kautz, J.: Pixel-adaptive convolutional neural networks. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.01142"},{"key":"13_CR35","doi-asserted-by":"crossref","unstructured":"Sun, L., Bian, J.W., Zhan, H., Yin, W., Reid, I., Shen, C.: SC-DepthV3: robust self-supervised monocular depth estimation for dynamic scenes. IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI) (2023)","DOI":"10.1109\/TPAMI.2023.3322549"},{"key":"13_CR36","doi-asserted-by":"crossref","unstructured":"Wang, C., Buenaposada, J.M., Zhu, R., Lucey, S.: Learning depth from monocular videos using direct methods. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2018)","DOI":"10.1109\/CVPR.2018.00216"},{"key":"13_CR37","doi-asserted-by":"crossref","unstructured":"Wang, L., Zhang, J., Wang, O., Lin, Z., Lu, H.: SDC-depth: semantic divide-and-conquer network for monocular depth estimation. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00062"},{"key":"13_CR38","doi-asserted-by":"crossref","unstructured":"Xian, K., Zhang, J., Wang, O., Mai, L., Lin, Z., Cao, Z.: Structure-guided ranking loss for single image depth prediction. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00069"},{"key":"13_CR39","unstructured":"Xie, E., Wang, W., Yu, Z., Anandkumar, A., Alvarez, J.M., Luo, P.: Segformer: simple and efficient design for semantic segmentation with transformers. arXiv preprint arXiv:2105.15203 (2021)"},{"key":"13_CR40","doi-asserted-by":"crossref","unstructured":"Yang, N., von Stumberg, L., Wang, R., Cremers, D.: D3vo: deep depth, deep pose and deep uncertainty for monocular visual odometry. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00136"},{"key":"13_CR41","doi-asserted-by":"crossref","unstructured":"Yu, Z., Jin, L., Gao, S.: P$$^{2}$$Net: patch-match and plane-regularization for unsupervised indoor depth estimation, pp. 206\u2013222 (2020)","DOI":"10.1007\/978-3-030-58586-0_13"},{"key":"13_CR42","doi-asserted-by":"crossref","unstructured":"Zhan, H., Weerasekera, C.S., Bian, J.W., Reid, I.: Visual odometry revisited: what should be learnt? In: 2020 IEEE International Conference on Robotics and Automation (ICRA) (2020)","DOI":"10.1109\/ICRA40945.2020.9197374"},{"key":"13_CR43","doi-asserted-by":"crossref","unstructured":"Zhao, W., Liu, S., Shu, Y., Liu, Y.J.: Towards better generalization: joint depth-pose learning without PoseNet. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00917"},{"key":"13_CR44","doi-asserted-by":"crossref","unstructured":"Zhou, B., Zhao, H., Puig, X., Fidler, S., Barriuso, A., Torralba, A.: Scene parsing through ADE20K dataset. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.544"},{"key":"13_CR45","doi-asserted-by":"crossref","unstructured":"Zhou, J., Wang, Y., Qin, K., Zeng, W.: Moving indoor: unsupervised video depth learning in challenging environments. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00871"},{"key":"13_CR46","doi-asserted-by":"crossref","unstructured":"Zhou, T., Fan, D.P., Cheng, M.M., Shen, J., Shao, L.: RGB-D salient object detection: a survey. Comput. Vis. Media 37\u201369 (2021)","DOI":"10.1007\/s41095-020-0199-z"},{"key":"13_CR47","doi-asserted-by":"crossref","unstructured":"Zhou, T., Brown, M., Snavely, N., Lowe, D.G.: Unsupervised learning of depth and ego-motion from video. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.700"},{"key":"13_CR48","doi-asserted-by":"crossref","unstructured":"Zhu, S., Brazil, G., Liu, X.: The edge of depth: explicit constraints between segmentation and depth. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.01313"}],"container-title":["Lecture Notes in Computer Science","Extended Reality"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-3679-2_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,10]],"date-time":"2025-04-10T09:38:19Z","timestamp":1744277899000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-3679-2_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819636785","9789819636792"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-3679-2_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"30 March 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors state that there are no known financial conflicts of interest or personal relationships that could have influenced the work presented in this paper.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration of Conflicts of Interest"}},{"value":"ICXR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Extended Reality","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Xiamen","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 November 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 November 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icxr2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icxr.net\/2024","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}