{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:44:42Z","timestamp":1776883482865,"version":"3.51.2"},"publisher-location":"Singapore","reference-count":30,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819666010","type":"print"},{"value":"9789819665990","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-6599-0_18","type":"book-chapter","created":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T22:21:17Z","timestamp":1751408477000},"page":"259-273","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Multi-modal Scene Global Fusion Framework for\u00a0Enhanced Depth Estimation"],"prefix":"10.1007","author":[{"given":"Anjie","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xujun","family":"Wei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingxuan","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoyan","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongbin","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhijun","family":"Fang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siwei","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,7,2]]},"reference":[{"key":"18_CR1","doi-asserted-by":"crossref","unstructured":"Zhou, T., Brown, M., Snavely, N., Lowe, D.G.: Unsupervised learning of depth and ego-motion from video. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1851\u20131858 (2017)","DOI":"10.1109\/CVPR.2017.700"},{"key":"18_CR2","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac\u00a0Aodha, O., Brostow, G.J.: Unsupervised monocular depth estimation with left-right consistency. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 270\u2013279 (2017)","DOI":"10.1109\/CVPR.2017.699"},{"key":"18_CR3","doi-asserted-by":"crossref","unstructured":"Ji, P., Li, R., Bhanu, B., Xu, Y.: Monoindoor: towards good practice of self-supervised monocular depth estimation for indoor environments. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12787\u201312796 (2021)","DOI":"10.1109\/ICCV48922.2021.01255"},{"key":"18_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"658","DOI":"10.1007\/978-3-030-58545-7_38","volume-title":"Computer Vision \u2013 ECCV 2020","author":"R Gao","year":"2020","unstructured":"Gao, R., Chen, C., Al-Halah, Z., Schissler, C., Grauman, K.: VisualEchoes: spatial image representation learning through echolocation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12354, pp. 658\u2013676. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58545-7_38"},{"key":"18_CR5","doi-asserted-by":"crossref","unstructured":"Li, Z., et al.: Revisiting stereo depth estimation from a sequence-to-sequence perspective with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6197\u20136206 (2021)","DOI":"10.1109\/ICCV48922.2021.00614"},{"key":"18_CR6","unstructured":"Straub, J., et\u00a0al.: The replica dataset: a digital replica of indoor spaces. arXiv preprint arXiv:1906.05797 (2019)"},{"key":"18_CR7","doi-asserted-by":"crossref","unstructured":"Chang, A., Dai, A., Funkhouser, T., Halber, M., Zhang, Y.: Matterport3d: Learning from rgb-d data in indoor environments. In: 2017 International Conference on 3D Vision (3DV) (2017)","DOI":"10.1109\/3DV.2017.00081"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Christensen, J.H., Hornauer, S., Stella, X.Y.: Batvision: learning to see 3d spatial layout with two ears. In: 2020 IEEE International Conference on Robotics and Automation (ICRA), pp. 1581\u20131587. IEEE (2020)","DOI":"10.1109\/ICRA40945.2020.9196934"},{"key":"18_CR9","unstructured":"Christensen, J.H., Hornauer, S., Yu, S.: Batvision with gcc-phat features for better sound to vision predictions. arXiv preprint arXiv:2006.07995 (2020)"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Vasudevan, A.B., Dai, D., Van\u00a0Gool, L.: Semantic object prediction and spatial sound super-resolution with binaural sounds. In: European Conference on Computer Vision, pp. 638\u2013655. Springer (2020)","DOI":"10.1007\/978-3-030-58548-8_37"},{"key":"18_CR11","doi-asserted-by":"crossref","unstructured":"Gao, R., Grauman, K.: 2.5 d visual sound. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 324\u2013333 (2019)","DOI":"10.1109\/CVPR.2019.00041"},{"key":"18_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"801","DOI":"10.1007\/978-3-319-46448-0_48","volume-title":"Computer Vision \u2013 ECCV 2016","author":"A Owens","year":"2016","unstructured":"Owens, A., Wu, J., McDermott, J.H., Freeman, W.T., Torralba, A.: Ambient sound provides supervision for visual learning. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 801\u2013816. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_48"},{"key":"18_CR13","doi-asserted-by":"crossref","unstructured":"Aytar, Y., Vondrick, C., Torralba, A.: Soundnet: learning sound representations from unlabeled video. In: Advances in Neural Information Processing Systems, vol. 29 (2016)","DOI":"10.1109\/CVPR.2016.18"},{"key":"18_CR14","doi-asserted-by":"crossref","unstructured":"Wang, A., Fang, Z., Jiang, X., Gao, Y., Cao, G., Ma, S.: Depth estimation of multi-modal scene based on multi-scale modulation. In: 2023 IEEE International Conference on Image Processing (ICIP), pp. 2795\u20132799. IEEE (2023)","DOI":"10.1109\/ICIP49359.2023.10222066"},{"key":"18_CR15","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., Zisserman, A.: Look, listen and learn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 609\u2013617 (2017)","DOI":"10.1109\/ICCV.2017.73"},{"key":"18_CR16","doi-asserted-by":"crossref","unstructured":"Parida, K.K., Srivastava, S., Sharma, G.: Beyond image to depth: improving depth prediction using echoes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8268\u20138277 (2021)","DOI":"10.1109\/CVPR46437.2021.00817"},{"key":"18_CR17","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1007\/978-3-030-58601-0_21","volume-title":"Computer Vision \u2013 ECCV 2020","author":"F-T Hong","year":"2020","unstructured":"Hong, F.-T., Huang, X., Li, W.-H., Zheng, W.-S.: MINI-net: multiple instance ranking network for video highlight detection. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12358, pp. 345\u2013360. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58601-0_21"},{"key":"18_CR18","doi-asserted-by":"crossref","unstructured":"Liu, J., et al.: Target-aware dual adversarial learning and a multi-scenario multi-modality benchmark to fuse infrared and visible for object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5802\u20135811 (2022)","DOI":"10.1109\/CVPR52688.2022.00571"},{"key":"18_CR19","doi-asserted-by":"crossref","unstructured":"Badamdorj, T., Rochan, M., Wang, Y., Cheng, L.: Joint visual and audio learning for video highlight detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8127\u20138137 (2021)","DOI":"10.1109\/ICCV48922.2021.00802"},{"key":"18_CR20","doi-asserted-by":"crossref","unstructured":"Ye, Q., et al.: Temporal cue guided video highlight detection with low-rank audio-visual fusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7950\u20137959 (2021)","DOI":"10.1109\/ICCV48922.2021.00785"},{"key":"18_CR21","doi-asserted-by":"crossref","unstructured":"Lu, G., Zhong, T., Geng, J., Hu, Q., Xu, D.: Learning based multi-modality image and video compression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6083\u20136092 (2022)","DOI":"10.1109\/CVPR52688.2022.00599"},{"key":"18_CR22","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"18_CR23","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1007\/978-3-030-58577-8_8","volume-title":"Computer Vision \u2013 ECCV 2020","author":"X Li","year":"2020","unstructured":"Li, X., et al.: Oscar: object-semantics aligned pre-training for vision-language tasks. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 121\u2013137. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_8"},{"key":"18_CR24","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"18_CR25","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"18_CR26","doi-asserted-by":"crossref","unstructured":"Li, Y., et\u00a0al.: Deepfusion: Lidar-camera deep fusion for multi-modal 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17182\u201317191 (2022)","DOI":"10.1109\/CVPR52688.2022.01667"},{"key":"18_CR27","doi-asserted-by":"crossref","unstructured":"Prakash, A., Chitta, K., Geiger, A.: Multi-modal fusion transformer for end-to-end autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7077\u20137087 (2021)","DOI":"10.1109\/CVPR46437.2021.00700"},{"key":"18_CR28","doi-asserted-by":"crossref","unstructured":"Sun, C., Myers, A., Vondrick, C., Murphy, K., Schmid, C.: Videobert: a joint model for video and language representation learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7464\u20137473 (2019)","DOI":"10.1109\/ICCV.2019.00756"},{"issue":"11","key":"18_CR29","doi-asserted-by":"publisher","first-page":"12878","DOI":"10.1109\/TPAMI.2022.3200245","volume":"45","author":"K Chitta","year":"2022","unstructured":"Chitta, K., Prakash, A., Jaeger, B., Yu, Z., Renz, K., Geiger, A.: Transfuser: imitation with transformer-based sensor fusion for autonomous driving. IEEE Trans. Pattern Anal. Mach. Intell. 45(11), 12878\u201312895 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"18_CR30","doi-asserted-by":"crossref","unstructured":"Irie, G., Shibata, T., Kimura, A.: Co-attention-guided bilinear model for echo-based depth estimation. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4648\u20134652. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746476"}],"container-title":["Lecture Notes in Computer Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-6599-0_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T22:21:24Z","timestamp":1751408484000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-6599-0_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819666010","9789819665990"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-6599-0_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"2 July 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Auckland","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"New Zealand","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iconip2024.org","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}