{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:45:45Z","timestamp":1777653945741,"version":"3.51.4"},"publisher-location":"Cham","reference-count":92,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030585440","type":"print"},{"value":"9783030585457","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-58545-7_38","type":"book-chapter","created":{"date-parts":[[2020,11,4]],"date-time":"2020-11-04T10:04:51Z","timestamp":1604484291000},"page":"658-676","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":55,"title":["VisualEchoes: Spatial Image Representation Learning Through Echolocation"],"prefix":"10.1007","author":[{"given":"Ruohan","family":"Gao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Changan","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ziad","family":"Al-Halah","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Carl","family":"Schissler","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kristen","family":"Grauman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,11,5]]},"reference":[{"key":"38_CR1","doi-asserted-by":"crossref","unstructured":"Agrawal, P., Carreira, J., Malik, J.: Learning to see by moving. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.13"},{"key":"38_CR2","unstructured":"Agrawal, P., Nair, A.V., Abbeel, P., Malik, J., Levine, S.: Learning to poke by poking: experiential learning of intuitive physics. In: NeurIPS (2016)"},{"issue":"8","key":"38_CR3","doi-asserted-by":"publisher","first-page":"1707","DOI":"10.1109\/TPAMI.2015.2496269","volume":"38","author":"X Alameda-Pineda","year":"2015","unstructured":"Alameda-Pineda, X., et al.: Salsa: a novel dataset for multimodal group behavior analysis. IEEE Trans. Pattern Anal. Mach. Intell. 38(8), 1707\u20131720 (2015)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"38_CR4","unstructured":"Anderson, P., et al.: On evaluation of embodied navigation agents (2018). arXiv preprint arXiv:1807.06757"},{"issue":"10","key":"38_CR5","doi-asserted-by":"publisher","first-page":"2683","DOI":"10.1109\/TASL.2012.2210877","volume":"20","author":"F Antonacci","year":"2012","unstructured":"Antonacci, F., et al.: Inference of room geometry from acoustic impulse responses. IEEE Trans. Audio Speech Lang. Process. 20(10), 2683\u20132695 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"38_CR6","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., Zisserman, A.: Look, listen and learn. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.73"},{"key":"38_CR7","doi-asserted-by":"crossref","unstructured":"Arandjelovi\u0107, R., Zisserman, A.: Objects that sound. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"38_CR8","doi-asserted-by":"crossref","unstructured":"Aytar, Y., Vondrick, C., Torralba, A.: Soundnet: learning sound representations from unlabeled video. In: NeurIPS (2016)","DOI":"10.1109\/CVPR.2016.18"},{"key":"38_CR9","doi-asserted-by":"crossref","unstructured":"Ban, Y., Li, X., Alameda-Pineda, X., Girin, L., Horaud, R.: Accounting for room acoustics in audio-visual multi-speaker tracking. In: ICASSP (2018)","DOI":"10.1109\/ICASSP.2018.8462100"},{"key":"38_CR10","doi-asserted-by":"crossref","unstructured":"Chang, A., et al.: Matterport3D: Learning from RGB-D data in indoor environments. 3DV (2017)","DOI":"10.1109\/3DV.2017.00081"},{"key":"38_CR11","doi-asserted-by":"crossref","unstructured":"Chen, C., et al.: Audio-visual embodied navigation. In: ECCV (2020)","DOI":"10.1109\/CVPR46437.2021.01526"},{"key":"38_CR12","doi-asserted-by":"crossref","unstructured":"Christensen, J., Hornauer, S., Yu, S.: Batvision - learning to see 3D spatial layout with two ears. In: ICRA (2020)","DOI":"10.1109\/ICRA40945.2020.9196934"},{"key":"38_CR13","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: CVPR (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"issue":"30","key":"38_CR14","doi-asserted-by":"publisher","first-page":"12186","DOI":"10.1073\/pnas.1221464110","volume":"110","author":"I Dokmani\u0107","year":"2013","unstructured":"Dokmani\u0107, I., Parhizkar, R., Walther, A., Lu, Y.M., Vetterli, M.: Acousticechoes reveal room shape. Proc. Natl. Acad. Sci. 110(30), 12186\u201312191 (2013)","journal-title":"Proc. Natl. Acad. Sci."},{"key":"38_CR15","doi-asserted-by":"crossref","unstructured":"Eigen, D., Fergus, R.: Predicting depth, surface normals and semantic labels with a common multi-scale convolutional architecture. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.304"},{"key":"38_CR16","unstructured":"Eigen, D., Puhrsch, C., Fergus, R.: Depth map prediction from a single image using a multi-scale deep network. In: NeurIPS (2014)"},{"issue":"9","key":"38_CR17","doi-asserted-by":"publisher","first-page":"e1006406","DOI":"10.1371\/journal.pcbi.1006406","volume":"14","author":"I Eliakim","year":"2018","unstructured":"Eliakim, I., Cohen, Z., Kosa, G., Yovel, Y.: A fully autonomous terrestrialbat-like acoustic robot. PLoS Comput. Biol. 14(9), e1006406 (2018)","journal-title":"PLoS Comput. Biol."},{"key":"38_CR18","doi-asserted-by":"crossref","unstructured":"Ephrat, A., et al.: Looking to listen at the cocktail party: a speaker-independent audio-visual model for speech separation. In: SIGGRAPH (2018)","DOI":"10.1145\/3197517.3201357"},{"key":"38_CR19","doi-asserted-by":"crossref","unstructured":"Feng, Z., Xu, C., Tao, D.: Self-supervised representation learning by rotation feature decoupling. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01061"},{"key":"38_CR20","doi-asserted-by":"crossref","unstructured":"Fernando, B., Bilen, H., Gavves, E., Gould, S.: Self-supervised video representation learning with odd-one-out networks. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.607"},{"key":"38_CR21","doi-asserted-by":"crossref","unstructured":"Fouhey, D.F., Gupta, A., Hebert, M.: Data-driven 3D primitives for single image understanding. In: ICCV (2013)","DOI":"10.1109\/ICCV.2013.421"},{"key":"38_CR22","doi-asserted-by":"crossref","unstructured":"Frank, N., Wolf, L., Olshansky, D., Boonman, A., Yovel, Y.: Comparing vision-based to sonar-based 3D reconstruction. ICCP (2020)","DOI":"10.1109\/ICCP48838.2020.9105273"},{"key":"38_CR23","doi-asserted-by":"crossref","unstructured":"Fu, H., Gong, M., Wang, C., Batmanghelich, K., Tao, D.: Deep ordinal regression network for monocular depth estimation. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00214"},{"key":"38_CR24","doi-asserted-by":"crossref","unstructured":"Gan, C., Huang, D., Zhao, H., Tenenbaum, J.B., Torralba, A.: Music gesture for visual sound separation. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01049"},{"key":"38_CR25","doi-asserted-by":"crossref","unstructured":"Gan, C., Zhang, Y., Wu, J., Gong, B., Tenenbaum, J.B.: Look, listen, and act: towards audio-visual embodied navigation. In: ICRA (2020)","DOI":"10.1109\/ICRA40945.2020.9197008"},{"key":"38_CR26","doi-asserted-by":"crossref","unstructured":"Gan, C., Zhao, H., Chen, P., Cox, D., Torralba, A.: Self-supervised moving vehicle tracking with stereo sound. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00715"},{"key":"38_CR27","doi-asserted-by":"crossref","unstructured":"Gandhi, D., Pinto, L., Gupta, A.: Learning to fly by crashing. In: IROS (2017)","DOI":"10.1109\/IROS.2017.8206247"},{"key":"38_CR28","doi-asserted-by":"crossref","unstructured":"Gao, R., Feris, R., Grauman, K.: Learning to separate object sounds by watching unlabeled video. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01219-9_3"},{"key":"38_CR29","doi-asserted-by":"crossref","unstructured":"Gao, R., Grauman, K.: 2.5D visual sound. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00041"},{"key":"38_CR30","doi-asserted-by":"crossref","unstructured":"Gao, R., Grauman, K.: Co-separating sounds of visual objects. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00398"},{"key":"38_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"248","DOI":"10.1007\/978-3-319-54193-8_16","volume-title":"Computer Vision \u2013 ACCV 2016","author":"R Gao","year":"2017","unstructured":"Gao, R., Jayaraman, D., Grauman, K.: Object-centric representation learning from unlabeled videos. In: Lai, S.-H., Lepetit, V., Nishino, K., Sato, Y. (eds.) ACCV 2016. LNCS, vol. 10115, pp. 248\u2013263. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-54193-8_16"},{"key":"38_CR32","doi-asserted-by":"crossref","unstructured":"Gao, R., Oh, T.H., Grauman, K., Torresani, L.: Listen to look: action recognition by previewing audio. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01047"},{"key":"38_CR33","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-46484-8_45","volume-title":"Computer Vision \u2013 ECCV 2016","author":"R Garg","year":"2016","unstructured":"Garg, R., B.G., V.K., Carneiro, G., Reid, I.: Unsupervised CNN for single view depth estimation: geometry to the rescue. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9912, pp. 740\u2013756. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46484-8_45"},{"key":"38_CR34","doi-asserted-by":"crossref","unstructured":"Gebru, I.D., Ba, S., Evangelidis, G., Horaud, R.: Tracking the active speaker based on a joint audio-visual observation model. In: ICCV Workshops (2015)","DOI":"10.1109\/ICCVW.2015.96"},{"key":"38_CR35","unstructured":"Gidaris, S., Singh, P., Komodakis, N.: Unsupervised representation learning by predicting image rotations. In: ICLR (2018)"},{"key":"38_CR36","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac Aodha, O., Brostow, G.J.: Unsupervised monocular depth estimation with left-right consistency. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.699"},{"key":"38_CR37","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac Aodha, O., Firman, M., Brostow, G.J.: Digging into self-supervised monocular depth estimation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00393"},{"key":"38_CR38","doi-asserted-by":"crossref","unstructured":"Goyal, P., Mahajan, D., Gupta, A., Misra, I.: Scaling and benchmarking self-supervised visual representation learning. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00649"},{"key":"38_CR39","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"38_CR40","unstructured":"Hershey, J.R., Movellan, J.R.: Audio vision: using audio-visual synchrony to locate sounds. In: NeurIPS (2000)"},{"key":"38_CR41","doi-asserted-by":"crossref","unstructured":"Hu, J., Ozay, M., Zhang, Y., Okatani, T.: Revisiting single image depth estimation: toward higher resolution maps with accurate object boundaries. In: WACV (2019)","DOI":"10.1109\/WACV.2019.00116"},{"key":"38_CR42","doi-asserted-by":"crossref","unstructured":"Irie, G., et al.: Seeing through sounds: predicting visual semantic segmentation results from multichannel audio signals. In: ICASSP (2019)","DOI":"10.1109\/ICASSP.2019.8683142"},{"key":"38_CR43","doi-asserted-by":"crossref","unstructured":"Jayaraman, D., Grauman, K.: Slow and steady feature analysis: higher order temporal coherence in video. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.418"},{"key":"38_CR44","doi-asserted-by":"crossref","unstructured":"Jayaraman, D., Gao, R., Grauman, K.: Shapecodes: self-supervised feature learning by lifting views to viewgrids. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01270-0_8"},{"key":"38_CR45","doi-asserted-by":"crossref","unstructured":"Jayaraman, D., Grauman, K.: Learning image representations equivariant to ego-motion. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.166"},{"key":"38_CR46","doi-asserted-by":"crossref","unstructured":"Jiang, H., Larsson, G., Maire Greg Shakhnarovich, M., Learned-Miller, E.: Self-supervised relative depth learning for urban scene understanding. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01252-6_2"},{"issue":"11","key":"38_CR47","doi-asserted-by":"publisher","first-page":"2144","DOI":"10.1109\/TPAMI.2014.2316835","volume":"36","author":"K Karsch","year":"2014","unstructured":"Karsch, K., Liu, C., Kang, S.B.: Depth transfer: depth extraction from video using non-parametric sampling. IEEE Trans. Pattern Anal. Mach. Intell. 36(11), 2144\u20132158 (2014)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"38_CR48","doi-asserted-by":"crossref","unstructured":"Kazakos, E., Nagrani, A., Zisserman, A., Damen, D.: Epic-fusion: Audio-visual temporal binding for egocentric action recognition. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00559"},{"key":"38_CR49","doi-asserted-by":"crossref","unstructured":"Kim, H., Remaggi, L., Jackson, P.J., Fazi, F.M., Hilton, A.: 3D room geometry reconstruction using audio-visual sensors. In: 3DV (2017)","DOI":"10.1109\/3DV.2017.00076"},{"key":"38_CR50","unstructured":"Korbar, B., Tran, D., Torresani, L.: Co-training of audio and video representations from self-supervised temporal synchronization. In: NeurIPS (2018)"},{"key":"38_CR51","first-page":"267","volume-title":"Room Acoustics","author":"H Kuttruff","year":"2017","unstructured":"Kuttruff, H.: Electroacoustical systems in rooms. Room Acoustics, pp. 267\u2013293. CRC Pres, Boca Raton (2017)"},{"key":"38_CR52","doi-asserted-by":"crossref","unstructured":"Larsson, G., Maire, M., Shakhnarovich, G.: Colorization as a proxy task for visual understanding. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.96"},{"key":"38_CR53","doi-asserted-by":"crossref","unstructured":"Liu, F., Shen, C., Lin, G.: Deep convolutional neural fields for depth estimation from a single image. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7299152"},{"issue":"35","key":"38_CR54","doi-asserted-by":"publisher","first-page":"eaaw9710","DOI":"10.1126\/scirobotics.aaw9710","volume":"4","author":"K McGuire","year":"2019","unstructured":"McGuire, K., De Wagter, C., Tuyls, K., Kappen, H., de Croon, G.: Minimal navigation solution for a swarm of tiny flying robots to explore an unknown environment. Sci. Robot. 4(35), eaaw9710 (2019)","journal-title":"Sci. Robot."},{"key":"38_CR55","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"527","DOI":"10.1007\/978-3-319-46448-0_32","volume-title":"Computer Vision \u2013 ECCV 2016","author":"I Misra","year":"2016","unstructured":"Misra, I., Zitnick, C.L., Hebert, M.: Shuffle and learn: unsupervised learning using temporal order verification. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 527\u2013544. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_32"},{"key":"38_CR56","unstructured":"Morgado, P., Vasconcelos, N., Langlois, T., Wang, O.: Self-supervised generation of spatial audio for 360$${}^\\circ $$ video. In: NeurIPS (2018)"},{"key":"38_CR57","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1007\/978-3-319-46466-4_5","volume-title":"Computer Vision \u2013 ECCV 2016","author":"M Noroozi","year":"2016","unstructured":"Noroozi, M., Favaro, P.: Unsupervised learning of visual representations by solving jigsaw puzzles. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9910, pp. 69\u201384. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46466-4_5"},{"key":"38_CR58","doi-asserted-by":"crossref","unstructured":"Owens, A., Efros, A.A.: Audio-visual scene analysis with self-supervised multisensory features. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01231-1_39"},{"key":"38_CR59","doi-asserted-by":"crossref","unstructured":"Owens, A., Isola, P., McDermott, J., Torralba, A., Adelson, E.H., Freeman, W.T.: Visually indicated sounds. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.264"},{"key":"38_CR60","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"801","DOI":"10.1007\/978-3-319-46448-0_48","volume-title":"Computer Vision \u2013 ECCV 2016","author":"A Owens","year":"2016","unstructured":"Owens, A., Wu, J., McDermott, J.H., Freeman, W.T., Torralba, A.: Ambient sound provides supervision for visual learning. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 801\u2013816. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_48"},{"issue":"5","key":"38_CR61","doi-asserted-by":"publisher","first-page":"8357","DOI":"10.1109\/JIOT.2019.2917066","volume":"6","author":"D Palossi","year":"2019","unstructured":"Palossi, D., Loquercio, A., Conti, F., Flamand, E., Scaramuzza, D., Benini, L.: A 64-mw DNN-based visual navigation engine for autonomous nano-drones. IEEE Internet Things J. 6(5), 8357\u20138371 (2019)","journal-title":"IEEE Internet Things J."},{"key":"38_CR62","doi-asserted-by":"crossref","unstructured":"Pinto, L., Gupta, A.: Supersizing self-supervision: learning to grasp from 50k tries and 700 robot hours. In: ICRA (2016)","DOI":"10.1109\/ICRA.2016.7487517"},{"key":"38_CR63","unstructured":"Purushwalkam, S., Gupta, A., Kaufman, D.M., Russell, B.: Bounce and learn: modeling scene dynamics with real-world bounces. In: ICLR (2019)"},{"key":"38_CR64","doi-asserted-by":"crossref","unstructured":"Quattoni, A., Torralba, A.: Recognizing indoor scenes. In: CVPR (2009)","DOI":"10.1109\/CVPRW.2009.5206537"},{"key":"38_CR65","doi-asserted-by":"crossref","unstructured":"Ranjan, A., et al.: Competitive collaboration: joint unsupervised learning of depth, camera motion, optical flow and motion segmentation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01252"},{"key":"38_CR66","doi-asserted-by":"crossref","unstructured":"Ren, Z., Jae Lee, Y.: Cross-domain self-supervised multi-task feature learning using synthetic imagery. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00086"},{"key":"38_CR67","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"234","DOI":"10.1007\/978-3-319-24574-4_28","volume-title":"Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2015","author":"O Ronneberger","year":"2015","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: convolutional networks for biomedical image segmentation. In: Navab, N., Hornegger, J., Wells, W.M., Frangi, A.F. (eds.) MICCAI 2015. LNCS, vol. 9351, pp. 234\u2013241. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28"},{"key":"38_CR68","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"234","DOI":"10.1007\/978-3-319-24574-4_28","volume-title":"Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2015","author":"O Ronneberger","year":"2015","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: convolutional networks for biomedical image segmentation. In: Navab, N., Hornegger, J., Wells, W.M., Frangi, A.F. (eds.) MICCAI 2015. LNCS, vol. 9351, pp. 234\u2013241. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28"},{"key":"38_CR69","unstructured":"de Sa, V.R.: Learning classification with unlabeled data. In: NeurIPS (1994)"},{"key":"38_CR70","doi-asserted-by":"crossref","unstructured":"Savva, M., et al.: Habitat: a platform for embodied AI research. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00943"},{"key":"38_CR71","doi-asserted-by":"crossref","unstructured":"Senocak, A., Oh, T.H., Kim, J., Yang, M.H., So Kweon, I.: Learning to localize sound source in visual scenes. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00458"},{"key":"38_CR72","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"746","DOI":"10.1007\/978-3-642-33715-4_54","volume-title":"Computer Vision \u2013 ECCV 2012","author":"N Silberman","year":"2012","unstructured":"Silberman, N., Hoiem, D., Kohli, P., Fergus, R.: Indoor segmentation and support inference from RGBD images. In: Fitzgibbon, A., Lazebnik, S., Perona, P., Sato, Y., Schmid, C. (eds.) ECCV 2012. LNCS, vol. 7576, pp. 746\u2013760. Springer, Heidelberg (2012). https:\/\/doi.org\/10.1007\/978-3-642-33715-4_54"},{"key":"38_CR73","unstructured":"Straub, J., et al.: The replica dataset: a digital replica of indoor spaces (2019). arXiv preprint arXiv:1906.05797"},{"issue":"3","key":"38_CR74","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1207\/s15326969eco0703_2","volume":"7","author":"TA Stroffregen","year":"1995","unstructured":"Stroffregen, T.A., Pittenger, J.B.: Human echolocation as a basic form of perception and action. Ecol. Psychol. 7(3), 181\u2013216 (1995)","journal-title":"Ecol. Psychol."},{"key":"38_CR75","doi-asserted-by":"crossref","unstructured":"Tian, Y., Shi, J., Li, B., Duan, Z., Xu, C.: Audio-visual event localization in unconstrained videos. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"38_CR76","doi-asserted-by":"crossref","unstructured":"Ummenhofer, B., et al.: Demon: Depth and motion network for learning monocular stereo. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.596"},{"issue":"10","key":"38_CR77","doi-asserted-by":"publisher","first-page":"e1004484","DOI":"10.1371\/journal.pcbi.1004484","volume":"11","author":"D Vanderelst","year":"2015","unstructured":"Vanderelst, D., Holderied, M.W., Peremans, H.: Sensorimotor model of obstacleavoidance in echolocating bats. PLoS Comput. Biol. 11(10), e1004484 (2015)","journal-title":"PLoS Comput. Biol."},{"key":"38_CR78","unstructured":"Vasiljevic, Iet al.: DIODE: A Dense Indoor and Outdoor DEpth Dataset (2019). arXiv preprint arXiv:1908.00463"},{"key":"38_CR79","doi-asserted-by":"crossref","unstructured":"Veach, E., Guibas, L.: Bidirectional estimators for light transport. In: Photorealistic Rendering Techniques (1995)","DOI":"10.1007\/978-3-642-87825-1_11"},{"key":"38_CR80","unstructured":"Vijayanarasimhan, S., Ricco, S., Schmid, C., Sukthankar, R., Fragkiadaki, K.: SFM-net: Learning of structure and motion from video (2017). arXiv preprint arXiv:1704.07804"},{"key":"38_CR81","unstructured":"Wang, P., Shen, X., Lin, Z., Cohen, S., Price, B., Yuille, A.L.: Towards unified depth and semantic prediction from a single image. In: CVPR (2015)"},{"key":"38_CR82","doi-asserted-by":"crossref","unstructured":"Wang, X., Gupta, A.: Unsupervised learning of visual representations using videos. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.320"},{"key":"38_CR83","doi-asserted-by":"crossref","unstructured":"Xia, F., Zamir, A.R., He, Z., Sax, A., Malik, J., Savarese, S.: Gibson env: Real-world perception for embodied agents. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00945"},{"key":"38_CR84","doi-asserted-by":"crossref","unstructured":"Xu, D., Ricci, E., Ouyang, W., Wang, X., Sebe, N.: Multi-scale continuous CRFS as sequential deep networks for monocular depth estimation. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.25"},{"key":"38_CR85","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, P., Xu, W., Zhao, L., Nevatia, R.: Unsupervised learning of geometry with edge-aware depth-normal consistency. In: AAAI (2018)","DOI":"10.1609\/aaai.v32i1.12257"},{"key":"38_CR86","doi-asserted-by":"crossref","unstructured":"Ye, M., Zhang, Y., Yang, R., Manocha, D.: 3d reconstruction in the presence of glasses by acoustic and stereo fusion. In: ICCV (2015)","DOI":"10.1109\/CVPR.2015.7299122"},{"key":"38_CR87","doi-asserted-by":"crossref","unstructured":"Zamir, A.R., Sax, A., Shen, W., Guibas, L.J., Malik, J., Savarese, S.: Taskonomy: Disentangling task transfer learning. In: CVPR (2018)","DOI":"10.24963\/ijcai.2019\/871"},{"key":"38_CR88","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"649","DOI":"10.1007\/978-3-319-46487-9_40","volume-title":"Computer Vision \u2013 ECCV 2016","author":"R Zhang","year":"2016","unstructured":"Zhang, R., Isola, P., Efros, A.A.: Colorful image colorization. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9907, pp. 649\u2013666. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46487-9_40"},{"key":"38_CR89","doi-asserted-by":"crossref","unstructured":"Zhao, H., Gan, C., Rouditchenko, A., Vondrick, C., McDermott, J., Torralba, A.: The sound of pixels. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01246-5_35"},{"key":"38_CR90","doi-asserted-by":"crossref","unstructured":"Zhao, H., Shi, J., Qi, X., Wang, X., Jia, J.: Pyramid scene parsing network. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.660"},{"key":"38_CR91","doi-asserted-by":"crossref","unstructured":"Zhou, T., Brown, M., Snavely, N., Lowe, D.G.: Unsupervised learning of depth and ego-motion from video. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.700"},{"key":"38_CR92","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Wang, Z., Fang, C., Bui, T., Berg, T.L.: Visual to sound: Generating natural sound for videos in the wild. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00374"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2020"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-58545-7_38","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,4]],"date-time":"2024-11-04T01:16:27Z","timestamp":1730682987000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-58545-7_38"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030585440","9783030585457"],"references-count":92,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-58545-7_38","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"5 November 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Glasgow","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 August 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2020.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"OpenReview","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5025","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1360","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"27% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"7","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held virtually due to the COVID-19 pandemic. From the ECCV Workshops 249 full papers, 18 short papers, and 21 further contributions were published out of a total of 467 submissions.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}