{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:11:47Z","timestamp":1784268707693,"version":"3.55.0"},"publisher-location":"Cham","reference-count":106,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030585389","type":"print"},{"value":"9783030585396","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-58539-6_2","type":"book-chapter","created":{"date-parts":[[2020,11,6]],"date-time":"2020-11-06T19:02:46Z","timestamp":1604689366000},"page":"17-36","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":149,"title":["SoundSpaces: Audio-Visual Navigation in\u00a03D\u00a0Environments"],"prefix":"10.1007","author":[{"given":"Changan","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Unnat","family":"Jain","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Carl","family":"Schissler","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sebastia Vicenc Amengual","family":"Gari","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziad","family":"Al-Halah","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Vamsi Krishna","family":"Ithapu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Philip","family":"Robinson","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kristen","family":"Grauman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2020,11,7]]},"reference":[{"key":"2_CR1","doi-asserted-by":"publisher","first-page":"437","DOI":"10.1177\/0278364914548050","volume":"34","author":"X Alameda-Pineda","year":"2015","unstructured":"Alameda-Pineda, X., Horaud, R.: Vision-guided robot hearing. Int. J. Robot. Res. 34, 437\u2013456 (2015)","journal-title":"Int. J. Robot. Res."},{"issue":"8","key":"2_CR2","doi-asserted-by":"publisher","first-page":"1707","DOI":"10.1109\/TPAMI.2015.2496269","volume":"38","author":"X Alameda-Pineda","year":"2015","unstructured":"Alameda-Pineda, X., et al.: Salsa: a novel dataset for multimodal group behavior analysis. IEEE Trans. Pattern Anal. Mach. intell. 38(8), 1707\u20131720 (2015)","journal-title":"IEEE Trans. Pattern Anal. Mach. intell."},{"key":"2_CR3","doi-asserted-by":"crossref","unstructured":"Ammirato, P., Poirson, P., Park, E., Kosecka, J., Berg, A.: A dataset for developing and benchmarking active vision. In: ICRA (2016)","DOI":"10.1109\/ICRA.2017.7989164"},{"key":"2_CR4","unstructured":"Anderson, P., et al.: On evaluation of embodied navigation agents. arXiv preprint arXiv:1807.06757 (2018)"},{"key":"2_CR5","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Vision-and-language navigation: interpreting visually-grounded navigation instructions in real environments. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00387"},{"key":"2_CR6","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., Zisserman, A.: Objects that sound. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"2_CR7","unstructured":"Armeni, I., Sax, A., Zamir, A.R., Savarese, S.: Joint 2D\u20133D-Semantic Data for Indoor Scene Understanding. ArXiv e-prints, February 2017"},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"Ban, Y., Girin, L., Alameda-Pineda, X., Horaud, R.: Exploiting the complementarity of audio and visual data in multi-speaker tracking. In: ICCV Workshop on Computer Vision for Audio-Visual Media. 2017 IEEE International Conference on Computer Vision Workshops (ICCVW) (2017). https:\/\/hal.inria.fr\/hal-01577965","DOI":"10.1109\/ICCVW.2017.60"},{"key":"2_CR9","doi-asserted-by":"crossref","unstructured":"Ban, Y., Li, X., Alameda-Pineda, X., Girin, L., Horaud, R.: Accounting for room acoustics in audio-visual multi-speaker tracking. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2018)","DOI":"10.1109\/ICASSP.2018.8462100"},{"key":"2_CR10","unstructured":"Brodeur, S., et al.: Home: a household multimodal environment. https:\/\/arxiv.org\/abs\/1711.11017 (2017)"},{"issue":"6","key":"2_CR11","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2980179.2982431","volume":"35","author":"C Cao","year":"2016","unstructured":"Cao, C., Ren, Z., Schissler, C., Manocha, D., Zhou, K.: Interactive sound propagation with bidirectional path tracing. ACM Trans. Graph. (TOG) 35(6), 1\u201311 (2016)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"2_CR12","doi-asserted-by":"crossref","unstructured":"Chang, A., et al.: Matterport3D: learning from RGB-D data in indoor environments. In: 3DV (2017)","DOI":"10.1109\/3DV.2017.00081"},{"key":"2_CR13","doi-asserted-by":"crossref","unstructured":"Chang, A., et al.: Matterport3D: learning from RGB-D data in indoor environments. In: Proceedings of the International Conference on 3D Vision (3DV) (2017)","DOI":"10.1109\/3DV.2017.00081"},{"key":"2_CR14","unstructured":"Chaplot, D.S., Gupta, S., Gupta, A., Salakhutdinov, R.: Learning to explore using active neural mapping. In: ICLR (2020)"},{"key":"2_CR15","doi-asserted-by":"crossref","unstructured":"Chen, H., Suhr, A., Misra, D., Snavely, N., Artzi, Y.: Touchdown: natural language navigation and spatial reasoning in visual street environments. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01282"},{"key":"2_CR16","doi-asserted-by":"crossref","unstructured":"Chen, L., Srivastava, S., Duan, Z., Xu, C.: Deep cross-modal audio-visual generation. In: Proceedings of the on Thematic Workshops of ACM Multimedia 2017. ACM (2017)","DOI":"10.1145\/3126686.3126723"},{"key":"2_CR17","unstructured":"Chen, T., Gupta, S., Gupta, A.: Learning exploration policies for navigation. http:\/\/arxiv.org\/abs\/1903.01959"},{"key":"2_CR18","unstructured":"Chung, J., Kastner, K., Dinh, L., Goel, K., Courville, A.C., Bengio, Y.: A recurrent latent variable model for sequential data. In: NeurIPS (2015)"},{"key":"2_CR19","first-page":"e50272","volume":"73","author":"EC Connors","year":"2013","unstructured":"Connors, E.C., Yazzolino, L.A., S\u00e1nchez, J., Merabet, L.B.: Development of an audio-based virtual gaming environment to assist with navigation skills in the blind. J. Vis. Exp. JoVE 73, e50272 (2013)","journal-title":"J. Vis. Exp. JoVE"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Das, A., Datta, S., Gkioxari, G., Lee, S., Parikh, D., Batra, D.: Embodied question answering. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00008"},{"key":"2_CR21","doi-asserted-by":"crossref","unstructured":"Das, A., Gkioxari, G., Lee, S., Parikh, D., Batra, D.: Neural modular control for embodied question answering. In: ECCV (2018)","DOI":"10.1109\/CVPR.2018.00008"},{"key":"2_CR22","unstructured":"Das, A., et al.: Probing emergent semantics in predictive agents via question answering. In: ICML (2020)"},{"key":"2_CR23","volume-title":"Architectural Acoustics","author":"MD Egan","year":"1989","unstructured":"Egan, M.D., Quirt, J., Rousseau, M.: Architectural Acoustics. Elsevier, Amsterdam (1989)"},{"key":"2_CR24","doi-asserted-by":"publisher","first-page":"731","DOI":"10.1002\/hipo.22449","volume":"25","author":"AD Ekstrom","year":"2015","unstructured":"Ekstrom, A.D.: Why vision is important to how we navigate. Hippocampus 25, 731\u2013735 (2015)","journal-title":"Hippocampus"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Ephrat, A., et al.: Looking to listen at the cocktail party: a speaker-independent audio-visual model for speech separation. In: SIGGRAPH (2018)","DOI":"10.1145\/3197517.3201357"},{"issue":"9","key":"2_CR26","doi-asserted-by":"publisher","first-page":"1484","DOI":"10.1109\/TASLP.2018.2828321","volume":"26","author":"C Evers","year":"2018","unstructured":"Evers, C., Naylor, P.: Acoustic slam. IEEE\/ACM Trans. Audio Speech Lang. Process. 26(9), 1484\u20131498 (2018)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"2_CR27","doi-asserted-by":"publisher","first-page":"2995","DOI":"10.1093\/brain\/awn250","volume":"131","author":"M Fortin","year":"2008","unstructured":"Fortin, M., et al.: Wayfinding in the blind: larger hippocampal volume and supranormal spatial navigation. Brain 131, 2995\u20133005 (2008)","journal-title":"Brain"},{"key":"2_CR28","doi-asserted-by":"crossref","unstructured":"Gan, C., Zhang, Y., Wu, J., Gong, B., Tenenbaum, J.: Look, listen, and act: towards audio-visual embodied navigation. In: ICRA (2020)","DOI":"10.1109\/ICRA40945.2020.9197008"},{"key":"2_CR29","doi-asserted-by":"crossref","unstructured":"Gao, R., Chen, C., Al-Halah, Z., Schissler, C., Grauman, K.: VisualEchoes: spatial image representation learning through echolocation. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58545-7_38"},{"key":"2_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1007\/978-3-030-01219-9_3","volume-title":"Computer Vision","author":"R Gao","year":"2018","unstructured":"Gao, R., Feris, R., Grauman, K.: Learning to separate object sounds by watching unlabeled video. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11207, pp. 36\u201354. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01219-9_3"},{"key":"2_CR31","doi-asserted-by":"crossref","unstructured":"Gao, R., Grauman, K.: 2.5 D visual sound. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00041"},{"key":"2_CR32","doi-asserted-by":"crossref","unstructured":"Gao, R., Grauman, K.: Co-separating sounds of visual objects. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00398"},{"key":"2_CR33","doi-asserted-by":"crossref","unstructured":"Gebru, I.D., Ba, S., Evangelidis, G., Horaud, R.: Tracking the active speaker based on a joint audio-visual observation model. In: Proceedings of the IEEE International Conference on Computer Vision Workshops, pp. 15\u201321 (2015)","DOI":"10.1109\/ICCVW.2015.96"},{"key":"2_CR34","doi-asserted-by":"crossref","unstructured":"Gordon, D., Kembhavi, A., Rastegari, M., Redmon, J., Fox, D., Farhadi, A.: IQA: visual question answering in interactive environments. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00430"},{"key":"2_CR35","doi-asserted-by":"crossref","unstructured":"Gordon, D., Kadian, A., Parikh, D., Hoffman, J., Batra, D.: SplitNet: Sim2Sim and Task2Task transfer for embodied visual navigation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00111"},{"issue":"2","key":"2_CR36","doi-asserted-by":"publisher","first-page":"e27","DOI":"10.1371\/journal.pbio.0030027","volume":"3","author":"F Gougoux","year":"2005","unstructured":"Gougoux, F., Zatorre, R.J., Lassonde, M., Voss, P., Lepore, F.: A functional neuroimaging study of sound localization: visual cortex activity predicts performance in early-blind individuals. PLoS Biol. 3(2), e27 (2005)","journal-title":"PLoS Biol."},{"issue":"6","key":"2_CR37","doi-asserted-by":"publisher","first-page":"435","DOI":"10.1080\/01449290410001723364","volume":"23","author":"R Gunther","year":"2010","unstructured":"Gunther, R., Kazman, R., MacGregor, C.: Using 3D sound as a navigational aid in virtual environments. Behav. Inf. Technol. 23(6), 435\u2013446 (2010). https:\/\/doi.org\/10.1080\/01449290410001723364","journal-title":"Behav. Inf. Technol."},{"key":"2_CR38","doi-asserted-by":"crossref","unstructured":"Gupta, S., Davidson, J., Levine, S., Sukthankar, R., Malik, J.: Cognitive mapping and planning for visual navigation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2616\u20132625 (2017)","DOI":"10.1109\/CVPR.2017.769"},{"key":"2_CR39","unstructured":"Gupta, S., Fouhey, D., Levine, S., Malik, J.: Unifying map and landmark based representations for visual navigation. arXiv preprint arXiv:1712.08125 (2017)"},{"key":"2_CR40","unstructured":"Haarnoja, T., Zhou, A., Abbeel, P., Levine, S.: Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: ICML (2018)"},{"key":"2_CR41","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511811685","volume-title":"Multiple View Geometry in Computer Vision","author":"R Hartley","year":"2004","unstructured":"Hartley, R., Zisserman, A.: Multiple View Geometry in Computer Vision. Cambridge University Press, Cambridge (2004)"},{"key":"2_CR42","doi-asserted-by":"crossref","unstructured":"Henriques, J.F., Vedaldi, A.: MapNet: an allocentric spatial memory for mapping environments. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00884"},{"key":"2_CR43","unstructured":"Hershey, J.R., Movellan, J.R.: Audio vision: using audio-visual synchrony to locate sounds. In: NeurIPS (2000)"},{"key":"2_CR44","doi-asserted-by":"crossref","unstructured":"Jain, U., et al.: A cordial sync: going beyond marginal policies for multi-agent embodied tasks. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58558-7_28"},{"key":"2_CR45","doi-asserted-by":"crossref","unstructured":"Jain, U., et al.: Two body problem: collaborative visual task completion. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00685"},{"issue":"7","key":"2_CR46","doi-asserted-by":"publisher","first-page":"1601","DOI":"10.1109\/TPAMI.2018.2840991","volume":"41","author":"D Jayaraman","year":"2018","unstructured":"Jayaraman, D., Grauman, K.: End-to-end policy learning for active visual categorization. TPAMI 41(7), 1601\u20131614 (2018)","journal-title":"TPAMI"},{"key":"2_CR47","unstructured":"Johnson, M., Hofmann, K., Hutton, T., Bignell, D.: The malmo platform for artificial intelligence experimentation. In: International Joint Conference on AI (2016)"},{"key":"2_CR48","doi-asserted-by":"crossref","unstructured":"Kempka, M., Wydmuch, M., Runc, G., Toczek, J., Jakowski, W.: ViZDoom: a doom-based AI research platform for visual reinforcement learning. In: Proceedings of the IEEE Conference on Computational Intelligence and Games (2016)","DOI":"10.1109\/CIG.2016.7860433"},{"key":"2_CR49","unstructured":"Kingma, D., Ba, J.: A method for stochastic optimization. In: CVPR (2017)"},{"key":"2_CR50","unstructured":"Kojima, N., Deng, J.: To learn or not to learn: analyzing the role of learning for navigation in virtual environments. arXiv preprint arXiv:1907.11770 (2019)"},{"key":"2_CR51","unstructured":"Kolve, E., et al.: AI2-THOR: an interactive 3D environment for visual AI. arXiv (2017)"},{"key":"2_CR52","doi-asserted-by":"publisher","DOI":"10.1201\/9781315372150","volume-title":"Room Acoustics","author":"H Kuttruff","year":"2016","unstructured":"Kuttruff, H.: Room Acoustics. CRC Press, Boca Raton (2016)"},{"key":"2_CR53","unstructured":"Lerer, A., Gross, S., Fergus, R.: Learning physical intuition of block towers by example. In: ICML (2016)"},{"key":"2_CR54","doi-asserted-by":"publisher","first-page":"278","DOI":"10.1038\/26228","volume":"395","author":"N Lessard","year":"1998","unstructured":"Lessard, N., Par\u00e9, M., Lepore, F., Lassonde, M.: Early-blind human subjects localize sound sources better than sighted subjects. Nature 395, 278\u2013280 (1998)","journal-title":"Nature"},{"key":"2_CR55","doi-asserted-by":"crossref","unstructured":"Savva, M., et al.: Habitat: a platform for embodied AI research. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00943"},{"issue":"7","key":"2_CR56","doi-asserted-by":"publisher","first-page":"e0199389","DOI":"10.1371\/journal.pone.0199389","volume":"13","author":"D Massiceti","year":"2018","unstructured":"Massiceti, D., Hicks, S.L., van Rheede, J.J.: Stereosonic vision: exploring visual-to-auditory sensory substitution mappings in an immersive virtual reality navigation paradigm. PLoS ONE 13(7), e0199389 (2018)","journal-title":"PLoS ONE"},{"key":"2_CR57","first-page":"128","volume":"2","author":"L Merabet","year":"2009","unstructured":"Merabet, L., Sanchez, J.: Audio-based navigation using virtual environments: combining technology and neuroscience. AER J. Res. Pract. Vis. Impair. Blind. 2, 128\u2013137 (2009)","journal-title":"AER J. Res. Pract. Vis. Impair. Blind."},{"key":"2_CR58","doi-asserted-by":"publisher","first-page":"44","DOI":"10.1038\/nrn2758","volume":"11","author":"LB Merabet","year":"2010","unstructured":"Merabet, L.B., Pascual-Leone, A.: Neural reorganization following sensory loss: the opportunity of change. Nat. Rev. Neurosci. 11, 44\u201352 (2010)","journal-title":"Nat. Rev. Neurosci."},{"key":"2_CR59","unstructured":"Mirowski, P., et al.: Learning to navigate in complex environments. In: ICLR (2017)"},{"key":"2_CR60","unstructured":"Mishkin, D., Dosovitskiy, A., Koltun, V.: Benchmarking classic and learned navigation in complex 3D environments. arXiv preprint arXiv:1901.10915 (2019)"},{"key":"2_CR61","unstructured":"Morgado, P., Nvasconcelos, N., Langlois, T., Wang, O.: Self-supervised generation of spatial audio for 360 video. In: NeurIPS (2018)"},{"key":"2_CR62","unstructured":"Murali, A. et al..: PyRobot: an open-source robotics framework for research and benchmarking. arXiv preprint arXiv:1906.08236 (2019)"},{"key":"2_CR63","unstructured":"Nakadai, K., Lourens, T., Okuno, H.G., Kitano, H.: Active audition for humanoid. In: AAAI (2000)"},{"key":"2_CR64","unstructured":"Nakadai, K., Nakamura, K.: Sound source localization and separation. Wiley Encyclopedia of Electrical and Electronics Engineering (1999)"},{"key":"2_CR65","unstructured":"Nakadai, K., Okuno, H.G., Kitano, H.: Epipolar geometry based sound localization and extraction for humanoid audition. In: IROS Workshops. IEEE (2001)"},{"key":"2_CR66","doi-asserted-by":"crossref","unstructured":"Owens, A., Efros, A.A.: Audio-visual scene analysis with self-supervised multisensory features. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01231-1_39"},{"key":"2_CR67","doi-asserted-by":"crossref","unstructured":"Owens, A., Isola, P., McDermott, J., Torralba, A., Adelson, E.H., Freeman, W.T.: Visually indicated sounds. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.264"},{"key":"2_CR68","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"801","DOI":"10.1007\/978-3-319-46448-0_48","volume-title":"Computer Vision","author":"A Owens","year":"2016","unstructured":"Owens, A., Wu, J., McDermott, J.H., Freeman, W.T., Torralba, A.: Ambient sound provides supervision for visual learning. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 801\u2013816. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_48"},{"issue":"4","key":"2_CR69","doi-asserted-by":"publisher","first-page":"393","DOI":"10.1016\/j.ijhcs.2013.12.008","volume":"72","author":"L Picinali","year":"2014","unstructured":"Picinali, L., Afonso, A., Denis, M., Katz, B.: Exploration of architectural spaces by blind people using auditory virtual reality for the construction of spatial knowledge. Int. J. Hum.-Comput. Stud. 72(4), 393\u2013407 (2014)","journal-title":"Int. J. Hum.-Comput. Stud."},{"key":"2_CR70","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1142\/S0219878906001003","volume":"3","author":"J Qin","year":"2006","unstructured":"Qin, J., Cheng, J., Wu, X., Xu, Y.: A learning based approach to audio surveillance in household environment. Int. J. Inf. Acquis. 3, 213\u2013219 (2006)","journal-title":"Int. J. Inf. Acquis."},{"key":"2_CR71","doi-asserted-by":"publisher","first-page":"184","DOI":"10.1016\/j.robot.2017.07.011","volume":"96","author":"C Rascon","year":"2017","unstructured":"Rascon, C., Meza, I.: Localization of sound sources in robotics: a review. Robot. Auton. Syst. 96, 184\u2013210 (2017)","journal-title":"Robot. Auton. Syst."},{"key":"2_CR72","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1038\/22106","volume":"400","author":"B Ro\u00c8der","year":"1999","unstructured":"Ro\u00c8der, B., Teder-Sa\u00c8leja\u00c8rvi, W., Sterr, A., Ro\u00c8sler, F., Hillyard, S.A., Neville, H.J.: Improved auditory spatial tuning in blind humans. Nature 400, 162\u2013166 (1999)","journal-title":"Nature"},{"key":"2_CR73","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1007\/s10514-013-9323-6","volume":"34","author":"JM Romano","year":"2013","unstructured":"Romano, J.M., Brindza, J.P., Kuchenbecker, K.J.: ROS open-source audio recognizer: ROAR environmental sound detection tools for robot programming. Auton. Robot. 34, 207\u2013215 (2013). https:\/\/doi.org\/10.1007\/s10514-013-9323-6","journal-title":"Auton. Robot."},{"key":"2_CR74","unstructured":"Savinov, N., Dosovitskiy, A., Koltun, V.: Semi-parametric topological memory for navigation. In: ICLR (2018)"},{"key":"2_CR75","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"2_CR76","doi-asserted-by":"crossref","unstructured":"Senocak, A., Oh, T.H., Kim, J., Yang, M.H., So Kweon, I.: Learning to localize sound source in visual scenes. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00458"},{"key":"2_CR77","unstructured":"Straub, J., et al.: The replica dataset: a digital replica of indoor spaces. arXiv preprint arXiv:1906.05797 (2019)"},{"key":"2_CR78","unstructured":"Sukhbaatar, S., Szlam, A., Synnaeve, G., Chintala, S., Fergus, R.: Mazebase: a sandbox for learning from games. arXiv preprint arXiv:1511.07401 (2015)"},{"key":"2_CR79","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1037\/0033-2909.121.1.20","volume":"121","author":"C Thinus-Blanc","year":"1997","unstructured":"Thinus-Blanc, C., Gaunet, F.: Representation of space in blind persons: vision as a spatial sense? Psychol. Bull. 121, 20 (1997)","journal-title":"Psychol. Bull."},{"key":"2_CR80","doi-asserted-by":"crossref","unstructured":"Thomason, J., Gordon, D., Bisk, Y.: Shifting the baseline: single modality performance on visual navigation & QA. In: NAACL-HLT (2019)","DOI":"10.18653\/v1\/N19-1197"},{"key":"2_CR81","volume-title":"Probabilistic Robotics","author":"S Thrun","year":"2005","unstructured":"Thrun, S., Burgard, W., Fox, D.: Probabilistic Robotics. MIT Press, Cambridge (2005)"},{"key":"2_CR82","doi-asserted-by":"crossref","unstructured":"Tian, Y., Shi, J., Li, B., Duan, Z., Xu, C.: Audio-visual event localization in unconstrained videos. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"2_CR83","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1037\/h0061626","volume":"55","author":"EC Tolman","year":"1948","unstructured":"Tolman, E.C.: Cognitive maps in rats and men. Psychol. Rev. 55, 189 (1948)","journal-title":"Psychol. Rev."},{"key":"2_CR84","first-page":"2579","volume":"9","author":"L van der Maaten","year":"2008","unstructured":"van der Maaten, L., Hinton, G.: Visualizing high-dimensional data using t-SNE. J. Mach. Learn. Res. 9, 2579\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."},{"key":"2_CR85","doi-asserted-by":"publisher","unstructured":"Veach, E., Guibas, L.: Bidirectional estimators for light transport. In: Sakas, G., Muller, S., Shirley, P. (eds) Photorealistic Rendering Techniques, pp. 145\u2013167. Springer, Heidelberg (1995). https:\/\/doi.org\/10.1007\/978-3-642-87825-1_11","DOI":"10.1007\/978-3-642-87825-1_11"},{"key":"2_CR86","doi-asserted-by":"publisher","first-page":"9522","DOI":"10.3390\/s140609522","volume":"14","author":"R Viciana-Abad","year":"2014","unstructured":"Viciana-Abad, R., Marfil, R., Perez-Lorenzo, J., Bandera, J., Romero-Garces, A., Reche-Lopez, P.: Audio-visual perception system for a humanoid robotic head. Sensors 14, 9522\u20139545 (2014)","journal-title":"Sensors"},{"issue":"19","key":"2_CR87","doi-asserted-by":"publisher","first-page":"1734","DOI":"10.1016\/j.cub.2004.09.051","volume":"14","author":"P Voss","year":"2004","unstructured":"Voss, P., Lassonde, M., Gougoux, F., Fortin, M., Guillemot, J.P., Lepore, F.: Early-and late-onset blind individuals show supra-normal auditory abilities in far-space. Curr. Biol. 14(19), 1734\u20131738 (2004)","journal-title":"Curr. Biol."},{"key":"2_CR88","doi-asserted-by":"crossref","unstructured":"Wang, Y., Kapadia, M., Huang, P., Kavan, L., Badler, N.: Sound localization and multi-modal steering for autonomous virtual agents. In: Symposium on Interactive 3D Graphics and Games (2014)","DOI":"10.1145\/2556700.2556718"},{"key":"2_CR89","doi-asserted-by":"crossref","unstructured":"Wijmans, E., et al.: Embodied question answering in photorealistic environments with point cloud perception. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00682"},{"key":"2_CR90","unstructured":"Wijmans, E., et al.: Decentralized distributed PPO: solving PointGoal navigation. In: ICLR (2020)"},{"key":"2_CR91","unstructured":"Wood, J., Magennis, M., Arias, E.F.C., Gutierrez, T., Graupp, H., Bergamasco, M.: The design and evaluation of a computer game for the blind in the GRAB haptic audio virtual environment. In: Proceedings of Eurohpatics (2003)"},{"key":"2_CR92","doi-asserted-by":"crossref","unstructured":"Wortsman, M., Ehsani, K., Rastegari, M., Farhadi, A., Mottaghi, R.: Learning to learn how to learn: self-adaptive visual navigation using meta-learning. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00691"},{"key":"2_CR93","unstructured":"Woubie, A., Kanervisto, A., Karttunen, J., Hautamaki, V.: Do autonomous agents benefit from hearing? arXiv preprint arXiv:1905.04192 (2019)"},{"key":"2_CR94","doi-asserted-by":"publisher","first-page":"403","DOI":"10.1007\/s10846-008-9297-3","volume":"55","author":"X Wu","year":"2009","unstructured":"Wu, X., Gong, H., Chen, P., Zhong, Z., Xu, Y.: Surveillance robot utilizing video and audio information. J. Intell. Robot. Syst. 55, 403\u2013421 (2009). https:\/\/doi.org\/10.1007\/s10846-008-9297-3","journal-title":"J. Intell. Robot. Syst."},{"key":"2_CR95","doi-asserted-by":"crossref","unstructured":"Wu, Y., Wu, Y., Tamar, A., Russell, S., Gkioxari, G., Tian, Y.: Bayesian relational memory for semantic visual navigation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00286"},{"key":"2_CR96","unstructured":"Wymann, B., Espi\u00e9, E., Guionneau, C., Dimitrakakis, C., Coulom, R., Sumner, A.: TORCS, the open racing car simulator (2013). http:\/\/www.torcs.org"},{"key":"2_CR97","unstructured":"Xia, F., et al.: Interactive Gibson: a benchmark for interactive navigation in cluttered environments. arXiv preprint arXiv:1910.14442 (2019)"},{"key":"2_CR98","doi-asserted-by":"crossref","unstructured":"Xia, F., Zamir, A.R., He, Z., Sax, A., Malik, J., Savarese, S.: Gibson Env: real-world perception for embodied agents. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00945"},{"key":"2_CR99","doi-asserted-by":"crossref","unstructured":"Yoshida, T., Nakadai, K., Okuno, H.G.: Automatic speech recognition improved by two-layered audio-visual integration for robot audition. In: 2009 9th IEEE-RAS International Conference on Humanoid Robots, pp. 604\u2013609. IEEE (2009)","DOI":"10.1109\/ICHR.2009.5379586"},{"key":"2_CR100","doi-asserted-by":"crossref","unstructured":"Aytar, Y., Vondrick, C., Torralba, A.: Learning sound representations from unlabeled video. In: NeurIPS (2016)","DOI":"10.1109\/CVPR.2016.18"},{"key":"2_CR101","unstructured":"Aytar, Y., Vondrick, C., Torralba, A.: See, hear, and read: deep aligned representations. arXiv:1706.00932 (2017)"},{"key":"2_CR102","doi-asserted-by":"publisher","first-page":"3616","DOI":"10.1121\/1.5040489","volume":"143","author":"M Zaunschirm","year":"2018","unstructured":"Zaunschirm, M., Sch\u00f6rkhuber, C., H\u00f6ldrich, R.: Binaural rendering of ambisonic signals by head-related impulse response time alignment and a diffuseness constraint. J. Acoust. Soc. Am. 143, 3616 (2018)","journal-title":"J. Acoust. Soc. Am."},{"key":"2_CR103","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"587","DOI":"10.1007\/978-3-030-01246-5_35","volume-title":"Computer Vision","author":"H Zhao","year":"2018","unstructured":"Zhao, H., Gan, C., Rouditchenko, A., Vondrick, C., McDermott, J., Torralba, A.: The sound of pixels. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11205, pp. 587\u2013604. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01246-5_35"},{"key":"2_CR104","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Wang, Z., Fang, C., Bui, T., Berg, T.L.: Visual to sound: generating natural sound for videos in the wild. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00374"},{"key":"2_CR105","doi-asserted-by":"crossref","unstructured":"Zhu, Y., et al.: Visual semantic planning using deep successor representations. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.60"},{"key":"2_CR106","doi-asserted-by":"crossref","unstructured":"Zhu, Y., et al.: Target-driven visual navigation in indoor scenes using deep reinforcement learning. In: ICRA (2017)","DOI":"10.1109\/ICRA.2017.7989381"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2020"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-58539-6_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,6]],"date-time":"2024-11-06T00:04:19Z","timestamp":1730851459000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-58539-6_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030585389","9783030585396"],"references-count":106,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-58539-6_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"7 November 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Glasgow","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 August 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2020.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"OpenReview","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5025","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1360","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"27% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"7","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held virtually due to the COVID-19 pandemic.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}