{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:56:15Z","timestamp":1777654575626,"version":"3.51.4"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2018,7,11]],"date-time":"2018-07-11T00:00:00Z","timestamp":1531267200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2018,7,11]],"date-time":"2018-07-11T00:00:00Z","timestamp":1531267200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"National Science Foundation","award":["1447476"],"award-info":[{"award-number":["1447476"]}]},{"name":"National Science Foundation","award":["1212849"],"award-info":[{"award-number":["1212849"]}]},{"name":"National Science Foundation","award":["1524817"],"award-info":[{"award-number":["1524817"]}]},{"DOI":"10.13039\/100006112","name":"Microsoft Research","doi-asserted-by":"publisher","award":["PhD Fellowship"],"award-info":[{"award-number":["PhD Fellowship"]}],"id":[{"id":"10.13039\/100006112","id-type":"DOI","asserted-by":"publisher"}]},{"name":"McDonnell Scholar Award"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2018,10]]},"DOI":"10.1007\/s11263-018-1083-5","type":"journal-article","created":{"date-parts":[[2018,7,11]],"date-time":"2018-07-11T04:55:04Z","timestamp":1531284904000},"page":"1120-1137","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":41,"title":["Learning Sight from Sound: Ambient Sound Provides Supervision for Visual Learning"],"prefix":"10.1007","volume":"126","author":[{"given":"Andrew","family":"Owens","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiajun","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Josh H.","family":"McDermott","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"William T.","family":"Freeman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Antonio","family":"Torralba","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,7,11]]},"reference":[{"key":"1083_CR1","doi-asserted-by":"crossref","unstructured":"Agrawal, P., Carreira, J., & Malik, J. (2015). Learning to see by moving. In IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2015.13"},{"key":"1083_CR2","unstructured":"Andrew, G., Arora, R., Bilmes, J. A., & Livescu, K. (2013). Deep canonical correlation analysis. In International conference on machine learning."},{"key":"1083_CR3","doi-asserted-by":"crossref","unstructured":"Arandjelovi\u0107, R., & Zisserman, A. (2017). Look, listen and learn. ICCV.","DOI":"10.1109\/ICCV.2017.73"},{"key":"1083_CR4","unstructured":"Aytar, Y., Vondrick, C., & Torralba, A. (2016). Soundnet: Learning sound representations from unlabeled video. In Advances in neural information processing systems."},{"key":"1083_CR5","doi-asserted-by":"crossref","unstructured":"Bau, D., Zhou, B., Khosla, A., Oliva, A., & Torralba, A. (2017). Network dissection: Quantifying interpretability of deep visual representations. CVPR.","DOI":"10.1109\/CVPR.2017.354"},{"key":"1083_CR6","unstructured":"de\u00a0Sa, V. R. (1994a). Learning classification with unlabeled data. Advances in neural information processing systems (pp 112)"},{"key":"1083_CR7","unstructured":"de\u00a0Sa, V. R. (1994b). Minimizing disagreement for self-supervised classification. In Proceedings of the 1993 Connectionist Models Summer School (pp. 300.). Psychology Press."},{"key":"1083_CR8","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L. J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1083_CR9","doi-asserted-by":"crossref","unstructured":"Doersch, C., Gupta, A., & Efros, A. A. (2015). Unsupervised visual representation learning by context prediction. In IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2015.167"},{"key":"1083_CR10","doi-asserted-by":"crossref","unstructured":"Doersch, C., & Zisserman, A. (2017). Multi-task self-supervised visual learning. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2051\u20132060).","DOI":"10.1109\/ICCV.2017.226"},{"key":"1083_CR11","unstructured":"Dosovitskiy, A., Springenberg, J. T., Riedmiller, M., & Brox, T. (2014). Discriminative unsupervised feature learning with convolutional neural networks. In Advances in neural information processing systems."},{"key":"1083_CR12","doi-asserted-by":"crossref","unstructured":"Ellis, D. P., Zeng, X., McDermott, J. H. (2011). Classifying soundtracks with audio texture features. In IEEE international conference on acoustics, speech, and signal processing.","DOI":"10.1109\/ICASSP.2011.5947699"},{"issue":"1","key":"1083_CR13","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1109\/TSA.2005.854103","volume":"14","author":"AJ Eronen","year":"2006","unstructured":"Eronen, A. J., Peltonen, V. T., Tuomi, J. T., Klapuri, A. P., Fagerlund, S., Sorsa, T., et al. (2006). Audio-based context recognition. IEEE\/ACM Transactions on Audio Speech and Language Processing, 14(1), 321\u2013329.","journal-title":"IEEE\/ACM Transactions on Audio Speech and Language Processing"},{"issue":"2","key":"1083_CR14","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C. K., Winn, J., & Zisserman, A. (2010). The pascal visual object classes (voc) challenge. International Journal of Computer Vision, 88(2), 303\u2013338.","journal-title":"International Journal of Computer Vision"},{"key":"1083_CR15","unstructured":"Fisher\u00a0III J. W., Darrell, T., Freeman, W. T., Viola, P. A. (2000). Learning joint statistical models for audio\u2013visual fusion and segregation. In Advances in neural information processing systems."},{"issue":"1","key":"1083_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1207\/s15326969eco0501_1","volume":"5","author":"WW Gaver","year":"1993","unstructured":"Gaver, W. W. (1993). What in the world do we hear?: An ecological approach to auditory event perception. Ecological psychology, 5(1), 1\u201329.","journal-title":"Ecological psychology"},{"key":"1083_CR17","doi-asserted-by":"crossref","unstructured":"Gemmeke, J. F., Ellis, D. P., Freedman, D., Jansen, A., Lawrence, W., Moore, R. C., Plakal, M., & Ritter, M. (2017). Audio set: An ontology and human-labeled dartaset for audio events. In IEEE international conference on acoustics, speech, and signal processing.","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"1083_CR18","doi-asserted-by":"crossref","unstructured":"Girshick, R. (2015). Fast r-cnn. In IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2015.169"},{"key":"1083_CR19","unstructured":"Goroshin, R., Bruna, J., Tompson, J., Eigen, D., & LeCun, Y. (2015). Unsupervised feature learning from temporal data. arXiv preprint \n                    a\n                    \n                  rXiv:1504.02518."},{"key":"1083_CR20","doi-asserted-by":"crossref","unstructured":"Gupta, S., Hoffman, J., & Malik, J. (2016). Cross modal distillation for supervision transfer. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2016.309"},{"key":"1083_CR21","unstructured":"Hershey, J. R., & Movellan, J. R. (1999). Audio vision: Using audio\u2013visual synchrony to locate sounds. In Advances in neural information processing systems."},{"key":"1083_CR22","doi-asserted-by":"crossref","unstructured":"Indyk, P., & Motwani, R. (1998). Approximate nearest neighbors: Towards removing the curse of dimensionality. In ACM symposium on theory of computing.","DOI":"10.1145\/276698.276876"},{"key":"1083_CR23","unstructured":"Ioffe, S., & Szegedy, C. (2015). Batch normalization: Accelerating deep network training by reducing internal covariate shift. In International conference on machine learning."},{"key":"1083_CR24","unstructured":"Isola, P. (2015). The discovery of perceptual structure from visual co-occurrences in space and time. PhD thesis"},{"key":"1083_CR25","unstructured":"Isola, P., Zoran, D., Krishnan, D., & Adelson, E.H. (2016). Learning visual groups from co-occurrences in space and time. In International conference on learning representations, Workshop."},{"key":"1083_CR26","doi-asserted-by":"crossref","unstructured":"Jayaraman, D., & Grauman, K. (2015). Learning image representations tied to ego-motion. In IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2015.166"},{"key":"1083_CR27","doi-asserted-by":"crossref","unstructured":"Jia, Y., Shelhamer, E., Donahue, J., Karayev, S., Long, J., Girshick, R., Guadarrama, S., & Darrell, T. (2014). Caffe: Convolutional architecture for fast feature embedding. In ACM multimedia conference.","DOI":"10.1145\/2647868.2654889"},{"key":"1083_CR28","doi-asserted-by":"crossref","unstructured":"Kidron, E., Schechner, Y. Y., & Elad, M. (2005). Pixels that sound. In IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2005.274"},{"key":"1083_CR29","unstructured":"Kr\u00e4henb\u00fchl, P., Doersch, C., Donahue, J., & Darrell, T. (2016). Data-dependent initializations of convolutional neural networks. In International conference on learning representations"},{"key":"1083_CR30","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). Imagenet classification with deep convolutional neural networks. In Advances in neural information processing systems."},{"key":"1083_CR31","doi-asserted-by":"crossref","unstructured":"Le, Q. V., Ranzato, M. A., Monga, R., Devin, M., Chen, K., Corrado, G. S, Dean, J., & Ng, A. Y. (2012). Building high-level features using large scale unsupervised learning. In International conference on machine learning","DOI":"10.1109\/ICASSP.2013.6639343"},{"key":"1083_CR32","doi-asserted-by":"crossref","unstructured":"Lee, K., Ellis, D. P., & Loui, A. C. (2010). Detecting local semantic concepts in environmental sounds using markov model based clustering. In IEEE international conference on acoustics, speech, and signal processing.","DOI":"10.1109\/ICASSP.2010.5495915"},{"issue":"1","key":"1083_CR33","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1023\/A:1011126920638","volume":"43","author":"T Leung","year":"2001","unstructured":"Leung, T., & Malik, J. (2001). Representing and recognizing the visual appearance of materials using three-dimensional textons. International Journal of Computer Vision, 43(1), 29\u201344.","journal-title":"International Journal of Computer Vision"},{"key":"1083_CR34","unstructured":"Lin, M., Chen, Q., & Yan, S. (2014). Network in network. International conference on learning representations."},{"issue":"5","key":"1083_CR35","doi-asserted-by":"publisher","first-page":"926","DOI":"10.1016\/j.neuron.2011.06.032","volume":"71","author":"JH McDermott","year":"2011","unstructured":"McDermott, J. H., & Simoncelli, E. P. (2011). Sound texture perception via statistics of the auditory periphery: Evidence from sound synthesis. Neuron, 71(5), 926\u2013940.","journal-title":"Neuron"},{"key":"1083_CR36","unstructured":"Mishkin, D., & Matas, J. (2015). All you need is a good init. arXiv preprint \n                    arXiv:1511.06422\n                    \n                  ."},{"key":"1083_CR37","doi-asserted-by":"crossref","unstructured":"Mobahi, H., Collobert, R., & Weston, J. (2009). Deep learning from temporal coherence in video. In International conference on machine learning.","DOI":"10.1145\/1553374.1553469"},{"key":"1083_CR38","unstructured":"Ngiam, J., Khosla, A., Kim, M., Nam, J., Lee, H., & Ng, A. Y. (2011). Multimodal deep learning. In International Conference on Machine Learning."},{"key":"1083_CR39","doi-asserted-by":"crossref","unstructured":"Oquab, M., Bottou, L., Laptev, I., & Sivic, J. (2015). Is object localization for free?-weakly-supervised learning with convolutional neural networks. In Conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2015.7298668"},{"key":"1083_CR40","doi-asserted-by":"crossref","unstructured":"Owens, A., Isola, P., McDermott, J., Torralba, A., Adelson, E. H., & Freeman, W. T. (2016a). Visually indicated sounds. In CVPR.","DOI":"10.1109\/CVPR.2016.264"},{"key":"1083_CR41","doi-asserted-by":"crossref","unstructured":"Owens, A., Wu, J., McDermott, J. H., Freeman, W. T., & Torralba, A. (2016b). Ambient sound provides supervision for visual learning. In European conference on computer vision.","DOI":"10.1007\/978-3-319-46448-0_48"},{"key":"1083_CR42","doi-asserted-by":"crossref","unstructured":"Pathak, D., Girshick, R., Doll\u00e1r, P., Darrell, T., & Hariharan, B. (2017). Learning features by watching objects move. In CVPR.","DOI":"10.1109\/CVPR.2017.638"},{"issue":"7","key":"1083_CR43","doi-asserted-by":"publisher","first-page":"969","DOI":"10.1016\/j.ijar.2008.11.006","volume":"50","author":"R Salakhutdinov","year":"2009","unstructured":"Salakhutdinov, R., & Hinton, G. (2009). Semantic hashing. International Journal of Approximate Reasoning, 50(7), 969\u2013978.","journal-title":"International Journal of Approximate Reasoning"},{"key":"1083_CR44","unstructured":"Slaney, M., & Covell, M. (2000). Facesync: A linear operator for measuring synchronization of video facial images and audio tracks. In Advances in neural information processing systems."},{"issue":"1\u20132","key":"1083_CR45","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1162\/1064546053278973","volume":"11","author":"L Smith","year":"2005","unstructured":"Smith, L., & Gasser, M. (2005). The development of embodied cognition: Six lessons from babies. Artificial life, 11(1\u20132), 13\u201329.","journal-title":"Artificial life"},{"key":"1083_CR46","unstructured":"Srivastava, N., & Salakhutdinov, R. R. (2012). Multimodal learning with deep boltzmann machines. In Advances in neural information processing systems."},{"key":"1083_CR47","unstructured":"Thomee, B., Shamma, D. A., Friedland, G., Elizalde, B., Ni, K., Poland, D., Borth, D., & Li, L. J. (2015). The new data and new challenges in multimedia research. arXiv preprint \n                    a\n                    \n                  rXiv:1503.01817."},{"key":"1083_CR48","doi-asserted-by":"crossref","unstructured":"Wang, X., & Gupta, A. (2015). Unsupervised learning of visual representations using videos. In IEEE international conference on computer vision","DOI":"10.1109\/ICCV.2015.320"},{"key":"1083_CR49","unstructured":"Weiss, Y., Torralba, A., & Fergus, R. (2009). Spectral hashing. In Advances in neural information processing systems."},{"key":"1083_CR50","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K. A., Oliva, A., & Torralba, A. (2010). Sun database: Large-scale scene recognition from abbey to zoo. In IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"1083_CR51","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., & Efros, A. A. (2016). Colorful image colorization. In European conference on computer vision pp. 649\u2013666. Springer","DOI":"10.1007\/978-3-319-46487-9_40"},{"key":"1083_CR52","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., & Efros, A. A. (2017). Split-brain autoencoders: Unsupervised learning by cross-channel prediction. In CVPR.","DOI":"10.1109\/CVPR.2017.76"},{"key":"1083_CR53","unstructured":"Zhou, B., Lapedriza, A., Xiao, J., Torralba, A., & Oliva, A. (2014). Learning deep features for scene recognition using places database. In Advances in neural information processing systems."},{"key":"1083_CR54","unstructured":"Zhou, B., Khosla, A., Lapedriza, A., Oliva, A., & Torralba, A. (2015). Object detectors emerge in deep scene cnns. In International conference on learning representations."},{"key":"1083_CR55","doi-asserted-by":"crossref","unstructured":"Zhou, B., Khosla, A., Lapedriza, A., Oliva, A., & Torralba, A. (2016). Learning deep features for discriminative localization. In The IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2016.319"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11263-018-1083-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-018-1083-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-018-1083-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,5,17]],"date-time":"2020-05-17T07:18:46Z","timestamp":1589699926000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11263-018-1083-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,7,11]]},"references-count":55,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2018,10]]}},"alternative-id":["1083"],"URL":"https:\/\/doi.org\/10.1007\/s11263-018-1083-5","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,7,11]]},"assertion":[{"value":"9 May 2017","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 March 2018","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2018","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}