{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T04:53:21Z","timestamp":1742964801618,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":31,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819786190"},{"type":"electronic","value":"9789819786206"}],"license":[{"start":{"date-parts":[[2024,10,20]],"date-time":"2024-10-20T00:00:00Z","timestamp":1729382400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,20]],"date-time":"2024-10-20T00:00:00Z","timestamp":1729382400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8620-6_36","type":"book-chapter","created":{"date-parts":[[2024,10,19]],"date-time":"2024-10-19T21:02:10Z","timestamp":1729371730000},"page":"526-540","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Robust Contrastive Learning Against Audio-Visual Noisy Correspondence"],"prefix":"10.1007","author":[{"given":"Yihan","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Xi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gairui","family":"Bai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinhui","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jizhong","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,20]]},"reference":[{"key":"36_CR1","first-page":"9758","volume":"33","author":"H Alwassel","year":"2020","unstructured":"Alwassel, H., Mahajan, D., Korbar, B., Torresani, L., Ghanem, B., Tran, D.: Self-supervised learning by cross-modal audio-video clustering. Adv. Neural. Inf. Process. Syst. 33, 9758\u20139770 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"36_CR2","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., Zisserman, A.: Look, listen and learn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 609\u2013617 (2017)","DOI":"10.1109\/ICCV.2017.73"},{"key":"36_CR3","doi-asserted-by":"crossref","unstructured":"Aytar, Y., Vondrick, C., Torralba, A.: Soundnet: learning sound representations from unlabeled video. Adv. Neural Inf. Process. Syst. 29 (2016)","DOI":"10.1109\/CVPR.2016.18"},{"issue":"3","key":"36_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3550307","volume":"6","author":"D Cao","year":"2022","unstructured":"Cao, D., Liu, R., Li, H., Wang, S., Jiang, W., Lu, C.X.: Cross vision-rf gait re-identification with low-cost rgb-d cameras and mmwave radars. Proc. ACM Interact. Mobile Wearable Ubiquit. Technol. 6(3), 1\u201325 (2022)","journal-title":"Proc. ACM Interact. Mobile Wearable Ubiquit. Technol."},{"key":"36_CR5","doi-asserted-by":"crossref","unstructured":"Chen, H., Xie, W., Afouras, T., Nagrani, A., Vedaldi, A., Zisserman, A.: Localizing visual sounds the hard way. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16867\u201316876 (2021)","DOI":"10.1109\/CVPR46437.2021.01659"},{"key":"36_CR6","doi-asserted-by":"crossref","unstructured":"Chen, H., Xie, W., Vedaldi, A., Zisserman, A.: Vggsound: a large-scale audio-visual dataset. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 721\u2013725. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"36_CR7","unstructured":"Gutmann, M., Hyv\u00e4rinen, A.: Noise-contrastive estimation: a new estimation principle for unnormalized statistical models. In: Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics, pp. 297\u2013304. JMLR Workshop and Conference Proceedings (2010)"},{"key":"36_CR8","unstructured":"Hu, D., Qian, R., Jiang, M., Tan, X., Wen, S., Ding, E., Lin, W., Dou, D.: Discriminative sounding objects localization via self-supervised audiovisual matching. Adv. Neural Inf. Process. Syst. 33 (2020)"},{"key":"36_CR9","unstructured":"Huang, Z., Niu, G., Liu, X., Ding, W., Xiao, X., Wu, H., Peng, X.: Learning with noisy correspondence for cross-modal matching. Adv. Neural Inf. Process. Syst. 34 (2021)"},{"key":"36_CR10","doi-asserted-by":"crossref","unstructured":"Jiang, W., Li, F., Mei, L., Liu, R., Wang, S.: Visble: Vision-enhanced BLE device tracking. In: 2022 19th Annual IEEE International Conference on Sensing, Communication, and Networking (SECON), pp. 217\u2013225. IEEE (2022)","DOI":"10.1109\/SECON55815.2022.9918581"},{"key":"36_CR11","doi-asserted-by":"crossref","unstructured":"Li, H., Liu, R., Wang, S., Jiang, W., Lu, C.X.: Pedestrian liveness detection based on mmwave radar and camera fusion. In: 2022 19th Annual IEEE International Conference on Sensing, Communication, and Networking (SECON), pp. 262\u2013270. IEEE (2022)","DOI":"10.1109\/SECON55815.2022.9918553"},{"key":"36_CR12","unstructured":"Lin, Y.B., Tseng, H.Y., Lee, H.Y., Lin, Y.Y., Yang, M.H.: Unsupervised sound localization via iterative contrastive learning. arXiv preprint arXiv:2104.00315 (2021)"},{"key":"36_CR13","doi-asserted-by":"crossref","unstructured":"Liu, J., Ju, C., Xie, W., Zhang, Y.: Exploiting transformation invariance and equivariance for self-supervised sound localisation. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 3742\u20133753 (2022)","DOI":"10.1145\/3503161.3548317"},{"issue":"10","key":"36_CR14","doi-asserted-by":"publisher","first-page":"5535","DOI":"10.1109\/TWC.2014.2336653","volume":"13","author":"K Liu","year":"2014","unstructured":"Liu, K., Ma, Q., Gong, W., Miao, X., Liu, Y.: Self-diagnosis for detecting system failures in large-scale wireless sensor networks. IEEE Trans. Wirel. Commun. 13(10), 5535\u20135545 (2014)","journal-title":"IEEE Trans. Wirel. Commun."},{"key":"36_CR15","doi-asserted-by":"crossref","unstructured":"Mo, S., Morgado, P.: Localizing visual sounds the easy way. In: Computer Vision\u2013ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXXVII, pp. 218\u2013234. Springer (2022)","DOI":"10.1007\/978-3-031-19836-6_13"},{"key":"36_CR16","doi-asserted-by":"crossref","unstructured":"Morgado, P., Misra, I., Vasconcelos, N.: Robust audio-visual instance discrimination. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12934\u201312945 (2021)","DOI":"10.1109\/CVPR46437.2021.01274"},{"key":"36_CR17","doi-asserted-by":"crossref","unstructured":"Morgado, P., Vasconcelos, N., Misra, I.: Audio-visual instance discrimination with cross-modal agreement. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12475\u201312486 (2021)","DOI":"10.1109\/CVPR46437.2021.01274"},{"key":"36_CR18","unstructured":"Oord, A.v.d., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)"},{"key":"36_CR19","doi-asserted-by":"crossref","unstructured":"Owens, A., Efros, A.A.: Audio-visual scene analysis with self-supervised multisensory features. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 631\u2013648 (2018)","DOI":"10.1007\/978-3-030-01231-1_39"},{"key":"36_CR20","doi-asserted-by":"crossref","unstructured":"Senocak, A., Oh, T.H., Kim, J., Yang, M.H., Kweon, I.S.: Learning to localize sound source in visual scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4358\u20134366 (2018)","DOI":"10.1109\/CVPR.2018.00458"},{"key":"36_CR21","doi-asserted-by":"crossref","unstructured":"Senocak, A., Ryu, H., Kim, J., Kweon, I.S.: Learning sound localization better from semantically similar samples. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4863\u20134867. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9747867"},{"key":"36_CR22","unstructured":"Song, Z., Wang, Y., Fan, J., Tan, T., Zhang, Z.: Self-supervised predictive learning: a negative-free method for sound source localization in visual scenes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3222\u20133231 (2022)"},{"key":"36_CR23","doi-asserted-by":"crossref","unstructured":"Sun, W., Zhang, J., Wang, J., Liu, Z., Zhong, Y., Feng, T., Guo, Y., Zhang, Y., Barnes, N.: Learning audio-visual source localization via false negative aware contrastive learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6420\u20136429 (2023)","DOI":"10.1109\/CVPR52729.2023.00621"},{"key":"36_CR24","doi-asserted-by":"crossref","unstructured":"Tian, Y., Hu, D., Xu, C.: Cyclic co-learning of sounding object visual grounding and sound separation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2745\u20132754 (2021)","DOI":"10.1109\/CVPR46437.2021.00277"},{"key":"36_CR25","doi-asserted-by":"crossref","unstructured":"Tian, Y., Shi, J., Li, B., Duan, Z., Xu, C.: Audio-visual event localization in unconstrained videos. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 247\u2013263 (2018)","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"36_CR26","first-page":"22099","volume":"36","author":"Z Wang","year":"2023","unstructured":"Wang, Z., Zhao, Y., Huang, H., Liu, J., Yin, A., Tang, L., Li, L., Wang, Y., Zhang, Z., Zhao, Z.: Connecting multi-modal contrastive representations. Adv. Neural. Inf. Process. Syst. 36, 22099\u201322114 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"3","key":"36_CR27","doi-asserted-by":"publisher","first-page":"1909","DOI":"10.1109\/COMST.2023.3282264","volume":"25","author":"LT Yang","year":"2023","unstructured":"Yang, L.T., Zhao, R., Liu, D., Lu, W., Deng, X.: Tensor-empowered federated learning for cyber-physical-social computing and communication systems. IEEE Commun. Surv. Tutor. 25(3), 1909\u20131940 (2023)","journal-title":"IEEE Commun. Surv. Tutor."},{"key":"36_CR28","doi-asserted-by":"crossref","unstructured":"Yang, M., Li, Y., Huang, Z., Liu, Z., Hu, P., Peng, X.: Partially view-aligned representation learning with noise-robust contrastive loss. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1134\u20131143 (2021)","DOI":"10.1109\/CVPR46437.2021.00119"},{"key":"36_CR29","unstructured":"Ye, J., Guo, J., Xiang, Y., Tan, K., Yu, Z.: Noise-robust cross-modal interactive learning with text2image mask for multi-modal neural machine translation. In: Proceedings of the 29th International Conference on Computational Linguistics, pp. 5098\u20135108 (2022)"},{"key":"36_CR30","doi-asserted-by":"crossref","unstructured":"Yuan, X., Lin, Z., Kuen, J., Zhang, J., Wang, Y., Maire, M., Kale, A., Faieta, B.: Multimodal contrastive training for visual representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6995\u20137004 (2021)","DOI":"10.1109\/CVPR46437.2021.00692"},{"key":"36_CR31","doi-asserted-by":"crossref","unstructured":"Zolfaghari, M., Zhu, Y., Gehler, P., Brox, T.: Crossclr: cross-modal contrastive learning for multi-modal video representations. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1450\u20131459 (2021)","DOI":"10.1109\/ICCV48922.2021.00148"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition and Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8620-6_36","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,14]],"date-time":"2025-01-14T20:16:43Z","timestamp":1736885803000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8620-6_36"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,20]]},"ISBN":["9789819786190","9789819786206"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8620-6_36","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,10,20]]},"assertion":[{"value":"20 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Pattern Recognition and Computer Vision  (PRCV)","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccprcv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2024.prcv.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}