{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T20:03:01Z","timestamp":1776888181168,"version":"3.51.2"},"publisher-location":"Cham","reference-count":47,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032212993","type":"print"},{"value":"9783032213006","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21300-6_16","type":"book-chapter","created":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T12:58:27Z","timestamp":1774357107000},"page":"254-270","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Learning Audio\u2013Visual Embeddings with\u00a0Inferred Latent Interaction Graphs"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6425-6270","authenticated-orcid":false,"given":"Donghuo","family":"Zeng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Niu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Masato","family":"Taya","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,25]]},"reference":[{"key":"16_CR1","unstructured":"Abu-El-Haija, S., et al.: Youtube-8m: a large-scale video classification benchmark. arXiv preprint arXiv:1609.08675 (2016)"},{"key":"16_CR2","doi-asserted-by":"crossref","unstructured":"Agarwal, V., Shetty, R., Fritz, M.: Towards causal VQA: revealing and reducing spurious correlations by invariant and covariant semantic editing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9690\u20139698 (2020)","DOI":"10.1109\/CVPR42600.2020.00971"},{"key":"16_CR3","unstructured":"Andrew, G., Arora, R., Bilmes, J., Livescu, K.: Deep canonical correlation analysis. In: Proceedings of the 30th International Conference on Machine Learning, vol.28 of Proceedings of Machine Learning Research, pp.1247\u20131255, Atlanta, Georgia, 17\u201319 Jun. PMLR (2013)"},{"key":"16_CR4","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., Zisserman, A.: Look, listen and learn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 609\u2013617 (2017)","DOI":"10.1109\/ICCV.2017.73"},{"key":"16_CR5","doi-asserted-by":"crossref","unstructured":"Cheng, Y., Wang, R., Pan, Z., Feng, R., Zhang, Y.: Look, listen, and attend: co-attention network for self-supervised audio-visual representation learning. In: Proceedings of the 28th ACM International Conference on Multimedia, MM \u201920, pp. 3884\u20133892, New York. Association for Computing Machinery (2020)","DOI":"10.1145\/3394171.3413869"},{"key":"16_CR6","doi-asserted-by":"crossref","unstructured":"David, H., R., S\u00e1ndor, S., John, S.T.: Canonical correlation analysis: an overview with application to learning methods. Neural Comput. 16(12), 2639\u20132664 (2004)","DOI":"10.1162\/0899766042321814"},{"key":"16_CR7","doi-asserted-by":"crossref","unstructured":"Hadsell, R., Chopra, S., LeCun, Y.: Dimensionality reduction by learning an invariant mapping. In: 2006 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR\u201906), vol.\u00a02, pp. 1735\u20131742. IEEE (2006)","DOI":"10.1109\/CVPR.2006.100"},{"key":"16_CR8","unstructured":"Hermans, A., Beyer, L., Leibe, B.: In defense of the triplet loss for person re-identification. arXiv preprint arXiv:1703.07737 (2017)"},{"key":"16_CR9","doi-asserted-by":"crossref","unstructured":"Hershey, S., et\u00a0al.: CNN architectures for large-scale audio classification. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (icassp), pp. 131\u2013135. IEEE (2017)","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"16_CR10","doi-asserted-by":"crossref","unstructured":"Hogan, J.W.: Causal inference in statistics: a primer Judea pearl, maria Glymour, and Nicholas Jewell, John Wiley & Sons, Ltd., chichester, UK. (2019)","DOI":"10.1111\/biom.13079"},{"key":"16_CR11","first-page":"18661","volume":"33","author":"P Khosla","year":"2020","unstructured":"Khosla, P., et al.: Supervised contrastive learning. Adv. Neural. Inf. Process. Syst. 33, 18661\u201318673 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR12","doi-asserted-by":"crossref","unstructured":"Kim, J.M., Koepke, A., Schmid, C., Akata, Z.: Exposing and mitigating spurious correlations for cross-modal retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2585\u20132595 (2023)","DOI":"10.1109\/CVPRW59228.2023.00257"},{"key":"16_CR13","unstructured":"Kingma, D.P.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"16_CR14","unstructured":"Korbar, B., Tran, D., Torresani, L.: Cooperative learning of audio and video models from self-supervised synchronization. Adv. Neural Inf. Process. Syst. 31 (2018)"},{"issue":"5","key":"16_CR15","doi-asserted-by":"publisher","first-page":"365","DOI":"10.1142\/S012906570000034X","volume":"10","author":"PL Lai","year":"2000","unstructured":"Lai, P.L., Fyfe, C.: Kernel and nonlinear canonical correlation analysis. Int. J. Neural Syst. 10(5), 365\u2013377 (2000)","journal-title":"Int. J. Neural Syst."},{"key":"16_CR16","unstructured":"Lam, W.Y., Andrews, B., Ramsey, J.: Greedy relaxations of the sparsest permutation algorithm. In: Uncertainty in Artificial Intelligence, pp. 1052\u20131062. PMLR (2022)"},{"key":"16_CR17","unstructured":"Li, L.H., Yatskar, M., Yin, D., Hsieh, C.J., Chang, K.W.: Visualbert: a simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557 (2019)"},{"key":"16_CR18","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"16_CR19","unstructured":"Pearl, J., Mackenzie, D.: The book of why: the new science of cause and effect. Basic books (2018)"},{"key":"16_CR20","doi-asserted-by":"crossref","unstructured":"Qi, J., Niu, Y., Huang, J., Zhang, H.: Two causal principles for improving visual dialog. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10860\u201310869 (2020)","DOI":"10.1109\/CVPR42600.2020.01087"},{"key":"16_CR21","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"16_CR22","unstructured":"Rasiwasia, N., Mahajan, D., Mahadevan, V., Aggarwal, G.: Cluster canonical correlation analysis. In: Proceedings of the Seventeenth International Conference on Artificial Intelligence and Statistics, pp. 823\u2013831, Reykjavik, Iceland. JMLR.org (2014)"},{"key":"16_CR23","doi-asserted-by":"crossref","unstructured":"Roth, J., et\u00a0al.: Ava active speaker: an audio-visual dataset for active speaker detection. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4492\u20134496. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053900"},{"key":"16_CR24","unstructured":"Ruder, S.: An overview of gradient descent optimization algorithms. arXiv preprint arXiv:1609.04747 (2016)"},{"key":"16_CR25","doi-asserted-by":"crossref","unstructured":"Schroff, F., Kalenichenko, D., Philbin, J.: Facenet: a unified embedding for face recognition and clustering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 815\u2013823 (2015)","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"16_CR26","unstructured":"Sohn, K.: Improved deep metric learning with multi-class n-pair loss objective. Adv. Neural Inf. Process. Syst., 29 (2016)"},{"key":"16_CR27","doi-asserted-by":"crossref","unstructured":"Sur\u00eds, D., Duarte, A., Salvador, A., Torres, J., Gir\u00f3-i-Nieto, X.: Cross-modal embeddings for video and audio retrieval. In: Proceedings of the European Conference on Computer Vision (ECCV) Workshops (2018)","DOI":"10.1007\/978-3-030-11018-5_62"},{"key":"16_CR28","unstructured":"Abbasnejad, E., Teney, D., Hengel, A.V.D.: Unshuffling data for improved generalization in visual question answering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 1417\u20131427 (2021)"},{"key":"16_CR29","doi-asserted-by":"crossref","unstructured":"Tian, Y., Shi, J., Li, B., Duan, Z., Xu, C.: Audio-visual event localization in unconstrained videos. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 247\u2013263 (2018)","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"16_CR30","doi-asserted-by":"crossref","unstructured":"Torfi, A., Iranmanesh, S.M., Nasrabadi, N., Dawson, J.: 3d convolutional neural networks for cross audio-visual matching recognition. IEEE Access 5, 22081\u201322091 (2017)","DOI":"10.1109\/ACCESS.2017.2761539"},{"key":"16_CR31","doi-asserted-by":"publisher","first-page":"51229","DOI":"10.1109\/ACCESS.2023.3280187","volume":"11","author":"Y Wang","year":"2023","unstructured":"Wang, Y., Zeng, D., Wada, S., Kurihara, S.: Videoadviser: video knowledge distillation for multimodal transfer learning. IEEE Access 11, 51229\u201351240 (2023)","journal-title":"IEEE Access"},{"issue":"528","key":"16_CR32","doi-asserted-by":"publisher","first-page":"1574","DOI":"10.1080\/01621459.2019.1686987","volume":"114","author":"Y Wang","year":"2019","unstructured":"Wang, Y., Blei, D.M.: The blessings of multiple causes. J. Am. Stat. Assoc. 114(528), 1574\u20131596 (2019)","journal-title":"J. Am. Stat. Assoc."},{"issue":"11","key":"16_CR33","first-page":"12996","volume":"45","author":"X Yang","year":"2021","unstructured":"Yang, X., Zhang, H., Cai, J.: Deconfounded image captioning: a causal retrospect. IEEE Trans. Pattern Anal. Mach. Intell. 45(11), 12996\u201313010 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"4","key":"16_CR34","doi-asserted-by":"publisher","first-page":"1250","DOI":"10.1109\/TNNLS.2018.2856253","volume":"30","author":"Y Yu","year":"2019","unstructured":"Yu, Y., Tang, S., Aizawa, K., Aizawa, A.: Category-based deep CCA for fine-grained venue discovery from multimodal data. IEEE Trans. Neural Netw. Learn. Syst. 30(4), 1250\u20131258 (2019)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"16_CR35","doi-asserted-by":"crossref","unstructured":"Zeng, D., Ikeda, K.: Two-stage triplet loss training with curriculum augmentation for audio-visual retrieval. arXiv preprint arXiv:2310.13451 (2023)","DOI":"10.1109\/ISM59092.2023.00038"},{"key":"16_CR36","doi-asserted-by":"crossref","unstructured":"Zeng, D., Ikeda, K.: Metric learning with progressive self-distillation for audio-visual embedding learning. In: ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2025)","DOI":"10.1109\/ICASSP49660.2025.10888698"},{"key":"16_CR37","doi-asserted-by":"crossref","unstructured":"Zeng, D., Oyama, K.: Learning joint embedding for cross-modal retrieval. In: International Conference on Data Mining Workshops (ICDMW), pp. 1070\u20131071. IEEE, Beijing (2019)","DOI":"10.1109\/ICDMW.2019.00156"},{"key":"16_CR38","doi-asserted-by":"crossref","unstructured":"Zeng, D., Wang, Y., Ikeda, K., Yu, Y.: Anchor-aware deep metric learning for audio-visual retrieval. In: Proceedings of the 2024 International Conference on Multimedia Retrieval, pp. 211\u2013219 (2024)","DOI":"10.1145\/3652583.3658067"},{"key":"16_CR39","doi-asserted-by":"crossref","unstructured":"Zeng, D., Wang, Y., Wu, J., Ikeda, K.: Complete cross-triplet loss in label space for audio-visual cross-modal retrieval. In: 2022 IEEE International Symposium on Multimedia (ISM), pp. 1\u20139. IEEE (2022)","DOI":"10.1109\/ISM55400.2022.00007"},{"key":"16_CR40","doi-asserted-by":"crossref","unstructured":"Zeng, D., Wu, J., Hattori, G., Xu, R., Yu, Y.: Learning explicit and implicit dual common subspaces for audio-visual cross-modal retrieval. ACM Trans. Multimedia Comput. Commun. Appl. 19(2s), 1\u201323 (2023)","DOI":"10.1145\/3564608"},{"key":"16_CR41","doi-asserted-by":"crossref","unstructured":"Zeng, D., Yu, Y., Oyama, K.: Audio-visual embedding for cross-modal music video retrieval through supervised deep CCA. In: 2018 IEEE International Symposium on Multimedia (ISM), pp. 143\u2013150. IEEE (2018)","DOI":"10.1109\/ISM.2018.00-21"},{"issue":"3","key":"16_CR42","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3387164","volume":"16","author":"D Zeng","year":"2020","unstructured":"Zeng, D., Yi, Yu., Oyama, K.: Deep triplet neural networks with cluster-CCA for audio-visual cross-modal retrieval. ACM Trans. Multimed. Comput. Commun. Appl. 16(3), 1\u201323 (2020)","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"16_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, J., Yu, Y., Tang, S., Wu, J., Li, W.: Multi-scale network with shared cross-attention for audio\u2013visual correlation learning. Neural Comput. Appl., 1\u201315 (2023)","DOI":"10.1007\/s00521-023-08817-1"},{"key":"16_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, J., Yu, Y., Tang, S., Wu, J., Li, W.: Variational autoencoder with CCA for audio\u2013visual cross-modal retrieval. ACM Trans. Multimedia Comput. Commun. Appl. 19(3s) (2023)","DOI":"10.1145\/3575658"},{"key":"16_CR45","unstructured":"Zhang, K., Xie, S., Ng, I., Zheng, Y.: Causal representation learning from multiple distributions: a general setting. arXiv preprint arXiv:2402.05052, 2024"},{"key":"16_CR46","doi-asserted-by":"crossref","unstructured":"Zheng, L., Cheng, H.Y., Cao, N., He, J.: Deep co-attention network for multi-view subspace learning. In: Proceedings of the Web Conference, pp. 1528\u20131539 (2021)","DOI":"10.1145\/3442381.3449801"},{"key":"16_CR47","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Wang, Z., Fang, C., Bui, T., erg, T.L.: Visual to sound: Generating natural sound for videos in the wild. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3550\u20133558 (2018)","DOI":"10.1109\/CVPR.2018.00374"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21300-6_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T12:58:54Z","timestamp":1774357134000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21300-6_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032212993","9783032213006"],"references-count":47,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21300-6_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"25 March 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors declare that they have no competing interests.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Delft","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 March 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"48","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2026.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}