{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T09:08:40Z","timestamp":1784279320187,"version":"3.55.0"},"reference-count":40,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372003"],"award-info":[{"award-number":["62372003"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100006247","name":"Anhui University of Science and Technology","doi-asserted-by":"publisher","award":["2024yjrc95"],"award-info":[{"award-number":["2024yjrc95"]}],"id":[{"id":"10.13039\/501100006247","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003995","name":"Natural Science Foundation of Anhui Province","doi-asserted-by":"publisher","award":["2308085Y40"],"award-info":[{"award-number":["2308085Y40"]}],"id":[{"id":"10.13039\/501100003995","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFC3807501"],"award-info":[{"award-number":["2023YFC3807501"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002729","name":"Anhui University","doi-asserted-by":"publisher","award":["MMC202508"],"award-info":[{"award-number":["MMC202508"]}],"id":[{"id":"10.13039\/501100002729","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113840","type":"journal-article","created":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T15:04:03Z","timestamp":1777129443000},"page":"113840","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["Bidirectional intervention attention network for audio\u2013visual matching"],"prefix":"10.1016","volume":"179","author":[{"given":"Jiaxiang","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9820-4743","authenticated-orcid":false,"given":"Aihua","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4329-864X","authenticated-orcid":false,"given":"Dequan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenglong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenjuan","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3807-991X","authenticated-orcid":false,"given":"Ran","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113840_b1","doi-asserted-by":"crossref","unstructured":"O.-B. Mercea, L. Riesch, A. Koepke, Z. Akata, Audio-visual Generalised Zero-shot Learning with Cross-modal Attention and Language, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 10553\u201310563.","DOI":"10.1109\/CVPR52688.2022.01030"},{"key":"10.1016\/j.patcog.2026.113840_b2","doi-asserted-by":"crossref","first-page":"4986","DOI":"10.1109\/TIFS.2024.3388949","article-title":"Attribute-guided cross-modal interaction and enhancement for audio-visual matching","volume":"19","author":"Wang","year":"2024","journal-title":"IEEE Trans. Inf. Forensics Secur."},{"key":"10.1016\/j.patcog.2026.113840_b3","doi-asserted-by":"crossref","first-page":"1763","DOI":"10.1109\/TMM.2021.3071243","article-title":"Disentangled representation learning for cross-modal biometric matching","volume":"24","author":"Ning","year":"2021","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.113840_b4","doi-asserted-by":"crossref","unstructured":"Z. Yu, X. Liu, Y.-M. Cheung, M. Zhu, X. Xu, N. Wang, T. Li, Detach and Enhance: Learning Disentangled Cross-modal Latent Representation for Efficient Face-Voice Association and Matching, in: Proceedings of the IEEE International Conference on Data Mining, 2022, pp. 648\u2013655.","DOI":"10.1109\/ICDM54844.2022.00075"},{"key":"10.1016\/j.patcog.2026.113840_b5","doi-asserted-by":"crossref","first-page":"338","DOI":"10.1109\/TMM.2021.3050089","article-title":"Adversarial-metric learning for audio-visual cross-modal matching","volume":"24","author":"Zheng","year":"2021","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.113840_b6","doi-asserted-by":"crossref","unstructured":"K. Cheng, X. Liu, Y.-m. Cheung, R. Wang, X. Xu, B. Zhong, Hearing like Seeing: Improving Voice-Face Interactions and Associations via Adversarial Deep Semantic Matching Network, in: Proceedings of the ACM International Conference on Multimedia, 2020, pp. 448\u2013455.","DOI":"10.1145\/3394171.3413710"},{"key":"10.1016\/j.patcog.2026.113840_b7","doi-asserted-by":"crossref","first-page":"7505","DOI":"10.1109\/TMM.2022.3222936","article-title":"Looking and hearing into details: Dual-enhanced siamese adversarial network for audio-visual matching","volume":"25","author":"Wang","year":"2023","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.113840_b8","doi-asserted-by":"crossref","unstructured":"P. Wen, Q. Xu, Y. Jiang, Z. Yang, Y. He, Q. Huang, Seeking the shape of sound: An adaptive framework for learning voice-face association, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 16347\u201316356.","DOI":"10.1109\/CVPR46437.2021.01608"},{"key":"10.1016\/j.patcog.2026.113840_b9","doi-asserted-by":"crossref","unstructured":"D. Yang, M. Li, D. Xiao, Y. Liu, K. Yang, Z. Chen, Y. Wang, P. Zhai, K. Li, L. Zhang, Towards multimodal sentiment analysis debiasing via bias purification, in: Proceedings of the European Conference on Computer Vision, 2025, pp. 464\u2013481.","DOI":"10.1007\/978-3-031-73636-0_27"},{"key":"10.1016\/j.patcog.2026.113840_b10","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111398","article-title":"Resolving semantic conflicts in RGB-t semantic segmentation","volume":"162","author":"Zhao","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113840_b11","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111463","article-title":"MDFCL: Multimodal data fusion-based graph contrastive learning framework for molecular property prediction","volume":"163","author":"Gong","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113840_b12","doi-asserted-by":"crossref","unstructured":"T. Sun, W. Wang, L. Jing, Y. Cui, X. Song, L. Nie, Counterfactual reasoning for out-of-distribution multimodal sentiment analysis, in: Proceedings of the ACM International Conference on Multimedia, 2022, pp. 15\u201323.","DOI":"10.1145\/3503161.3548211"},{"key":"10.1016\/j.patcog.2026.113840_b13","doi-asserted-by":"crossref","unstructured":"X. Li, Y. Lu, B. Liu, Y. Liu, G. Yin, Q. Chu, J. Huang, F. Zhu, R. Zhao, N. Yu, Counterfactual intervention feature transfer for visible-infrared person re-identification, in: Proceedings of the European Conference on Computer Vision, 2022, pp. 381\u2013398.","DOI":"10.1007\/978-3-031-19809-0_22"},{"issue":"12","key":"10.1016\/j.patcog.2026.113840_b14","doi-asserted-by":"crossref","first-page":"10663","DOI":"10.1109\/TPAMI.2024.3443129","article-title":"Towards context-aware emotion recognition debiasing from a causal demystification perspective via de-confounded training","volume":"46","author":"Yang","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113840_b15","doi-asserted-by":"crossref","unstructured":"M. Ye, J. Shen, D. J. Crandall, L. Shao, J. Luo, Dynamic dual-attentive aggregation learning for visible-infrared person re-identification, in: Proceedings of the European Conference on Computer Vision, 2020, pp. 229\u2013247.","DOI":"10.1007\/978-3-030-58520-4_14"},{"key":"10.1016\/j.patcog.2026.113840_b16","doi-asserted-by":"crossref","unstructured":"A. Nagrani, J.S. Chung, A. Zisserman, Voxceleb: a large-scale speaker identification dataset, in: Proceedings of the International Speech Communication Association, 2017, pp. 2616\u20132620.","DOI":"10.21437\/Interspeech.2017-950"},{"key":"10.1016\/j.patcog.2026.113840_b17","doi-asserted-by":"crossref","unstructured":"J.S. Chung, A. Nagrani, A. Zisserman, Voxceleb2: Deep speaker recognition, in: Proceedings of the International Speech Communication Association, 2018, pp. 1086\u20131090.","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"10.1016\/j.patcog.2026.113840_b18","doi-asserted-by":"crossref","unstructured":"A. Nagrani, S. Albanie, A. Zisserman, Seeing voices and hearing faces: Cross-modal biometric matching, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2018, pp. 8427\u20138436.","DOI":"10.1109\/CVPR.2018.00879"},{"key":"10.1016\/j.patcog.2026.113840_b19","unstructured":"Y. Wen, M.A. Ismail, W. Liu, B. Raj, R. Singh, Disjoint mapping network for cross-modal matching of voices and faces, in: Proceedings of the International Conference on Learning Representations, 2019."},{"key":"10.1016\/j.patcog.2026.113840_b20","doi-asserted-by":"crossref","unstructured":"R. Wang, X. Liu, Y.-m. Cheung, K. Cheng, N. Wang, W. Fan, Learning discriminative joint embeddings for efficient face and voice association, in: Proceedings of the International ACM SIGIR Conference on Research and Development in Information Retrieval, 2020, pp. 1881\u20131884.","DOI":"10.1145\/3397271.3401302"},{"key":"10.1016\/j.patcog.2026.113840_b21","doi-asserted-by":"crossref","unstructured":"S. Nawaz, M.K. Janjua, I. Gallo, A. Mahmood, A. Calefati, Deep latent space learning for cross-modal mapping of audio and visual signals, in: Proceedings of the Digital Image Computing: Techniques and Applications, 2019, pp. 1\u20137.","DOI":"10.1109\/DICTA47822.2019.8945863"},{"issue":"5","key":"10.1016\/j.patcog.2026.113840_b22","doi-asserted-by":"crossref","first-page":"1025","DOI":"10.1109\/TPAMI.2019.2961900","article-title":"Adversarial cross-spectral face completion for NIR-vis face recognition","volume":"42","author":"He","year":"2019","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"3","key":"10.1016\/j.patcog.2026.113840_b23","doi-asserted-by":"crossref","first-page":"351","DOI":"10.1007\/s11633-021-1293-0","article-title":"Deep audio-visual learning: A survey","volume":"18","author":"Zhu","year":"2021","journal-title":"Int. J. Autom. Comput."},{"issue":"9","key":"10.1016\/j.patcog.2026.113840_b24","doi-asserted-by":"crossref","first-page":"8698","DOI":"10.1109\/TCSVT.2024.3390573","article-title":"Public-private attributes-based variational adversarial network for audio-visual cross-modal matching","volume":"34","author":"Zheng","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.113840_b25","unstructured":"I. Higgins, L. Matthey, A. Pal, C. Burgess, X. Glorot, M. Botvinick, S. Mohamed, A. Lerchner, beta-vae: Learning basic visual concepts with a constrained variational framework, in: Proceedings of the International Conference on Learning Representations, 2016."},{"key":"10.1016\/j.patcog.2026.113840_b26","unstructured":"H. Kim, A. Mnih, Disentangling by factorising, in: Proceedings of the International Conference on Machine Learning, 2018, pp. 2649\u20132658."},{"key":"10.1016\/j.patcog.2026.113840_b27","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111394","article-title":"Unbiased VQA via modal information interaction and question transformation","author":"Peng","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113840_b28","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102725","article-title":"Atcaf: Attention-based causality-aware fusion network for multimodal sentiment analysis","volume":"114","author":"Huang","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.patcog.2026.113840_b29","doi-asserted-by":"crossref","unstructured":"L. Wang, Z. He, R. Dang, M. Shen, C. Liu, Q. Chen, Vision-and-language navigation via causal learning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13139\u201313150.","DOI":"10.1109\/CVPR52733.2024.01248"},{"key":"10.1016\/j.patcog.2026.113840_b30","doi-asserted-by":"crossref","unstructured":"Y. Wang, L. Meng, H. Ma, Y. Wang, H. Huang, X. Meng, Modeling event-level causal representation for video classification, in: Proceedings of the ACM International Conference on Multimedia, 2024, pp. 3936\u20133944.","DOI":"10.1145\/3664647.3681547"},{"key":"10.1016\/j.patcog.2026.113840_b31","doi-asserted-by":"crossref","unstructured":"Y. Niu, K. Tang, H. Zhang, Z. Lu, X.-S. Hua, J.-R. Wen, Counterfactual vqa: A cause-effect look at language bias, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 12700\u201312710.","DOI":"10.1109\/CVPR46437.2021.01251"},{"issue":"8","key":"10.1016\/j.patcog.2026.113840_b32","doi-asserted-by":"crossref","first-page":"10317","DOI":"10.1109\/TPAMI.2023.3261659","article-title":"Progressive instance-aware feature learning for compositional action recognition","volume":"45","author":"Yan","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113840_b33","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep residual learning for image recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2016, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.patcog.2026.113840_b34","doi-asserted-by":"crossref","unstructured":"J. Hu, L. Shen, G. Sun, Squeeze-and-excitation networks, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2018, pp. 7132\u20137141.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"10.1016\/j.patcog.2026.113840_b35","first-page":"6","article-title":"Cross-modality person re-identification with generative adversarial training","volume":"vol. 1","author":"Dai","year":"2018"},{"key":"10.1016\/j.patcog.2026.113840_b36","doi-asserted-by":"crossref","unstructured":"A. Andonian, S. Chen, R. Hamid, Robust Cross-Modal Representation Learning with Progressive Self-Distillation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 16430\u201316441.","DOI":"10.1109\/CVPR52688.2022.01594"},{"key":"10.1016\/j.patcog.2026.113840_b37","doi-asserted-by":"crossref","unstructured":"J. Deng, W. Dong, R. Socher, L.-J. Li, K. Li, L. Fei-Fei, Imagenet: A large-scale hierarchical image database, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2009, pp. 248\u2013255.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"10.1016\/j.patcog.2026.113840_b38","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.patcog.2026.113840_b39","doi-asserted-by":"crossref","unstructured":"A. Nagrani, S. Albanie, A. Zisserman, Learnable pins: Cross-modal embeddings for person identity, in: Proceedings of the European Conference on Computer Vision, 2018, pp. 71\u201388.","DOI":"10.1007\/978-3-030-01261-8_5"},{"key":"10.1016\/j.patcog.2026.113840_b40","doi-asserted-by":"crossref","unstructured":"B. Zhu, K. Xu, C. Wang, Z. Qin, T. Sun, H. Wang, Y. Peng, Unsupervised Voice-Face Representation Learning by Cross-Modal Prototype Contrast, in: Proceedings of the International Joint Conference on Artificial Intelligence, 2022, pp. 3787\u20133794.","DOI":"10.24963\/ijcai.2022\/526"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326008058?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326008058?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T08:36:13Z","timestamp":1784277373000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326008058"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":40,"alternative-id":["S0031320326008058"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113840","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Bidirectional intervention attention network for audio\u2013visual matching","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113840","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113840"}}