{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,19]],"date-time":"2025-12-19T07:03:37Z","timestamp":1766127817114,"version":"3.48.0"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T00:00:00Z","timestamp":1759968000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T00:00:00Z","timestamp":1759968000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Shaanxi provincial special fund for Technological innovation guidance","award":["2022CGBX-29"],"award-info":[{"award-number":["2022CGBX-29"]}]},{"name":"Research and Application of Key Technologies for Drone Swarm Networking and Efficient Data Backhaul","award":["2023KXJ-157"],"award-info":[{"award-number":["2023KXJ-157"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62371392"],"award-info":[{"award-number":["62371392"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Key Program for International S & T Cooperation Projects of Shaanxi Province","award":["2023-GHZD-37"],"award-info":[{"award-number":["2023-GHZD-37"]}]},{"name":"Key Industrial Chain Projects of Shaanxi Province","award":["2023-ZDLGY-49"],"award-info":[{"award-number":["2023-ZDLGY-49"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1007\/s00530-025-01999-9","type":"journal-article","created":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T09:38:31Z","timestamp":1760002711000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing cross-modal voice-face association with heterogeneous hashing network"],"prefix":"10.1007","volume":"31","author":[{"given":"Yanxia","family":"Liang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huanhuan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaopeng","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fuping","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,10,9]]},"reference":[{"issue":"1\u20132","key":"1999_CR1","doi-asserted-by":"publisher","first-page":"23","DOI":"10.1016\/S0167-6393(98)00048-X","volume":"26","author":"H Yehia","year":"1998","unstructured":"Yehia, H., Rubin, P., Vatikiotis-Bateson, E.: Quantitative association of vocal-tract and facial behavior. Speech Commun. 26(1\u20132), 23\u201343 (1998)","journal-title":"Speech Commun."},{"issue":"12","key":"1999_CR2","doi-asserted-by":"publisher","first-page":"535","DOI":"10.1016\/j.tics.2007.10.001","volume":"11","author":"S Campanella","year":"2007","unstructured":"Campanella, S., Belin, P.: Integrating face and voice in person perception. Trends Cogn. Sci. 11(12), 535\u2013543 (2007)","journal-title":"Trends Cogn. Sci."},{"issue":"36","key":"1999_CR3","doi-asserted-by":"publisher","first-page":"12906","DOI":"10.1523\/JNEUROSCI.2091-11.2011","volume":"31","author":"H Blank","year":"2011","unstructured":"Blank, H., Anwander, A., Von Kriegstein, K.: Direct structural connections between voice-and face-recognition areas. J. Neurosci. 31(36), 12906\u201312915 (2011)","journal-title":"J. Neurosci."},{"issue":"19","key":"1999_CR4","doi-asserted-by":"publisher","first-page":"1709","DOI":"10.1016\/j.cub.2003.09.005","volume":"13","author":"M Kamachi","year":"2003","unstructured":"Kamachi, M., Hill, H., Lander, K., Vatikiotis-Bateson, E.: Putting the face to the voice\u2019: matching identity across modality. Curr. Biol. 13(19), 1709\u20131714 (2003)","journal-title":"Curr. Biol."},{"issue":"11","key":"1999_CR5","doi-asserted-by":"publisher","first-page":"14470","DOI":"10.1007\/s10489-022-04216-6","volume":"53","author":"Z Fang","year":"2023","unstructured":"Fang, Z., Liu, Z., Hung, C.-C., Sekhavat, Y.A., Liu, T., Wang, X.: Learning coordinated emotion representation between voice and face. Appl. Intell. 53(11), 14470\u201314492 (2023)","journal-title":"Appl. Intell."},{"key":"1999_CR6","doi-asserted-by":"publisher","unstructured":"Oh, T.-H., Dekel, T., Kim, C., Mosseri, I., Freeman, W.T., Rubinstein, M., Matusik, W.: Speech2face: Learning the face behind a voice. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7531\u20137540 (2019). https:\/\/doi.org\/10.1109\/CVPR.2019.00772","DOI":"10.1109\/CVPR.2019.00772"},{"key":"1999_CR7","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2023.110124","volume":"136","author":"H Ilyas","year":"2023","unstructured":"Ilyas, H., Javed, A., Malik, K.M.: Avfakenet: a unified end-to-end dense swin transformer deep learning model for audio-visual deepfakes detection. Appl. Soft Comput. 136, 110124 (2023)","journal-title":"Appl. Soft Comput."},{"key":"1999_CR8","unstructured":"Saeed, M.S., Nawaz, S., Khan, M.H., Javed, S., Yousaf, M.H., Bue, A.D.: Learning Branched Fusion and Orthogonal Projection for Face-Voice Association (2022). https:\/\/arxiv.org\/abs\/2208.10238"},{"issue":"12","key":"1999_CR9","doi-asserted-by":"publisher","first-page":"8717","DOI":"10.1109\/TPAMI.2018.2889052","volume":"44","author":"T Afouras","year":"2018","unstructured":"Afouras, T., Chung, J.S., Senior, A., Vinyals, O., Zisserman, A.: Deep audio-visual speech recognition. IEEE Trans. Pattern Anal. Mach. Intell. 44(12), 8717\u20138727 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"4","key":"1999_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3197517.3201357","volume":"37","author":"A Ephrat","year":"2018","unstructured":"Ephrat, A., Mosseri, I., Lang, O., Dekel, T., Wilson, K., Hassidim, A., Freeman, W.T., Rubinstein, M.: Looking to listen at the cocktail party: a speaker-independent audio-visual model for speech separation. ACM Trans. Gr. 37(4), 1\u201311 (2018). https:\/\/doi.org\/10.1145\/3197517.3201357","journal-title":"ACM Trans. Gr."},{"key":"1999_CR11","unstructured":"Nagrani, A., Chung, J.S., Zisserman, A.: Voxceleb: a large-scale speaker identification dataset. Telephony 3, 33\u2013039"},{"key":"1999_CR12","doi-asserted-by":"crossref","unstructured":"Parkhi, O., Vedaldi, A., Zisserman, A.: Deep face recognition. In: BMVC 2015-Proceedings of the British Machine Vision Conference 2015 (2015). British Machine Vision Association","DOI":"10.5244\/C.29.41"},{"key":"1999_CR13","doi-asserted-by":"publisher","unstructured":"Nagrani, A., Albanie, S., Zisserman, A.: Seeing voices and hearing faces: Cross-modal biometric matching. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8427\u20138436 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00879","DOI":"10.1109\/CVPR.2018.00879"},{"key":"1999_CR14","unstructured":"Wen, Y., Ismail, M.A., Liu, W., Raj, B., Singh, R.: Disjoint Mapping Network for Cross-modal Matching of Voices and Faces (2018). https:\/\/arxiv.org\/abs\/1807.04836"},{"key":"1999_CR15","doi-asserted-by":"crossref","unstructured":"Tao, R., Das, R.K., Li, H.: Audio-visual Speaker Recognition with a Cross-modal Discriminative Network (2020). https:\/\/arxiv.org\/abs\/2008.03894","DOI":"10.21437\/Interspeech.2020-1814"},{"key":"1999_CR16","unstructured":"Xiong, C., Zhang, D., Liu, T., Du, X.: Voice-Face Cross-modal Matching and Retrieval: A Benchmark (2019). https:\/\/arxiv.org\/abs\/1911.09338"},{"key":"1999_CR17","doi-asserted-by":"crossref","unstructured":"Wang, R., Liu, X., Cheung, Y.-m., Cheng, K., Wang, N., Fan, W.: Learning discriminative joint embeddings for efficient face and voice association. In: Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 1881\u20131884 (2020)","DOI":"10.1145\/3397271.3401302"},{"key":"1999_CR18","doi-asserted-by":"publisher","unstructured":"Nawaz, S., Janjua, M.K., Gallo, I., Mahmood, A., Calefati, A.: Deep latent space learning for cross-modal mapping of audio and visual signals. In: 2019 Digital Image Computing: Techniques and Applications (DICTA), pp. 1\u20137 (2019). https:\/\/doi.org\/10.1109\/DICTA47822.2019.8945863","DOI":"10.1109\/DICTA47822.2019.8945863"},{"key":"1999_CR19","doi-asserted-by":"crossref","unstructured":"Zhu, B., Xu, K., Wang, C., Qin, Z., Sun, T., Wang, H., Peng, Y.: Unsupervised Voice-Face Representation Learning by Cross-Modal Prototype Contrast (2022). https:\/\/arxiv.org\/abs\/2204.14057","DOI":"10.24963\/ijcai.2022\/526"},{"key":"1999_CR20","doi-asserted-by":"crossref","unstructured":"Kim, C., Shin, H.V., Oh, T.-H., Kaspar, A., Elgharib, M., Matusik, W.: On learning associations of faces and voices. In: Computer Vision\u2013ACCV 2018: 14th Asian Conference on Computer Vision, Perth, Australia, December 2\u20136, 2018, Revised Selected Papers, Part V 14, pp. 276\u2013292 (2019). Springer","DOI":"10.1007\/978-3-030-20873-8_18"},{"key":"1999_CR21","doi-asserted-by":"crossref","unstructured":"Horiguchi, S., Kanda, N., Nagamatsu, K.: Face-voice matching using cross-modal embeddings. In: Proceedings of the 26th ACM International Conference on Multimedia, pp. 1011\u20131019 (2018)","DOI":"10.1145\/3240508.3240601"},{"key":"1999_CR22","doi-asserted-by":"publisher","unstructured":"Li, J., Tan, L., Zhou, Y., Mao, J., Liu, Z., Bu, F.: Voice-face cross-modal association learning based on deep residual shrinkage network. In: 2023 IEEE International Conference on Image Processing and Computer Applications (ICIPCA), pp. 140\u2013145 (2023). https:\/\/doi.org\/10.1109\/ICIPCA59209.2023.10257804","DOI":"10.1109\/ICIPCA59209.2023.10257804"},{"key":"1999_CR23","doi-asserted-by":"publisher","first-page":"338","DOI":"10.1109\/TMM.2021.3050089","volume":"24","author":"A Zheng","year":"2022","unstructured":"Zheng, A., Hu, M., Jiang, B., Huang, Y., Yan, Y., Luo, B.: Adversarial-metric learning for audio-visual cross-modal matching. IEEE Trans. Multimed. 24, 338\u2013351 (2022). https:\/\/doi.org\/10.1109\/TMM.2021.3050089","journal-title":"IEEE Trans. Multimed."},{"issue":"2","key":"1999_CR24","doi-asserted-by":"publisher","first-page":"560","DOI":"10.1109\/TKDE.2020.2987312","volume":"34","author":"R-C Tu","year":"2022","unstructured":"Tu, R.-C., Mao, X.-L., Ma, B., Hu, Y., Yan, T., Wei, W., Huang, H.: Deep cross-modal hashing with hashing functions and unified hash codes jointly learning. IEEE Trans. Knowl. Data Eng. 34(2), 560\u2013572 (2022). https:\/\/doi.org\/10.1109\/TKDE.2020.2987312","journal-title":"IEEE Trans. Knowl. Data Eng."},{"issue":"4","key":"1999_CR25","doi-asserted-by":"publisher","first-page":"1838","DOI":"10.1109\/TNNLS.2020.2997020","volume":"34","author":"L Jin","year":"2023","unstructured":"Jin, L., Li, Z., Tang, J.: Deep semantic multimodal hashing network for scalable image-text and video-text retrievals. IEEE Trans. Neural Netw. Learn. Syst. 34(4), 1838\u20131851 (2023). https:\/\/doi.org\/10.1109\/TNNLS.2020.2997020","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"1999_CR26","doi-asserted-by":"publisher","first-page":"5881","DOI":"10.1109\/TIP.2022.3203216","volume":"31","author":"C Zheng","year":"2022","unstructured":"Zheng, C., Zhu, L., Zhang, Z., Li, J., Yu, X.: Efficient semi-supervised multimodal hashing with importance differentiation regression. IEEE Trans. Image Process. 31, 5881\u20135892 (2022). https:\/\/doi.org\/10.1109\/TIP.2022.3203216","journal-title":"IEEE Trans. Image Process."},{"issue":"10","key":"1999_CR27","doi-asserted-by":"publisher","first-page":"4540","DOI":"10.1109\/TIP.2016.2592800","volume":"25","author":"D Wang","year":"2016","unstructured":"Wang, D., Gao, X., Wang, X., He, L., Yuan, B.: Multimodal discriminative binary embedding for large-scale cross-modal retrieval. IEEE Trans. Image Process. 25(10), 4540\u20134554 (2016). https:\/\/doi.org\/10.1109\/TIP.2016.2592800","journal-title":"IEEE Trans. Image Process."},{"key":"1999_CR28","doi-asserted-by":"crossref","unstructured":"Cao, Y., Long, M., Wang, J., Yang, Q., Yu, P.S.: Deep visual-semantic hashing for cross-modal retrieval. In: Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, pp. 1445\u20131454 (2016)","DOI":"10.1145\/2939672.2939812"},{"key":"1999_CR29","doi-asserted-by":"crossref","unstructured":"Jiang, Q.-Y., Li, W.-J.: Deep cross-modal hashing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3232\u20133240 (2017)","DOI":"10.1109\/CVPR.2017.348"},{"key":"1999_CR30","doi-asserted-by":"publisher","unstructured":"Ding, G., Guo, Y., Zhou, J.: Collective matrix factorization hashing for multimodal data. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition, pp. 2083\u20132090 (2014). https:\/\/doi.org\/10.1109\/CVPR.2014.267","DOI":"10.1109\/CVPR.2014.267"},{"key":"1999_CR31","doi-asserted-by":"crossref","unstructured":"Zhang, D., Wang, F., Si, L.: Composite hashing with multiple information sources. In: Proceedings of the 34th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 225\u2013234 (2011)","DOI":"10.1145\/2009916.2009950"},{"key":"1999_CR32","doi-asserted-by":"crossref","unstructured":"Wang, B., Yang, Y., Xu, X., Hanjalic, A., Shen, H.T.: Adversarial cross-modal retrieval. In: Proceedings of the 25th ACM International Conference on Multimedia, pp. 154\u2013162 (2017)","DOI":"10.1145\/3123266.3123326"},{"key":"1999_CR33","unstructured":"Kingma, D.P., Ba, J.: Adam: A Method for Stochastic Optimization (2017). https:\/\/arxiv.org\/abs\/1412.6980"},{"key":"1999_CR34","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Killeen, T., Lin, Z., Gimelshein, N., Antiga, L., et al.: Pytorch: An imperative style, high-performance deep learning library. Adv Neural Inf Process Syst32 (2019)"},{"key":"1999_CR35","doi-asserted-by":"crossref","unstructured":"Desplanques, B., Thienpondt, J., Demuynck, K.: Ecapa-tdnn: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification. Interspeech 2020 (2020)","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"1999_CR36","doi-asserted-by":"crossref","unstructured":"Chung, J.S., Nagrani, A., Zisserman, A.: Voxceleb2: Deep speaker recognition. arXiv preprint arXiv:1806.05622 (2018)","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"1999_CR37","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., Rabinovich, A.: Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1\u20139 (2015)","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"1999_CR38","doi-asserted-by":"crossref","unstructured":"Cao, Q., Shen, L., Xie, W., Parkhi, O.M., Zisserman, A.: VGGFace2: A dataset for recognising faces across pose and age (2018). https:\/\/arxiv.org\/abs\/1710.08092","DOI":"10.1109\/FG.2018.00020"},{"key":"1999_CR39","doi-asserted-by":"publisher","unstructured":"Saeed, M.S., Nawaz, S., Khan, M.H., Zaigham\u00a0Zaheer, M., Nandakumar, K., Yousaf, M.H., Mahmood, A.: Single-branch network for multimodal training. In: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023). https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10097207","DOI":"10.1109\/ICASSP49357.2023.10097207"},{"key":"1999_CR40","doi-asserted-by":"publisher","first-page":"1763","DOI":"10.1109\/TMM.2021.3071243","volume":"24","author":"H Ning","year":"2021","unstructured":"Ning, H., Zheng, X., Lu, X., Yuan, Y.: Disentangled representation learning for cross-modal biometric matching. IEEE Trans. Multimed. 24, 1763\u20131774 (2021). https:\/\/doi.org\/10.1109\/TMM.2021.3071243","journal-title":"IEEE Trans. Multimed."},{"issue":"7","key":"1999_CR41","first-page":"2804","volume":"24","author":"J Li","year":"2024","unstructured":"Li, J., Bu, F., Tan, L., Zhou, Y., Mao, J.: Self-supervised voice-face cross-modal association learning method via multi-modal shared network. Sci. Technol. Eng. 24(7), 2804\u20132812 (2024). ((in Chinese))","journal-title":"Sci. Technol. Eng."},{"key":"1999_CR42","doi-asserted-by":"publisher","unstructured":"Yu, Z., Liu, X., Cheung, Y.-M., Zhu, M., Xu, X., Wang, N., Li, T.: Detach and enhance: Learning disentangled cross-modal latent representation for efficient face-voice association and matching. In: 2022 IEEE International Conference on Data Mining (ICDM), pp. 648\u2013655 (2022). https:\/\/doi.org\/10.1109\/ICDM54844.2022.00075","DOI":"10.1109\/ICDM54844.2022.00075"},{"key":"1999_CR43","doi-asserted-by":"crossref","unstructured":"Chen, G., Liu, X., Xu, X., Cheung, Y.-m., Li, T.: Taking a part for the whole: An archetype-agnostic framework for voice-face association. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 7056\u20137064 (2023)","DOI":"10.1145\/3581783.3611938"},{"key":"1999_CR44","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Albanie, S., Zisserman, A.: Learnable pins: Cross-modal embeddings for person identity. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 71\u201388 (2018)","DOI":"10.1007\/978-3-030-01261-8_5"},{"key":"1999_CR45","unstructured":"Maaten, L., Hinton, G.: Visualizing data using t-sne. J. Mach. Learn. Res. 9(11) (2008)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01999-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-01999-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01999-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,19]],"date-time":"2025-12-19T06:58:47Z","timestamp":1766127527000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-01999-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,9]]},"references-count":45,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2025,12]]}},"alternative-id":["1999"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-01999-9","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"type":"print","value":"0942-4962"},{"type":"electronic","value":"1432-1882"}],"subject":[],"published":{"date-parts":[[2025,10,9]]},"assertion":[{"value":"9 February 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 August 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 October 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"411"}}