{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T06:34:35Z","timestamp":1763102075681,"version":"3.45.0"},"reference-count":44,"publisher":"Tech Science Press","issue":"3","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["CMC"],"published-print":{"date-parts":[[2025]]},"DOI":"10.32604\/cmc.2025.061187","type":"journal-article","created":{"date-parts":[[2025,2,2]],"date-time":"2025-02-02T04:18:26Z","timestamp":1738469906000},"page":"5169-5184","source":"Crossref","is-referenced-by-count":1,"title":["Cross-Modal Simplex Center Learning for Speech-Face Association"],"prefix":"10.32604","volume":"82","author":[{"given":"Qiming","family":"Ma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fanliang","family":"Bu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lingbin","family":"Bu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiyuan","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"17807","published-online":{"date-parts":[[2025]]},"reference":[{"key":"ref1","series-title":"2021 IEEE Winter Conference on Applications of Computer Vision (WACV)","first-page":"1547","article-title":"FairFace: face attribute dataset for balanced race, gender, and age for bias measurement and mitigation","author":"K\u00e4rkk\u00e4inen","year":"2021"},{"key":"ref2","series-title":"2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","first-page":"687","article-title":"Voxceleb enrichment for age and gender recognition","author":"Hechmi","year":"2021"},{"key":"ref3","doi-asserted-by":"crossref","first-page":"1709","DOI":"10.1016\/j.cub.2003.09.005","article-title":"Putting the face to the voice: matching identity across modality","volume":"13","author":"Kamachi","year":"2003","journal-title":"Curr Biol"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"342","DOI":"10.1152\/jn.00459.2022","article-title":"Electrocorticography reveals the dynamics of famous voice responses in human fusiform gyrus","volume":"129","author":"Rhone","year":"2022 Dec","journal-title":"J Neurophysiol"},{"key":"ref5","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"73","article-title":"Learnable PINs: cross-modal embeddings for person identity","author":"Nagrani","year":"2018"},{"key":"ref6","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"8427","article-title":"Seeing voices and hearing faces: cross-modal biometric matching","author":"Nagrani","year":"2018"},{"key":"ref7","series-title":"Proceedings of the 26th ACM International Conference on Multimedia","first-page":"1011","article-title":"Face-voice matching using cross-modal embeddings","author":"Horiguchi","year":"2018"},{"key":"ref8","series-title":"Computer Vision-ACCV: 14th Asian Conference on Computer Vision","first-page":"276","article-title":"On learning associations of faces and voices","author":"Kim","year":"2019"},{"key":"ref9","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1945","article-title":"Triplet-center loss for multi-view 3D object retrieval","author":"He","year":"2018"},{"key":"ref10","unstructured":"Yang S, Tantrawenith M, Zhuang H, Zhiyong W, Aolan S, Jianzong W, et al. Speech representation disentanglement with adversarial mutual information learning for one-shot voice conversion. doi:10.48550\/arXiv.2208.08757."},{"key":"ref11","series-title":"Proceedings of the 46th International ACM SIGIR Conference on Research and Development in Information Retrieval","first-page":"1252","article-title":"Learnable pillar-based re-ranking for image-text retrieval","author":"Qu","year":"2023"},{"key":"ref12","series-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"18749","article-title":"FaceFormer: speech-driven 3D facial animation with transformers","author":"Fan","year":"2022"},{"key":"ref13","series-title":"Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence","first-page":"5410","article-title":"Image-text retrieval: a survey on recent research and development","author":"Cao","year":"2022"},{"key":"ref14","series-title":"Proceedings of the Sixteenth ACM International Conference on Web Search and Data Mining","first-page":"456","article-title":"Aligning cross-modal entities for image-text retrieval upon vision-language pre-trained models","author":"Wang","year":"2023"},{"key":"ref15","series-title":"Advances in Neural Information Processing Systems","first-page":"63529","article-title":"Achieving cross modal generalization with multimodal unified representation","author":"Xia","year":"2023"},{"key":"ref16","series-title":"2021 IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"9103","article-title":"CDS: cross-domain self-supervised pre-training","author":"Kim","year":"2021"},{"key":"ref17","series-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"15159","article-title":"Learning semantic relationship among instances for image-text matching","author":"Fu","year":"2023"},{"key":"ref18","series-title":"Computer Vision-ECCV 2022: 17th European Conference","first-page":"69","article-title":"Learning visual representation from modality-shared contrastive language-image pre-training","author":"You","year":"2022"},{"key":"ref19","series-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"2787","article-title":"Cross-modal implicit relation reasoning and aligning for text-to-image person retrieval","author":"Jiang","year":"2023"},{"key":"ref20","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"707","article-title":"Deep cross-modal projection learning for image-text matching","author":"Zhang","year":"2018"},{"key":"ref21","doi-asserted-by":"crossref","first-page":"22","DOI":"10.1145\/3284750","article-title":"CM-GANs: cross-modal generative adversarial networks for common representation learning","volume":"15","author":"Peng","year":"2019","journal-title":"ACM Trans Multimedia Comput Commun Appl"},{"key":"ref22","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"261","article-title":"Supervised cross-modal contrastive learning for audio-visual coding","author":"Sun","year":"2023"},{"key":"ref23","doi-asserted-by":"crossref","first-page":"6216","DOI":"10.1109\/TPAMI.2024.3379752","article-title":"Causality-invariant interactive mining for cross-modal similarity learning","volume":"46","author":"Yan","year":"2024","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"ref24","series-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics","first-page":"3013","article-title":"Cross-modal discrete representation learning","author":"Liu","year":"2021"},{"key":"ref25","series-title":"2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"3141","article-title":"Cross-modal center loss for 3D cross-modal retrieval","author":"Jing","year":"2021"},{"key":"ref26","series-title":"Proceedings of the European Conference on Computer Vision (ECCV) Workshops","first-page":"711","article-title":"Cross-modal embeddings for video and audio retrieval","author":"Sur\u00eds","year":"2018"},{"key":"ref27","series-title":"2022 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","first-page":"2111","article-title":"Masking modalities for cross-modal video retrieval","author":"Gabeur","year":"2022"},{"key":"ref28","series-title":"2022 IEEE International Conference on Multimedia and Expo (ICME)","first-page":"1","article-title":"Listen and look: multi-modal aggregation and co-attention network for video-audio retrieval","author":"Hao","year":"2022"},{"key":"ref29","doi-asserted-by":"crossref","first-page":"112076","DOI":"10.1016\/j.knosys.2024.112076","article-title":"Video and audio are images: a cross-modal mixer for original data on video-audio retrieval","volume":"299","author":"Yuan","year":"2024","journal-title":"Knowl Based Syst"},{"key":"ref30","series-title":"2019 Digital Image Computing: Techniques and Applications (DICTA)","first-page":"1","article-title":"Deep latent space learning for cross-modal mapping of audio and visual signals","author":"Nawaz","year":"2019"},{"key":"ref31","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"16347","article-title":"Seeking the shape of sound: an adaptive framework for learning voice-face association","author":"Wen","year":"2021"},{"key":"ref32","unstructured":"Wen Y, Al Ismail M, Liu W, Raj B, Singh R. Disjoint mapping network for cross-modal matching of voices and faces. In: ICLR 2019; 2019 [cited 2024 Nov 11]. p. 1\u201317. Available from: https:\/\/api.semanticscholar.org\/Corpus."},{"key":"ref33","doi-asserted-by":"crossref","first-page":"338","DOI":"10.1109\/TMM.2021.3050089","article-title":"Adversarial-metric learning for audio-visual cross-modal matching","volume":"24","author":"Zheng","year":"2022","journal-title":"IEEE Trans Multimed"},{"key":"ref34","series-title":"British Machine Vision Conference","article-title":"Deep face recognition","author":"Parkhi","year":"2015 [cited 2024 Nov 11]"},{"key":"ref35","series-title":"2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"131","article-title":"CNN architectures for large-scale audio classification","author":"Hershey","year":"2017"},{"key":"ref36","series-title":"Computer Vision-ECCV 2016: 14th European Conference","first-page":"499","article-title":"A discriminative feature learning approach for deep face recognition","author":"Wen","year":"2016"},{"key":"ref37","doi-asserted-by":"crossref","first-page":"4373","DOI":"10.1109\/TNNLS.2021.3056762","article-title":"Regular polytope networks","volume":"33","author":"Pernici","year":"2022","journal-title":"IEEE Trans Neural Net Learn Syst"},{"key":"ref38","series-title":"Image analysis","first-page":"16","article-title":"Prototype softmax cross entropy: a new perspective on softmax cross entropy","author":"Bytyqi","year":"2023"},{"key":"ref39","series-title":"Image analysis","first-page":"91","article-title":"Deep simplex classifier for maximizing the margin in both Euclidean and angular spaces","author":"Cevikalp","year":"2023"},{"key":"ref40","doi-asserted-by":"crossref","first-page":"427","DOI":"10.1111\/j.1467-9868.2005.00510.x","article-title":"Geometric representation of high dimension, low sample size data","volume":"67","author":"Hall","year":"2005","journal-title":"J R Stat Soc Ser B: Stat Methodol"},{"key":"ref41","series-title":"2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"815","article-title":"FaceNet: a unified embedding for face recognition and clustering","author":"Schroff","year":"2015"},{"key":"ref42","series-title":"2023 IEEE 32nd International Symposium on Industrial Electronics (ISIE)","first-page":"1","article-title":"ARTriViT: automatic face recognition system using ViT-based siamese neural networks with a triplet loss","author":"Khan","year":"2023"},{"key":"ref43","series-title":"Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval","first-page":"1881","article-title":"Learning discriminative joint embeddings for efficient face and voice association","author":"Wang","year":"2020"},{"key":"ref44","doi-asserted-by":"crossref","unstructured":"Nagrani A, Chung JS, Zisserman A. VoxCeleb: a Large-Scale Speaker Identification Dataset. 2017. doi:10.48550\/arXiv.1706.08612.","DOI":"10.21437\/Interspeech.2017-950"}],"container-title":["Computers, Materials &amp; Continua"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/cdn.techscience.cn\/files\/cmc\/2025\/TSP_CMC-82-3\/TSP_CMC_61187\/TSP_CMC_61187.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T06:31:27Z","timestamp":1763101887000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.techscience.com\/cmc\/v82n3\/59935"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":44,"journal-issue":{"issue":"3","published-online":{"date-parts":[[2025]]},"published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.32604\/cmc.2025.061187","relation":{},"ISSN":["1546-2226"],"issn-type":[{"type":"electronic","value":"1546-2226"}],"subject":[],"published":{"date-parts":[[2025]]}}}