{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T11:17:21Z","timestamp":1783423041837,"version":"3.54.6"},"reference-count":69,"publisher":"Springer Science and Business Media LLC","issue":"32","license":[{"start":{"date-parts":[[2024,2,26]],"date-time":"2024-02-26T00:00:00Z","timestamp":1708905600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,2,26]],"date-time":"2024-02-26T00:00:00Z","timestamp":1708905600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-024-18575-4","type":"journal-article","created":{"date-parts":[[2024,2,26]],"date-time":"2024-02-26T07:02:11Z","timestamp":1708930931000},"page":"78563-78576","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Multimodal pre-train then transfer learning approach for speaker recognition"],"prefix":"10.1007","volume":"83","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0655-8414","authenticated-orcid":false,"given":"Summaira","family":"Jabeen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Muhammad Shoib","family":"Amin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xi","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,2,26]]},"reference":[{"key":"18575_CR1","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1016\/j.neunet.2021.03.004","volume":"140","author":"Z Bai","year":"2021","unstructured":"Bai Z, Zhang XL (2021) Speaker recognition based on deep learning: an overview. Neural Netw 140:65\u201399","journal-title":"Neural Netw"},{"key":"18575_CR2","doi-asserted-by":"crossref","unstructured":"Chung JS, Nagrani A, Zisserman A (2018) VoxCeleb2: deep speaker recognition. In: INTERSPEECH","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"18575_CR3","doi-asserted-by":"crossref","unstructured":"Nagrani A, Chung JS, Zisserman A (2017) VoxCeleb: a large-scale speaker identification dataset. In: INTERSPEECH","DOI":"10.21437\/Interspeech.2017-950"},{"key":"18575_CR4","doi-asserted-by":"crossref","unstructured":"Jung Jw, Kim YJ, Heo HS, Lee BJ, Kwon Y, Chung JS (2022) Pushing the limits of raw waveform speaker recognition. In: Proc. Interspeech","DOI":"10.21437\/Interspeech.2022-126"},{"key":"18575_CR5","unstructured":"Stoll LL (2011) Finding difficult speakers in automatic speaker recognition. PhD thesis, EECS Department, University of California, Berkeley. http:\/\/www2.eecs.berkeley.edu\/Pubs\/TechRpts\/2011\/EECS-2011-152.html"},{"issue":"1\u20133","key":"18575_CR6","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1006\/dspr.1999.0361","volume":"10","author":"DA Reynolds","year":"2000","unstructured":"Reynolds DA, Quatieri TF, Dunn RB (2000) Speaker verification using adapted Gaussian mixture models. Digit Sig Process 10(1\u20133):19\u201341","journal-title":"Digit Sig Process"},{"key":"18575_CR7","unstructured":"Kenny P (2005) Joint factor analysis of speaker and session variability: theory and algorithms. CRIM, Montreal,(Report) CRIM-06\/08\u201313 14(28\u201329):2"},{"key":"18575_CR8","doi-asserted-by":"crossref","unstructured":"Chatfield K, Simonyan K, Vedaldi A, Zisserman A (2014) Return of the devil in the details: delving deep into convolutional nets. In: British machine vision conference","DOI":"10.5244\/C.28.6"},{"key":"18575_CR9","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"18575_CR10","doi-asserted-by":"crossref","unstructured":"Nagrani A, Albanie S, Zisserman A (2018) Seeing voices and hearing faces: cross-modal biometric matching. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 8427\u20138436","DOI":"10.1109\/CVPR.2018.00879"},{"key":"18575_CR11","doi-asserted-by":"crossref","unstructured":"Nagrani A, Albanie S, Zisserman A (2018) Learnable pins: cross-modal embeddings for person identity. In: Proceedings of the European conference on computer vision (ECCV). pp 71\u201388","DOI":"10.1007\/978-3-030-01261-8_5"},{"key":"18575_CR12","doi-asserted-by":"crossref","unstructured":"Saeed MS, Nawaz S, Yousaf Khan MH, Zaheer MZ, Nandakumar K, Yousaf MH, Mahmood A (2023) Single-branch network for multimodal training. In: ICASSP 2023\u20132023 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE","DOI":"10.1109\/ICASSP49357.2023.10097207"},{"key":"18575_CR13","doi-asserted-by":"crossref","unstructured":"Horiguchi S, Kanda N, Nagamatsu K (2018) Face-voice matching using cross-modal embeddings. In: Proceedings of the 26th ACM international conference on multimedia. pp 1011\u20131019","DOI":"10.1145\/3240508.3240601"},{"key":"18575_CR14","doi-asserted-by":"crossref","unstructured":"Nawaz S, Janjua MK, Gallo I, Mahmood A, Calefati A (2019) Deep latent space learning for cross-modal mapping of audio and visual signals. In: 2019 digital image computing: techniques and applications (DICTA). IEEE, pp 1\u20137","DOI":"10.1109\/DICTA47822.2019.8945863"},{"key":"18575_CR15","doi-asserted-by":"crossref","unstructured":"Wen P, Xu Q, Jiang Y, Yang Z, He Y, Huang Q (2021) Seeking the shape of sound: an adaptive framework for learning voice-face association. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 16347\u201316356","DOI":"10.1109\/CVPR46437.2021.01608"},{"key":"18575_CR16","unstructured":"Wen Y, Ismail MA, Liu W, Raj B, Singh R (2019) Disjoint mapping network for cross-modal matching of voices and faces. In: 7th International conference on learning representations, ICLR 2019, New Orleans, LA, USA"},{"key":"18575_CR17","doi-asserted-by":"crossref","unstructured":"Shah SH, Saeed MS, Nawaz S, Yousaf MH (2023) Speaker recognition in realistic scenario using multimodal data. In: 2023 3rd international conference on artificial intelligence (ICAI). IEEE, pp 209\u2013213","DOI":"10.1109\/ICAI58407.2023.10136626"},{"key":"18575_CR18","doi-asserted-by":"crossref","unstructured":"Saeed MS, Nawaz S, Khan MH, Javed S, Yousaf MH, Del Bue A (2022) Learning branched fusion and orthogonal projection for face-voice association. arXiv:2208.10238","DOI":"10.1109\/ICASSP43922.2022.9747704"},{"key":"18575_CR19","doi-asserted-by":"crossref","unstructured":"Nawaz S, Saeed MS, Morerio P, Mahmood A, Gallo I, Yousaf MH, Del Bue A (2021) Cross-modal speaker verification and recognition: a multilingual perspective. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 1682\u20131691","DOI":"10.1109\/CVPRW53098.2021.00184"},{"key":"18575_CR20","doi-asserted-by":"crossref","unstructured":"Albanie S, Nagrani A, Vedaldi A, Zisserman A (2018) Emotion recognition in speech using cross-modal transfer in the wild. In: Proceedings of the 26th ACM international conference on multimedia. pp 292\u2013301","DOI":"10.1145\/3240508.3240578"},{"key":"18575_CR21","doi-asserted-by":"crossref","unstructured":"Afouras T, Chung JS, Zisserman A (2018) The conversation: deep audio-visual speech enhancement. In: INTERSPEECH","DOI":"10.21437\/Interspeech.2018-1400"},{"key":"18575_CR22","doi-asserted-by":"crossref","unstructured":"Koepke AS, Wiles O, Zisserman A (2018) Self-supervised learning of a facial attribute embedding from video. In: BMVC. pp 302","DOI":"10.1109\/ICCVW.2019.00364"},{"key":"18575_CR23","doi-asserted-by":"crossref","unstructured":"Ellis AW (1989) Neuro-cognitive processing of faces and voices. In: Handbook of research on face processing. Elsevier, pp 207\u2013215","DOI":"10.1016\/B978-0-444-87143-5.50017-2"},{"issue":"19","key":"18575_CR24","doi-asserted-by":"publisher","first-page":"1709","DOI":"10.1016\/j.cub.2003.09.005","volume":"13","author":"M Kamachi","year":"2003","unstructured":"Kamachi M, Hill H, Lander K, Vatikiotis-Bateson E (2003) Putting the face to the voice\u2019: matching identity across modality. Curr Biol 13(19):1709\u20131714","journal-title":"Curr Biol"},{"key":"18575_CR25","doi-asserted-by":"crossref","unstructured":"Kim C, Shin HV, Oh TH, Kaspar A, Elgharib M, Matusik W (2018) On learning associations of faces and voices. In: Asian conference on computer vision. Springer, pp 276\u2013292","DOI":"10.1007\/978-3-030-20873-8_18"},{"issue":"3","key":"18575_CR26","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1121\/1.1918467","volume":"35","author":"S Pruzansky","year":"1963","unstructured":"Pruzansky S (1963) Pattern-matching procedure for automatic talker recognition. J Acoust Soc Am 35(3):354\u2013358","journal-title":"J Acoust Soc Am"},{"key":"18575_CR27","doi-asserted-by":"crossref","unstructured":"Dehak N, Kenny P, Dehak R, Glembek O, Dumouchel P, Burget L, Hubeika V, Castaldo F (2009) Support vector machines and joint factor analysis for speaker verification. In: 2009 IEEE international conference on acoustics, speech and signal processing. IEEE, pp 4237\u20134240","DOI":"10.1109\/ICASSP.2009.4960564"},{"issue":"4","key":"18575_CR28","doi-asserted-by":"publisher","first-page":"788","DOI":"10.1109\/TASL.2010.2064307","volume":"19","author":"N Dehak","year":"2020","unstructured":"Dehak N, Kenny PJ, Dehak R, Dumouchel P, Ouellet P (2020) Front-end factor analysis for speaker verification. IEEE Trans Audio Speech Lang Process 19(4):788\u2013798","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"18575_CR29","doi-asserted-by":"crossref","unstructured":"Yapanel U, Zhang X, Hansen JH (2002) High performance digit recognition in real car environments. In: Seventh international conference on spoken language processing","DOI":"10.21437\/ICSLP.2002-276"},{"issue":"7553","key":"18575_CR30","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1038\/nature14539","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun Y, Bengio Y, Hinton G (2015) Deep learning. Nature 521(7553):436\u2013444","journal-title":"Nature"},{"key":"18575_CR31","doi-asserted-by":"crossref","unstructured":"Lei Y, Scheffer N, Ferrer L, McLaren M (2014) A novel scheme for speaker recognition using a phonetically-aware deep neural network. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 1695\u20131699","DOI":"10.1109\/ICASSP.2014.6853887"},{"key":"18575_CR32","doi-asserted-by":"crossref","unstructured":"Snyder D, Garcia-Romero D, Sell G, Povey D, Khudanpur S (2018) X-vectors: Robust DNN embeddings for speaker recognition. In: 2018 IEEE International conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5329\u20135333","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"18575_CR33","doi-asserted-by":"crossref","unstructured":"Salman A, Chen K (2011) Exploring speaker-specific characteristics with deep learning. In: The 2011 international joint conference on neural networks. IEEE, pp 103\u2013110","DOI":"10.1109\/IJCNN.2011.6033207"},{"key":"18575_CR34","doi-asserted-by":"crossref","unstructured":"Xie W, Nagrani A, Chung JS, Zisserman A (2019) Utterance-level aggregation for speaker recognition in the wild. In: ICASSP 2019-2019 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5791\u20135795","DOI":"10.1109\/ICASSP.2019.8683120"},{"key":"18575_CR35","doi-asserted-by":"crossref","unstructured":"Arandjelovic R, Gronat P, Torii A, Pajdla T, Sivic J (2016) NetVLAD: CNN architecture for weakly supervised place recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 5297\u20135307","DOI":"10.1109\/CVPR.2016.572"},{"key":"18575_CR36","doi-asserted-by":"crossref","unstructured":"Zhong Y, Arandjelovi\u0107 R, Zisserman A (2019) GhostVLAD for set-based face recognition. In: Computer vision\u2013ACCV 2018: 14th Asian conference on computer vision, Perth, Australia, December 2\u20136, 2018, Revised Selected Papers, Part II 14. Springer, pp 35\u201350","DOI":"10.1007\/978-3-030-20890-5_3"},{"key":"18575_CR37","doi-asserted-by":"crossref","unstructured":"Wang R, Ao J, Zhou L, Liu S, Wei Z, Ko T, Li Q, Zhang Y (2022) Multi-view self-attention based transformer for speaker recognition. In: ICASSP 2022\u20132022 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6732\u20136736","DOI":"10.1109\/ICASSP43922.2022.9746639"},{"key":"18575_CR38","doi-asserted-by":"crossref","unstructured":"India M, Safari P, Hernando J (2021) Double multi-head attention for speaker verification. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6144\u20136148","DOI":"10.1109\/ICASSP39728.2021.9414877"},{"key":"18575_CR39","doi-asserted-by":"publisher","unstructured":"Zhu H, Lee KA, Li H (2021) Serialized multi-layer multi-head attention for neural speaker embedding. In: Proc. Interspeech 2021. pp 106\u2013110. https:\/\/doi.org\/10.21437\/Interspeech.2021-2210","DOI":"10.21437\/Interspeech.2021-2210"},{"key":"18575_CR40","doi-asserted-by":"crossref","unstructured":"Wu CY, Hsu CC, Neumann U (2022) Cross-modal perceptionist: can face geometry be gleaned from voices? In: CVPR","DOI":"10.1109\/CVPR52688.2022.01020"},{"key":"18575_CR41","doi-asserted-by":"crossref","unstructured":"Wang J, Li C, Zheng A, Tang J, Luo B (2022) Looking and hearing into details: dual-enhanced Siamese adversarial network for audio-visual matching. IEEE Transactions on Multimedia","DOI":"10.1109\/TMM.2022.3222936"},{"key":"18575_CR42","doi-asserted-by":"crossref","unstructured":"Saeed MS, Khan MH, Nawaz S, Yousaf MH, Del Bue A (2022) Fusion and orthogonal projection for improved face-voice association. In: ICASSP 2022\u20132022 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 7057\u20137061","DOI":"10.1109\/ICASSP43922.2022.9747704"},{"issue":"2","key":"18575_CR43","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1109\/TPAMI.2018.2798607","volume":"41","author":"T Baltru\u0161aitis","year":"2018","unstructured":"Baltru\u0161aitis T, Ahuja C, Morency LP (2018) Multimodal machine learning: a survey and taxonomy. IEEE Trans Pattern Anal Mach Intell 41(2):423\u2013443","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"18575_CR44","doi-asserted-by":"crossref","unstructured":"Vielzeuf V, Lechervy A, Pateux S, Jurie F (2018) Centralnet: a multilayer approach for multimodal fusion. In: Proceedings of the European conference on computer vision (ECCV) workshops. pp 0\u20130","DOI":"10.1007\/978-3-030-11024-6_44"},{"key":"18575_CR45","doi-asserted-by":"crossref","unstructured":"Kiela D, Grave E, Joulin A, Mikolov T (2018) Efficient large-scale multi-modal classification. arXiv:1802.02892","DOI":"10.1609\/aaai.v32i1.11945"},{"key":"18575_CR46","first-page":"2611","volume":"33","author":"D Kiela","year":"2020","unstructured":"Kiela D, Firooz H, Mohan A, Goswami V, Singh A, Ringshia P, Testuggine D (2020) The hateful memes challenge: detecting hate speech in multimodal memes. Adv Neural Inf Process Syst 33:2611\u20132624","journal-title":"Adv Neural Inf Process Syst"},{"key":"18575_CR47","doi-asserted-by":"crossref","unstructured":"Gallo I, Calefati A, Nawaz S (2017) Multimodal classification fusion in real-world scenarios. In: 2017 14th IAPR international conference on document analysis and recognition (ICDAR), vol 5. IEEE, pp 36\u201341","DOI":"10.1109\/ICDAR.2017.326"},{"key":"18575_CR48","doi-asserted-by":"crossref","unstructured":"Arshad O, Gallo I, Nawaz S, Calefati A (2019) Aiding intra-text representations with visual context for multimodal named entity recognition. In: 2019 international conference on document analysis and recognition (ICDAR). IEEE, pp 337\u2013342","DOI":"10.1109\/ICDAR.2019.00061"},{"key":"18575_CR49","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 6077\u20136086","DOI":"10.1109\/CVPR.2018.00636"},{"key":"18575_CR50","doi-asserted-by":"crossref","unstructured":"Fukui A, Park DH, Yang D, Rohrbach A, Darrell T, Rohrbach M (2016) Multimodal compact bilinear pooling for visual question answering and visual grounding. arXiv:1606.01847","DOI":"10.18653\/v1\/D16-1044"},{"key":"18575_CR51","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan D (2015) Show and tell: a neural image caption generator. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 3156\u20133164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"18575_CR52","doi-asserted-by":"crossref","unstructured":"Yan L, Han C, Xu Z, Liu D, Wang Q (2023) Prompt learns prompt: exploring knowledge-aware generative prompt collaboration for video captioning. In: Proceedings of international joint conference on artificial intelligence (IJCAI). pp 1622\u20131630","DOI":"10.24963\/ijcai.2023\/180"},{"key":"18575_CR53","doi-asserted-by":"crossref","unstructured":"Yan L, Wang Q, Cui Y, Feng F, Quan X, Zhang X, Liu D (2022) GL-RG: global-local representation granularity for video captioning. arXiv:2205.10706","DOI":"10.24963\/ijcai.2022\/384"},{"key":"18575_CR54","doi-asserted-by":"crossref","unstructured":"Popattia M, Rafi M, Qureshi R, Nawaz S (2022) Guiding attention using partial order relationships for image captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 4671\u20134680","DOI":"10.1109\/CVPRW56347.2022.00513"},{"key":"18575_CR55","doi-asserted-by":"crossref","unstructured":"Yan L, Liu D, Song Y, Yu C (2020) Multimodal aggregation approach for memory vision-voice indoor navigation with meta-learning. In: 2020 IEEE\/RSJ international conference on intelligent robots and systems (IROS). IEEE, pp 5847\u20135854","DOI":"10.1109\/IROS45743.2020.9341398"},{"key":"18575_CR56","doi-asserted-by":"crossref","unstructured":"Nawaz S, Cavazza J, Del Bue A (2022) Semantically grounded visual embeddings for zero-shot learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 4589\u20134599","DOI":"10.1109\/CVPRW56347.2022.00505"},{"key":"18575_CR57","doi-asserted-by":"crossref","unstructured":"Wang L, Li Y, Lazebnik S (2016) Learning deep structure-preserving image-text embeddings. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 5005\u20135013","DOI":"10.1109\/CVPR.2016.541"},{"key":"18575_CR58","doi-asserted-by":"crossref","unstructured":"Nagrani A, Chung JS, Albanie S, Zisserman A (2020) Disentangled speech embeddings using cross-modal self-supervision. In: ICASSP 2020\u20132020 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6829\u20136833","DOI":"10.1109\/ICASSP40776.2020.9054057"},{"key":"18575_CR59","doi-asserted-by":"crossref","unstructured":"Hajavi A, Etemad A (2023) Audio representation learning by distilling video as privileged information. IEEE Transactions on Artificial Intelligence","DOI":"10.1109\/TAI.2023.3243596"},{"key":"18575_CR60","unstructured":"Nawaz S (2019) Multimodal representation and learning. PhD thesis, Universit\u00e1 degli Studi dell\u2019Insubria"},{"key":"18575_CR61","doi-asserted-by":"crossref","unstructured":"Szegedy C, Ioffe S, Vanhoucke V, Alemi AA (2017) Inception-v4, inception-resnet and the impact of residual connections on learning. In: Thirty-first AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"18575_CR62","doi-asserted-by":"crossref","unstructured":"Schroff F, Kalenichenko D, Philbin J (2015) FaceNet: a unified embedding for face recognition and clustering. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 815\u2013823","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"18575_CR63","unstructured":"Ioffe S, Szegedy C (2015) Batch normalization: accelerating deep network training by reducing internal covariate shift. In: International conference on machine learning. pp 448\u2013456. PMLR"},{"key":"18575_CR64","unstructured":"Calefati A, Janjua MK, Nawaz S, Gallo I (2018) Git loss for deep face recognition. In: Proceedings of the British machine vision conference (BMVC)"},{"key":"18575_CR65","doi-asserted-by":"crossref","unstructured":"Sar\u0131 L, Singh K, Zhou J, Torresani L, Singhal N, Saraf Y (2021) A multi-view approach to audio-visual speaker verification. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6194\u20136198","DOI":"10.1109\/ICASSP39728.2021.9414260"},{"key":"18575_CR66","doi-asserted-by":"publisher","first-page":"338","DOI":"10.1109\/TMM.2021.3050089","volume":"24","author":"A Zheng","year":"2021","unstructured":"Zheng A, Hu M, Jiang B, Huang Y, Yan Y, Luo B (2021) Adversarial-metric learning for audio-visual cross-modal matching. IEEE Trans Multimedia 24:338\u2013351","journal-title":"IEEE Trans Multimedia"},{"key":"18575_CR67","doi-asserted-by":"publisher","first-page":"1763","DOI":"10.1109\/TMM.2021.3071243","volume":"24","author":"H Ning","year":"2021","unstructured":"Ning H, Zheng X, Lu X, Yuan Y (2021) Disentangled representation learning for cross-modal biometric matching. IEEE Trans Multimedia 24:1763\u20131774","journal-title":"IEEE Trans Multimedia"},{"key":"18575_CR68","doi-asserted-by":"crossref","unstructured":"Deng J, Guo J, Xue N, Zafeiriou S (2019) ArcFace: additive angular margin loss for deep face recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 4690\u20134699","DOI":"10.1109\/CVPR.2019.00482"},{"key":"18575_CR69","unstructured":"VGG Dataset Privacy Notice\u2013robots.ox.ac.uk. https:\/\/www.robots.ox.ac.uk\/~vgg\/terms\/url-lists-privacy-notice.html. Accessed 01 Jan 2024"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18575-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-024-18575-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18575-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,4]],"date-time":"2024-09-04T04:23:54Z","timestamp":1725423834000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-024-18575-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,26]]},"references-count":69,"journal-issue":{"issue":"32","published-online":{"date-parts":[[2024,9]]}},"alternative-id":["18575"],"URL":"https:\/\/doi.org\/10.1007\/s11042-024-18575-4","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,2,26]]},"assertion":[{"value":"22 November 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 January 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 February 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 February 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}]}}