{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T21:20:00Z","timestamp":1776979200258,"version":"3.51.4"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T00:00:00Z","timestamp":1774396800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T00:00:00Z","timestamp":1774396800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Aeronautical Science Foundation of China","award":["202400080U0001"],"award-info":[{"award-number":["202400080U0001"]}]},{"name":"Fundamental Research Program of Shanxi Province","award":["202403021222185"],"award-info":[{"award-number":["202403021222185"]}]},{"DOI":"10.13039\/100014718","name":"Innovative Research Group Project of the National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["51821003"],"award-info":[{"award-number":["51821003"]}],"id":[{"id":"10.13039\/100014718","id-type":"DOI","asserted-by":"publisher"}]},{"name":"the fundamental research program of Shanxi Province","award":["202303021211150"],"award-info":[{"award-number":["202303021211150"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s11760-026-05203-x","type":"journal-article","created":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T21:11:05Z","timestamp":1774473065000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Audio-face fusion of multi-methods for speaker verification in strong interference"],"prefix":"10.1007","volume":"20","author":[{"given":"Xuan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaochen","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huijun","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenguang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chong","family":"Shen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,25]]},"reference":[{"issue":"1","key":"5203_CR1","doi-asserted-by":"publisher","first-page":"1175","DOI":"10.1109\/JIOT.2023.3290001","volume":"11","author":"X Li","year":"2024","unstructured":"Li, X., Zheng, Z., Yan, C., Li, C., Ji, X., Xu, W.: Toward Pitch-Insensitive Speaker Verification via Soundfield. IEEE Internet Things J. 11(1), 1175\u20131189 (2024)","journal-title":"IEEE Internet Things J."},{"key":"5203_CR2","doi-asserted-by":"crossref","unstructured":"Desplanques, B., Thienpondt, J., Demuynck, K.: ECAPATDNN: emphasized channel attention, propagation and aggregation in TDNN based speaker verification., in Interspeech, pp. 3830\u20133834 (2020)","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"5203_CR3","unstructured":"Haibo, W., et al.: CAM++: A Fast and Efficient Network for Speaker Verification Using Context-Aware Masking. in arXiv:2303.00332. (2023): n. pag"},{"key":"5203_CR4","unstructured":"Yang, Z., et al.: Mfa-conformer: Multi-scale feature aggregation conformer for automatic speaker verification. in arXiv preprint, (2022). arXiv:2203.15249"},{"key":"5203_CR5","unstructured":"Ivan, Y., et al.: Reshape dimensions network for speaker recognition. in (2024). arXiv:2407.18223"},{"key":"5203_CR6","unstructured":"Tianchi, L., et al.: Golden gemini is all you need: Finding the sweet spots for speaker verification. in IEEE\/ACM Transactions on Audio, Speech, and Language Processing, (2024)"},{"key":"5203_CR7","doi-asserted-by":"publisher","first-page":"124159","DOI":"10.1016\/j.eswa.2024.124159","volume":"252","author":"R Dmitry","year":"2024","unstructured":"Dmitry, R., et al.: Audio\u2013visual speech recognition based on regulated transformer and spatio\u2013temporal fusion strategy for driver assistive systems. in Expert Systems with Applications 252, 124159 (2024)","journal-title":"in Expert Systems with Applications"},{"key":"5203_CR8","unstructured":"Debang, L., et al.: Audio-Visual Fusion With Temporal Convolutional Attention Network for Speech Separation. in IEEE\/ACM Transactions on Audio, Speech, and Language Processing, APA (2024)"},{"key":"5203_CR9","unstructured":"Xiaoqin, Z., et al.: Transformer-based multimodal emotional perception for dynamic facial expression recognition in the wild. in IEEE Transactions on Circuits and Systems for Video Technology, (2023)"},{"key":"5203_CR10","doi-asserted-by":"crossref","unstructured":"Sell, G., Duh, K., Snyder, D., Etter, D., Garcia-Romero, D.: Audio-visual person recognition in multimedia data from the iarpa janus program. in ICASSP, pp. 3031\u20133035 (2018)","DOI":"10.1109\/ICASSP.2018.8462122"},{"key":"5203_CR11","doi-asserted-by":"crossref","unstructured":"Sadjadi, S.O., Greenberg, C.S., Singer, E., Reynolds, D.A., Mason, L., Hernandez-Cordero, J.: The 2019 nist audio-visual speaker recognition evaluation, in Proc. Odyssey 2020 The Speaker and Language Recogn. Workshop, pp. 259\u2013265 (2020)","DOI":"10.21437\/Odyssey.2020-37"},{"issue":"8","key":"5203_CR12","doi-asserted-by":"publisher","first-page":"9454","DOI":"10.1109\/TPAMI.2023.3243048","volume":"45","author":"Z Peng","year":"2023","unstructured":"Peng, Z., et al.: Conformer: Local Features Coupling Global Representations for Recognition and Detection. IEEE Trans. Pattern Anal. Mach. Intell. 45(8), 9454\u20139468 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"5203_CR13","doi-asserted-by":"crossref","unstructured":"He, K., et al.: (2016) Deep residual learning for image recognition. in Proceedings of the IEEE conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2016.90"},{"key":"5203_CR14","doi-asserted-by":"crossref","unstructured":"Finder S.E., Amoyal, R., Treister, E., et al.: Wavelet convolutions for large receptive fields.in European Conference on Computer Vision, Cham: Springer Nature Switzerland: 363\u2013380 (2024)","DOI":"10.1007\/978-3-031-72949-2_21"},{"key":"5203_CR15","doi-asserted-by":"crossref","unstructured":"Deng, J., Guo, J., Xue, N., Zafeiriou, S.: Arc-face: Additive angular margin loss for deep face recognition. in IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4690\u20134699 (2019)","DOI":"10.1109\/CVPR.2019.00482"},{"key":"5203_CR16","doi-asserted-by":"crossref","unstructured":"Wang, H., Wang, Y., Zhou, Z., et al.: Cosface: Large margin cosine loss for deep face recognition. in Proceedings of the IEEE conference on computer vision and pattern recognition, 5265\u20135274 (2018)","DOI":"10.1109\/CVPR.2018.00552"},{"key":"5203_CR17","doi-asserted-by":"crossref","unstructured":"Wang X., Zhang S., Wang S., et al.: Mis-classified vector guided softmax loss for face recognition., in Proceedings of the AAAI conference on artificial intelligence, 34 (07): 12241\u201312248 (2020)","DOI":"10.1609\/aaai.v34i07.6906"},{"key":"5203_CR18","doi-asserted-by":"crossref","unstructured":"Huang, Y., et al.: Curricularface: adaptive curriculum learning loss for deep face recognition., in proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, (2020)","DOI":"10.1109\/CVPR42600.2020.00594"},{"key":"5203_CR19","doi-asserted-by":"crossref","unstructured":"Kim, M., Jain, A.K., Liu, X.: Adaface: Quality adaptive margin for face recognition. in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, (2022)","DOI":"10.1109\/CVPR52688.2022.01819"},{"key":"5203_CR20","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Chung, J.S., Zisserman, A.: VoxCeleb: A largescale speaker identification dataset. in Interspeech, pp. 2616\u20132620 (2017)","DOI":"10.21437\/Interspeech.2017-950"},{"key":"5203_CR21","doi-asserted-by":"crossref","unstructured":"Fan, Y., et al., CN-Celeb: A Challenging Chinese Speaker Recognition Dataset. in ICASSP, IEEE, (2020)","DOI":"10.1109\/ICASSP40776.2020.9054017"},{"key":"5203_CR22","doi-asserted-by":"crossref","unstructured":"Scheibler, R., Bezzam, E., Dokmanic, I.: Pyroomacoustics: A python package for audio room simulation and array processing algorithms. in Proc. IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 351\u2013355, (2018)","DOI":"10.1109\/ICASSP.2018.8461310"},{"key":"5203_CR23","unstructured":"Yi, D., Lei, Z., Liao, S., Li, S.Z.: Learning face representation from scratch., in arXiv preprint, (2014) arXiv:1411.7923"},{"issue":"10","key":"5203_CR24","doi-asserted-by":"publisher","first-page":"1499","DOI":"10.1109\/LSP.2016.2603342","volume":"23","author":"K Zhang","year":"2016","unstructured":"Zhang, K., et al.: Joint face detection and alignment using multitask cascaded convolutional networks. IEEE Signal Process. Lett. 23(10), 1499\u20131503 (2016)","journal-title":"IEEE Signal Process. Lett."},{"key":"5203_CR25","doi-asserted-by":"crossref","unstructured":"Schroff, F., Kalenichenko, D., Philbin, J.: Facenet: A unified embedding for face recognition and clustering. in Proceedings of the IEEE conference on computer vision and pattern recognition, 815\u2013823. MLA (2015)","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"5203_CR26","unstructured":"Wang, F., et al.: Residual attention network for image classification. in Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 3156\u20133164 (2017)"},{"key":"5203_CR27","doi-asserted-by":"crossref","unstructured":"Liu, X., et al.: Efficientvit: Memory efficient vision transformer with cascaded group attention. in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (2023)","DOI":"10.1109\/CVPR52729.2023.01386"},{"key":"5203_CR28","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.Y., et al.: Cbam: Convolutional block attention module, in Proceedings of the European conference on computer vision (ECCV), 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"5203_CR29","unstructured":"Gao, S., Cheng, M.-M., Zhao, K., Zhang, X., Yang, M.-H., Torr, P.H.S.: Res2Net: A new multi-scale backbone architecture, in IEEE TPAMI, (2019)"},{"issue":"10","key":"5203_CR30","doi-asserted-by":"publisher","first-page":"844","DOI":"10.1007\/s11760-025-04438-4","volume":"19","author":"V Chandrabanshi","year":"2025","unstructured":"Chandrabanshi, V., Domnic, S.: Leveraging 3D-CNN and graph neural network with attention mechanism for visual speech recognition, in Signal. Image and Video Processing 19(10), 844 (2025)","journal-title":"Image and Video Processing"},{"issue":"6","key":"5203_CR31","doi-asserted-by":"publisher","first-page":"5433","DOI":"10.1007\/s11760-024-03245-7","volume":"18","author":"V Chandrabanshi","year":"2024","unstructured":"Chandrabanshi, V., Domnic, S.: A novel framework using 3D-CNN and BiLSTM model with dynamic learning rate scheduler for visual speech recognition. SIViP 18(6), 5433\u20135448 (2024)","journal-title":"SIViP"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05203-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-026-05203-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05203-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T20:32:20Z","timestamp":1776976340000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-026-05203-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,25]]},"references-count":31,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["5203"],"URL":"https:\/\/doi.org\/10.1007\/s11760-026-05203-x","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,25]]},"assertion":[{"value":"6 August 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 January 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 February 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 March 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have disclosed no relevant Conflict of interest for the content of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"209"}}