{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T12:45:50Z","timestamp":1770813950838,"version":"3.50.1"},"reference-count":57,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T00:00:00Z","timestamp":1764806400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T00:00:00Z","timestamp":1764806400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s00530-025-02091-y","type":"journal-article","created":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T07:19:22Z","timestamp":1764832762000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Domain adaptative keyword spotting with multimodal enhancement"],"prefix":"10.1007","volume":"32","author":[{"given":"Longxi","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Han","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,12,4]]},"reference":[{"key":"2091_CR1","doi-asserted-by":"publisher","unstructured":"Praveen, R.G., Melo, W.C., Ullah, N., Aslam, H., Zeeshan, O., Denorme, T., Pedersoli, M., Koerich, A.L., Bacon, S., Cardinal, P., Granger, E.: A joint cross-attention model for audio-visual fusion in dimensional emotion recognition. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, CVPR Workshops 2022, New Orleans, LA, USA, June 19-20, 2022, pp. 2485\u20132494 (2022). https:\/\/doi.org\/10.1109\/CVPRW56347.2022.00278","DOI":"10.1109\/CVPRW56347.2022.00278"},{"key":"2091_CR2","doi-asserted-by":"publisher","first-page":"39753","DOI":"10.1109\/ACCESS.2025.3546610","volume":"13","author":"J Zhang","year":"2025","unstructured":"Zhang, J., Yu, Y., Tang, S., Li, W.: Adversarial contrastive autoencoder with shared attention for audio-visual correlation learning. IEEE Access 13, 39753\u201339764 (2025). https:\/\/doi.org\/10.1109\/ACCESS.2025.3546610","journal-title":"IEEE Access"},{"key":"2091_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/J.NEUCOM.2025.129750","volume":"634","author":"M Alsuwat","year":"2025","unstructured":"Alsuwat, M., Al-Shareef, S., Alghamdi, M.: Audio-visual self-supervised representation learning: a survey. Neurocomputing 634, 129750 (2025). https:\/\/doi.org\/10.1016\/J.NEUCOM.2025.129750","journal-title":"Neurocomputing"},{"key":"2091_CR4","doi-asserted-by":"publisher","unstructured":"Zhang, A., Wang, H., Guo, P., Fu, Y., Xie, L., Gao, Y., Zhang, S., Feng, J.: VE-KWS: visual modality enhanced end-to-end keyword spotting. In: IEEE International Conference on Acoustics, Speech and Signal Processing ICASSP 2023, Rhodes Island, Greece, June 4-10, 2023, pp. 1\u20135 (2023). https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10096858","DOI":"10.1109\/ICASSP49357.2023.10096858"},{"key":"2091_CR5","doi-asserted-by":"publisher","first-page":"23498","DOI":"10.1109\/ACCESS.2025.3536470","volume":"13","author":"Z Wu","year":"2025","unstructured":"Wu, Z., Chan, S., Wubet, Y.A., Lian, K.: Dligru-x: efficient x-vector-based embeddings for small-footprint keyword spotting system. IEEE Access 13, 23498\u201323507 (2025). https:\/\/doi.org\/10.1109\/ACCESS.2025.3536470","journal-title":"IEEE Access"},{"key":"2091_CR6","doi-asserted-by":"publisher","first-page":"91289","DOI":"10.1109\/ACCESS.2024.3421605","volume":"12","author":"M Pudo","year":"2024","unstructured":"Pudo, M., Wosik, M., Janicki, A.: Improved small-footprint asr-based solution for open vocabulary keyword spotting. IEEE Access 12, 91289\u201391299 (2024). https:\/\/doi.org\/10.1109\/ACCESS.2024.3421605","journal-title":"IEEE Access"},{"key":"2091_CR7","doi-asserted-by":"publisher","first-page":"4909","DOI":"10.1109\/TMM.2022.3183830","volume":"25","author":"D Wang","year":"2023","unstructured":"Wang, D., Liu, S., Wang, Q., Tian, Y., He, L., Gao, X.: Cross-modal enhancement network for multimodal sentiment analysis. IEEE Trans. Multim. 25, 4909\u20134921 (2023). https:\/\/doi.org\/10.1109\/TMM.2022.3183830","journal-title":"IEEE Trans. Multim."},{"issue":"2","key":"2091_CR8","doi-asserted-by":"publisher","first-page":"188","DOI":"10.1007\/S10489-024-06150-1","volume":"55","author":"Z Li","year":"2025","unstructured":"Li, Z., Liu, P., Pan, Y., Yu, J., Liu, W., Chen, H., Luo, Y., Wang, H.: Text-dominant multimodal perception network for sentiment analysis based on cross-modal semantic enhancements. Appl. Intell. 55(2), 188 (2025). https:\/\/doi.org\/10.1007\/S10489-024-06150-1","journal-title":"Appl. Intell."},{"key":"2091_CR9","doi-asserted-by":"publisher","DOI":"10.1016\/J.DSP.2025.105081","volume":"161","author":"Z Sun","year":"2025","unstructured":"Sun, Z., Liu, H., Li, H., Li, Y., Zhang, W.: Averformer: End-to-end audio-visual emotion recognition transformer framework with balanced modal contributions. Digit. Signal Process. 161, 105081 (2025). https:\/\/doi.org\/10.1016\/J.DSP.2025.105081","journal-title":"Digit. Signal Process."},{"key":"2091_CR10","doi-asserted-by":"publisher","unstructured":"Chen, H., Zhou, H., Du, J., Lee, C., Chen, J., Watanabe, S., Siniscalchi, S.M., Scharenborg, O., Liu, D., Yin, B., Pan, J., Gao, J., Liu, C.: The first multimodal information based speech processing (misp) challenge: Data, tasks, baselines and results. In: IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2022, Virtual and Singapore, 23-27 May 2022, pp. 9266\u20139270 (2022). https:\/\/doi.org\/10.1109\/ICASSP43922.2022.9746683","DOI":"10.1109\/ICASSP43922.2022.9746683"},{"key":"2091_CR11","doi-asserted-by":"publisher","unstructured":"Kim, M., Yeo, J.H., Ro, Y.M.: Distinguishing homophenes using multi-head visual-audio memory for lip reading. In: Thirty-Sixth AAAI Conference on Artificial Intelligence, AAAI 2022, Thirty-Fourth Conference on Innovative Applications of Artificial Intelligence, IAAI 2022, The Twelveth Symposium on Educational Advances in Artificial Intelligence, EAAI 2022 Virtual Event, February 22 - March 1, 2022, pp. 1174\u20131182 (2022). https:\/\/doi.org\/10.1609\/AAAI.V36I1.20003","DOI":"10.1609\/AAAI.V36I1.20003"},{"key":"2091_CR12","doi-asserted-by":"publisher","unstructured":"Zhu, Q., Zhang, J., Gu, Y., Hu, Y., Dai, L.: Multichannel av-wav2vec2: A framework for learning multichannel multi-modal speech representation. In: Wooldridge, M.J., Dy, J.G., Natarajan, S. (eds.) Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI 2024, Thirty-Sixth Conference on Innovative Applications of Artificial Intelligence, IAAI 2024, Fourteenth Symposium on Educational Advances in Artificial Intelligence, EAAI 2014, February 20-27, 2024, Vancouver, Canada, pp. 19768\u201319776 (2024). https:\/\/doi.org\/10.1609\/AAAI.V38I17.29951","DOI":"10.1609\/AAAI.V38I17.29951"},{"key":"2091_CR13","doi-asserted-by":"publisher","unstructured":"Wang, H., Guo, P., Zhou, P., Xie, L.: MLCA-AVSR: multi-layer cross attention fusion based audio-visual speech recognition. In: IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2024, Seoul, Republic of Korea, April 14-19, 2024, pp. 8150\u20138154 (2024). https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10446769","DOI":"10.1109\/ICASSP48485.2024.10446769"},{"key":"2091_CR14","doi-asserted-by":"publisher","first-page":"39753","DOI":"10.1109\/ACCESS.2025.3546610","volume":"13","author":"J Zhang","year":"2025","unstructured":"Zhang, J., Yu, Y., Tang, S., Li, W.: Adversarial contrastive autoencoder with shared attention for audio-visual correlation learning. IEEE Access 13, 39753\u201339764 (2025). https:\/\/doi.org\/10.1109\/ACCESS.2025.3546610","journal-title":"IEEE Access"},{"key":"2091_CR15","doi-asserted-by":"publisher","unstructured":"Wang, H., Cheng, M., Fu, Q., Li, M.: Robust wake word spotting with frame-level cross-modal attention based audio-visual conformer. In: IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2024, Seoul, Republic of Korea, April 14-19, 2024, pp. 11556\u201311560 (2024). https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10446074","DOI":"10.1109\/ICASSP48485.2024.10446074"},{"key":"2091_CR16","doi-asserted-by":"publisher","unstructured":"Guo, Y., Sun, S., Ma, S., Zheng, K., Bao, X., Ma, S., Zou, W., Zheng, Y.: Crossmae: Cross-modality masked autoencoders for region-aware audio-visual pre-training. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024, Seattle, WA, USA, June 16-22, 2024, pp. 26711\u201326721 (2024). https:\/\/doi.org\/10.1109\/CVPR52733.2024.02523","DOI":"10.1109\/CVPR52733.2024.02523"},{"key":"2091_CR17","doi-asserted-by":"publisher","unstructured":"Wilkinghoff, K., Cornaggia-Urrigshardt, A.: Tacos: Learning temporally structured embeddings for few-shot keyword spotting with dynamic time warping. In: IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2024, Seoul, Republic of Korea, April 14-19, 2024, pp. 9941\u20139945 (2024). https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10445814","DOI":"10.1109\/ICASSP48485.2024.10445814"},{"key":"2091_CR18","doi-asserted-by":"publisher","DOI":"10.1016\/J.COMPBIOMED.2025.110019","volume":"190","author":"GM Rahmatullah","year":"2025","unstructured":"Rahmatullah, G.M., Ruan, S., Wisesa, I.W.W., Li, L.P.: Enhancing visual speech perception through deep automatic lipreading: a systematic review. Comput. Biol. Med. 190, 110019 (2025). https:\/\/doi.org\/10.1016\/J.COMPBIOMED.2025.110019","journal-title":"Comput. Biol. Med."},{"key":"2091_CR19","doi-asserted-by":"publisher","unstructured":"Wei, K., Li, B., Lv, H., Lu, Q., Jiang, N., Xie, L.: Conversational speech recognition by learning audio-textual cross-modal contextual representation. IEEE ACM Trans. Audio Speech Lang. Process. 32, 2432\u20132444 (2024) https:\/\/doi.org\/10.1109\/TASLP.2024.3389630","DOI":"10.1109\/TASLP.2024.3389630"},{"key":"2091_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/J.INFFUS.2025.103077","volume":"120","author":"R Liu","year":"2025","unstructured":"Liu, R., Yuan, H., Gao, G., Li, H.: Listening and seeing again: Generative error correction for audio-visual speech recognition. Inf. Fusion 120, 103077 (2025). https:\/\/doi.org\/10.1016\/J.INFFUS.2025.103077","journal-title":"Inf. Fusion"},{"key":"2091_CR21","doi-asserted-by":"publisher","unstructured":"Chen, H., Li, W., Cheng, Z., Liang, X., Zhang, Q.: Tcs-lipnet: Temporal & channel & spatial attention-based lip reading network. In: Iliadis, L., Papaleonidas, A., Angelov, P., Jayne, C. (eds.) Artificial Neural Networks and Machine Learning - ICANN 2023 - 32nd International Conference on Artificial Neural Networks, Heraklion, Crete, Greece, September 26-29, 2023, Proceedings, Part IX. Lecture Notes in Computer Science, vol. 14262, pp. 413\u2013424 (2023). https:\/\/doi.org\/10.1007\/978-3-031-44201-8_34","DOI":"10.1007\/978-3-031-44201-8_34"},{"key":"2091_CR22","doi-asserted-by":"publisher","unstructured":"Teo, W.S., Minami, Y.: CIF-RNNT: streaming ASR via acoustic word embeddings with continuous integrate-and-fire and rnn-transducers. In: IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2024, Seoul, Republic of Korea, April 14-19, 2024, pp. 10561\u201310565 (2024). https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10448492","DOI":"10.1109\/ICASSP48485.2024.10448492"},{"key":"2091_CR23","doi-asserted-by":"publisher","unstructured":"Zhao, Y., Xu, R., Song, M.: A cascade sequence-to-sequence model for chinese mandarin lip reading. In: Xu, C., Kankanhalli, M.S., Aizawa, K., Jiang, S., Zimmermann, R., Cheng, W. (eds.) MMAsia \u201919: ACM Multimedia Asia, Beijing, China, December 16-18, 2019, pp. 32\u20131326 (2019). https:\/\/doi.org\/10.1145\/3338533.3366579","DOI":"10.1145\/3338533.3366579"},{"key":"2091_CR24","doi-asserted-by":"publisher","unstructured":"Dibbo, S.V., Moore, J.S., Kenyon, G.T., Teti, M.A.: Lcanets++: Robust audio classification using multi-layer neural networks with lateral competition. In: IEEE International Conference on Acoustics, Speech, and Signal Processing, ICASSP 2024 - Workshops, Seoul, Republic of Korea, April 14-19, 2024, pp. 129\u2013133 (2024). https:\/\/doi.org\/10.1109\/ICASSPW62465.2024.10627668","DOI":"10.1109\/ICASSPW62465.2024.10627668"},{"issue":"1s","key":"2091_CR25","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1145\/3524620","volume":"19","author":"F Xue","year":"2023","unstructured":"Xue, F., Yang, T., Liu, K., Hong, Z., Cao, M., Guo, D., Hong, R.: Lcsnet: End-to-end lipreading with channel-aware feature selection. ACM Trans. Multimed. Comput. Commun. Appl. 19(1s), 28\u201312821 (2023). https:\/\/doi.org\/10.1145\/3524620","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"2091_CR26","doi-asserted-by":"publisher","unstructured":"Chen, W., Tan, X., Xia, Y., Qin, T., Wang, Y., Liu, T.: Duallip: A system for joint lip reading and generation. In: Chen, C.W., Cucchiara, R., Hua, X., Qi, G., Ricci, E., Zhang, Z., Zimmermann, R. (eds.) MM \u201920: The 28th ACM International Conference on Multimedia, Virtual Event \/ Seattle, WA, USA, October 12-16, 2020, pp. 1985\u20131993 (2020). https:\/\/doi.org\/10.1145\/3394171.3413623","DOI":"10.1145\/3394171.3413623"},{"key":"2091_CR27","doi-asserted-by":"publisher","DOI":"10.1016\/J.ESWA.2025.126741","volume":"272","author":"J Zhang","year":"2025","unstructured":"Zhang, J., Mao, T., Guo, L., Li, J., Zhang, L.: Target speaker lipreading by audio-visual self-distillation pretraining and speaker adaptation. Expert Syst. Appl. 272, 126741 (2025). https:\/\/doi.org\/10.1016\/J.ESWA.2025.126741","journal-title":"Expert Syst. Appl."},{"key":"2091_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/J.NEUCOM.2025.129750","volume":"634","author":"M Alsuwat","year":"2025","unstructured":"Alsuwat, M., Al-Shareef, S., Alghamdi, M.: Audio-visual self-supervised representation learning: a survey. Neurocomputing 634, 129750 (2025). https:\/\/doi.org\/10.1016\/J.NEUCOM.2025.129750","journal-title":"Neurocomputing"},{"issue":"12","key":"2091_CR29","doi-asserted-by":"publisher","first-page":"8717","DOI":"10.1109\/TPAMI.2018.2889052","volume":"44","author":"T Afouras","year":"2022","unstructured":"Afouras, T., Chung, J.S., Senior, A.W., Vinyals, O., Zisserman, A.: Deep audio-visual speech recognition. IEEE Trans. Pattern Anal. Mach. Intell. 44(12), 8717\u20138727 (2022). https:\/\/doi.org\/10.1109\/TPAMI.2018.2889052","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2091_CR30","doi-asserted-by":"publisher","unstructured":"Haliassos, A., Zinonos, A., Mira, R., Petridis, S., Pantic, M.: Braven: Improving self-supervised pre-training for visual and auditory speech recognition. In: IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2024, Seoul, Republic of Korea, April 14-19, 2024, pp. 11431\u201311435 (2024). https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10448473","DOI":"10.1109\/ICASSP48485.2024.10448473"},{"key":"2091_CR31","doi-asserted-by":"publisher","unstructured":"Wan, G., Ye, Z.: Multi-modal knowledge transfer for target speaker lipreading with improved audio-visual pretraining and cross-lingual fine-tuning. In: IEEE International Conference on Multimedia and Expo, ICME 2024 - Workshops, Niagara Falls, ON, Canada, July 15-19, 2024, pp. 1\u20136 (2024). https:\/\/doi.org\/10.1109\/ICMEW63481.2024.10645443","DOI":"10.1109\/ICMEW63481.2024.10645443"},{"key":"2091_CR32","doi-asserted-by":"publisher","unstructured":"Huang, Y., Liang, X., Fang, C.: Callip: Lipreading using contrastive and attribute learning. In: Shen, H.T., Zhuang, Y., Smith, J.R., Yang, Y., C\u00e9sar, P., Metze, F., Prabhakaran, B. (eds.) MM \u201921: ACM Multimedia Conference, Virtual Event, China, October 20 - 24, 2021, pp. 2492\u20132500 (2021). https:\/\/doi.org\/10.1145\/3474085.3475420","DOI":"10.1145\/3474085.3475420"},{"issue":"3","key":"2091_CR33","doi-asserted-by":"publisher","first-page":"2398","DOI":"10.1109\/TCSVT.2024.3486344","volume":"35","author":"T Chen","year":"2025","unstructured":"Chen, T., Tan, Z., Gong, T., Chu, Q., Wu, Y., Liu, B., Yu, N., Lu, L., Ye, J.: Bootstrapping audio-visual video segmentation by strengthening audio cues. IEEE Trans. Circuits Syst. Video Technol. 35(3), 2398\u20132409 (2025). https:\/\/doi.org\/10.1109\/TCSVT.2024.3486344","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"2091_CR34","doi-asserted-by":"publisher","first-page":"11623","DOI":"10.1109\/ACCESS.2024.3470521","volume":"13","author":"AWA Ameer","year":"2025","unstructured":"Ameer, A.W.A., Salehpour, P., Asadpour, M.: Deep transfer learning for lip reading based on nasnetmobile pretrained model in wild dataset. IEEE Access 13, 11623\u201311638 (2025). https:\/\/doi.org\/10.1109\/ACCESS.2024.3470521","journal-title":"IEEE Access"},{"key":"2091_CR35","doi-asserted-by":"publisher","first-page":"20871","DOI":"10.1109\/ACCESS.2025.3532397","volume":"13","author":"A Albladi","year":"2025","unstructured":"Albladi, A., Islam, M., Das, A., Bigonah, M., Zhang, Z., Jamshidi, F., Rahgouy, M., Raychawdhary, N., Marghitu, D., Seals, C.D.: Hate speech detection using large language models: a comprehensive review. IEEE Access 13, 20871\u201320892 (2025). https:\/\/doi.org\/10.1109\/ACCESS.2025.3532397","journal-title":"IEEE Access"},{"key":"2091_CR36","doi-asserted-by":"publisher","first-page":"8482","DOI":"10.1109\/ACCESS.2025.3528054","volume":"13","author":"S Lee","year":"2025","unstructured":"Lee, S., Kim, D.: Multi-view prototypical transport for unsupervised domain adaptation. IEEE Access 13, 8482\u20138494 (2025). https:\/\/doi.org\/10.1109\/ACCESS.2025.3528054","journal-title":"IEEE Access"},{"key":"2091_CR37","doi-asserted-by":"publisher","DOI":"10.1016\/J.NEUCOM.2025.129750","volume":"634","author":"M Alsuwat","year":"2025","unstructured":"Alsuwat, M., Al-Shareef, S., Alghamdi, M.: Audio-visual self-supervised representation learning: a survey. Neurocomputing 634, 129750 (2025). https:\/\/doi.org\/10.1016\/J.NEUCOM.2025.129750","journal-title":"Neurocomputing"},{"issue":"1","key":"2091_CR38","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1049\/CIT2.12212","volume":"9","author":"Y Li","year":"2024","unstructured":"Li, Y., Ren, J., Wang, Y., Wang, G., Li, X., Liu, H.: Audio-visual keyword transformer for unconstrained sentence-level keyword spotting. CAAI Trans. Intell. Technol. 9(1), 142\u2013152 (2024). https:\/\/doi.org\/10.1049\/CIT2.12212","journal-title":"CAAI Trans. Intell. Technol."},{"key":"2091_CR39","doi-asserted-by":"publisher","unstructured":"Li, C., Su, F., Liu, J.: Audio-visual wake-up word spotting under noisy and multi-person scenarios. In: Antonacopoulos, A., Chaudhuri, S., Chellappa, R., Liu, C., Bhattacharya, S., Pal, U. (eds.) Pattern Recognition - 27th International Conference, ICPR 2024, Kolkata, India, December 1-5, 2024, Proceedings, Part XXXIII. Lecture Notes in Computer Science, vol. 15333, pp. 170\u2013184 (2024). https:\/\/doi.org\/10.1007\/978-3-031-80136-5_12","DOI":"10.1007\/978-3-031-80136-5_12"},{"key":"2091_CR40","doi-asserted-by":"publisher","DOI":"10.1016\/J.SPECOM.2023.103019","volume":"156","author":"F Ma","year":"2024","unstructured":"Ma, F., Wang, C., Li, X., Zeng, Z.: Selective transfer subspace learning for small-footprint end-to-end cross-domain keyword spotting. Speech Commun. 156, 103019 (2024). https:\/\/doi.org\/10.1016\/J.SPECOM.2023.103019","journal-title":"Speech Commun."},{"key":"2091_CR41","doi-asserted-by":"publisher","unstructured":"Ozay, M.: Joint embedding learning and latent subspace probing for cross-domain few-shot keyword spotting. In: IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2024, Seoul, Republic of Korea, April 14-19, 2024, pp. 6425\u20136429 (2024). https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10446764","DOI":"10.1109\/ICASSP48485.2024.10446764"},{"key":"2091_CR42","doi-asserted-by":"publisher","unstructured":"Chen, Q., Chang, Y., Kim, K., Gao, C., Liu, S.: An area-efficient ultra-low-power time-domain feature extractor for edge keyword spotting. In: IEEE International Symposium on Circuits and Systems, ISCAS 2023, Monterey, CA, USA, May 21-25, 2023, pp. 1\u20135 (2023). https:\/\/doi.org\/10.1109\/ISCAS46773.2023.10181602","DOI":"10.1109\/ISCAS46773.2023.10181602"},{"issue":"1","key":"2091_CR43","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1186\/S13636-023-00293-8","volume":"2023","author":"X Liang","year":"2023","unstructured":"Liang, X., Zhang, Z., Xu, R.: Multi-task deep cross-attention networks for far-field speaker verification and keyword spotting. EURASIP J. Audio Speech Music. Process. 2023(1), 28 (2023). https:\/\/doi.org\/10.1186\/S13636-023-00293-8","journal-title":"EURASIP J. Audio Speech Music. Process."},{"key":"2091_CR44","doi-asserted-by":"publisher","unstructured":"Li, X., Liu, R., Huang, H., Wu, Q.: Contrastive learning for target speaker extraction with attention-based fusion. IEEE ACM Trans. Audio Speech Lang. Process. 32, 178\u2013188 (2024) https:\/\/doi.org\/10.1109\/TASLP.2023.3324550","DOI":"10.1109\/TASLP.2023.3324550"},{"key":"2091_CR45","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TIM.2025.3551848","volume":"74","author":"J Gao","year":"2025","unstructured":"Gao, J., Zhang, W., Hu, D., Pan, M.: Speech enhancement for VHF communication audio empowered by feature consistency and contrastive learning. IEEE Trans. Instrum. Meas. 74, 1\u201311 (2025). https:\/\/doi.org\/10.1109\/TIM.2025.3551848","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"2091_CR46","doi-asserted-by":"publisher","unstructured":"Ghosh, S., Kumar, S., Seth, A., Chiniya, P., Tyagi, U., Duraiswami, R., Manocha, D.: Lipger: Visually-conditioned generative error correction for robust automatic speech recognition. CoRR abs\/2406.04432 (2024) https:\/\/doi.org\/10.48550\/ARXIV.2406.04432arXiV:2406.04432","DOI":"10.48550\/ARXIV.2406.04432"},{"key":"2091_CR47","doi-asserted-by":"publisher","DOI":"10.1016\/J.KNOSYS.2024.112735","volume":"307","author":"J Wu","year":"2025","unstructured":"Wu, J., Wu, C., Shen, X., Wang, L.: Adaptive cross-modal experts network with uncertainty-driven fusion for vision-language navigation. Knowl. Based Syst. 307, 112735 (2025). https:\/\/doi.org\/10.1016\/J.KNOSYS.2024.112735","journal-title":"Knowl. Based Syst."},{"key":"2091_CR48","unstructured":"Mu, Z., Yang, X.: Separate in the speech chain: Cross-modal conditional audio-visual target speech extraction. In: Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, IJCAI 2024, Jeju, South Korea, August 3-9, 2024, pp. 6415\u20136423 (2024). https:\/\/www.ijcai.org\/proceedings\/2024\/709"},{"issue":"5","key":"2091_CR49","doi-asserted-by":"publisher","first-page":"2779","DOI":"10.1007\/S11042-024-20450-1","volume":"84","author":"MS Hosen","year":"2025","unstructured":"Hosen, M.S., Basir, S., Khan, M.F., Asaduzzaman, A.O.M., Islam, M.M., Islam, M.S.: Supervised single-channel dual domains speech enhancement technique using bidirectional long short-term memory. Multim. Tools Appl. 84(5), 2779\u20132803 (2025). https:\/\/doi.org\/10.1007\/S11042-024-20450-1","journal-title":"Multim. Tools Appl."},{"key":"2091_CR50","doi-asserted-by":"publisher","unstructured":"Xin, Y., Yang, D., Zou, Y.: Audio pyramid transformer with domain adaption for weakly supervised sound event detection and audio classification. In: Ko, H., Hansen, J.H.L. (eds.) 23rd Annual Conference of the International Speech Communication Association, Interspeech 2022, Incheon, Korea, September 18-22, 2022, pp. 1546\u20131550 (2022). https:\/\/doi.org\/10.21437\/INTERSPEECH.2022-10057","DOI":"10.21437\/INTERSPEECH.2022-10057"},{"key":"2091_CR51","doi-asserted-by":"publisher","unstructured":"Jung, Y., Lee, J., Lee, S., Jung, M., Lee, Y., Cho, H.: Text-aware adapter for few-shot keyword spotting. In: 2025 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2025, Hyderabad, India, April 6-11, 2025, pp. 1\u20135. IEEE (2025). https:\/\/doi.org\/10.1109\/ICASSP49660.2025.10890609","DOI":"10.1109\/ICASSP49660.2025.10890609"},{"key":"2091_CR52","doi-asserted-by":"publisher","unstructured":"Yang, M., He, Q., Huang, J., Chen, Y., Liu, Z., Li, Y.: Cross-domain few-shot open-set keyword spotting using keyword adaptation and prototype reprojection. In: 2025 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2025, Hyderabad, India, April 6-11, 2025, pp. 1\u20135. IEEE (2025). https:\/\/doi.org\/10.1109\/ICASSP49660.2025.10888051","DOI":"10.1109\/ICASSP49660.2025.10888051"},{"key":"2091_CR53","doi-asserted-by":"publisher","unstructured":"Xi, Y., Li, H., Li, H., Guo, J., Li, X., Ding, W., Yu, K.: NTC-KWS: noise-aware CTC for robust keyword spotting. In: 2025 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2025, Hyderabad, India, April 6-11, 2025, pp. 1\u20135. IEEE (2025). https:\/\/doi.org\/10.1109\/ICASSP49660.2025.10889332","DOI":"10.1109\/ICASSP49660.2025.10889332"},{"key":"2091_CR54","doi-asserted-by":"publisher","unstructured":"Ai, Z., Chen, Z., Xu, S.: MM-KWS: Multi-modal Prompts for Multilingual User-defined Keyword Spotting. In: Interspeech 2024, pp. 2415\u20132419 (2024). https:\/\/doi.org\/10.21437\/Interspeech.2024-10","DOI":"10.21437\/Interspeech.2024-10"},{"issue":"1","key":"2091_CR55","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1049\/cit2.12212","volume":"9","author":"Y Li","year":"2024","unstructured":"Li, Y., Ren, J., Wang, Y., Wang, G., Li, X., Liu, H.: Audio-visual keyword transformer for unconstrained sentence-level keyword spotting. CAAI Trans. Intell. Technol. 9(1), 142\u2013152 (2024). https:\/\/doi.org\/10.1049\/cit2.12212","journal-title":"CAAI Trans. Intell. Technol."},{"key":"2091_CR56","doi-asserted-by":"publisher","unstructured":"Valentini-Botinhao, C., Aldana\u00a0Blanco, A.L., Klejch, O., Bell, P.: Efficient intelligibility evaluation using keyword spotting: A study on audio-visual speech enhancement. In: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023). https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10096479","DOI":"10.1109\/ICASSP49357.2023.10096479"},{"key":"2091_CR57","doi-asserted-by":"publisher","unstructured":"Zhang, A., Wang, H., Guo, P., Fu, Y., Xie, L., Gao, Y., Zhang, S., Feng, J.: Ve-kws: Visual modality enhanced end-to-end keyword spotting. In: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023). https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10096858","DOI":"10.1109\/ICASSP49357.2023.10096858"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02091-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-02091-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02091-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T04:17:43Z","timestamp":1770783463000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-02091-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,4]]},"references-count":57,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["2091"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-02091-y","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,4]]},"assertion":[{"value":"10 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"26"}}