{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,12]],"date-time":"2026-03-12T05:05:33Z","timestamp":1773291933727,"version":"3.50.1"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T00:00:00Z","timestamp":1773187200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T00:00:00Z","timestamp":1773187200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62373084"],"award-info":[{"award-number":["62373084"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-026-08307-w","type":"journal-article","created":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T12:41:27Z","timestamp":1773232887000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Accurate target speaker extraction method with adaptive information interaction and updating in complex scenarios"],"prefix":"10.1007","volume":"82","author":[{"given":"Yangjie","family":"Wei","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ben","family":"Niu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuqiao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoli","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,11]]},"reference":[{"issue":"10","key":"8307_CR1","doi-asserted-by":"publisher","first-page":"8193","DOI":"10.1007\/s11227-019-02785-x","volume":"76","author":"H-G Kim","year":"2020","unstructured":"Kim H-G, Jang G-J, Oh Y-H, Choi H-J (2020) Speech and music pitch trajectory classification using recurrent neural networks for monaural speech segregation. J Supercomput 76(10):8193","journal-title":"J Supercomput"},{"key":"8307_CR2","doi-asserted-by":"publisher","DOI":"10.7717\/peerj-cs.3326","volume":"11","author":"X Ye","year":"2025","unstructured":"Ye X, Lin L, Jiang G, Yuan Z (2025) Addressing human speech characteristics in single-channel speaker extraction networks. PeerJ Comput Sci 11:e3326","journal-title":"PeerJ Comput Sci"},{"issue":"6","key":"8307_CR3","doi-asserted-by":"publisher","first-page":"8751","DOI":"10.1007\/s11227-021-04251-z","volume":"78","author":"S Cai","year":"2022","unstructured":"Cai S, Han D, Li D, Zheng Z, Crespi N (2022) An reinforcement learning-based speech censorship chatbot system. J Supercomput 78(6):8751\u20138773","journal-title":"J Supercomput"},{"key":"8307_CR4","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2025.103321","volume":"175","author":"M Zhang","year":"2025","unstructured":"Zhang M, Jia X, Guo Y (2025) Dynamic graph learning with gated convolutions for single-channel speech separation. Speech Commun 175:103321","journal-title":"Speech Commun"},{"key":"8307_CR5","doi-asserted-by":"publisher","first-page":"2015","DOI":"10.1109\/LSP.2025.3560237","volume":"32","author":"D Liu","year":"2025","unstructured":"Liu D, Zhang T, Wei Y, Yi C, Christensen MG (2025) Speech conv-mamba: selective structured state space model with temporal dilated convolution for efficient speech separation. IEEE Signal Process Lett 32:2015","journal-title":"IEEE Signal Process Lett"},{"key":"8307_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102550","volume":"112","author":"F Hao","year":"2024","unstructured":"Hao F, Li X, Zheng C (2024) X-tf-gridnet: a time-frequency domain target speaker extraction network with adaptive speaker embedding fusion. Inf Fusion 112:102550","journal-title":"Inf Fusion"},{"key":"8307_CR7","doi-asserted-by":"crossref","unstructured":"Liu X, Li X, Serr\u00e0 J (2023) Quantitative evidence on overlooked aspects of enrollment speaker embeddings for target speaker separation. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 1\u20135","DOI":"10.1109\/ICASSP49357.2023.10096478"},{"issue":"2","key":"8307_CR8","doi-asserted-by":"publisher","first-page":"307","DOI":"10.3390\/electronics13020307","volume":"13","author":"J Wang","year":"2024","unstructured":"Wang J, Lai Y, Tai T, Le PT, Pham T, Wang Z, Li Y, Wang J, Chang P (2024) Target speaker extraction using attention-enhanced temporal convolutional network. Electronics 13(2):307","journal-title":"Electronics"},{"key":"8307_CR9","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2025.107388","volume":"188","author":"S He","year":"2025","unstructured":"He S, Xue W, Yang Y, Zhang H, Pan J, Zhang X (2025) Enhancing target speaker extraction with hierarchical speaker representation learning. Neural Netw 188:107388","journal-title":"Neural Netw"},{"key":"8307_CR10","doi-asserted-by":"publisher","first-page":"646","DOI":"10.1109\/JSTSP.2025.3560513","volume":"19","author":"W Wu","year":"2025","unstructured":"Wu W, Chen X, Wang S, Wang J, Meng L, Wu X, Meng H, Li H (2025) Av-tse: context and confidence-aware audio visual target speaker extraction. IEEE J Sel Top Signal Process 19:646","journal-title":"IEEE J Sel Top Signal Process"},{"key":"8307_CR11","doi-asserted-by":"publisher","first-page":"797","DOI":"10.1109\/TASLPRO.2025.3527766","volume":"33","author":"R Tao","year":"2025","unstructured":"Tao R, Qian X, Jiang Y, Li J, Wang J, Li H (2025) Audio-visual target speaker extraction with selective auditory attention. IEEE Trans Audio Speech Lang Process 33:797","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"8307_CR12","doi-asserted-by":"publisher","first-page":"3167","DOI":"10.1109\/LSP.2025.3591408","volume":"32","author":"X Zhu","year":"2025","unstructured":"Zhu X, Qian X, Liang D (2025) Ssdq: target speaker extraction via semantic and spatial dual querying. IEEE Signal Process Lett 32:3167","journal-title":"IEEE Signal Process Lett"},{"key":"8307_CR13","doi-asserted-by":"publisher","first-page":"3350","DOI":"10.1109\/LSP.2025.3600168","volume":"32","author":"S Zhang","year":"2025","unstructured":"Zhang S, Zhang J, Wang Y, Yan H (2025) Doa or speaker embedding: which is better for multi-microphone target speaker extraction. IEEE Signal Process Lett 32:3350","journal-title":"IEEE Signal Process Lett"},{"key":"8307_CR14","doi-asserted-by":"crossref","unstructured":"King B, Chen IF, Vaizman Y, Liu Y, Maas R, Parthasarathi SHK, Hoffmeister B (2017) Robust speech recognition via anchor word representations. In: Interspeech 2017","DOI":"10.21437\/Interspeech.2017-1570"},{"key":"8307_CR15","doi-asserted-by":"publisher","unstructured":"Delcroix M, Zmolikova K, Kinoshita K, Ogawa A, Nakatani T (2018) Single channel target speaker extraction and recognition with speaker beam. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), vol 2018. IEEE, pp 5554\u20135558. https:\/\/doi.org\/10.1109\/ICASSP.2018.8462661","DOI":"10.1109\/ICASSP.2018.8462661"},{"key":"8307_CR16","doi-asserted-by":"crossref","unstructured":"Wang Q, Muckenhirn H, Wilson K, Sridhar P, Wu Z, Hershey J, Saurous RA, Weiss RJ, Jia Y, Moreno IL (2018) Voicefilter: targeted voice separation by speaker-conditioned spectrogram masking. arXiv preprint arXiv:1810.04826","DOI":"10.21437\/Interspeech.2019-1101"},{"issue":"14","key":"8307_CR17","doi-asserted-by":"publisher","first-page":"816","DOI":"10.1049\/el.2019.1228","volume":"55","author":"W Li","year":"2019","unstructured":"Li W, Zhang P, Yan Y (2019) Tenet: target speaker extraction network with accumulated speaker embedding for automatic speech recognition. Electron Lett 55(14):816\u2013819","journal-title":"Electron Lett"},{"key":"8307_CR18","doi-asserted-by":"publisher","first-page":"1370","DOI":"10.1109\/TASLP.2020.2987429","volume":"28","author":"C Xu","year":"2020","unstructured":"Xu C, Rao W, Chng ES, Li H (2020) Spex: multi-scale time domain speaker extraction network. IEEE\/ACM Trans Audio Speech Language Process 28:1370\u20131384","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"key":"8307_CR19","doi-asserted-by":"publisher","unstructured":"He S, Li H, Zhang X (2020) Speakerfilter: deep learning-based target speaker extraction using anchor speech. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 376\u2013380. https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9054222","DOI":"10.1109\/ICASSP40776.2020.9054222"},{"issue":"2","key":"8307_CR20","doi-asserted-by":"publisher","first-page":"1153","DOI":"10.1007\/s40747-021-00565-w","volume":"8","author":"A Mittal","year":"2022","unstructured":"Mittal A, Dua M (2022) Static-dynamic features and hybrid deep learning models based spoof detection system for ASV. Complex Intell Syst 8(2):1153\u20131166","journal-title":"Complex Intell Syst"},{"key":"8307_CR21","doi-asserted-by":"publisher","first-page":"3830","DOI":"10.21437\/Interspeech.2020-2650","volume":"2020","author":"B Desplanques","year":"2020","unstructured":"Desplanques B, Thienpondt J, Demuynck K (2020) Ecapa-tdnn Emphasized channel attention, propagation and aggregation in tdnn based speaker verification. Interspeech 2020:3830\u20133834. https:\/\/doi.org\/10.21437\/Interspeech.2020-2650","journal-title":"Interspeech"},{"key":"8307_CR22","doi-asserted-by":"publisher","first-page":"471","DOI":"10.1109\/LSP.2024.3358754","volume":"31","author":"J Li","year":"2024","unstructured":"Li J, Duan Z, Li S, Yu X, Yang G (2024) Esaformer: enhanced self-attention for automatic speech recognition. IEEE Signal Process Lett 31:471\u2013475","journal-title":"IEEE Signal Process Lett"},{"issue":"8","key":"8307_CR23","doi-asserted-by":"publisher","first-page":"3471","DOI":"10.3390\/app14083471","volume":"14","author":"M Gao","year":"2024","unstructured":"Gao M, Zhang X (2024) Improved convolutional neural network-time-delay neural network structure with repeated feature fusions for speaker verification. Appl Sci 14(8):3471","journal-title":"Appl Sci"},{"key":"8307_CR24","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2024.111735","volume":"161","author":"N Saleem","year":"2024","unstructured":"Saleem N, Elmannai H, Bourouis S, Trigui A (2024) Squeeze-and-excitation 3D convolutional attention recurrent network for end-to-end speech emotion recognition. Appl Soft Comput 161:111735","journal-title":"Appl Soft Comput"},{"key":"8307_CR25","first-page":"1060","volume":"2023","author":"Y Wang","year":"2023","unstructured":"Wang Y, Zhang X (2023) MFT-CRN: multi-scale Fourier transform for monaural speech enhancement. Interspeech 2023:1060\u20131064","journal-title":"Interspeech"},{"issue":"5","key":"8307_CR26","doi-asserted-by":"publisher","first-page":"4237","DOI":"10.1007\/s40747-022-00713-w","volume":"8","author":"M Swain","year":"2022","unstructured":"Swain M, Maji B, Kabisatpathy P, Routray A (2022) A DCRNN-based ensemble classifier for speech emotion recognition in Odia language. Complex Intell Syst 8(5):4237\u20134249","journal-title":"Complex Intell Syst"},{"issue":"7","key":"8307_CR27","doi-asserted-by":"publisher","first-page":"4588","DOI":"10.1007\/s00034-024-02677-3","volume":"43","author":"C Lan","year":"2024","unstructured":"Lan C, Chen H, Zhang L, Zhao S, Guo R, Fan Z (2024) Research on speech enhancement algorithm by fusing improved EMD and GCRN networks. Circuits Syst Signal Process 43(7):4588\u20134604","journal-title":"Circuits Syst Signal Process"},{"key":"8307_CR28","doi-asserted-by":"publisher","unstructured":"He S, Li H, Zhang X (2022) Speakerfilter-pro: an improved target speaker extractor combines the time domain and frequency domain. In: 2022 13th International Symposium on Chinese Spoken Language Processing (ISCSLP), pp 473\u2013477. https:\/\/doi.org\/10.1109\/ISCSLP57327.2022.10037794","DOI":"10.1109\/ISCSLP57327.2022.10037794"},{"key":"8307_CR29","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2024.120131","volume":"660","author":"H Shen","year":"2024","unstructured":"Shen H, Wang Z, Zhang J, Zhang M (2024) L-net: a lightweight convolutional neural network for devices with low computing power. Inf Sci 660:120131","journal-title":"Inf Sci"},{"key":"8307_CR30","first-page":"66","volume":"9","author":"Z Ye","year":"2023","unstructured":"Ye Z, Saleem N, Ali H et al (2023) Efficient gated convolutional recurrent neural networks for real-time speech enhancement. Int J Interact Multimed Artif Intell 9:66","journal-title":"Int J Interact Multimed Artif Intell"},{"key":"8307_CR31","doi-asserted-by":"crossref","unstructured":"Mun S, Choe S, Huh J, Chung JS (2020) The sound of my voice: speaker representation loss for target voice separation. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp 7289\u20137293","DOI":"10.1109\/ICASSP40776.2020.9053521"},{"key":"8307_CR32","doi-asserted-by":"publisher","unstructured":"Luo Y, Mesgarani N (2018) Tasnet: time-domain audio separation network for real-time, single-channel speech separation. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp 696\u2013700. https:\/\/doi.org\/10.1109\/ICASSP.2018.8462116","DOI":"10.1109\/ICASSP.2018.8462116"},{"key":"8307_CR33","doi-asserted-by":"publisher","first-page":"2598","DOI":"10.1109\/TASLP.2020.3016498","volume":"28","author":"S Wang","year":"2020","unstructured":"Wang S, Yang Y, Wu Z, Qian Y, Yu K (2020) Data augmentation using deep generative models for embedding based speaker recognition. IEEE\/ACM Trans Audio Speech Lang Process 28:2598\u20132609. https:\/\/doi.org\/10.1109\/TASLP.2020.3016498","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"8307_CR34","unstructured":"Snyder D, Chen G, Povey D (2015) Musan: a music, speech, and noise corpus. arXiv preprint arXiv:1510.08484"},{"key":"8307_CR35","doi-asserted-by":"publisher","unstructured":"Ko T, Peddinti V, Povey D, Seltzer ML, Khudanpur S, A study on data augmentation of reverberant speech for robust speech recognition, in, (2017) IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE 2017:5220\u20135224. https:\/\/doi.org\/10.1109\/ICASSP.2017.7953152","DOI":"10.1109\/ICASSP.2017.7953152"},{"key":"8307_CR36","doi-asserted-by":"publisher","unstructured":"Hoffer E, Ben\u00a0Nun T, Hubara I, Giladi N, Hoefler T, Soudry D (2020) Augment your batch: improving generalization through instance repetition. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 8126\u20138135. https:\/\/doi.org\/10.1109\/CVPR42600.2020.00815","DOI":"10.1109\/CVPR42600.2020.00815"},{"key":"8307_CR37","doi-asserted-by":"crossref","unstructured":"Kamo N, Delcroix M, Nakatani T (2023) Target speech extraction with conditional diffusion model. arXiv preprint arXiv:2308.03987","DOI":"10.21437\/Interspeech.2023-1234"},{"key":"8307_CR38","doi-asserted-by":"crossref","unstructured":"Zhao C, He S, Zhang X (2024) Sicrn: advancing speech enhancement through state space model and inplace convolution techniques. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp 10506\u201310510","DOI":"10.1109\/ICASSP48485.2024.10446396"},{"key":"8307_CR39","doi-asserted-by":"publisher","first-page":"2110","DOI":"10.1109\/TASLPRO.2025.3572756","volume":"33","author":"B Zeng","year":"2025","unstructured":"Zeng B, Li M (2025) Usef-tse: universal speaker embedding free target speaker extraction. IEEE Trans Audio Speech Lang Process 33:2110","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"8307_CR40","doi-asserted-by":"publisher","unstructured":"Panayotov V, Chen G, Povey D, Khudanpur S (2015) Librispeech: an ASR corpus based on public domain audio books. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), vol 2015. IEEE, pp 5206\u20135210. https:\/\/doi.org\/10.1109\/ICASSP.2015.7178964","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"8307_CR41","unstructured":"Paszke A, Gross S, Massa F, Lerer A, Bradbury J, Chanan G, Killeen T, Lin Z, Gimelshein N, Antiga L et al (2019) Pytorch: an imperative style, high-performance deep learning library. In: Advances in Neural Information Processing Systems, vol 32"},{"issue":"4","key":"8307_CR42","doi-asserted-by":"publisher","first-page":"1462","DOI":"10.1109\/TSA.2005.858005","volume":"14","author":"E Vincent","year":"2006","unstructured":"Vincent E, Gribonval R, F\u00e9votte C (2006) Performance measurement in blind audio source separation. IEEE\/ACM Trans Audio Speech Lang Process 14(4):1462\u20131469. https:\/\/doi.org\/10.1109\/TSA.2005.858005","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"issue":"7","key":"8307_CR43","doi-asserted-by":"publisher","first-page":"2125","DOI":"10.1109\/TASL.2011.2114881","volume":"19","author":"CH Taal","year":"2011","unstructured":"Taal CH, Hendriks RC, Heusdens R, Jensen J (2011) An algorithm for intelligibility prediction of time-frequency weighted noisy speech. IEEE\/ACM Trans Audio Speech Lang Process 19(7):2125\u20132136. https:\/\/doi.org\/10.1109\/TASL.2011.2114881","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"issue":"11","key":"8307_CR44","doi-asserted-by":"publisher","first-page":"2009","DOI":"10.1109\/TASLP.2016.2585878","volume":"24","author":"J Jensen","year":"2016","unstructured":"Jensen J, Taal CH (2016) An algorithm for predicting the intelligibility of speech masked by modulated noise maskers. IEEE\/ACM Trans Audio Speech Lang Process 24(11):2009\u20132022. https:\/\/doi.org\/10.1109\/TASLP.2016.2585878","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"issue":"10","key":"8307_CR45","first-page":"755","volume":"50","author":"AW Rix","year":"2002","unstructured":"Rix AW, Hollier MP, Hekstra AP, Beerends JG (2002) Perceptual evaluation of speech quality (PESQ) the new itu standard for end-to-end speech quality assessment part i-time-delay compensation. J Audio Eng Soc 50(10):755\u2013764","journal-title":"J Audio Eng Soc"},{"key":"8307_CR46","doi-asserted-by":"publisher","unstructured":"Snyder D, Garcia-Romero D, Sell G, Povey D, Khudanpur S (2018) X-vectors: robust DNN embeddings for speaker recognition. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 2018. IEEE, pp 5329\u20135333. https:\/\/doi.org\/10.1109\/ICASSP.2018.8461375","DOI":"10.1109\/ICASSP.2018.8461375"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08307-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-026-08307-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08307-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T12:41:32Z","timestamp":1773232892000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-026-08307-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,11]]},"references-count":46,"journal-issue":{"issue":"4","published-online":{"date-parts":[[2026,3]]}},"alternative-id":["8307"],"URL":"https:\/\/doi.org\/10.1007\/s11227-026-08307-w","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,11]]},"assertion":[{"value":"24 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 January 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"241"}}