{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T08:16:00Z","timestamp":1783671360122,"version":"3.55.0"},"reference-count":60,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2027,2]]},"DOI":"10.1016\/j.csl.2026.102022","type":"journal-article","created":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T06:52:44Z","timestamp":1783061564000},"page":"102022","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Selective teaching yourself: An efficient self-learning approach for speech separation"],"prefix":"10.1016","volume":"102","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-7490-3926","authenticated-orcid":false,"given":"Ha Minh","family":"Tan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3807-0464","authenticated-orcid":false,"given":"Diem Thi","family":"Tran","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.102022_b1","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"690","article-title":"Teacher-student deep clustering for low-delay single channel speech separation","author":"Aihara","year":"2019"},{"key":"10.1016\/j.csl.2026.102022_b2","series-title":"INTERSPEECH","article-title":"Dual-path transformer network: Direct context-aware modeling for end-to-end monaural speech separation","author":"jing Chen","year":"2020"},{"key":"10.1016\/j.csl.2026.102022_b3","series-title":"AAAI","first-page":"3430","article-title":"Online knowledge distillation with diverse peers","volume":"Vol. 34","author":"Chen","year":"2020"},{"key":"10.1016\/j.csl.2026.102022_b4","article-title":"Ultra fast speech separation model with teacher student learning","author":"Chen","year":"2022","journal-title":"Interspeech"},{"key":"10.1016\/j.csl.2026.102022_b5","series-title":"INTERSPEECH","first-page":"926","article-title":"Cross-layer similarity knowledge distillation for speech enhancement","author":"Cheng","year":"2022"},{"key":"10.1016\/j.csl.2026.102022_b6","series-title":"CVPR","first-page":"7882","article-title":"Mars: Motion-augmented rgb stream for action recognition","author":"Crasto","year":"2019"},{"key":"10.1016\/j.csl.2026.102022_b7","doi-asserted-by":"crossref","unstructured":"Deng, C., Ma, S., Sha, Y., Zhang, Y., Zhang, H., Song, H., Wang, F., 2021. Robust Speaker Extraction Network Based on Iterative Refined Adaptation. In: Proc. Interspeech 2021. pp. 3530\u20133534.","DOI":"10.21437\/Interspeech.2021-2250"},{"key":"10.1016\/j.csl.2026.102022_b8","first-page":"26","article-title":"Learning with learned loss function: Speech enhancement with quality-net to improve perceptual evaluation of speech quality","volume":"27","author":"Fu","year":"2019","journal-title":"SPL"},{"key":"10.1016\/j.csl.2026.102022_b9","series-title":"ICML","first-page":"2031","article-title":"Metricgan: Generative adversarial networks based black-box metric scores optimization for speech enhancement","author":"Fu","year":"2019"},{"key":"10.1016\/j.csl.2026.102022_b10","series-title":"CVPR","first-page":"10478","article-title":"Music gesture for visual sound separation","author":"Gan","year":"2020"},{"key":"10.1016\/j.csl.2026.102022_b11","series-title":"Conference on Computer Vision and Pattern Recognition","first-page":"15490","article-title":"Visualvoice: Audio-visual speech separation with cross-modal consistency","author":"Gao","year":"2021"},{"key":"10.1016\/j.csl.2026.102022_b12","article-title":"Continuous speech recognition (CSR-I) wall street journal (WSJ0) news, complete","author":"Garofalo","year":"1993","journal-title":"Linguist. Data Consort."},{"key":"10.1016\/j.csl.2026.102022_b13","series-title":"Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics","first-page":"315","article-title":"Deep sparse rectifier neural networks","author":"Glorot","year":"2011"},{"key":"10.1016\/j.csl.2026.102022_b14","doi-asserted-by":"crossref","DOI":"10.1007\/s11263-021-01453-z","article-title":"Knowledge distillation: A survey","volume":"129","author":"Gou","year":"2021","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.csl.2026.102022_b15","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"31","article-title":"Deep clustering: Discriminative embeddings for segmentation and separation","author":"Hershey","year":"2016"},{"key":"10.1016\/j.csl.2026.102022_b16","article-title":"Distilling the knowledge in a neural network","author":"Hinton","year":"2015","journal-title":"Conf. Neural Inf. Process. Syst. (NeurIPS)"},{"key":"10.1016\/j.csl.2026.102022_b17","series-title":"Interspeech","article-title":"Single-channel multi-speaker separation using deep clustering","author":"Isik","year":"2016"},{"key":"10.1016\/j.csl.2026.102022_b18","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"1396","article-title":"Improving target sound extraction with timestamp knowledge distillation","author":"Kim","year":"2024"},{"key":"10.1016\/j.csl.2026.102022_b19","series-title":"ICLR (Poster)","article-title":"Adam: A method for Stochastic optimization","author":"Kingma","year":"2015"},{"key":"10.1016\/j.csl.2026.102022_b20","doi-asserted-by":"crossref","first-page":"825","DOI":"10.1109\/TASLP.2020.2968738","article-title":"On loss functions for supervised monaural time-domain speech enhancement","volume":"28","author":"Kolb\u00e6k","year":"2020","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102022_b21","doi-asserted-by":"crossref","first-page":"1901","DOI":"10.1109\/TASLP.2017.2726762","article-title":"Multitalker speech separation with utterance-level permutation invariant training of deep recurrent neural networks","volume":"25","author":"Kolb\u00e6k","year":"2017","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102022_b22","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"101","article-title":"Source separation with weakly labelled data: An approach to computational auditory scene analysis","author":"Kong","year":"2020"},{"key":"10.1016\/j.csl.2026.102022_b23","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"626","article-title":"SDR\u2013half-baked or well done?","author":"Le Roux","year":"2019"},{"issue":"11","key":"10.1016\/j.csl.2026.102022_b24","doi-asserted-by":"crossref","first-page":"1697","DOI":"10.1109\/TASLP.2019.2928140","article-title":"Audio\u2013visual deep clustering for speech separation","volume":"27","author":"Lu","year":"2019","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"4","key":"10.1016\/j.csl.2026.102022_b25","doi-asserted-by":"crossref","first-page":"787","DOI":"10.1109\/TASLP.2018.2795749","article-title":"Speaker-independent speech separation with deep attractor network","volume":"26","author":"Luo","year":"2018","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102022_b26","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"46","article-title":"Dual-path rnn: efficient long sequence modeling for time-domain single-channel speech separation","author":"Luo","year":"2020"},{"key":"10.1016\/j.csl.2026.102022_b27","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"696","article-title":"Tasnet: time-domain audio separation network for real-time, single-channel speech separation","author":"Luo","year":"2018"},{"issue":"8","key":"10.1016\/j.csl.2026.102022_b28","doi-asserted-by":"crossref","first-page":"1256","DOI":"10.1109\/TASLP.2019.2915167","article-title":"Conv-tasnet: Surpassing ideal time\u2013frequency magnitude masking for speech separation","volume":"27","author":"Luo","year":"2019","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102022_b29","series-title":"International Conference on Acoustics, Speech and Signal Processing","first-page":"696","article-title":"Whamr!: Noisy and reverberant single-channel speech separation","author":"Maciejewski","year":"2020"},{"key":"10.1016\/j.csl.2026.102022_b30","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"6673","article-title":"Audio-visual speech separation using cross-modal correspondence loss","author":"Makishima","year":"2021"},{"key":"10.1016\/j.csl.2026.102022_b31","doi-asserted-by":"crossref","unstructured":"Mirzadeh, S.I., Farajtabar, M., Li, A., Levine, N., Matsukawa, A., Ghasemzadeh, H., 2020. Improved knowledge distillation via teacher assistant. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 34, pp. 5191\u20135198.","DOI":"10.1609\/aaai.v34i04.5963"},{"key":"10.1016\/j.csl.2026.102022_b32","series-title":"ICML","first-page":"7164","article-title":"Voice separation with an unknown number of multiple speakers","author":"Nachmani","year":"2020"},{"key":"10.1016\/j.csl.2026.102022_b33","unstructured":"Nair, V., Hinton, G.E., 2010. Rectified linear units improve restricted boltzmann machines. In: Proceedings of the 27th International Conference on Machine Learning. ICML-10, pp. 807\u2013814."},{"key":"10.1016\/j.csl.2026.102022_b34","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"661","article-title":"Teacher-student learning for low-latency online speech enhancement using wave-u-net","author":"Nakaoka","year":"2021"},{"key":"10.1016\/j.csl.2026.102022_b35","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"10141","article-title":"Two-step knowledge distillation for tiny speech enhancement","author":"Nathoo","year":"2024"},{"key":"10.1016\/j.csl.2026.102022_b36","first-page":"2386","article-title":"Finding strength in weakness: Learning to separate sounds with weak supervision","volume":"28","author":"Pishdadian","year":"2020","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102022_b37","doi-asserted-by":"crossref","first-page":"763","DOI":"10.1109\/LSP.2023.3289110","article-title":"Noise-separated adaptive feature distillation for robust speech recognition","volume":"30","author":"Qu","year":"2023","journal-title":"IEEE Signal Process. Lett."},{"key":"10.1016\/j.csl.2026.102022_b38","series-title":"AAAI","first-page":"11220","article-title":"Sfsrnet: Super-resolution for single-channel audio source separation","volume":"Vol. 36","author":"Rixen","year":"2022"},{"key":"10.1016\/j.csl.2026.102022_b39","article-title":"Fitnets: Hints for thin deep nets","author":"Romero","year":"2015","journal-title":"ICLR"},{"key":"10.1016\/j.csl.2026.102022_b40","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"21","article-title":"Attention is all you need in speech separation","author":"Subakan","year":"2021"},{"key":"10.1016\/j.csl.2026.102022_b41","article-title":"Recursive speech separation for unknown number of speakers","author":"Takahashi","year":"2019","journal-title":"InterSpeech"},{"key":"10.1016\/j.csl.2026.102022_b42","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"1","article-title":"Discriminative vector learning with application to single channel speech separation","author":"Tan","year":"2023"},{"key":"10.1016\/j.csl.2026.102022_b43","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"3678","article-title":"Selective mutual learning: An efficient approach for single channel speech separation","author":"Tan","year":"2022"},{"key":"10.1016\/j.csl.2026.102022_b44","series-title":"ICCV","first-page":"1105","article-title":"The right to talk: An audio-visual transformer approach","author":"Truong","year":"2021"},{"key":"10.1016\/j.csl.2026.102022_b45","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"561","article-title":"Online target sound extraction with knowledge distillation from partially non-causal teacher","author":"Wakayama","year":"2024"},{"key":"10.1016\/j.csl.2026.102022_b46","doi-asserted-by":"crossref","DOI":"10.1109\/TASLP.2023.3304482","article-title":"Tf-gridnet: Integrating full-and sub-band modeling for speech separation","author":"Wang","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102022_b47","series-title":"AAAI","first-page":"13961","article-title":"Tune-in: Training under negative environments with interference for attention networks simulating cocktail party effect","volume":"Vol. 35","author":"Wang","year":"2021"},{"key":"10.1016\/j.csl.2026.102022_b48","series-title":"Moving speaker separation via parallel spectral-spatial processing","author":"Wang","year":"2026"},{"key":"10.1016\/j.csl.2026.102022_b49","doi-asserted-by":"crossref","unstructured":"Wichern, G., Antognini, J., Flynn, M., Zhu, L.R., McQuinn, E., Crow, D., Manilow, E., Roux, J.L., 2019. WHAM!: Extending Speech Separation to Noisy Environments. In: Proc. Interspeech 2019. pp. 1368\u20131372.","DOI":"10.21437\/Interspeech.2019-2821"},{"key":"10.1016\/j.csl.2026.102022_b50","first-page":"3846","article-title":"Unsupervised sound separation using mixture invariant training","volume":"33","author":"Wisdom","year":"2020","journal-title":"NIPS"},{"key":"10.1016\/j.csl.2026.102022_b51","doi-asserted-by":"crossref","unstructured":"Xu, J., Shi, J., Liu, G., Chen, X., Xu, B., 2018. Modeling attention and memory for auditory selection in a cocktail party environment. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 32.","DOI":"10.1609\/aaai.v32i1.11879"},{"key":"10.1016\/j.csl.2026.102022_b52","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"6842","article-title":"Tfpsnet: Time-frequency domain path scanning network for speech separation","author":"Yang","year":"2022"},{"key":"10.1016\/j.csl.2026.102022_b53","series-title":"CVPR","first-page":"8227","article-title":"Audio-visual speech codecs: Rethinking audio-visual speech enhancement by re-synthesis","author":"Yang","year":"2022"},{"key":"10.1016\/j.csl.2026.102022_b54","doi-asserted-by":"crossref","first-page":"2840","DOI":"10.1109\/TASLP.2021.3099291","article-title":"Wavesplit: End-to-end speech separation by speaker clustering","volume":"29","author":"Zeghidour","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102022_b55","series-title":"IEEE CVPR","first-page":"4320","article-title":"Deep mutual learning","author":"Zhang","year":"2018"},{"key":"10.1016\/j.csl.2026.102022_b56","article-title":"Teacher-student mixit for unsupervised and semi-supervised speech separation","author":"Zhang","year":"2021","journal-title":"Interspeech"},{"key":"10.1016\/j.csl.2026.102022_b57","series-title":"IJCAI","first-page":"3251","article-title":"Multi-scale group transformer for long sequence modeling in speech separation","author":"Zhao","year":"2021"},{"key":"10.1016\/j.csl.2026.102022_b58","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"1","article-title":"MossFormer: Pushing the performance limit of monaural speech separation using gated single-head transformer with convolution-augmented joint self-attentions","author":"Zhao","year":"2023"},{"key":"10.1016\/j.csl.2026.102022_b59","series-title":"International Conference on Acoustics, Speech and Signal Processing","first-page":"10356","article-title":"Mossformer2: Combining transformer and rnn-free recurrent network for enhanced time-domain monaural speech separation","author":"Zhao","year":"2024"},{"key":"10.1016\/j.csl.2026.102022_b60","series-title":"Findings of the Association for Computational Linguistics: ACL 2024","first-page":"265","article-title":"Teaching-assistant-in-the-loop: Improving knowledge distillation from imperfect teacher models in low-budget scenarios","author":"Zhou","year":"2024"}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000859?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000859?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T07:46:10Z","timestamp":1783669570000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000859"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,2]]},"references-count":60,"alternative-id":["S0885230826000859"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.102022","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2027,2]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Selective teaching yourself: An efficient self-learning approach for speech separation","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.102022","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"102022"}}