{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T13:20:21Z","timestamp":1783171221692,"version":"3.54.6"},"reference-count":73,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T00:00:00Z","timestamp":1780704000000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"DOI":"10.13039\/100003187","name":"NSF","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100003187","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100017186","name":"Ohio Supercomputer Center","doi-asserted-by":"publisher","award":["ACI-1928147"],"award-info":[{"award-number":["ACI-1928147"]}],"id":[{"id":"10.13039\/100017186","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["ECCS-2125074"],"award-info":[{"award-number":["ECCS-2125074"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Speech Communication"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.specom.2026.103427","type":"journal-article","created":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T03:53:59Z","timestamp":1780718039000},"page":"103427","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Elevating robust multi-talker ASR by decoupling speaker separation and speech recognition"],"prefix":"10.1016","volume":"182","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5921-6541","authenticated-orcid":false,"given":"Yufeng","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hassan","family":"Taherian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Vahid Ahmadi","family":"Kalkhorani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"DeLiang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.specom.2026.103427_b1","doi-asserted-by":"crossref","first-page":"1185","DOI":"10.1109\/TASLP.2024.3350887","article-title":"TS-SEP: Joint diarization and separation conditioned on estimated speaker embeddings","volume":"32","author":"Boeddeker","year":"2024","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b2","doi-asserted-by":"crossref","unstructured":"Chang, X., Maekaku, T., Fujita, Y., Watanabe, S., 2022. End-to-End Integration of Speech Recognition, Speech Enhancement, and Self-Supervised Learning Representation. In: Proc. INTERSPEECH. pp. 3819\u20133823.","DOI":"10.21437\/Interspeech.2022-10839"},{"key":"10.1016\/j.specom.2026.103427_b3","doi-asserted-by":"crossref","unstructured":"Chang, X., Zhang, W., Qian, Y., Le Roux, J., Watanabe, S., 2020. End-to-end multi-speaker speech recognition with Transformer. In: Proc. IEEE ICASSP. pp. 6134\u20136138.","DOI":"10.1109\/ICASSP40776.2020.9054029"},{"key":"10.1016\/j.specom.2026.103427_b4","doi-asserted-by":"crossref","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","article-title":"WavLM: Large-scale self-supervised pre-training for full stack speech processing","author":"Chen","year":"2022","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.specom.2026.103427_b5","doi-asserted-by":"crossref","unstructured":"Chen, Z., Yoshioka, T., Lu, L., et al., 2020. Continuous speech separation: Dataset and analysis. In: Proc. IEEE ICASSP. pp. 7284\u20137288.","DOI":"10.1109\/ICASSP40776.2020.9053426"},{"key":"10.1016\/j.specom.2026.103427_b6","doi-asserted-by":"crossref","unstructured":"Cord-Landwehr, T., Gburrek, T., Deegen, M., Haeb-Umbach, R., 2025. Spatio-spectral diarization of meetings by combining TDOA-based segmentation and speaker embedding-based clustering. In: Proc. Interspeech. pp. 5223\u20135227.","DOI":"10.21437\/Interspeech.2025-1663"},{"key":"10.1016\/j.specom.2026.103427_b7","series-title":"LibriMix: An open-source dataset for generalizable speech separation","author":"Cosentino","year":"2020"},{"key":"10.1016\/j.specom.2026.103427_b8","doi-asserted-by":"crossref","unstructured":"Delcroix, M., Zmolikova, K., Ochiai, T., Kinoshita, K., Nakatani, T., 2021. Speaker activity driven neural speech extraction. In: Proc. IEEE ICASSP. pp. 6099\u20136103.","DOI":"10.1109\/ICASSP39728.2021.9414998"},{"key":"10.1016\/j.specom.2026.103427_b9","series-title":"SMS-WSJ: Database, performance measures, and baseline recipe for multi-channel source separation and recognition","author":"Drude","year":"2019"},{"key":"10.1016\/j.specom.2026.103427_b10","doi-asserted-by":"crossref","first-page":"198","DOI":"10.1109\/TASLP.2020.3039600","article-title":"Gated recurrent fusion with joint training framework for robust end-to-end speech recognition","volume":"29","author":"Fan","year":"2020","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b11","doi-asserted-by":"crossref","unstructured":"Gulati, A., Qin, J., Chiu, C.-C., et al., 2020. Conformer: Convolution-augmented transformer for speech recognition. In: Proc. INTERSPEECH. pp. 5036\u20135040.","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"10.1016\/j.specom.2026.103427_b12","doi-asserted-by":"crossref","unstructured":"Guo, P., Chang, X., Watanabe, S., Xie, L., 2021. Multi-speaker ASR combining non-autoregressive Conformer CTC and conditional speaker chain. In: Proc. INTERSPEECH. pp. 3720\u20133724.","DOI":"10.21437\/Interspeech.2021-2155"},{"key":"10.1016\/j.specom.2026.103427_b13","unstructured":"Heymann, J., Drude, L., Haeb-Umbach, R., 2016. Wide residual BLSTM network with discriminative speaker adaptation for robust speech recognition. In: Proc. CHiME-4 Workshop. Vol. 78, p. 79."},{"key":"10.1016\/j.specom.2026.103427_b14","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","article-title":"Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups","volume":"29","author":"Hinton","year":"2012","journal-title":"IEEE Signal Process. Mag."},{"key":"10.1016\/j.specom.2026.103427_b15","doi-asserted-by":"crossref","unstructured":"Kalda, J., Alum\u00e4e, T., 2022. Collar-Aware Training for Streaming Speaker Change Detection in Broadcast Speech. In: Proc. Speaker Odyssey. pp. 141\u2013147.","DOI":"10.21437\/Odyssey.2022-20"},{"key":"10.1016\/j.specom.2026.103427_b16","doi-asserted-by":"crossref","first-page":"4999","DOI":"10.1109\/TASLP.2024.3492803","article-title":"TF-CrossNet: Leveraging global, cross-band, narrow-band, and positional encoding for single- and multi-channel speaker separation","volume":"32","author":"Kalkhorani","year":"2024","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b17","doi-asserted-by":"crossref","unstructured":"Kanda, N., Gaur, Y., Wang, X., Meng, Z., Yoshioka, T., 2020. Serialized Output Training for End-to-End Overlapped Speech Recognition. In: Proc. INTERSPEECH. pp. 2797\u20132801.","DOI":"10.21437\/Interspeech.2020-999"},{"key":"10.1016\/j.specom.2026.103427_b18","doi-asserted-by":"crossref","unstructured":"Kanda, N., Xiao, X., Gaur, Y., et al., 2022. Transcribe-to-diarize: Neural speaker diarization for unlimited number of speakers using end-to-end speaker-attributed ASR. In: Proc. IEEE ICASSP. pp. 8082\u20138086.","DOI":"10.1109\/ICASSP43922.2022.9746225"},{"key":"10.1016\/j.specom.2026.103427_b19","unstructured":"Kingma, D.P., Ba, J., 2015. Adam: A Method for Stochastic Optimization. In: Proc. Int. Conf. Learn. Representations. pp. 1\u201315."},{"key":"10.1016\/j.specom.2026.103427_b20","doi-asserted-by":"crossref","first-page":"1901","DOI":"10.1109\/TASLP.2017.2726762","article-title":"Multitalker speech separation with utterance-level permutation invariant training of deep recurrent neural networks","volume":"25","author":"Kolb\u00e6k","year":"2017","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b21","doi-asserted-by":"crossref","unstructured":"Le Roux, J., Wisdom, S., Erdogan, H., Hershey, J.R., 2019. SDR\u2013half-baked or well done?. In: Proc. IEEE ICASSP. pp. 626\u2013630.","DOI":"10.1109\/ICASSP.2019.8683855"},{"key":"10.1016\/j.specom.2026.103427_b22","doi-asserted-by":"crossref","first-page":"803","DOI":"10.1109\/LSP.2021.3070817","article-title":"Streaming end-to-end multi-talker speech recognition","volume":"28","author":"Lu","year":"2021","journal-title":"IEEE Signal Process. Lett."},{"key":"10.1016\/j.specom.2026.103427_b23","doi-asserted-by":"crossref","unstructured":"Luo, Y., Chen, Z., Mesgarani, N., Yoshioka, T., 2020a. End-to-end microphone permutation and number invariant multi-channel speech separation. In: Proc. IEEE ICASSP. pp. 6394\u20136398.","DOI":"10.1109\/ICASSP40776.2020.9054177"},{"key":"10.1016\/j.specom.2026.103427_b24","doi-asserted-by":"crossref","unstructured":"Luo, Y., Chen, Z., Yoshioka, T., 2020b. Dual-path RNN: efficient long sequence modeling for time-domain single-channel speech separation. In: Proc. IEEE ICASSP. pp. 46\u201350.","DOI":"10.1109\/ICASSP40776.2020.9054266"},{"key":"10.1016\/j.specom.2026.103427_b25","doi-asserted-by":"crossref","first-page":"1256","DOI":"10.1109\/TASLP.2019.2915167","article-title":"Conv-TasNet: Surpassing ideal time-frequency magnitude masking for speech separation","volume":"27","author":"Luo","year":"2019","journal-title":"IEEE\/ACM Trans. Audio Speech Language Process."},{"key":"10.1016\/j.specom.2026.103427_b26","doi-asserted-by":"crossref","unstructured":"Masuyama, Y., Chang, X., Cornell, S., Watanabe, S., Ono, N., 2023a. End-to-end integration of speech recognition, dereverberation, beamforming, and self-supervised learning representation. In: Proc. IEEE Spoken Language Technology Workshop. SLT, pp. 260\u2013265.","DOI":"10.1109\/SLT54892.2023.10023199"},{"key":"10.1016\/j.specom.2026.103427_b27","doi-asserted-by":"crossref","unstructured":"Masuyama, Y., Chang, X., Zhang, W., et al., 2023b. Exploring the Integration of Speech Separation and Recognition with Self-Supervised Learning Representation. In: Proc. IEEE WASPAA. pp. 1\u20135.","DOI":"10.1109\/WASPAA58266.2023.10248096"},{"key":"10.1016\/j.specom.2026.103427_b28","doi-asserted-by":"crossref","DOI":"10.1109\/TASLPRO.2025.3600913","article-title":"DCF-DS: Deep cascade fusion of diarization and separation for speech recognition under realistic single-channel conditions","author":"Niu","year":"2025","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b29","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S., 2015. LibriSpeech: an ASR corpus based on public domain audio books. In: Proc. IEEE Int. Conf. Acoust. Speech Signal Process.. pp. 5206\u20135210.","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"10.1016\/j.specom.2026.103427_b30","doi-asserted-by":"crossref","first-page":"1374","DOI":"10.1109\/TASLP.2022.3161143","article-title":"Self-attending RNN for speech enhancement to improve cross-corpus generalization","volume":"30","author":"Pandey","year":"2022","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b31","doi-asserted-by":"crossref","unstructured":"Paul, D.B., Baker, J.M., 1992. The design for the Wall Street Journal-based CSR corpus. In: Proc. Workshop on Speech and Natural Language. pp. 357\u2013362.","DOI":"10.3115\/1075527.1075614"},{"key":"10.1016\/j.specom.2026.103427_b32","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., et al., 2011. The Kaldi speech recognition toolkit. In: Proc. IEEE ASRU."},{"key":"10.1016\/j.specom.2026.103427_b33","doi-asserted-by":"crossref","first-page":"325","DOI":"10.1109\/TASLP.2023.3328283","article-title":"End-to-end speech recognition: A survey","volume":"32","author":"Prabhavalkar","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b34","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.specom.2018.09.003","article-title":"Single-channel multi-talker speech recognition with permutation invariant training","volume":"104","author":"Qian","year":"2018","journal-title":"Speech Commun."},{"key":"10.1016\/j.specom.2026.103427_b35","doi-asserted-by":"crossref","first-page":"1310","DOI":"10.1109\/TASLP.2024.3357036","article-title":"SpatialNet: Extensively learning spatial information for multichannel joint speech separation, denoising and dereverberation","volume":"32","author":"Quan","year":"2024","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b36","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1109\/5.18626","article-title":"A tutorial on hidden Markov models and selected applications in speech recognition","volume":"77","author":"Rabiner","year":"1989","journal-title":"Proc. IEEE"},{"key":"10.1016\/j.specom.2026.103427_b37","unstructured":"Radford, A., Kim, J.W., Xu, T., Brockman, G., McLeavey, C., Sutskever, I., 2023. Robust speech recognition via large-scale weak supervision. In: Proc. Int. Conf. Mach. Learn.. pp. 28492\u201328518."},{"key":"10.1016\/j.specom.2026.103427_b38","doi-asserted-by":"crossref","unstructured":"Raj, D., Denisov, P., Chen, Z., et al., 2021. Integration of speech separation, diarization, and recognition for multi-speaker meetings: System description, comparison, and analysis. In: Proc. IEEE SLT. pp. 897\u2013904.","DOI":"10.1109\/SLT48900.2021.9383556"},{"key":"10.1016\/j.specom.2026.103427_b39","doi-asserted-by":"crossref","unstructured":"Rix, A.W., Beerends, J.G., Hollier, M.P., Hekstra, A.P., 2001. Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs. In: Proc. IEEE Int. Conf. Acoust. Speech Signal Process.. Vol. 2, pp. 749\u2013752.","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"10.1016\/j.specom.2026.103427_b40","doi-asserted-by":"crossref","unstructured":"Seong, J.-S., Choi, J.-H., Jeoung, Y.-R., Kim, I., Chang, J.-H., 2025. Enhancing Target-speaker Automatic Speech Recognition Using Multiple Speaker Embedding Extractors with Virtual Speaker Embedding. In: Proc. Interspeech. pp. 4918\u20134922.","DOI":"10.21437\/Interspeech.2025-2486"},{"key":"10.1016\/j.specom.2026.103427_b41","doi-asserted-by":"crossref","unstructured":"Shakeel, M., Sudo, Y., Peng, Y., Lin, C.-J., Watanabe, S., 2025. Unifying diarization, separation, and ASR with multi-speaker encoder. In: Proc. IEEE ASRU. pp. 1\u20137.","DOI":"10.1109\/ASRU65441.2025.11434654"},{"key":"10.1016\/j.specom.2026.103427_b42","doi-asserted-by":"crossref","DOI":"10.1016\/j.csl.2022.101387","article-title":"Train from scratch: Single-stage joint training of speech separation and recognition","volume":"76","author":"Shi","year":"2022","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.specom.2026.103427_b43","doi-asserted-by":"crossref","unstructured":"Shi, H., Gao, Y., Ni, Z., Kawahara, T., 2024. Serialized Speech Information Guidance with Overlapped Encoding Separation for Multi-Speaker Automatic Speech Recognition. In: Proc. IEEE SLT. pp. 193\u2013199.","DOI":"10.1109\/SLT61566.2024.10832330"},{"key":"10.1016\/j.specom.2026.103427_b44","doi-asserted-by":"crossref","unstructured":"Shi, M., Jin, Z., Xu, Y., Xu, Y., Zhang, S.-X., Wei, K., Shao, Y., Zhang, C., Yu, D., 2024. Advancing Multi-Talker ASR Performance With Large Language Models. In: Proc. IEEE SLT. pp. 14\u201321.","DOI":"10.1109\/SLT61566.2024.10832362"},{"key":"10.1016\/j.specom.2026.103427_b45","doi-asserted-by":"crossref","first-page":"1945","DOI":"10.1109\/LSP.2024.3432324","article-title":"Keyword guided target speech recognition","volume":"31","author":"Shi","year":"2024","journal-title":"IEEE Signal Process. Lett."},{"key":"10.1016\/j.specom.2026.103427_b46","doi-asserted-by":"crossref","first-page":"2125","DOI":"10.1109\/TASL.2011.2114881","article-title":"An algorithm for intelligibility prediction of time\u2013frequency weighted noisy speech","volume":"19","author":"Taal","year":"2011","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b47","doi-asserted-by":"crossref","unstructured":"Taherian, H., Pandey, A., Wong, D., Xu, B., Wang, D., 2024. Leveraging sound localization to improve continuous speaker separation. In: Proc. IEEE ICASSP. pp. 621\u2013625.","DOI":"10.1109\/ICASSP48485.2024.10446934"},{"key":"10.1016\/j.specom.2026.103427_b48","doi-asserted-by":"crossref","first-page":"2791","DOI":"10.1109\/TASLP.2022.3202129","article-title":"Multi-channel talker-independent speaker separation through location-based training","volume":"30","author":"Taherian","year":"2022","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b49","doi-asserted-by":"crossref","first-page":"2467","DOI":"10.1109\/TASLP.2024.3393726","article-title":"Multi-channel conversational speaker separation via neural diarization","volume":"32","author":"Taherian","year":"2024","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b50","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et al., 2017. Attention is all you need. In: Proc. Adv. Neural Inf. Process. Syst.. pp. 5998\u20136008."},{"key":"10.1016\/j.specom.2026.103427_b51","series-title":"Error analysis in a modular meeting transcription system","author":"Vieting","year":"2025"},{"key":"10.1016\/j.specom.2026.103427_b52","doi-asserted-by":"crossref","unstructured":"Vincent, E., Barker, J., Watanabe, S., et al., 2013. The second \u2018CHiME\u2019 speech separation and recognition challenge: Datasets, tasks and baselines. In: Proc. IEEE ICASSP. pp. 126\u2013130.","DOI":"10.1109\/ICASSP.2013.6637622"},{"key":"10.1016\/j.specom.2026.103427_b53","doi-asserted-by":"crossref","first-page":"535","DOI":"10.1016\/j.csl.2016.11.005","article-title":"An analysis of environment, microphone and data simulation mismatches in robust speech recognition","volume":"46","author":"Vincent","year":"2017","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.specom.2026.103427_b54","doi-asserted-by":"crossref","first-page":"1702","DOI":"10.1109\/TASLP.2018.2842159","article-title":"Supervised speech separation based on deep learning: An overview","volume":"26","author":"Wang","year":"2018","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b55","doi-asserted-by":"crossref","first-page":"3221","DOI":"10.1109\/TASLP.2023.3304482","article-title":"TF-GridNet: Integrating full- and sub-band modeling for speech separation","volume":"31","author":"Wang","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b56","doi-asserted-by":"crossref","unstructured":"Wang, W., Mo, S., Dong, L., Yu, Z., Guo, J., Huang, Y., 2024. DGSRN: Noise-Robust Speech Recognition Method with Dual-Path Gated Spectral Refinement Network. In: Proc. INTERSPEECH. pp. 5018\u20135022.","DOI":"10.21437\/Interspeech.2024-1796"},{"key":"10.1016\/j.specom.2026.103427_b57","doi-asserted-by":"crossref","first-page":"39","DOI":"10.1109\/TASLP.2019.2946789","article-title":"Bridging the gap between monaural speech enhancement and recognition with distortion-independent acoustic modeling","volume":"28","author":"Wang","year":"2019","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b58","doi-asserted-by":"crossref","unstructured":"Wang, Z.-Q., Wang, D., 2020. Multi-microphone complex spectral mapping for speech dereverberation. In: Proc. IEEE ICASSP. pp. 486\u2013490.","DOI":"10.1109\/ICASSP40776.2020.9053610"},{"key":"10.1016\/j.specom.2026.103427_b59","doi-asserted-by":"crossref","unstructured":"Wang, Z.-Q., Wang, D., 2022. Localization based sequential grouping for continuous speech separation. In: Proc. IEEE ICASSP. pp. 281\u2013285.","DOI":"10.1109\/ICASSP43922.2022.9746896"},{"key":"10.1016\/j.specom.2026.103427_b60","doi-asserted-by":"crossref","first-page":"1778","DOI":"10.1109\/TASLP.2020.2998279","article-title":"Complex spectral mapping for single-and multi-channel speech enhancement and robust ASR","volume":"28","author":"Wang","year":"2020","journal-title":"IEEE\/ACM Trans. Audio Speech Language Process."},{"key":"10.1016\/j.specom.2026.103427_b61","doi-asserted-by":"crossref","first-page":"2001","DOI":"10.1109\/TASLP.2021.3083405","article-title":"Multi-microphone complex spectral mapping for utterance-wise and continuous speech separation","volume":"29","author":"Wang","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b62","doi-asserted-by":"crossref","unstructured":"Wang, Z.-Q., Wichern, G., Le Roux, J., 2021b. Convolutive prediction for reverberant speech separation. In: Proc. IEEE WASPAA. pp. 56\u201360.","DOI":"10.1109\/WASPAA52581.2021.9632667"},{"key":"10.1016\/j.specom.2026.103427_b63","doi-asserted-by":"crossref","unstructured":"Watanabe, S., Hori, T., Karita, S., et al., 2018. ESPnet: End-to-end speech processing toolkit. In: Proc. INTERSPEECH. pp. 2207\u20132211.","DOI":"10.21437\/Interspeech.2018-1456"},{"key":"10.1016\/j.specom.2026.103427_b64","doi-asserted-by":"crossref","unstructured":"Watanabe, S., Mandel, M., Barker, J., et al., 2020. CHiME-6 challenge: tackling multispeaker speech recognition for unsegmented recordings. In: Proc. CHiME-6. pp. 1\u20137.","DOI":"10.21437\/CHiME.2020-1"},{"key":"10.1016\/j.specom.2026.103427_b65","doi-asserted-by":"crossref","unstructured":"Wichern, G., Antognini, J., Flynn, M., Zhu, L.R., McQuinn, E., Crow, D., Manilow, E., Roux, J.L., 2019. WHAM!: Extending speech separation to noisy environments. In: Proc. INTERSPEECH. pp. 1368\u20131372.","DOI":"10.21437\/Interspeech.2019-2821"},{"key":"10.1016\/j.specom.2026.103427_b66","doi-asserted-by":"crossref","first-page":"483","DOI":"10.1109\/TASLP.2015.2512042","article-title":"Complex ratio masking for monaural speech separation","volume":"24","author":"Williamson","year":"2016","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103427_b67","doi-asserted-by":"crossref","unstructured":"Yang, Y., Pandey, A., Wang, D.L., 2023. Time-Domain Speech Enhancement for Robust Automatic Speech Recognition. In: Proc. INTERSPEECH. pp. 4913\u20134917.","DOI":"10.21437\/Interspeech.2023-167"},{"key":"10.1016\/j.specom.2026.103427_b68","doi-asserted-by":"crossref","DOI":"10.1016\/j.csl.2025.101821","article-title":"Towards decoupling frontend enhancement and backend recognition in monaural robust ASR","volume":"95","author":"Yang","year":"2026","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.specom.2026.103427_b69","doi-asserted-by":"crossref","unstructured":"Yang, Y., Taherian, H., Kalkhorani, V.A., Wang, D.L., 2025. Elevating Robust ASR By Decoupling Multi-Channel Speaker Separation and Speech Recognition. In: Proc. IEEE ICASSP. pp. 1\u20135.","DOI":"10.1109\/ICASSP49660.2025.10888074"},{"key":"10.1016\/j.specom.2026.103427_b70","series-title":"A conformer based acoustic model for robust automatic speech recognition","author":"Yang","year":"2022"},{"key":"10.1016\/j.specom.2026.103427_b71","doi-asserted-by":"crossref","unstructured":"Zhang, W., Qian, Y., 2023. Weakly-supervised speech pre-training: A case study on target speech recognition. In: Proc. INTERSPEECH. pp. 3517\u20133521.","DOI":"10.21437\/Interspeech.2023-1280"},{"key":"10.1016\/j.specom.2026.103427_b72","doi-asserted-by":"crossref","unstructured":"Zhang, W., Yang, L., Qian, Y., 2023. Exploring Time-Frequency Domain Target Speaker Extraction For Causal and Non-Causal Processing. In: Proc. IEEE ASRU. pp. 1\u20136.","DOI":"10.1109\/ASRU57964.2023.10389752"},{"key":"10.1016\/j.specom.2026.103427_b73","doi-asserted-by":"crossref","unstructured":"Zhang, J., Zoril\u0103, C., Doddipatla, R., Barker, J., 2020. On end-to-end multi-channel time domain speech separation in reverberant environments. In: Proc. IEEE ICASSP. pp. 6389\u20136393.","DOI":"10.1109\/ICASSP40776.2020.9053833"}],"container-title":["Speech Communication"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000750?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000750?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T12:53:39Z","timestamp":1783169619000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167639326000750"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":73,"alternative-id":["S0167639326000750"],"URL":"https:\/\/doi.org\/10.1016\/j.specom.2026.103427","relation":{},"ISSN":["0167-6393"],"issn-type":[{"value":"0167-6393","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Elevating robust multi-talker ASR by decoupling speaker separation and speech recognition","name":"articletitle","label":"Article Title"},{"value":"Speech Communication","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.specom.2026.103427","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"103427"}}