{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:05Z","timestamp":1779228365345,"version":"3.51.4"},"reference-count":45,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100020595","name":"National Science and Technology Council","doi-asserted-by":"publisher","award":["114-2221-E-027-081-MY2"],"award-info":[{"award-number":["114-2221-E-027-081-MY2"]}],"id":[{"id":"10.13039\/100020595","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101984","type":"journal-article","created":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T16:01:37Z","timestamp":1774368097000},"page":"101984","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Scaling multi-speaker speech recognition with high-quality synthetic data"],"prefix":"10.1016","volume":"100","author":[{"given":"Shao-Jung","family":"Chan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5692-9279","authenticated-orcid":false,"given":"Kuang-Yow","family":"Lian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101984_bib0001","series-title":"5th International Workshop on Speech Processing in Everyday Environments","first-page":"35","article-title":"Front-end processing for the CHiME-5 dinner party scenario","author":"Boeddecker","year":"2018"},{"key":"10.1016\/j.csl.2026.101984_bib0002","article-title":"The AMI Meeting Corpus: a pre-announcement","volume":"3869","author":"Carletta","year":"2006"},{"key":"10.1016\/j.csl.2026.101984_bib0003","series-title":"2016 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"4960","article-title":"Listen, attend and spell: a neural network for large vocabulary conversational speech recognition","author":"Chan","year":"2016"},{"key":"10.1016\/j.csl.2026.101984_bib0004","series-title":"ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6256","article-title":"End-to-end monaural multi-speaker ASR system without pretraining","author":"Chang","year":"2019"},{"key":"10.1016\/j.csl.2026.101984_bib0005","series-title":"2019 IEEE Automatic Speech Recognition and Understanding Workshop","first-page":"237","article-title":"MIMO-speech: end-to-end multi-channel multi-speaker speech recognition","author":"Chang","year":"2019"},{"key":"10.1016\/j.csl.2026.101984_bib0006","series-title":"ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6134","article-title":"End-to-end multi-speaker speech recognition with transformer","author":"Chang","year":"2020"},{"key":"10.1016\/j.csl.2026.101984_bib0007","series-title":"ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5749","article-title":"Continuous speech separation with conformer","author":"Chen","year":"2021"},{"key":"10.1016\/j.csl.2026.101984_bib0008","series-title":"ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7284","article-title":"Continuous speech separation: dataset and analysis","author":"Chen","year":"2020"},{"key":"10.1016\/j.csl.2026.101984_bib0009","article-title":"Attention-based models for speech recognition","volume":"28","author":"Chorowski","year":"2015"},{"key":"10.1016\/j.csl.2026.101984_bib0010","series-title":"ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"11856","article-title":"One model to rule them all ? Towards end-to-end joint speaker diarization and speech recognition","author":"Cornell","year":"2024"},{"key":"10.1016\/j.csl.2026.101984_bib0011","series-title":"ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"9986","article-title":"SA-SOT: speaker-aware serialized output training for multi-talker ASR","author":"Fan","year":"2024"},{"key":"10.1016\/j.csl.2026.101984_bib0012","doi-asserted-by":"crossref","unstructured":"Graves, A., 2012. Sequence Transduction With Recurrent Neural Networks. arXiv:1211.3711. https:\/\/arxiv.org\/abs\/1211.3711.","DOI":"10.1007\/978-3-642-24797-2_3"},{"key":"10.1016\/j.csl.2026.101984_bib0013","series-title":"Proceedings of the 23rd International Conference on Machine Learning","first-page":"369","article-title":"Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks","author":"Graves","year":"2006"},{"key":"10.1016\/j.csl.2026.101984_bib0014","series-title":"Proceedings of the 31st International Conference on Machine Learning","first-page":"1764","article-title":"Towards end-to-end speech recognition with recurrent neural networks","author":"Graves","year":"2014"},{"key":"10.1016\/j.csl.2026.101984_bib0015","series-title":"CMT-LLM: Contextual Multi-Talker ASR Utilizing Large Language Models","first-page":"2575","author":"He","year":"2025"},{"key":"10.1016\/j.csl.2026.101984_bib0016","article-title":"Survey of end-to-end multi-speaker automatic speech recognition for monaural audio","author":"He","year":"2025","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.csl.2026.101984_bib0017","series-title":"Guided Source Separation Meets a Strong ASR backend: Hitachi\/Paderborn university Joint Investigation For Dinner Party ASR","first-page":"1248","author":"Kanda","year":"2019"},{"key":"10.1016\/j.csl.2026.101984_bib0018","series-title":"Joint Speaker counting, Speech recognition, and Speaker Identification For Overlapped Speech of Any Number of Speakers","first-page":"36","author":"Kanda","year":"2020"},{"key":"10.1016\/j.csl.2026.101984_bib0019","series-title":"Serialized Output Training For End-To-End Overlapped Speech Recognition","first-page":"2797","author":"Kanda","year":"2020"},{"key":"10.1016\/j.csl.2026.101984_bib0020","series-title":"Auxiliary Interference Speaker Loss For Target-Speaker Speech Recognition","first-page":"236","author":"Kanda","year":"2019"},{"key":"10.1016\/j.csl.2026.101984_bib0021","series-title":"Streaming Multi-Talker ASR With Token-Level Serialized Output Training","first-page":"3774","author":"Kanda","year":"2022"},{"key":"10.1016\/j.csl.2026.101984_bib0022","series-title":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"8082","article-title":"Transcribe-to-diarize: neural speaker diarization for unlimited number of speakers using end-to-end speaker-attributed ASR","author":"Kanda","year":"2022"},{"key":"10.1016\/j.csl.2026.101984_bib0023","series-title":"End-to-end Speaker-Attributed ASR With Transformer","first-page":"4413","author":"Kanda","year":"2021"},{"key":"10.1016\/j.csl.2026.101984_bib0024","series-title":"Large-Scale Pre-Training of End-to-End Multi-Talker ASR For Meeting Transcription With Single Distant Microphone","first-page":"3430","author":"Kanda","year":"2021"},{"key":"10.1016\/j.csl.2026.101984_bib0025","series-title":"Adapting Multi-Lingual ASR Models for Handling Multiple Talkers","first-page":"1314","author":"Li","year":"2023"},{"key":"10.1016\/j.csl.2026.101984_bib0026","series-title":"BA-SOT: Boundary-aware serialized Output Training For Multi-Talker ASR","first-page":"3487","author":"Liang","year":"2023"},{"key":"10.1016\/j.csl.2026.101984_bib0027","series-title":"Streaming Multi-Talker Speech Recognition With Joint Speaker Identification","first-page":"1782","author":"Lu","year":"2021"},{"key":"10.1016\/j.csl.2026.101984_bib0028","series-title":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7312","article-title":"Endpoint detection for streaming end-to-end multi-talker ASR","author":"Lu","year":"2022"},{"key":"10.1016\/j.csl.2026.101984_bib0029","series-title":"Empowering Whisper As a Joint Multi-Talker and Target-Talker Speech Recognition System","first-page":"4653","author":"Meng","year":"2024"},{"key":"10.1016\/j.csl.2026.101984_bib0030","series-title":"Whisper: General-purpose speech Recognition Model [Computer Software]","year":"2022"},{"key":"10.1016\/j.csl.2026.101984_bib0031","series-title":"2015 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5206","article-title":"Librispeech: an ASR corpus based on public domain audio books","author":"Panayotov","year":"2015"},{"key":"10.1016\/j.csl.2026.101984_bib0032","series-title":"IEEE 2011 workshop on automatic speech recognition and understanding","article-title":"The Kaldi speech recognition toolkit","author":"Povey","year":"2011"},{"key":"10.1016\/j.csl.2026.101984_bib0033","series-title":"Proceedings of the 40th International Conference on Machine Learning","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume":"202","author":"Radford","year":"2023"},{"key":"10.1016\/j.csl.2026.101984_bib0034","series-title":"2021 IEEE Spoken Language Technology Workshop","first-page":"897","article-title":"Integration of speech separation, diarization, and recognition for multi-speaker meetings: system description, comparison, and analysis","author":"Raj","year":"2021"},{"key":"10.1016\/j.csl.2026.101984_bib0035","series-title":"SpeechBrain: A general-Purpose Speech Toolkit","author":"Ravanelli","year":"2021"},{"key":"10.1016\/j.csl.2026.101984_bib0036","series-title":"Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics","first-page":"2620","article-title":"A purely end-to-end system for multi-speaker speech recognition","volume":"1","author":"Seki","year":"2018"},{"key":"10.1016\/j.csl.2026.101984_bib0037","series-title":"ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6903","article-title":"Streaming multi-speaker ASR with RNN-T","author":"Sklyar","year":"2021"},{"key":"10.1016\/j.csl.2026.101984_bib0038","series-title":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"8402","article-title":"Multi-turn RNN-T for streaming recognition of multi-party speech","author":"Sklyar","year":"2022"},{"key":"10.1016\/j.csl.2026.101984_bib0039","series-title":"ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6129","article-title":"End-to-end multi-talker overlapping speech recognition","author":"Tripathi","year":"2020"},{"key":"10.1016\/j.csl.2026.101984_bib0040","first-page":"6000","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.csl.2026.101984_bib0041","series-title":"6th International Workshop on Speech Processing in Everyday Environments","first-page":"1","article-title":"CHiME-6 challenge: tackling multispeaker speech recognition for unsegmented recordings","author":"Watanabe","year":"2020"},{"issue":"2","key":"10.1016\/j.csl.2026.101984_bib0042","doi-asserted-by":"crossref","first-page":"270","DOI":"10.1162\/neco.1989.1.2.270","article-title":"A learning algorithm for continually running fully recurrent neural networks","volume":"1","author":"Williams","year":"1989","journal-title":"Neural Comput"},{"key":"10.1016\/j.csl.2026.101984_bib0043","series-title":"Recognizing Multi-Talker Speech With Permutation Invariant Training","first-page":"2456","author":"Yu","year":"2017"},{"key":"10.1016\/j.csl.2026.101984_bib0044","series-title":"2017 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"241","article-title":"Permutation invariant training of deep models for speaker-independent multi-talker speech separation","author":"Yu","year":"2017"},{"key":"10.1016\/j.csl.2026.101984_bib0045","doi-asserted-by":"crossref","first-page":"1385","DOI":"10.1109\/TASLP.2020.2988423","article-title":"Improving end-to-end single-channel multi-talker speech recognition","volume":"28","author":"Zhang","year":"2020","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process"}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000471?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000471?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:10:47Z","timestamp":1779225047000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000471"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":45,"alternative-id":["S0885230826000471"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101984","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Scaling multi-speaker speech recognition with high-quality synthetic data","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101984","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"101984"}}