{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:36Z","timestamp":1779228396251,"version":"3.51.4"},"reference-count":23,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001872","name":"CDTI","doi-asserted-by":"publisher","award":["IDI-20240449"],"award-info":[{"award-number":["IDI-20240449"]}],"id":[{"id":"10.13039\/501100001872","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101966","type":"journal-article","created":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T08:50:59Z","timestamp":1772095859000},"page":"101966","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Improvements in Spanish audio transcription workflows: Integrating preprocessing, LLM-based correction, and speaker diarization and identification"],"prefix":"10.1016","volume":"100","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-4265-1447","authenticated-orcid":false,"given":"Gonzalo","family":"Nieto Montero","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Santiago","family":"Hern\u00e1ndez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Juan","family":"Casal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101966_b1","series-title":"Proceedings of the Annual Conference of the International Speech Communication Association","first-page":"4489","article-title":"Whisperx: Time-accurate speech transcription of long-form audio","author":"Bain","year":"2023"},{"key":"10.1016\/j.csl.2026.101966_b2","series-title":"ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7124","article-title":"Pyannote.audio: Neural building blocks for speaker diarization","author":"Bredin","year":"2020"},{"key":"10.1016\/j.csl.2026.101966_b3","series-title":"Llm-based speaker diarization correction: A generalizable approach","author":"Efstathiadis","year":"2024"},{"key":"10.1016\/j.csl.2026.101966_b4","series-title":"A comparative analysis of speaker diarization models: Creating a dataset for German dialectal speech","first-page":"43","author":"Fischbach","year":"2024"},{"key":"10.1016\/j.csl.2026.101966_b5","series-title":"The llama 3 herd of models","author":"Grattafiori","year":"2024"},{"key":"10.1016\/j.csl.2026.101966_b6","series-title":"Listen again and choose the right answer: A new paradigm for automatic speech recognition with large language models","first-page":"666","author":"Hu","year":"2024"},{"key":"10.1016\/j.csl.2026.101966_b7","series-title":"12th International Conference on Learning Representations","article-title":"Large language models are efficient learners of noise-robust speech recognition","author":"Hu","year":"2024"},{"key":"10.1016\/j.csl.2026.101966_b8","doi-asserted-by":"crossref","first-page":"3450","DOI":"10.1109\/TASLP.2024.3422818","article-title":"Diaper: End-to-end neural diarization with perceiver-based attractors","volume":"32","author":"Landini","year":"2024","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.101966_b9","series-title":"RTVE 2024 databases description","author":"Lleida","year":"2024"},{"key":"10.1016\/j.csl.2026.101966_b10","doi-asserted-by":"crossref","first-page":"8577","DOI":"10.3390\/app13158577","article-title":"An overview of the iberspeech-rtve 2022 challenges on speech technologies","volume":"13","author":"Lleida","year":"2023","journal-title":"Appl. Sci."},{"key":"10.1016\/j.csl.2026.101966_b11","doi-asserted-by":"crossref","first-page":"1389","DOI":"10.1109\/TASLPRO.2025.3551083","article-title":"Asr error correction using large language models","volume":"33","author":"Ma","year":"2025","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.101966_b12","series-title":"Can generative large language models perform asr error correction?","author":"Ma","year":"2023"},{"key":"10.1016\/j.csl.2026.101966_b13","series-title":"Proceedings of the 7th International Conference on Natural Language and Speech Processing","first-page":"214","article-title":"Casca: Leveraging role-based lexical cues for robust multimodal speaker diarization via large language models","author":"Nehrboss","year":"2024"},{"key":"10.1016\/j.csl.2026.101966_b14","series-title":"Identifying speakers in dialogue transcripts: A text-based approach using pretrained language models","author":"Nguyen","year":"2024"},{"key":"10.1016\/j.csl.2026.101966_b15","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume":"202","author":"Radford","year":"2022","journal-title":"Proc. Mach. Learn. Res."},{"key":"10.1016\/j.csl.2026.101966_b16","doi-asserted-by":"crossref","DOI":"10.3390\/app13137820","article-title":"Target selection strategies for demucs-based speech enhancement","volume":"13","author":"Rascon","year":"2023","journal-title":"Appl. Sci."},{"key":"10.1016\/j.csl.2026.101966_b17","series-title":"Proc. Interspeech 2021","first-page":"4204","article-title":"Personalized keyphrase detection using speaker and environment information","author":"Rikhye","year":"2021"},{"key":"10.1016\/j.csl.2026.101966_b18","series-title":"ICASSP, IEEE International Conference on Acoustics, Speech and Signal Processing - Proceedings","article-title":"Hybrid transformers for music source separation","author":"Rouard","year":"2023"},{"key":"10.1016\/j.csl.2026.101966_b19","series-title":"Canary-1b-v2 & parakeet-tdt-0.6b-v3: Efficient and high-performance models for multilingual asr and ast","author":"Sekoyan","year":"2025"},{"key":"10.1016\/j.csl.2026.101966_b20","series-title":"Diarizationlm: Speaker diarization post-processing with large language models","first-page":"3754","author":"Wang","year":"2024"},{"key":"10.1016\/j.csl.2026.101966_b21","series-title":"Proc. Interspeech 2019","first-page":"2728","article-title":"VoiceFilter: Targeted voice separation by speaker-conditioned spectrogram masking","author":"Wang","year":"2019"},{"key":"10.1016\/j.csl.2026.101966_b22","series-title":"ACL-IJCNLP 2021 - 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, Proceedings of the Conference","first-page":"993","article-title":"Voxpopuli: A large-scale multilingual speech corpus for representation learning, semi-supervised learning and interpretation","author":"Wang","year":"2021"},{"key":"10.1016\/j.csl.2026.101966_b23","series-title":"Using pretrained language models for improved speaker identification","first-page":"51","author":"Zamana","year":"2024"}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S088523082600029X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S088523082600029X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:13:52Z","timestamp":1779225232000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S088523082600029X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":23,"alternative-id":["S088523082600029X"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101966","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Improvements in Spanish audio transcription workflows: Integrating preprocessing, LLM-based correction, and speaker diarization and identification","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101966","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"101966"}}