{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T03:28:20Z","timestamp":1782358100750,"version":"3.54.5"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,10,16]],"date-time":"2024-10-16T00:00:00Z","timestamp":1729036800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2024,10,16]],"date-time":"2024-10-16T00:00:00Z","timestamp":1729036800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"funder":[{"DOI":"10.13039\/501100004895","name":"European Social Fund","doi-asserted-by":"publisher","award":["N\/A"],"award-info":[{"award-number":["N\/A"]}],"id":[{"id":"10.13039\/501100004895","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100008530","name":"European Regional Development Fund","doi-asserted-by":"publisher","award":["N\/A"],"award-info":[{"award-number":["N\/A"]}],"id":[{"id":"10.13039\/501100008530","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Lang Resources &amp; Evaluation"],"published-print":{"date-parts":[[2025,12]]},"abstract":"<jats:title>Abstract<\/jats:title>\n                  <jats:p>This paper presents our progress in developing and maintaining a public speech and speaker recognition platform for the Estonian language. The platform consists of a speech processing pipeline and a web-based user interface for end-users, offering transcript post-editing functionality. It is offered for free as a public service and is in active use. The service provides significantly higher speech recognition accuracy than commercial alternatives. We discuss the switch to a workflow management system and how it has improved the core speech processing pipeline. The core systems behind the platform have been made available as open-source code and deployed internally by multiple public and private institutions.<\/jats:p>","DOI":"10.1007\/s10579-024-09777-1","type":"journal-article","created":{"date-parts":[[2024,10,16]],"date-time":"2024-10-16T07:02:20Z","timestamp":1729062140000},"page":"4421-4438","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Open source platform for Estonian speech transcription"],"prefix":"10.1007","volume":"59","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7213-1622","authenticated-orcid":false,"given":"Aivo","family":"Olev","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tanel","family":"Alum\u00e4e","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,16]]},"reference":[{"key":"9777_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/ICAICTA.2017.8090955","volume":"2017","author":"R Adianto","year":"2017","unstructured":"Adianto, R., Satriawan, C. H., & Lestari, D. P. (2017). Transcriber: An Android application that automates the transcription of interviews in Indonesian. Proc. ICAICTA, 2017, 1\u20136. https:\/\/doi.org\/10.1109\/ICAICTA.2017.8090955","journal-title":"Proc. ICAICTA"},{"key":"9777_CR2","unstructured":"Alum\u00e4e, T. (2014). Recent improvements in Estonian LVCSR. In: Proc. SLTU 2014"},{"key":"9777_CR3","doi-asserted-by":"publisher","unstructured":"Alum\u00e4e, T., & Tilk, O. (2016). Automatic speech recognition system for Lithuanian broadcast audio. In: Proc. Baltic HLT 2016, IOS Press, vol 289, p\u00a039, https:\/\/doi.org\/10.3233\/978-1-61499-701-6-39","DOI":"10.3233\/978-1-61499-701-6-39"},{"key":"9777_CR4","doi-asserted-by":"publisher","unstructured":"Alum\u00e4e, T., Tilk, O., & Asadullah,. (2018). Advanced rich transcription system for Estonian speech. Proc. Baltic HLT, 2018, 1\u20138. https:\/\/doi.org\/10.3233\/978-1-61499-912-6-1","DOI":"10.3233\/978-1-61499-912-6-1"},{"key":"9777_CR5","doi-asserted-by":"crossref","unstructured":"Asadullah, Alum\u00e4e, T. (2018). Data augmentation and teacher-student training for LF-MMI. In: Proc. TSD 2018","DOI":"10.1007\/978-3-030-00794-2_43"},{"key":"9777_CR6","doi-asserted-by":"publisher","first-page":"452","DOI":"10.1038\/533452a","volume":"533","author":"M Baker","year":"2016","unstructured":"Baker, M. (2016). 1,500 scientists lift the lid on reproducibility. Nature, 533, 452\u2013454.","journal-title":"Nature"},{"key":"9777_CR7","doi-asserted-by":"crossref","unstructured":"Belz, A., Agarwal, S., Shimorina, A., & Reiter, E. (2021). A systematic review of reproducibility research in natural language processing. ArXiv:2103.07929","DOI":"10.18653\/v1\/2021.eacl-main.29"},{"key":"9777_CR8","doi-asserted-by":"publisher","first-page":"284","DOI":"10.1016\/j.future.2017.01.012","volume":"75","author":"S Cohen-Boulakia","year":"2017","unstructured":"Cohen-Boulakia, S., Belhajjame, K., Collin, O., Chopard, J., Froidevaux, C., Gaignard, A., Hinsen, K., Larmande, P., Bras, Y. L., Lemoine, F., Mareuil, F., M\u00e9nager, H., Pradal, C., & Blanchet, C. (2017). Scientific workflows for computational reproducibility in the life sciences: Status, challenges and opportunities. Future Generation Computer Systems, 75, 284\u2013298. https:\/\/doi.org\/10.1016\/j.future.2017.01.012","journal-title":"Future Generation Computer Systems"},{"key":"9777_CR9","doi-asserted-by":"crossref","unstructured":"Conneau, A., Baevski, A., Collobert, R., Mohamed, A., Auli, M. (2021). Unsupervised Cross-Lingual Representation Learning for Speech Recognition. In: Proc. Interspeech 2021, https:\/\/doi.org\/10.21437\/Interspeech.2021-329","DOI":"10.21437\/Interspeech.2021-329"},{"key":"9777_CR10","doi-asserted-by":"crossref","unstructured":"Desplanques, B., Thienpondt, J., & Demuynck, K. (2020). ECAPA-TDNN: Emphasized channel attention, propagation and aggregation in TDNN based speaker verification. In: Proc. Interspeech 2020","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"9777_CR11","doi-asserted-by":"publisher","first-page":"316","DOI":"10.1038\/nbt.3820","volume":"35","author":"P Di Tommaso","year":"2017","unstructured":"Di Tommaso, P., Chatzou, M., Floden, E. W., Barja, P., Palumbo, E., & Notredame, C. (2017). Nextflow enables reproducible computational workflows. Nature Biotechnology, 35, 316\u2013319. https:\/\/doi.org\/10.1038\/nbt.3820","journal-title":"Nature Biotechnology"},{"issue":"3","key":"9777_CR12","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1093\/jamia\/ocaa261","volume":"28","author":"W Digan","year":"2020","unstructured":"Digan, W., N\u00e9v\u00e9ol, A., Neuraz, A., Wack, M., Baudoin, D., Burgun, A., & Rance, B. (2020). Can reproducibility be improved in clinical natural language processing? A study of 7 clinical NLP suites. Journal of the American Medical Informatics Association, 28(3), 504\u2013515. https:\/\/doi.org\/10.1093\/jamia\/ocaa261","journal-title":"Journal of the American Medical Informatics Association"},{"key":"9777_CR13","unstructured":"Fokkens, A., van Erp, M., Postma, M., Pedersen, T., Vossen, P., & Freire, N. (2013). Offspring from reproduction problems: What replication failure teaches us. In: Proc. ACL 2013, pp 1691\u20131701, https:\/\/aclanthology.org\/P13-1166"},{"key":"9777_CR14","doi-asserted-by":"crossref","unstructured":"Gorman, K. (2016). Pynini: A Python library for weighted finite-state grammar compilation. In: Proc. SIGFSM Workshop on Statistical NLP and Weighted Automata, pp 75\u201380","DOI":"10.18653\/v1\/W16-2409"},{"key":"9777_CR15","first-page":"6873","volume":"2021","author":"KJ Han","year":"2021","unstructured":"Han, K. J., Pan, J., Tadala, V. K. N., Ma, T., & Povey, D. (2021). Multistream CNN for robust acoustic modeling. Proc. ICASSP, 2021, 6873\u20136877.","journal-title":"Proc. ICASSP"},{"key":"9777_CR16","doi-asserted-by":"publisher","first-page":"419","DOI":"10.1109\/ASRU46091.2019.9003788","volume":"2019","author":"K Irie","year":"2019","unstructured":"Irie, K., Zeyer, A., Schl\u00fcter, R., & Ney, H. (2019). Training language models for long-span cross-sentence evaluation. Proc. ASR, 2019, 419\u2013426. https:\/\/doi.org\/10.1109\/ASRU46091.2019.9003788","journal-title":"Proc. ASR"},{"key":"9777_CR17","doi-asserted-by":"publisher","unstructured":"Kallas, J., & Koppel, K. (2019). Estonian National Corpus 2019. https:\/\/doi.org\/10.15155\/3-00-0000-0000-0000-08565L","DOI":"10.15155\/3-00-0000-0000-0000-08565L"},{"key":"9777_CR18","doi-asserted-by":"crossref","unstructured":"Karu, M., & Alum\u00e4e, T. (2018). Weakly supervised training of speaker identification models. In: Proc. Speaker Odyssey, The Speaker and Language Recognition Workshop 2018","DOI":"10.21437\/Odyssey.2018-4"},{"key":"9777_CR19","doi-asserted-by":"publisher","DOI":"10.1007\/s10758-021-09522-5","volume-title":"Do teachers find dashboards trustworthy, actionable and useful?","author":"R Kasepalu","year":"2021","unstructured":"Kasepalu, R., Chejara, P., Prieto, L., & Ley, T. (2021). Do teachers find dashboards trustworthy, actionable and useful? A vignette study using a logs and audio dashboard: Technology, Knowledge and Learning. https:\/\/doi.org\/10.1007\/s10758-021-09522-5"},{"key":"9777_CR20","doi-asserted-by":"crossref","unstructured":"Ko, T., Peddinti, V., Povey, D., Seltzer, M.L., Khudanpur, S. (2017). A study on data augmentation of reverberant speech for robust speech recognition. In: Proc. ICASSP 2017","DOI":"10.1109\/ICASSP.2017.7953152"},{"key":"9777_CR21","doi-asserted-by":"publisher","unstructured":"Kukk, K., & Alum\u00e4e, T. (2022). Improving language identification of accented speech. In: Proc. Interspeech 2022, pp 1288\u20131292, https:\/\/doi.org\/10.21437\/Interspeech.2022-10455","DOI":"10.21437\/Interspeech.2022-10455"},{"key":"9777_CR22","unstructured":"K\u00e4ver, A. (2021). Efficient population based data augmentation in speaker verification. Master\u2019s thesis, Tallinn University of Technology"},{"issue":"1","key":"9777_CR23","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s41687-021-00288-z","volume":"5","author":"M Laissaar","year":"2021","unstructured":"Laissaar, M., Hallik, R., Sillaste, P., Ragun, U., P\u00e4rn, M. L., & Suija, K. (2021). Translation and cultural adaptation of IPOS (integrated palliative care outcome scale) in Estonia. Journal of Patient-Reported Outcomes, 5(1), 1\u201312. https:\/\/doi.org\/10.1186\/s41687-021-00288-z","journal-title":"Journal of Patient-Reported Outcomes"},{"key":"9777_CR24","first-page":"7152","volume":"2020","author":"S Laur","year":"2020","unstructured":"Laur, S., Orasmaa, S., S\u00e4rg, D., & Tammo, P. (2020). EstNLTK 1.6: Remastered Estonian NLP pipeline. Proc. LREC, 2020, 7152\u20137160.","journal-title":"Proc. LREC"},{"key":"9777_CR25","unstructured":"Lippus, P. (2011). The acoustic features and perception of the Estonian quantity system. PhD thesis, University of Tartu"},{"key":"9777_CR26","unstructured":"Lison, P., & Tiedemann, J. (2016). OpenSubtitles2016: Extracting large parallel corpora from movie and TV subtitles. In: Proc. LREC 2016"},{"key":"9777_CR27","unstructured":"Meignier, S., & Merlin, T. (2010). LIUM SpkDiarization: an open source toolkit for diarization. In: Proc. CMU SPUD Workshop"},{"key":"9777_CR28","doi-asserted-by":"publisher","unstructured":"Meister, E. (2021). A corpus of elderly estonian speech (under development). https:\/\/doi.org\/10.15155\/9-00-0000-0000-0000-00220L","DOI":"10.15155\/9-00-0000-0000-0000-00220L"},{"key":"9777_CR29","first-page":"45","volume":"2015","author":"E Meister","year":"2015","unstructured":"Meister, E., & Meister, L. (2015). Development and use of the Estonian L2 corpus. Proc. Workshop on Phonetic Learner Corpora, 2015, 45\u201347.","journal-title":"Proc. Workshop on Phonetic Learner Corpora"},{"key":"9777_CR30","unstructured":"Meister, E., Meister, L., & Metsvahi, R. (2012). New speech corpora at IoC. In: XXVII Fonetiikan p\u00e4iv\u00e4t"},{"key":"9777_CR31","unstructured":"Mieskes, M., Fort, K., N\u00e9v\u00e9ol, A., Grouin, C., & Cohen, K.B. (2019). NLP Community Perspectives on Replicability. In: Recent Advances in Natural Language Processing, Varna, Bulgaria"},{"key":"9777_CR32","unstructured":"Olev, A. (2019). Web application for authoring speech transcriptions. Master\u2019s thesis, Tallinn University of Technology"},{"key":"9777_CR33","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., Silovsky, J., Stemmer, G., Vesely, K. (2011a). The Kaldi speech recognition toolkit. In: Proc. ASRU 2011"},{"key":"9777_CR34","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motl\u00ed\u010dek, P., Qian, Y., Schwarz, P., Silovsk\u00fd, J., Stemmer, G., & Vesel, K. (2011b). The Kaldi speech recognition toolkit. In: Proc. ASRU 2011"},{"key":"9777_CR35","doi-asserted-by":"crossref","unstructured":"Povey, D., Cheng, G., Wang, Y., Li, K., Xu, H., Yarmohamadi, M., Khudanpur, S. (2018). Semi-orthogonal low-rank matrix factorization for deep neural networks. In: Proc. Interspeech 2018","DOI":"10.21437\/Interspeech.2018-1417"},{"key":"9777_CR36","unstructured":"Ravanelli, M., Parcollet, T., Plantinga, P., Rouhe, A., Cornell, S., Lugosch, L., Subakan, C., Dawalatabad, N., Heba, A., Zhong, J., Chou, J.C., Yeh, S.L., Fu, S.W., Liao, C.F., Rastorgueva, E., Grondin, F., Aris, W., Na, H., Gao, Y., Mori, R.D., & Bengio, Y. (2021). SpeechBrain: A general-purpose speech toolkit. ArXiv:2106.04624"},{"key":"9777_CR37","doi-asserted-by":"publisher","unstructured":"Rehm, G., & Way, A. (2023). European Language Equality: Introduction, Springer International Publishing, Cham, pp 1\u201310. https:\/\/doi.org\/10.1007\/978-3-031-28819-7_1, https:\/\/doi.org\/10.1007\/978-3-031-28819-7_1","DOI":"10.1007\/978-3-031-28819-7_1"},{"key":"9777_CR38","unstructured":"Reynaert, M., Van\u00a0Gompel, M., Sloot, K., Van\u00a0den Bosch, A. (2015). Piccl: Philosophical integrator of computational and corpus libraries. In: Proc. CLARIN Annual Conference 2015"},{"key":"9777_CR39","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1163\/26668912-bja10004","volume":"1","author":"A Saks","year":"2021","unstructured":"Saks, A. (2021). Digitalisation of work in Estonian parliament Riigikogu. International Journal of Parliamentary Studies, 1, 181\u2013188. https:\/\/doi.org\/10.1163\/26668912-bja10004","journal-title":"International Journal of Parliamentary Studies"},{"key":"9777_CR40","unstructured":"Synder, D., Chen, G., Povey, D. (2015). MUSAN: A music, speech, and noise corpus. ArXiv:1510.08484"},{"key":"9777_CR41","doi-asserted-by":"crossref","unstructured":"Tilk, O., & Alum\u00e4e, T. (2016). Bidirectional recurrent neural network with attention mechanism for punctuation restoration. In: Proc. Interspeech 2016","DOI":"10.21437\/Interspeech.2016-1517"},{"key":"9777_CR42","first-page":"652","volume":"2021","author":"J Valk","year":"2021","unstructured":"Valk, J., & Alum\u00e4e, T. (2021). VoxLingua107: a dataset for spoken language recognition. Proc. Spoken Language Technology Workshop (SLT), 2021, 652\u2013658.","journal-title":"Proc. Spoken Language Technology Workshop (SLT)"},{"key":"9777_CR43","doi-asserted-by":"publisher","unstructured":"Watanabe, S., Hori, T., Karita, S., Hayashi, T., Nishitoba, J., Unno, Y., Enrique, Yalta Soplin N., Heymann, J., Wiesner, M., Chen, N., Renduchintala, A., &Ochiai, T. (2018). ESPnet: End-to-end speech processing toolkit. In: Proc. Interspeech 2018, https:\/\/doi.org\/10.21437\/Interspeech.2018-1456","DOI":"10.21437\/Interspeech.2018-1456"},{"key":"9777_CR44","doi-asserted-by":"crossref","unstructured":"Xu, H., Li, K., Wang, Y., Wang, J., Kang, S., Chen, X., Povey, D., & Khudanpur, S. (2018). Neural network language modeling with letter-based features and importance sampling. In: Proc. ICASSP 2018","DOI":"10.1109\/ICASSP.2018.8461704"}],"container-title":["Language Resources and Evaluation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10579-024-09777-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10579-024-09777-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10579-024-09777-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T05:15:54Z","timestamp":1765257354000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10579-024-09777-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,16]]},"references-count":44,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,12]]}},"alternative-id":["9777"],"URL":"https:\/\/doi.org\/10.1007\/s10579-024-09777-1","relation":{},"ISSN":["1574-020X","1574-0218"],"issn-type":[{"value":"1574-020X","type":"print"},{"value":"1574-0218","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,16]]},"assertion":[{"value":"5 September 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 October 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration"}},{"value":"This work has been partially conducted in the project \u201cICT programme\u201d which was supported by the European Union through the European Social Fund. The study has been supported by the European Regional Development Fund (the project \u201cCentre of Excellence in Estonian Studies\u201d).","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"This platform was developed and deployed in accordance with the ethical guidelines set by Tallinn University of Technology. The system adheres to all relevant ethical standards for data handling and user privacy.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}]}}