{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T04:13:12Z","timestamp":1781151192941,"version":"3.54.1"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s10772-024-10141-5","type":"journal-article","created":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T14:02:42Z","timestamp":1730469762000},"page":"923-934","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Whisper for L2 speech scoring"],"prefix":"10.1007","volume":"27","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2179-1043","authenticated-orcid":false,"given":"Nicolas","family":"Ballier","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0576-0669","authenticated-orcid":false,"given":"Taylor","family":"Arnold","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Adrien","family":"M\u00e9li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tori","family":"Thurston","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0291-4263","authenticated-orcid":false,"given":"Jean-Baptiste","family":"Yun\u00e8s","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"10141_CR1","unstructured":"Aks\u00ebnova, A., Chen, Z., Chiu, C.-C., Esch, D., Golik, P., Han, W., King, L., Ramabhadran, B., Rosenberg, A., Schwartz, S., & Wang, G. (2022). Accented speech recognition: Benchmarking, pre-training, and diverse data. arXiv preprint. arXiv:2205.08014"},{"issue":"1","key":"10141_CR2","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1121\/1.5017834","volume":"143","author":"V Arora","year":"2018","unstructured":"Arora, V., Lahiri, A., & Reetz, H. (2018). Phonological feature-based speech recognition system for pronunciation training in non-native language learning. The Journal of the Acoustical Society of America, 143(1), 98\u2013108.","journal-title":"The Journal of the Acoustical Society of America"},{"key":"10141_CR3","first-page":"5","volume":"27","author":"E Atwell","year":"2003","unstructured":"Atwell, E., Howarth, P., & Souter, D. (2003). The isle corpus: Italian and German spoken learner\u2019s English. ICAME Journal: International Computer Archive of Modern and Medieval English Journal, 27, 5\u201318.","journal-title":"ICAME Journal: International Computer Archive of Modern and Medieval English Journal"},{"key":"10141_CR4","first-page":"12449","volume":"33","author":"A Baevski","year":"2020","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., & Auli, M. (2020). wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in Neural Information Processing Systems, 33, 12449\u201312460.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10141_CR5","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1017\/CBO9781139649414.006","volume-title":"The Cambridge handbook of learner corpus research","author":"N Ballier","year":"2015","unstructured":"Ballier, N., & Martin, P. (2015). Speech annotation of learner corpora. In S. Granger, G. Gilquin, & F. Meunier (Eds.), The Cambridge handbook of learner corpus research (pp. 107\u2013134). Cambridge University Press."},{"key":"10141_CR6","unstructured":"Ballier, N., M\u00e9li, A., Amand, M., & Yun\u00e8s, J.-B. (2023). Using whisper llm for automatic phonetic diagnosis of L2 speech, a case study with French learners of English. In Proceedings of the 6th international conference on natural language and speech processing (ICNLSP 2023) (pp. 282\u2013292)."},{"key":"10141_CR7","unstructured":"Ballier, N., Namdarzadeh, B., & Zimina-Poirot, M. (2023). Translating dislocations or parentheticals: Investigating the role of prosodic boundaries for spoken language translation from French into English. In Machine translation summit 2023 (Vol. 19, pp. 119\u2013131)."},{"key":"10141_CR8","unstructured":"Casanova, E., Weber, J., Shulby, C. D., Junior, A. C., G\u00f6lge, E., & Ponti, M. A. (2022). YourTTS: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone. In International conference on machine learning (pp. 2709\u2013 2720)."},{"key":"10141_CR9","doi-asserted-by":"publisher","unstructured":"Chan, M. P. Y., Choe, J., Li, A., Chen, Y., Gao, X., & Holliday, N. (2022). Training and typological bias in ASR performance for world Englishes. In Proceedings of Interspeech, 2022 (pp. 1273\u20131277). https:\/\/doi.org\/10.21437\/Interspeech.2022-10869","DOI":"10.21437\/Interspeech.2022-10869"},{"key":"10141_CR10","doi-asserted-by":"crossref","unstructured":"Chanethom, V., & Henderson, A. (2022). Alignment in ASR and L1 listeners\u2019 recognition of L2 learner speech: A replication study. In 15th International conference on native and non-native accents of English, Universit\u00e9 de \u0141\u00f3d\u017a, \u0141\u00f3d\u017a, Poland. https:\/\/hal.science\/hal-03929160","DOI":"10.18778\/1731-7533.21.3.03"},{"issue":"3","key":"10141_CR11","doi-asserted-by":"publisher","first-page":"425","DOI":"10.1558\/cj.v16i3.425-445","volume":"16","author":"J Dalby","year":"1999","unstructured":"Dalby, J., & Kewley-Port, D. (1999). Explicit pronunciation training using automatic speech recognition technology. CALICO, 16(3), 425\u2013445.","journal-title":"CALICO"},{"key":"10141_CR12","unstructured":"Gerganov, G. (2003). whisper.cpp: A high-performance inference of OpenAI\u2019s whisper automatic speech recognition (ASR) model."},{"issue":"1","key":"10141_CR21","doi-asserted-by":"publisher","first-page":"89","DOI":"10.1017\/S0958344022000192","volume":"35","author":"S Inceoglu","year":"2023","unstructured":"Inceoglu, S., Chen, W.-H., & Lim, H. (2023). Assessment of L2 intelligibility: Comparing l1 listeners and automatic speech recognition. ReCALL, 35(1), 89\u2013104. https:\/\/doi.org\/10.1017\/S0958344022000192","journal-title":"ReCALL"},{"issue":"3","key":"10141_CR13","first-page":"824","volume":"17","author":"S Inceoglu","year":"2020","unstructured":"Inceoglu, S., Lim, H., & Chen, W.-H. (2020). ASR for EFL pronunciation practice: Segmental development and learners\u2019 beliefs. The Journal of Asia TEFL, 17(3), 824\u2013840.","journal-title":"The Journal of Asia TEFL"},{"key":"10141_CR14","doi-asserted-by":"crossref","unstructured":"Islam, E., Park, C., & Hain, T. (2023). Exploring speech representations for proficiency assessment in language learning. In 9th Workshop on speech and language technology in education (SLaTE) proceedings (pp. 151\u2013 155). International Speech Communication Association (ISCA).","DOI":"10.21437\/SLaTE.2023-29"},{"key":"10141_CR15","doi-asserted-by":"publisher","unstructured":"Javed, T., Joshi, S., Nagarajan, V., Sundaresan, S., Nawale, J., Raman, A., Bhogale, K., Kumar, P., & Khapra, M. M. (2023). Svarah: Evaluating English ASR systems on Indian accents. In Proceedings of Interspeech  (pp. 5087\u2013 5091). https:\/\/doi.org\/10.21437\/Interspeech.2023-2588","DOI":"10.21437\/Interspeech.2023-2588"},{"key":"10141_CR16","unstructured":"Jiang, Z., Ren, Y., Ye, Z., Liu, J., Zhang, C., Yang, Q., Ji, S., Huang, R., Wang, C., Yin, X., Ma, Z., & Zhao, Z. (2023). Mega-TTS: Zero-shot text-to-speech at scale with intrinsic inductive bias. arXiv:2306.03509v1"},{"key":"10141_CR17","first-page":"707","volume":"10","author":"V Levenshtein","year":"1966","unstructured":"Levenshtein, V. (1966). Binary codes capable of correcting deletions, insertions, and reversals. Soviet Physics Doklady, 10, 707\u2013710.","journal-title":"Soviet Physics Doklady"},{"key":"10141_CR18","doi-asserted-by":"publisher","unstructured":"Levinstein, B. A.,  & Herrmann, D. A. (2024). Still no lie detector for language models: Probing empirical and conceptual roadblocks. Philosophical Studies. https:\/\/doi.org\/10.1007\/s11098-023-02094-3","DOI":"10.1007\/s11098-023-02094-3"},{"key":"10141_CR19","unstructured":"Martin, A., Daniel, E., & Ward, N. (1998). The use of the word error rate for evaluating automatic speech recognition systems. Proceedings of the IEEE International conference on acoustics, speech, and signal processing (Vol. 1, pp. 77\u201380)."},{"key":"10141_CR20","unstructured":"Menzel, W., Atwell, E., Bonaventura, P., Herron, D., Howarth, P., Morton, R., & Souter, C. (2000). The ISLE corpus of non-native spoken English. In Proceedings of LREC 2000: Language resources and evaluation conference (Vol. 2,  pp. 957\u2013964). European Language Resources Association."},{"key":"10141_CR22","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., Silovsk\u00fd, J., Stemmer, G., & Vesel\u00fd, K. (2011). The Kaldi speech recognition toolkit. In IEEE 2011 workshop on automatic speech recognition and understanding. IEEE Signal Processing Society. https:\/\/infoscience.epfl.ch\/record\/192584\/files\/Povey_ASRU2011_2011.pdf"},{"key":"10141_CR27","unstructured":"R Core Team. (2024). R: A language and environment for statistical computing [computer software manual]. R Core Team."},{"key":"10141_CR25","doi-asserted-by":"publisher","unstructured":"Radford, A., Kim, J. W., Xu, T., Brockman, G., McLeavey, C., & Sutskever, I. (2022). Robust speech recognition via large-scale weak supervision. https:\/\/doi.org\/10.48550\/arXiv.2212.04356","DOI":"10.48550\/arXiv.2212.04356"},{"key":"10141_CR23","unstructured":"Radford, A., Kim, J. W., Xu, T., Brockman, G., McLeavey, C., & Sutskever, I. (2023). Robust speech recognition via large-scale weak supervision. In International conference on machine learning (pp. 28492\u201328518)."},{"issue":"5","key":"10141_CR26","doi-asserted-by":"publisher","first-page":"3348","DOI":"10.1121\/1.410623","volume":"96","author":"CL Rogers","year":"1994","unstructured":"Rogers, C. L., Dalby, J. M., & DeVane, G. (1994). Intelligibility training for foreign-accented speech: A preliminary study. JASA, 96(5), 3348. https:\/\/doi.org\/10.1121\/1.410623","journal-title":"JASA"},{"key":"10141_CR28","doi-asserted-by":"publisher","unstructured":"Tortel, A., & Hirst, D. (2010). Rhythm metrics and the production of English L1\/L2. In Proceedings of speech prosody 2010 (p. 959). https:\/\/doi.org\/10.21437\/SpeechProsody.2010-49","DOI":"10.21437\/SpeechProsody.2010-49"},{"issue":"2","key":"10141_CR36","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1044\/jshr.3202.245","volume":"32","author":"CS Watson","year":"1989","unstructured":"Watson, C. S., Reed, D. J., Kewley-Port, D., & Maki, D. (1989). The Indiana Speech Training Aid (ISTRA) I: Comparisons between human and computer-based evaluation of speech quality.\nJournal of Speech, Language, and Hearing Research, 32(2), 245\u2013251.","journal-title":"Journal of Speech, Language, and Hearing Research"},{"key":"10141_CR29","unstructured":"Weinberger, S. (2015). Speech accent archive. George Mason University. http:\/\/accent.gmu.edu"},{"issue":"2\u20133","key":"10141_CR30","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1016\/S0167-6393(99)00044-8","volume":"30","author":"SM Witt","year":"2000","unstructured":"Witt, S. M., & Young, S. J. (2000). Phone-level pronunciation scoring and assessment for interactive language learning. Speech Communication, 30(2\u20133), 95\u2013108.","journal-title":"Speech Communication"},{"key":"10141_CR31","unstructured":"Zhang, Z., Zhou, L., Wang, C., Chen, S., Wu, Y., Liu, S., Chen, Z., Liu, Y., Wang, H., Li, J., He, L., Zhao, S., & Wei, F. (2023). Speak foreign languages with your own voice: Cross-lingual neural codec language modeling. arXiv preprint. arXiv:2303.03926"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10141-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10141-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10141-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T10:09:51Z","timestamp":1734343791000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10141-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"references-count":31,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["10141"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10141-5","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 September 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 November 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}