{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T09:11:31Z","timestamp":1784797891461,"version":"3.55.0"},"reference-count":26,"publisher":"Walter de Gruyter GmbH","issue":"1","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,1,23]]},"abstract":"<jats:title>Abstract<\/jats:title>\n                  <jats:p>Building reliable automatic speech recognition (ASR) systems for low-resource languages such as Kannada is challenging due to the scarcity of training data and the cross-lingual translation requirements. This study investigates fine-tuning Whisper models \u2013 Tiny, Small, and Medium \u2013 to achieve end-to-end Kannada-to-English speech transcription. Using a dataset of Kannada speech, the multilingual Whisper models are fine-tuned to improve transcription and translation performance. The accuracy of the fine-tuned models was evaluated using word error rate (WER) and character error rate (CER), revealing clear improvements across the different model sizes. Based on the analysis of the experimental results, the Whisper-Medium model was found to perform the best, with significant reductions in WER and CER compared to the zero-shot baselines. The results highlight the feasibility of lightweight and midsize Whisper architectures for cross-lingual ASR in low-resource Indian languages and indicate that these models are more suitable for real-world applications in multilingual speech systems.<\/jats:p>","DOI":"10.1515\/comp-2025-0060","type":"journal-article","created":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T08:57:44Z","timestamp":1784797064000},"source":"Crossref","is-referenced-by-count":0,"title":["Fine-tuning Whisper model for\u00a0end-to-end Kannada to\u00a0English speech transcription"],"prefix":"10.1515","volume":"16","author":[{"given":"Raghurama","family":"Holla","sequence":"first","affiliation":[{"name":"Manipal Institute of Technology , Manipal Academy of Higher Education , Manipal , India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"R.","family":"Agam","sequence":"additional","affiliation":[{"name":"Department of CSE , NMAMIT, Nitte , Karkala , India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bianca Gelesia","family":"Martis","sequence":"additional","affiliation":[{"name":"Department of ISE , NMAMIT, Nitte , Karkala , India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Britney Genelia","family":"Martis","sequence":"additional","affiliation":[{"name":"Department of CSE , NMAMIT, Nitte , Karkala , India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"S.N.","family":"Muralikrishna","sequence":"additional","affiliation":[{"name":"Manipal Institute of Technology , Manipal Academy of Higher Education , Manipal , India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Raghavendra","family":"Ganiga","sequence":"additional","affiliation":[{"name":"Manipal Institute of Technology , Manipal Academy of Higher Education , Manipal , India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"374","published-online":{"date-parts":[[2026,7,23]]},"reference":[{"key":"2026072308573865790_j_comp-2025-0060_ref_001","unstructured":"A. Radford, J. W. Kim, T. Xu, G. Brockman, C. McLeavey, and I. Sutskever, \u201cRobust speech recognition via large-scale weak supervision,\u201d in Proceedings of the 40th International Conference on Machine Learning, ser. Proceedings of Machine Learning Research, vol.\u00a0202, PMLR, 2023, pp.\u00a028492\u201328518."},{"key":"2026072308573865790_j_comp-2025-0060_ref_002","doi-asserted-by":"crossref","unstructured":"A. Diwan et al.., \u201cMultilingual and code-switching ASR challenges for low resource Indian languages,\u201d arXiv preprint arXiv:2104.00235, 2021.","DOI":"10.21437\/Interspeech.2021-1339"},{"key":"2026072308573865790_j_comp-2025-0060_ref_003","doi-asserted-by":"crossref","unstructured":"Y. Liu, X. Yang, and D. Qu, \u201cExploration of whisper fine-tuning strategies for low-resource ASR,\u201d EURASIP J. Audio Speech Music Process., vol.\u00a02024, no.\u00a01, p.\u00a029, 2024, https:\/\/doi.org\/10.1186\/s13636-024-00349-3.","DOI":"10.1186\/s13636-024-00349-3"},{"key":"2026072308573865790_j_comp-2025-0060_ref_004","unstructured":"L. G. Pillai, K. Manohar, B. K. Raju, and E. Sherly, \u201cMultistage fine-tuning strategies for automatic speech recognition in low-resource languages,\u201d arXiv preprint arXiv:2411.04573, 2024."},{"key":"2026072308573865790_j_comp-2025-0060_ref_005","doi-asserted-by":"crossref","unstructured":"K. Tripathi, R. Gothi, and P. Wasnik, \u201cEnhancing whisper\u2019s accuracy and speed for Indian languages through prompt-tuning and tokenization,\u201d in ICASSP 2025 \u2013 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, 2025, pp.\u00a01\u20135.","DOI":"10.1109\/ICASSP49660.2025.10887595"},{"key":"2026072308573865790_j_comp-2025-0060_ref_006","doi-asserted-by":"crossref","unstructured":"P. Guo, X. Chang, H. Lv, S. Watanabe, and L. Xie, \u201cSQ-Whisper: Speaker-querying based whisper model for target-speaker ASR,\u201d IEEE Trans. Audio Speech Lang. Process., vol.\u00a033, pp.\u00a0175\u2013185, 2025, https:\/\/doi.org\/10.1109\/taslp.2024.3513835.","DOI":"10.1109\/TASLP.2024.3513835"},{"key":"2026072308573865790_j_comp-2025-0060_ref_007","doi-asserted-by":"crossref","unstructured":"H.-J. Chang, H. Gong, C. Wang, J. Glass, and Y.-A. Chung, \u201cDC-Spin: A speaker-invariant speech tokenizer for spoken language models,\u201d arXiv preprint arXiv:2410.24177, 2024.","DOI":"10.21437\/Interspeech.2025-246"},{"key":"2026072308573865790_j_comp-2025-0060_ref_008","doi-asserted-by":"crossref","unstructured":"A. Babu et al.., \u201cXLS-R: Self-supervised cross-lingual speech representation learning at scale,\u201d arXiv preprint arXiv:2111.09296, 2021.","DOI":"10.21437\/Interspeech.2022-143"},{"key":"2026072308573865790_j_comp-2025-0060_ref_009","doi-asserted-by":"crossref","unstructured":"A. Conneau, A. Baevski, R. Collobert, A. Mohamed, and M. Auli, \u201cUnsupervised cross-lingual representation learning for speech recognition,\u201d arXiv preprint arXiv:2006.13979, 2020.","DOI":"10.21437\/Interspeech.2021-329"},{"key":"2026072308573865790_j_comp-2025-0060_ref_010","doi-asserted-by":"crossref","unstructured":"D. Mach\u00e1\u010dek, R. Dabre, and O. Bojar, \u201cTurning whisper into real-time transcription system,\u201d in Proceedings of the 13th International Joint Conference on Natural Language Processing and the 3rd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics: System Demonstrations, Association for Computational Linguistics, 2023, pp.\u00a017\u201324.","DOI":"10.18653\/v1\/2023.ijcnlp-demo.3"},{"key":"2026072308573865790_j_comp-2025-0060_ref_011","doi-asserted-by":"crossref","unstructured":"R. Jairam, G. Jyothish, and B. Premjith, \u201cA few-shot multi-accented speech classification for Indian languages using transformers and LLM\u2019s fine-tuning approaches,\u201d in Proceedings of the Fourth Workshop on Speech, Vision, and Language Technologies for Dravidian Languages, 2024, pp.\u00a01\u20139.","DOI":"10.18653\/v1\/2024.dravidianlangtech-1.1"},{"key":"2026072308573865790_j_comp-2025-0060_ref_012","doi-asserted-by":"crossref","unstructured":"D. E. Chowdary, G. S. V. Kumar, A. Ravi, M. Pawan, and G. Praveen, \u201cTransformer-based multilingual automatic speech recognition (ASR) model for Dravidian languages,\u201d in Automatic Speech Recognition and Translation for Low Resource Languages, Springer, 2024, pp.\u00a0259\u2013273.","DOI":"10.1002\/9781394214624.ch13"},{"key":"2026072308573865790_j_comp-2025-0060_ref_013","unstructured":"T. P. Ferraz, \u201cEfficient compression of multitask multilingual speech models,\u201d arXiv preprint arXiv:2405.00966, 2024."},{"key":"2026072308573865790_j_comp-2025-0060_ref_014","doi-asserted-by":"crossref","unstructured":"R. E. Zezario, Y. E. Raharjo, A. K. Nugraha, and A. Purwarianti, \u201cA study on incorporating Whisper for robust speech assessment,\u201d in Proceedings of the 2024 IEEE International Conference on Multimedia and Expo (ICME), IEEE, 2024.","DOI":"10.1109\/ICME57554.2024.10688047"},{"key":"2026072308573865790_j_comp-2025-0060_ref_015","doi-asserted-by":"crossref","unstructured":"K. Padmanandam, U. Shivani, T. Pooja, P. Girija, and G. Vaishnavi, \u201cSpoken language identification and translation using deep learning,\u201d in 2023 7th International Conference on Electronics, Communication and Aerospace Technology (ICECA), IEEE, 2023, pp.\u00a0535\u2013540.","DOI":"10.1109\/ICECA58529.2023.10395377"},{"key":"2026072308573865790_j_comp-2025-0060_ref_016","doi-asserted-by":"crossref","unstructured":"H. Ma, X. Guo, H.-Y. Lee, Z. Meng, and S.-W. Lin, \u201cExtending Whisper with prompt tuning to target-speaker ASR,\u201d in ICASSP 2024 \u2013 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, 2024, pp.\u00a0859\u2013863.","DOI":"10.1109\/ICASSP48485.2024.10447492"},{"key":"2026072308573865790_j_comp-2025-0060_ref_017","unstructured":"V. Timmel, C. Paonessa, R. Kakooee, M. Vogel, and D. Perruchoud, \u201cFine-tuning Whisper on low-resource languages for real-world applications,\u201d arXiv preprint arXiv:2412.15726, 2024."},{"key":"2026072308573865790_j_comp-2025-0060_ref_018","doi-asserted-by":"crossref","unstructured":"J. Philip, A. Kunchukuttan, and P. Bhattacharyya, \u201cRevisiting low resource status of Indian languages in machine translation,\u201d in Proceedings of the 3rd ACM India Joint International Conference on Data Science & Management of Data (8th ACM IKDD CODS & 26th COMAD), ACM, 2021, pp.\u00a091\u2013100.","DOI":"10.1145\/3430984.3431026"},{"key":"2026072308573865790_j_comp-2025-0060_ref_019","doi-asserted-by":"crossref","unstructured":"S. Srivatsa and S. Srinivasa, \u201cKnowledge management framework over low resource Indian colloquial language audio contents,\u201d in Proceedings of the 7th Joint International Conference on Data Science & Management of Data (11th ACM IKDD CODS and 29th COMAD), ACM, 2024.","DOI":"10.1145\/3632410.3632483"},{"key":"2026072308573865790_j_comp-2025-0060_ref_020","unstructured":"H. F. Schiffman, \u201cA grammar of Kannada: Phonology, morphophonemics, and brief morphology,\u201d Manuscript\/Unpublished Working Paper, 1970, pDF retrieved as \u201cKaGram1.pdf\u201d; referenced September 2025."},{"key":"2026072308573865790_j_comp-2025-0060_ref_021","unstructured":"V. Team, \u201cVAANI: Capturing the language landscape for an inclusive digital India (phase 1),\u201d 2025. Available at: https:\/\/vaani.iisc.ac.in\/."},{"key":"2026072308573865790_j_comp-2025-0060_ref_022","doi-asserted-by":"crossref","unstructured":"S. R. Aithal, S. N. Muralikrishna, R. Ganiga, A. B. Rao, and G. K. Hegde, \u201cKannadaLex: A lexical database with psycholinguistic information,\u201d ACM Trans. Asian Low-Resour. Lang. Inf. Process., vol.\u00a023, no.\u00a07, 2024, https:\/\/doi.org\/10.1145\/3670688.","DOI":"10.1145\/3670688"},{"key":"2026072308573865790_j_comp-2025-0060_ref_023","doi-asserted-by":"crossref","unstructured":"C. Toraman, E. H. Yilmaz, F. \u015eahinu\u00e7, and O. Ozcelik, \u201cImpact of tokenization on language models: An analysis for Turkish,\u201d ACM Trans. Asian Low-Resour. Lang. Inf. Process., vol.\u00a022, no.\u00a04, pp.\u00a01\u201321, 2023, https:\/\/doi.org\/10.1145\/3578707.","DOI":"10.1145\/3578707"},{"key":"2026072308573865790_j_comp-2025-0060_ref_024","unstructured":"A. Conneau et al.., \u201cFLEURS: Few-shot learning evaluation of universal representations of speech,\u201d 2022, https:\/\/arxiv.org\/abs\/2205.12446."},{"key":"2026072308573865790_j_comp-2025-0060_ref_025","doi-asserted-by":"crossref","unstructured":"R. Sennrich, B. Haddow, and A. Birch, \u201cNeural machine translation of rare words with subword units,\u201d in Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), Association for Computational Linguistics, 2016, pp.\u00a01715\u20131725.","DOI":"10.18653\/v1\/P16-1162"},{"key":"2026072308573865790_j_comp-2025-0060_ref_026","doi-asserted-by":"crossref","unstructured":"T. Kudo and J. Richardson, \u201cSentencePiece: A simple and language independent subword tokenizer and detokenizer for neural text processing,\u201d in Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, Association for Computational Linguistics, 2018, pp.\u00a066\u201371.","DOI":"10.18653\/v1\/D18-2012"}],"container-title":["Open Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.degruyterbrill.com\/document\/doi\/10.1515\/comp-2025-0060\/xml","content-type":"application\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.degruyterbrill.com\/document\/doi\/10.1515\/comp-2025-0060\/pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T08:58:01Z","timestamp":1784797081000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.degruyterbrill.com\/document\/doi\/10.1515\/comp-2025-0060\/html"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,1]]},"references-count":26,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,7,23]]},"published-print":{"date-parts":[[2026,1,23]]}},"alternative-id":["10.1515\/comp-2025-0060"],"URL":"https:\/\/doi.org\/10.1515\/comp-2025-0060","relation":{},"ISSN":["2299-1093"],"issn-type":[{"value":"2299-1093","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,1]]},"article-number":"20250060"}}