{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T14:58:47Z","timestamp":1782399527841,"version":"3.54.5"},"reference-count":29,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,7,27]],"date-time":"2024-07-27T00:00:00Z","timestamp":1722038400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,27]],"date-time":"2024-07-27T00:00:00Z","timestamp":1722038400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001843","name":"Science & Engineering Research Board","doi-asserted-by":"crossref","award":["CRG\/2021\/000930"],"award-info":[{"award-number":["CRG\/2021\/000930"]}],"id":[{"id":"10.13039\/501100001843","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s10772-024-10132-6","type":"journal-article","created":{"date-parts":[[2024,7,27]],"date-time":"2024-07-27T12:01:55Z","timestamp":1722081715000},"page":"717-728","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":16,"title":["An automatic speech recognition system in Odia language using attention mechanism and data augmentation"],"prefix":"10.1007","volume":"27","author":[{"given":"Malay Kumar","family":"Majhi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7275-7980","authenticated-orcid":false,"given":"Sujan Kumar","family":"Saha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,7,27]]},"reference":[{"key":"10132_CR1","doi-asserted-by":"publisher","first-page":"165","DOI":"10.1007\/s10772-012-9131-y","volume":"15","author":"RK Aggarwal","year":"2012","unstructured":"Aggarwal, R. K., & Dave, M. (2012). Integration of multiple acoustic and language models for improved Hindi speech recognition system. International Journal of Speech Technology, 15, 165\u2013180. https:\/\/doi.org\/10.1007\/s10772-012-9131-y","journal-title":"International Journal of Speech Technology"},{"key":"10132_CR2","doi-asserted-by":"publisher","first-page":"1457","DOI":"10.1007\/s11235-011-9623-0","volume":"52","author":"RK Aggarwal","year":"2013","unstructured":"Aggarwal, R. K., & Dave, M. (2013). Performance evaluation of sequentially combined heterogeneous feature streams for Hindi speech recognition system. Telecommunication Systems, 52, 1457\u20131466. https:\/\/doi.org\/10.1007\/s11235-011-9623-0","journal-title":"Telecommunication Systems"},{"key":"10132_CR3","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3112535","author":"S Alharbi","year":"2021","unstructured":"Alharbi, S., Alrazgan, M., Alrashed, A., Alnomasi, T., Almojel, R., Alharbi, R., Alharbi, S., Alturki, S., Alshehri, F., & Almojil, M. (2021). Automatic speech recognition: Systematic literature review. IEEE Access. https:\/\/doi.org\/10.1109\/ACCESS.2021.3112535","journal-title":"IEEE Access"},{"key":"10132_CR4","doi-asserted-by":"publisher","unstructured":"Anoop, S. C., & Ramakrishnan, A. G. (2021). CTC-based end-to-end ASR for the low resource sanskrit language with spectrogram augmentation. In 2021 national conference on communications (NCC). IEEE. https:\/\/doi.org\/10.1109\/ncc52529.2021.9530162","DOI":"10.1109\/ncc52529.2021.9530162"},{"key":"10132_CR5","unstructured":"Chadha, H. S., Shah, P., Dhuriya, A., Chhimwal, N., Gupta, A., & Raghavan, V. (2022). Code switched and code mixed speech recognition for Indic languages. arXiv preprint arXiv:2203.16578."},{"key":"10132_CR6","doi-asserted-by":"publisher","unstructured":"Das, B., Mandal, S., & Mitra, P. (2011). Bengali speech corpus for continuous automatic speech recognition system. In 2011 international conference on speech database and assessments (Oriental COCOSDA), Hsinchu, Taiwan (pp. 51\u201355). https:\/\/doi.org\/10.1109\/ICSDA.2011.6085979","DOI":"10.1109\/ICSDA.2011.6085979"},{"key":"10132_CR7","doi-asserted-by":"publisher","first-page":"2446","DOI":"10.21437\/Interspeech.2021-1339","volume":"2021","author":"A Diwan","year":"2021","unstructured":"Diwan, A., Vaideeswaran, R., Shah, S., Singh, A., Raghavan, S., Khare, S., Unni, V., Vyas, S., Rajpuria, A., Yarra, C., Mittal, A., Ghosh, P. K., Jyothi, P., Bali, K., Seshadri, V., Sitaram, S., Bharadwaj, S., Nanavati, J., Nanavati, R., & Sankaranarayanan, K. (2021). MUCS 2021: Multilingual and code-switching ASR challenges for low resource Indian languages. In\u00a0Proceedings of Interspeech 2021, (pp. 2446\u20132450). https:\/\/doi.org\/10.21437\/Interspeech.2021-1339","journal-title":"Proceedings of Interspeech"},{"key":"10132_CR8","doi-asserted-by":"crossref","unstructured":"Fathima, N., Patel, T., Mahima, C., & Iyengar, A. (2018). TDNN-based multilingual speech recognition system for low resource Indian languages. In Interspeech (pp. 3197\u20133201).","DOI":"10.21437\/Interspeech.2018-2117"},{"key":"10132_CR9","doi-asserted-by":"publisher","unstructured":"Karan, B., Sahoo, J., & Sahu, P. K. (2015). Automatic speech recognition based Odia system. In Proceedings of the international conference on microwave, optical and communication engineering (ICMOCE), Bhubaneswar, India. https:\/\/doi.org\/10.1109\/ICMOCE.2015.7489765","DOI":"10.1109\/ICMOCE.2015.7489765"},{"issue":"1\u20132","key":"10132_CR10","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1016\/S0167-6393(01)00041-3","volume":"38","author":"D Klakow","year":"2002","unstructured":"Klakow, D., & Jochen, P. (2002). Testing the correlation of word error rate and perplexity. Speech Communication, 38(1\u20132), 19\u201328.","journal-title":"Speech Communication"},{"key":"10132_CR11","unstructured":"Krishna, D. N. (2021). A dual-decoder conformer for multilingual speech recognition. arXiv preprint arXiv:2109.03277."},{"key":"10132_CR12","unstructured":"Liu, C., Zhang, Q., Zhang, X., Singh, K., Saraf, Y., & Zweig, G. (2020). Multilingual graphemic hybrid ASR with massive data augmentation. In D. Beermann, L. Besacier, S. Sakti, & C. Soria (Eds.), Proceedings of the 1st joint workshop on spoken language technologies for under-resourced languages (SLTU) and collaboration and computing for under-resourced languages (CCURL) (pp. 46\u201352)."},{"issue":"11","key":"10132_CR13","doi-asserted-by":"publisher","first-page":"1946","DOI":"10.1109\/TASLP.2016.2593800","volume":"24","author":"Y Liu","year":"2016","unstructured":"Liu, Y., & Kirchoff, K. (2016). Graph-based semisupervised learning for acoustic modeling in automatic speech recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 24(11), 1946\u20131956.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10132_CR14","unstructured":"Mirishkar, G., Yadavalli, A., & Vuppala, A. K. (2021). An investigation of hybrid architectures for low resource multilingual speech recognition system in Indian context. In 18th international conference on natural language processing (ICON) (pp. 205\u2013212). Silchar, India. https:\/\/aclanthology.org\/2021.icon-main.25."},{"key":"10132_CR15","unstructured":"Naman, A., & Deepshikha, K. (2021). Indic languages automatic speech recognition using meta-learning approach. In Proceedings of the 4th international conference on natural language and speech processing (ICNLSP 2021) (pp. 219\u2013225). Trento, Italy: Association for Computational Linguistics."},{"key":"10132_CR16","doi-asserted-by":"publisher","unstructured":"Nguyen, T. S., St\u00fcker, S., Niehues, J., & Waibel, A. (2020). Improving sequence-to-sequence speech recognition training with on-the-fly data augmentation. In 2020 IEEE international conference on acoustics, speech and signal processing (ICASSP 2020) (pp. 7689\u20137693). IEEE. https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9054130","DOI":"10.1109\/ICASSP40776.2020.9054130"},{"key":"10132_CR17","doi-asserted-by":"crossref","unstructured":"Ochiai, T., Watanabe, S., Hori, T., Hershey, J., & Xiao, X. (2017). Unified architecture for multichannel end-to-end speech recognition with neural beamforming. IEEE Journal of Selected Topics in Signal Processing, 11(8), 1274.","DOI":"10.1109\/JSTSP.2017.2764276"},{"key":"10132_CR18","doi-asserted-by":"crossref","unstructured":"Park, D.S., Chan, W., Zhang, Y., Chiu, C.-C., Zoph, B., Cubuk, E. D., & Le, Q. V. (2019). Specaugment: A simple data augmentation method for automatic speech recognition. In Interspeech 2019 (pp. 2613\u20132617).","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"10132_CR19","doi-asserted-by":"publisher","unstructured":"Paul, A. K., Das, D., & Kamal, M. M. (2009). Bangla speech recognition system using LPC and ANN. In 2009 7th international conference on advances in pattern recognition (pp. 171\u2013174). Kolkata, India. https:\/\/doi.org\/10.1109\/ICAPR.2009.80","DOI":"10.1109\/ICAPR.2009.80"},{"key":"10132_CR20","doi-asserted-by":"publisher","first-page":"394","DOI":"10.1109\/TASLP.2022.3140552","volume":"30","author":"Y Qian","year":"2022","unstructured":"Qian, Y., & Zhou, Z. (2022). Optimizing data usage for low-resource speech recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 30, 394\u2013403.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"issue":"3","key":"10132_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3483446","volume":"21","author":"D Raval","year":"2022","unstructured":"Raval, D., Pathak, V., Patel, M., & Bhatt, B. (2022). Improving deep learning based automatic speech recognition for Gujarati. ACM Transactions on Asian and Low-Resource Language Information Processing, 21(3), 1\u201318.","journal-title":"ACM Transactions on Asian and Low-Resource Language Information Processing"},{"key":"10132_CR22","doi-asserted-by":"crossref","unstructured":"Renduchintala, A., Ding, S., Wiesner, M., & Watanabe, S. (2018). Multi-modal data augmentation for end-to-end ASR. In Interspeech 2018. Johns Hopkins University.","DOI":"10.21437\/Interspeech.2018-2456"},{"key":"10132_CR23","doi-asserted-by":"crossref","unstructured":"Saraswathi, S., & Geetha, T. V. (2004). Implementation of Tamil speech recognition system using neural networks. In Lecture notes in computer science (Vol. 3285).","DOI":"10.1007\/978-3-540-30176-9_22"},{"key":"10132_CR24","doi-asserted-by":"publisher","first-page":"16173","DOI":"10.1007\/s11042-022-14019-z","volume":"82","author":"U Sharma","year":"2023","unstructured":"Sharma, U., Om, H., & Mishra, A. N. (2023). HindiSpeech-Net: A deep learning based robust automatic speech recognition system for Hindi language. Multimedia Tools and Applications, 82, 16173\u201316193. https:\/\/doi.org\/10.1007\/s11042-022-14019-z","journal-title":"Multimedia Tools and Applications"},{"key":"10132_CR25","doi-asserted-by":"publisher","unstructured":"Srivastava, B. M. L., Sitaram, S., Mehta, R. K., Mohan, K. D., Matani, P., Satpal, S., Bali, K., Srikanth, R., & Nayak, N. (2018). Interspeech 2018 low resource automatic speech recognition challenge for Indian languages. In Proceedings of the 6th workshop on spoken language technologies for under-resourced languages (SLTU 2018) (pp. 11\u201314). https:\/\/doi.org\/10.21437\/SLTU.2018-3","DOI":"10.21437\/SLTU.2018-3"},{"key":"10132_CR26","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1007\/s10772-009-9058-0","volume":"12","author":"R Thangarajan","year":"2009","unstructured":"Thangarajan, R., Natarajan, A. M., & Selvam, M. (2009). Syllable modeling in continuous speech recognition for Tamil language. International Journal of Speech Technology, 12, 47\u201357. https:\/\/doi.org\/10.1007\/s10772-009-9058-0","journal-title":"International Journal of Speech Technology"},{"key":"10132_CR27","doi-asserted-by":"crossref","unstructured":"Toshniwal, S., Sainath, T. N., Weiss, R. J., Li, B., Moreno, P., Weinstein, E., & Rao, K. (2017). Multilingual speech recognition with a single end-to-end model. arXiv preprint arXiv:1711.01694.","DOI":"10.1109\/ICASSP.2018.8461972"},{"key":"10132_CR28","doi-asserted-by":"publisher","unstructured":"Tripathy, S., Baranwal, N., & Nandi, G. (2013). A MFCC based Hindi speech recognition technique using HTK Toolkit. In Proceedings of the 2013 IEEE 2nd international conference on image information processing (ICIIP-2013). Shimla, India. https:\/\/doi.org\/10.1109\/ICIIP.2013.6707650","DOI":"10.1109\/ICIIP.2013.6707650"},{"key":"10132_CR29","doi-asserted-by":"crossref","unstructured":"Zeyer, A., Irie, K., Schl\u00fcter, R., & Ney, H. (2018). Improved training of end-to-end attention models for speech recognition. In Interspeech 2018 (pp. 7\u201311).","DOI":"10.21437\/Interspeech.2018-1616"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10132-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10132-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10132-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,12]],"date-time":"2024-09-12T12:11:49Z","timestamp":1726143109000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10132-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,27]]},"references-count":29,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["10132"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10132-6","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7,27]]},"assertion":[{"value":"10 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 July 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 July 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no Conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}