{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T06:22:30Z","timestamp":1757312550664,"version":"3.37.3"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2020,10,16]],"date-time":"2020-10-16T00:00:00Z","timestamp":1602806400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,10,16]],"date-time":"2020-10-16T00:00:00Z","timestamp":1602806400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2022,3]]},"DOI":"10.1007\/s10772-020-09757-0","type":"journal-article","created":{"date-parts":[[2020,10,16]],"date-time":"2020-10-16T15:03:02Z","timestamp":1602860582000},"page":"67-78","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":14,"title":["Hindi speech recognition using time delay neural network acoustic modeling with i-vector adaptation"],"prefix":"10.1007","volume":"25","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4713-5047","authenticated-orcid":false,"given":"Ankit","family":"Kumar","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rajesh Kumar","family":"Aggarwal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,10,16]]},"reference":[{"key":"9757_CR1","doi-asserted-by":"crossref","unstructured":"Abraham, B., Seeram, T., & Umesh, S. (2017). Transfer learning and distillation techniques to improve the acoustic modeling of low resource languages. In INTERSPEECH (pp. 2158\u20132162).","DOI":"10.21437\/Interspeech.2017-1009"},{"issue":"4","key":"9757_CR2","doi-asserted-by":"publisher","first-page":"309","DOI":"10.1007\/s10772-011-9106-4","volume":"14","author":"RK Aggarwal","year":"2011","unstructured":"Aggarwal, R. K., & Dave, M. (2011). Acoustic modeling problem for automatic speech recognition system: Advances and refinements (Part II). International Journal of Speech Technology, 14(4), 309.","journal-title":"International Journal of Speech Technology"},{"issue":"2","key":"9757_CR3","doi-asserted-by":"publisher","first-page":"191","DOI":"10.1007\/s10772-012-9133-9","volume":"15","author":"RK Aggarwal","year":"2012","unstructured":"Aggarwal, R. K., & Dave, M. (2012). Filterbank optimization for robust ASR using GA and PSO. International Journal of Speech Technology, 15(2), 191\u2013201.","journal-title":"International Journal of Speech Technology"},{"issue":"3","key":"9757_CR4","doi-asserted-by":"publisher","first-page":"1457","DOI":"10.1007\/s11235-011-9623-0","volume":"52","author":"RK Aggarwal","year":"2013","unstructured":"Aggarwal, R. K., & Dave, M. (2013). Performance evaluation of sequentially combined heterogeneous feature streams for hindi speech recognition system. Telecommunication Systems, 52(3), 1457\u20131466.","journal-title":"Telecommunication Systems"},{"key":"9757_CR5","doi-asserted-by":"crossref","unstructured":"An, G., Brizan, D. G., Ma, M., Morales, M., Syed, A.R., & Rosenberg, A. (2015). Automatic recognition of unified parkinson\u2019s disease rating from speech with acoustic, i-vector and phonotactic features. In Sixteenth Annual Conference of the International Speech Communication Association.","DOI":"10.21437\/Interspeech.2015-185"},{"key":"9757_CR6","doi-asserted-by":"crossref","unstructured":"Biswas, A., Menon, R., van der Westhuizen, E., & Niesler, T. (2019). Improved low-resource somali speech recognition by semi-supervised acoustic and language model training. arXiv preprint arXiv:1907.03064.","DOI":"10.21437\/Interspeech.2019-1328"},{"issue":"8","key":"9757_CR7","doi-asserted-by":"publisher","first-page":"902","DOI":"10.1049\/iet-spr.2015.0488","volume":"10","author":"A Biswas","year":"2016","unstructured":"Biswas, A., Sahu, P. K., & Chandra, M. (2016). Admissible wavelet packet sub-band based harmonic energy features using anova fusion techniques for hindi phoneme recognition. IET Signal Processing, 10(8), 902\u2013911.","journal-title":"IET Signal Processing"},{"key":"9757_CR8","doi-asserted-by":"crossref","unstructured":"Chellapriyadharshini, M., Toffy, A., & Ramasubramanian, V., et al. (2018). Semi-supervised and active-learning scenarios: Efficient acoustic model refinement for a low resource indian language. arXiv preprint arXiv:1810.06635.","DOI":"10.21437\/Interspeech.2018-2486"},{"issue":"3","key":"9757_CR9","first-page":"501","volume":"26","author":"NF Chen","year":"2017","unstructured":"Chen, N. F., Lim, B. P., Hasegawa-Johnson, M. A., et al. (2017). Multitask learning for phone recognition of underresourced languages using mismatched transcription. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 26(3), 501\u2013514.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"9757_CR10","unstructured":"Chuangsuwanich, E. (2016). Multilingual techniques for low resource automatic speech recognition. Massachusetts Institute of Technology Cambridge United States: Tech. rep."},{"key":"9757_CR11","doi-asserted-by":"crossref","unstructured":"Dahl, G. E., Sainath, T. N., & Hinton, G. E. (2013). Improving deep neural networks for LVCSR using rectified linear units and dropout. In 2013 IEEE international conference on acoustics, speech and signal processing (pp. 8609\u20138613). IEEE.","DOI":"10.1109\/ICASSP.2013.6639346"},{"issue":"1","key":"9757_CR12","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"GE Dahl","year":"2011","unstructured":"Dahl, G. E., Yu, D., Deng, L., & Acero, A. (2011a). Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Transactions on Audio, Speech, and Language Processing, 20(1), 30\u201342.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9757_CR13","doi-asserted-by":"crossref","unstructured":"Dahl, G. E., Yu, D., Deng, L., & Acero, A. (2011b). Large vocabulary continuous speech recognition with context-dependent DBN-HMMS. In 2011 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 4688\u20134691). IEEE.","DOI":"10.1109\/ICASSP.2011.5947401"},{"key":"9757_CR14","doi-asserted-by":"crossref","unstructured":"Dua, M., Aggarwal, R. K., & Biswas, M. (2017). Discriminative training using heterogeneous feature vector for hindi automatic speech recognition system. In 2017 International Conference on Computer and Applications (ICCA) (pp. 158\u2013162). IEEE.","DOI":"10.1109\/COMAPP.2017.8079777"},{"issue":"1","key":"9757_CR15","doi-asserted-by":"publisher","first-page":"327","DOI":"10.1515\/jisys-2017-0618","volume":"29","author":"M Dua","year":"2018","unstructured":"Dua, M., Aggarwal, R. K., & Biswas, M. (2018a). Discriminative training using noise robust integrated features and refined hmm modeling. Journal of Intelligent Systems, 29(1), 327\u2013344.","journal-title":"Journal of Intelligent Systems"},{"issue":"3","key":"9757_CR16","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1016\/j.jestch.2018.04.005","volume":"21","author":"M Dua","year":"2018","unstructured":"Dua, M., Aggarwal, R. K., & Biswas, M. (2018b). Performance evaluation of hindi speech recognition system using optimized filterbanks. Engineering Science and Technology, an International Journal, 21(3), 389\u2013398.","journal-title":"Engineering Science and Technology, an International Journal"},{"key":"9757_CR17","first-page":"5024","volume":"6","author":"H Eghbal-Zadeh","year":"2016","unstructured":"Eghbal-Zadeh, H., Lehner, B., Dorfer, M., & Widmer, G. (2016). CP-JKU submissions for dcase-2016: A hybrid approach using binaural i-vectors and deep convolutional neural networks. IEEE AASP Challenge on Detection and Classification of Acoustic Scenes and Events (DCASE), 6, 5024\u20135028.","journal-title":"IEEE AASP Challenge on Detection and Classification of Acoustic Scenes and Events (DCASE)"},{"key":"9757_CR18","doi-asserted-by":"crossref","unstructured":"Ghalehjegh, S. H., & Rose, R. C. (2015). Deep bottleneck features for i-vector based text-independent speaker verification. In 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU) (pp. 555\u2013560). IEEE.","DOI":"10.1109\/ASRU.2015.7404844"},{"key":"9757_CR19","doi-asserted-by":"crossref","unstructured":"Hartmann, W., Hsiao, R., & Tsakalidis, S. (2017). Alternative networks for monolingual bottleneck features. In 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 5290\u20135294). IEEE.","DOI":"10.1109\/ICASSP.2017.7953166"},{"issue":"6","key":"9757_CR20","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G. E., Mohamed, Ar, Jaitly, N., et al. (2012). Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups. IEEE Signal Processing Magazine, 29(6), 82\u201397.","journal-title":"IEEE Signal Processing Magazine"},{"key":"9757_CR21","doi-asserted-by":"crossref","unstructured":"Jaitly, N., & Hinton, G. (2011). Learning a better representation of speech soundwaves using restricted boltzmann machines. In 2011 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 5884\u20135887). IEEE.","DOI":"10.1109\/ICASSP.2011.5947700"},{"key":"9757_CR22","unstructured":"Jaitly, N., & Hinton, G. E. (2013). Vocal tract length perturbation (VTLP) improves speech recognition. In: Proc. ICML Workshop on Deep Learning for Audio, Speech and Language (Vol. 117)."},{"key":"9757_CR23","doi-asserted-by":"crossref","unstructured":"Karafi\u00e1t, M., Burget, L., Mat\u011bjka, P., Glembek, O., & \u010cernock\u1ef3, J. (2011). ivector-based discriminative adaptation for automatic speech recognition. In 2011 IEEE Workshop on Automatic Speech Recognition & Understanding (pp. 152\u2013157). IEEE.","DOI":"10.1109\/ASRU.2011.6163922"},{"key":"9757_CR24","doi-asserted-by":"crossref","unstructured":"Ko, T., Peddinti, V., Povey, D., Seltzer, M. L., & Khudanpur, S. (2017). A study on data augmentation of reverberant speech for robust speech recognition. In 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 5220\u20135224). IEEE.","DOI":"10.1109\/ICASSP.2017.7953152"},{"key":"9757_CR25","doi-asserted-by":"crossref","unstructured":"Kreyssig, F.L., Zhang, C., & Woodland, P. C. (2018). Improved tdnns using deep kernels and frequency dependent grid-RNNS. In 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 4864\u20134868). IEEE.","DOI":"10.1109\/ICASSP.2018.8462523"},{"key":"9757_CR26","doi-asserted-by":"crossref","unstructured":"Liu, B., Zhang, W., Xu, X., & Chen, D. (2019). Time delay recurrent neural network for speech recognition. In Journal of Physics: Conference Series (Vol. 1229, p. 012078). IOP Publishing.","DOI":"10.1088\/1742-6596\/1229\/1\/012078"},{"key":"9757_CR27","doi-asserted-by":"crossref","unstructured":"Peddinti, V., Chen, G., Manohar, V., Ko, T., Povey, D., & Khudanpur, S. (2015a). JHU aspire system: Robust LVCSR with TDNNS, ivector adaptation and rnn-lms. In ASRU (pp. 539\u2013546).","DOI":"10.1109\/ASRU.2015.7404842"},{"key":"9757_CR28","doi-asserted-by":"crossref","unstructured":"Peddinti, V., Chen, G., Povey, D., & Khudanpur, S. (2015b). Reverberation robust acoustic modeling using i-vectors with time delay neural networks. In Sixteenth Annual Conference of the International Speech Communication Association.","DOI":"10.21437\/Interspeech.2015-527"},{"key":"9757_CR29","doi-asserted-by":"crossref","unstructured":"Peddinti, V., Povey, D., & Khudanpur, S. (2015c). A time delay neural network architecture for efficient modeling of long temporal contexts. In Sixteenth Annual Conference of the International Speech Communication Association.","DOI":"10.21437\/Interspeech.2015-647"},{"key":"9757_CR30","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., & Schwarz, P., et al. (2011). The kaldi speech recognition toolkit. In IEEE 2011 workshop on automatic speech recognition and understanding, CONF. IEEE Signal Processing Society."},{"key":"9757_CR31","doi-asserted-by":"crossref","unstructured":"Povey, D., Peddinti, V., Galvez, D., Ghahremani, P., Manohar, V., Na, X., Wang, Y., & Khudanpur, S. (2016). Purely sequence-trained neural networks for ASR based on lattice-free MMI. In Interspeech (pp. 2751\u20132755).","DOI":"10.21437\/Interspeech.2016-595"},{"key":"9757_CR32","doi-asserted-by":"crossref","unstructured":"Ragni, A., Knill, K., Rath, S. P., & Gales, M. (2014). Data augmentation for low resource languages.","DOI":"10.21437\/Interspeech.2014-207"},{"issue":"6088","key":"9757_CR33","doi-asserted-by":"publisher","first-page":"533","DOI":"10.1038\/323533a0","volume":"323","author":"DE Rumelhart","year":"1986","unstructured":"Rumelhart, D. E., Hinton, G. E., & Williams, R. J. (1986). Learning representations by back-propagating errors. Nature, 323(6088), 533\u2013536.","journal-title":"Nature"},{"key":"9757_CR34","doi-asserted-by":"crossref","unstructured":"Sak, H., Senior, A., & Beaufays, F. (2014). Long short-term memory based recurrent neural network architectures for large vocabulary speech recognition. arXiv preprint arXiv:1402.1128.","DOI":"10.21437\/Interspeech.2014-80"},{"key":"9757_CR35","doi-asserted-by":"crossref","unstructured":"Samudravijaya, K., Rao, P., & Agrawal, S. (2000). Hindi speech database. In Sixth International Conference on Spoken Language Processing.","DOI":"10.21437\/ICSLP.2000-847"},{"key":"9757_CR36","doi-asserted-by":"crossref","unstructured":"Saon, G., Soltau, H., Nahamoo, D., & Picheny, M. (2013). Speaker adaptation of neural network acoustic models using i-vectors. In 2013 IEEE Workshop on Automatic Speech Recognition and Understanding (pp. 55\u201359). IEEE.","DOI":"10.1109\/ASRU.2013.6707705"},{"key":"9757_CR37","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., & Yu, D. (2011). Feature engineering in context-dependent deep neural networks for conversational speech transcription. In 2011 IEEE Workshop on Automatic Speech Recognition & Understanding (pp. 24\u201329). IEEE.","DOI":"10.1109\/ASRU.2011.6163899"},{"key":"9757_CR38","doi-asserted-by":"crossref","unstructured":"Sercu, T., Puhrsch, C., Kingsbury, B., & LeCun, Y. (2016). Very deep multilingual convolutional neural networks for LVCSR. In 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 4955\u20134959). IEEE.","DOI":"10.1109\/ICASSP.2016.7472620"},{"key":"9757_CR39","doi-asserted-by":"crossref","unstructured":"Stolcke, A. (2002). Srilm-an extensible language modeling toolkit. In Seventh international conference on spoken language processing.","DOI":"10.21437\/ICSLP.2002-303"},{"key":"9757_CR40","unstructured":"Trmal, J., Kumar, G., Manohar, V., Khudanpur, S., Post, M., & McNamee, P. (2017). Using of heterogeneous corpora for training of an ASR system. arXiv preprint arXiv:1706.00321."},{"issue":"3","key":"9757_CR41","doi-asserted-by":"publisher","first-page":"328","DOI":"10.1109\/29.21701","volume":"37","author":"A Waibel","year":"1989","unstructured":"Waibel, A., Hanazawa, T., Hinton, G., Shikano, K., & Lang, K. J. (1989). Phoneme recognition using time-delay neural networks. IEEE Transactions on Acoustics, Speech, and Signal Processing, 37(3), 328\u2013339.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"9757_CR42","unstructured":"Weninger, F., Watanabe, S., Le Roux, J., Hershey, J., Tachioka, Y., Geiger, J., Schuller, B., & Rigoll, G. (2014). The merl\/melco\/tum system for the reverb challenge using deep recurrent neural network feature enhancement. In Proc. REVERB Workshop (pp. 1\u20138)."},{"key":"9757_CR43","doi-asserted-by":"crossref","unstructured":"Xu, H., Su, H., Ni, C., Xiao, X., Huang, H., Chng, E. S., & Li, H. (2016). Semi-supervised and cross-lingual knowledge transfer learnings for DNN hybrid acoustic models under low-resource conditions. In INTERSPEECH (pp. 1315\u20131319).","DOI":"10.21437\/Interspeech.2016-1099"},{"issue":"12","key":"9757_CR44","doi-asserted-by":"publisher","first-page":"1713","DOI":"10.1109\/TASLP.2014.2346313","volume":"22","author":"S Xue","year":"2014","unstructured":"Xue, S., Abdel-Hamid, O., Jiang, H., Dai, L., & Liu, Q. (2014). Fast adaptation of deep neural network based on discriminant codes for speech recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 22(12), 1713\u20131725.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-020-09757-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-020-09757-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-020-09757-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,23]],"date-time":"2022-11-23T17:28:31Z","timestamp":1669224511000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-020-09757-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10,16]]},"references-count":44,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2022,3]]}},"alternative-id":["9757"],"URL":"https:\/\/doi.org\/10.1007\/s10772-020-09757-0","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2020,10,16]]},"assertion":[{"value":"26 November 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 September 2020","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 October 2020","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}