{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T04:16:20Z","timestamp":1742271380626,"version":"3.40.1"},"reference-count":69,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T00:00:00Z","timestamp":1730505600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T00:00:00Z","timestamp":1730505600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Circuits Syst Signal Process"],"published-print":{"date-parts":[[2025,3]]},"DOI":"10.1007\/s00034-024-02896-8","type":"journal-article","created":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T00:05:01Z","timestamp":1730505901000},"page":"1914-1937","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Exploring the Role of Data Augmentation and Acoustic Feature Concatenation in the Context of Zero-Resource Children\u2019s ASR"],"prefix":"10.1007","volume":"44","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-6688-0872","authenticated-orcid":false,"family":"Ankita","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"S.","family":"Shahnawazuddin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,2]]},"reference":[{"issue":"1","key":"2896_CR1","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1186\/s40537-023-00727-2","volume":"10","author":"L Alzubaidi","year":"2023","unstructured":"L. Alzubaidi, J. Bai, A. Al-Sabaawi, J. Santamar\u00eda, A.S. Albahri, B.S.N. Al-dabbagh, M.A. Fadhel, M. Manoufali, J. Zhang, A.H. Al-Timemy et al., A survey on deep learning tools dealing with data scarcity: definitions, challenges, solutions, tips, and applications. J. Big Data 10(1), 46 (2023)","journal-title":"J. Big Data"},{"key":"2896_CR2","doi-asserted-by":"crossref","unstructured":"Ankita, S. Shahnawazuddin, Developing children\u2019s ASR system under low-resource conditions using end-to-end architecture. Digit. Sign. Process. 146, 104385 (2024)","DOI":"10.1016\/j.dsp.2024.104385"},{"key":"2896_CR3","doi-asserted-by":"crossref","unstructured":"Ankita, S. Shahnawazuddin, Studying the effect of frame-level concatenation of GFCC and TS-MFCC features on zero-shot children\u2019s ASR. in Proc. SPECOM (2023), pp. 140\u2013150","DOI":"10.1007\/978-3-031-48312-7_11"},{"key":"2896_CR4","doi-asserted-by":"crossref","unstructured":"Ankita, S. Shahnawazuddin, Effect of modeling glottal activity parameters on zero-shot children\u2019s ASR. IEEE\/ACM Trans. Audio, Speech Lang. Process. 32, 3039\u20133048 (2024)","DOI":"10.1109\/TASLP.2024.3407576"},{"key":"2896_CR5","doi-asserted-by":"publisher","first-page":"109420","DOI":"10.1016\/j.apacoust.2023.109420","volume":"209","author":"S Aziz","year":"2023","unstructured":"S. Aziz, S. Shahnawazuddin, Effective preservation of higher-frequency contents in the context of short utterance based children\u2019s speaker verification system. Appl. Acoust. 209, 109420 (2023)","journal-title":"Appl. Acoust."},{"key":"2896_CR6","doi-asserted-by":"publisher","first-page":"109783","DOI":"10.1016\/j.apacoust.2023.109783","volume":"216","author":"S Aziz","year":"2024","unstructured":"S. Aziz, S. Shahnawazuddin, Experimental studies for improving the performance of children\u2019s speaker verification system using short utterances. Appl. Acoust. 216, 109783 (2024)","journal-title":"Appl. Acoust."},{"issue":"5","key":"2896_CR7","doi-asserted-by":"publisher","first-page":"3139","DOI":"10.1007\/s00034-024-02598-1","volume":"43","author":"S Aziz","year":"2024","unstructured":"S. Aziz, S. Shahnawazuddin, Role of data augmentation and effective conservation of high-frequency contents in the context children\u2019s speaker verification system. Circuits, Syst. Sign. Process. 43(5), 3139\u20133159 (2024)","journal-title":"Circuits, Syst. Sign. Process."},{"key":"2896_CR8","doi-asserted-by":"crossref","unstructured":"A. Batliner, M. Blomberg, S. D\u2019Arcy, D. Elenius, D. Giuliani, M. Gerosa, C. Hacker, M. Russell, S. Steidl, M. Wong, The pf_star children\u2019s speech corpus (2005)","DOI":"10.21437\/Interspeech.2005-705"},{"key":"2896_CR9","doi-asserted-by":"crossref","unstructured":"L. Bell, J. Gustafson, Children\u2019s convergence in referring expressions to graphical objects in a speech-enabled computer game. in Proc. INTERSPEECH (2007), pp. 2209\u20132212","DOI":"10.21437\/Interspeech.2007-601"},{"key":"2896_CR10","doi-asserted-by":"crossref","unstructured":"X. Chen, Y. Wu, Z. Wang, S. Liu, J. Li, Developing real-time streaming transformer transducer for speech recognition on large-scale dataset. in Proc. ICASSP (IEEE. 2021), pp. 5904\u20135908 (2021)","DOI":"10.1109\/ICASSP39728.2021.9413535"},{"key":"2896_CR11","doi-asserted-by":"publisher","first-page":"1360","DOI":"10.1109\/TASLP.2022.3161159","volume":"30","author":"G Cheng","year":"2022","unstructured":"G. Cheng, H. Miao, R. Yang, K. Deng, Y. Yan, Eteh: unified attention-based end-to-end ASR and KWS architecture. IEEE\/ACM Trans. Audio Speech Lang. Process. 30, 1360\u20131373 (2022)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"1","key":"2896_CR12","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"G Dahl","year":"2012","unstructured":"G. Dahl, D. Yu, L. Deng, A. Acero, Context-dependent pre-trained deep neural networks for large vocabulary speech recognition. IEEE Trans. Speech Audio Process. 20(1), 30\u201342 (2012)","journal-title":"IEEE Trans. Speech Audio Process."},{"issue":"12","key":"2896_CR13","doi-asserted-by":"publisher","first-page":"1293","DOI":"10.3390\/app7121293","volume":"7","author":"EP Damsk\u00e4gg","year":"2017","unstructured":"E.P. Damsk\u00e4gg, V. V\u00e4lim\u00e4ki, Audio time stretching using fuzzy classification of spectral bins. Appl. Sci. 7(12), 1293 (2017)","journal-title":"Appl. Sci."},{"key":"2896_CR14","doi-asserted-by":"crossref","unstructured":"M. Delcroix, S. Watanabe, A. Ogawa, S. Karita, T. Nakatani, Auxiliary feature based adaptation of end-to-end ASR systems. in Proc. INTERSPEECH, vol. 2018 (2018). pp. 2444\u20132448","DOI":"10.21437\/Interspeech.2018-1438"},{"key":"2896_CR15","doi-asserted-by":"crossref","unstructured":"R. Duan, N.F. Chen, Unsupervised feature adaptation using adversarial multi-task training for automatic evaluation of children\u2019s speech. in Proc. INTERSPEECH (2020), pp. 3037\u20133041","DOI":"10.21437\/Interspeech.2020-1657"},{"key":"2896_CR16","doi-asserted-by":"publisher","first-page":"62","DOI":"10.1016\/j.csl.2017.12.006","volume":"50","author":"S Dudy","year":"2018","unstructured":"S. Dudy, S. Bedrick, M. Asgari, A. Kain, Automatic analysis of pronunciations for children with speech sound disorders. Comput. Speech Lang. 50, 62\u201384 (2018)","journal-title":"Comput. Speech Lang."},{"key":"2896_CR17","unstructured":"M. Eskenazi, J. Mostow, D. Graff, The CMU Kids Corpus LDC97S63. https:\/\/catalog.ldc.upenn.edu\/LDC97S63 (1997)"},{"key":"2896_CR18","doi-asserted-by":"publisher","first-page":"16721","DOI":"10.1007\/s11042-017-5237-1","volume":"77","author":"M Fedila","year":"2018","unstructured":"M. Fedila, M. Bengherabi, A. Amrouche, Gammatone filterbank and symbiotic combination of amplitude and phase-based spectra for robust speaker verification under noisy conditions and compression artifacts. Multimed. Tools Appl. 77, 16721\u201316739 (2018)","journal-title":"Multimed. Tools Appl."},{"issue":"10\u201311","key":"2896_CR19","doi-asserted-by":"publisher","first-page":"847","DOI":"10.1016\/j.specom.2007.01.002","volume":"49","author":"M Gerosa","year":"2007","unstructured":"M. Gerosa, D. Giuliani, F. Brugnara, Acoustic variability and automatic recognition of children\u2019s speech. Speech Commun. 49(10\u201311), 847\u2013860 (2007)","journal-title":"Speech Commun."},{"issue":"3","key":"2896_CR20","doi-asserted-by":"publisher","first-page":"1861","DOI":"10.1121\/1.4742973","volume":"132","author":"B Gold","year":"2012","unstructured":"B. Gold, N. Morgan, D. Ellis, D. O\u2019Shaughnessy, Speech and audio signal processing: processing and perception of speech and music. J. Acoust. Soc. Am. 132(3), 1861 (2012)","journal-title":"J. Acoust. Soc. Am."},{"key":"2896_CR21","doi-asserted-by":"crossref","unstructured":"A. Graves, N. Jaitly, A.R. Mohamed, Hybrid speech recognition with deep bidirectional LSTM. in Proc. ASRU (IEEE, 2013), pp. 273\u2013278","DOI":"10.1109\/ASRU.2013.6707742"},{"key":"2896_CR22","doi-asserted-by":"crossref","unstructured":"A. Graves, A.R. Mohamed, G. Hinton, Speech recognition with deep recurrent neural networks. in Proc. ICASSP (IEEE, 2013), pp. 6645\u20136649","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"2896_CR23","unstructured":"S.S. Gray, D. Willett, J. Pinto, J. Lu, P. Maergner, N. Bodenstab, Child automatic speech recognition for US English: child interaction with living-room-electronic-devices. in Proc. INTERSPEECH, Workshop on Child, Computer and Interaction (2014)"},{"key":"2896_CR24","unstructured":"A. Hagen, B. Pellom, R. Cole, Children\u2019s speech recognition with application to interactive books and tutors. in Proc. ASRU (2003), pp. 186\u2013191"},{"issue":"12","key":"2896_CR25","doi-asserted-by":"publisher","first-page":"861","DOI":"10.1016\/j.specom.2007.05.004","volume":"49","author":"A Hagen","year":"2007","unstructured":"A. Hagen, B. Pellom, R. Cole, Highly accurate children\u2019s speech recognition for interactive reading tutors using subword units. Speech Commun. 49(12), 861\u2013873 (2007)","journal-title":"Speech Commun."},{"issue":"6","key":"2896_CR26","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"GE Hinton","year":"2012","unstructured":"G.E. Hinton, L. Deng, D. Yu, G. Dahl, A.R. Mohamed, N. Jaitly, A. Senior, V. Vanhoucke, P. Nguyen, T. Sainath, B. Kingsbury, Deep neural networks for acoustic modeling in speech recognition. Sign. Process. Mag. 29(6), 82\u201397 (2012)","journal-title":"Sign. Process. Mag."},{"key":"2896_CR27","doi-asserted-by":"crossref","unstructured":"W.R. Huang, S.Y. Chang, D. Rybach, T. Sainath, R. Prabhavalkar, C. Peyser, Z. Lu, C. Allauzen, E2E segmenter: joint segmenting and decoding for long-form ASR. in Proc. INTERSPEECH (2022) pp 4995\u20134999","DOI":"10.21437\/Interspeech.2022-38"},{"key":"2896_CR28","doi-asserted-by":"crossref","unstructured":"R. Jain, A. Barcovschi, M. Yiwere, P. Corcoran, H. Cucu, Adaptation of whisper models to child speech recognition. in Proc. INTERSPEECH (2023)","DOI":"10.21437\/Interspeech.2023-935"},{"key":"2896_CR29","doi-asserted-by":"crossref","unstructured":"A. Johnson, R. Fan, R. Morris, A. Alwan, LPC augment: an LPC-based ASR data augmentation algorithm for low and zero-resource children\u2019s dialects. in ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (IEEE, 2022), pp. 8577\u20138581","DOI":"10.1109\/ICASSP43922.2022.9746281"},{"key":"2896_CR30","doi-asserted-by":"crossref","unstructured":"T. Kaneko, H. Kameoka, Parallel-data-free voice conversion using cycle-consistent adversarial networks. arXiv preprint arXiv:1711.11293 (2017)","DOI":"10.23919\/EUSIPCO.2018.8553236"},{"key":"2896_CR31","doi-asserted-by":"publisher","first-page":"713","DOI":"10.1007\/s12046-011-0043-3","volume":"36","author":"H Kawahara","year":"2011","unstructured":"H. Kawahara, M. Morise, Technical foundations of tandem-straight, a speech analysis, modification and synthesis framework. Sadhana 36, 713\u2013727 (2011)","journal-title":"Sadhana"},{"issue":"5","key":"2896_CR32","doi-asserted-by":"publisher","first-page":"713","DOI":"10.1007\/s12046-011-0043-3","volume":"36","author":"H Kawahara","year":"2011","unstructured":"H. Kawahara, M. Morise, Technical foundations of TANDEM-STRAIGHTs, a speech analysis, modification and synthesis framework. Sadhana 36(5), 713\u2013727 (2011)","journal-title":"Sadhana"},{"key":"2896_CR33","doi-asserted-by":"crossref","unstructured":"K. Kim, K. Lee, D. Gowda, J. Park, S. Kim, S. Jin, Y.Y. Lee, J. Yeo, D. Kim, S. Jung, et\u00a0al. Attention based on-device streaming speech recognition with large speech corpus. in Proc. ASRU (IEEE, 2019), pp. 956\u2013963","DOI":"10.1109\/ASRU46091.2019.9004027"},{"issue":"4","key":"2896_CR34","doi-asserted-by":"publisher","first-page":"2205","DOI":"10.1007\/s00034-021-01885-5","volume":"41","author":"V Kumar","year":"2022","unstructured":"V. Kumar, A. Kumar, S. Shahnawazuddin, Creating robust children\u2019s ASR system in zero-resource condition through out-of-domain data augmentation. Circuits Syst. Sign. Process 41(4), 2205\u20132220 (2022)","journal-title":"Circuits Syst. Sign. Process"},{"key":"2896_CR35","doi-asserted-by":"crossref","unstructured":"H. Kumar Kathania, S. Reddy Kadiri, P. Alku, M. Kurimo, Study of formant modification for children ASR. in Proc. ICASSP (2020), pp. 7429\u20137433","DOI":"10.1109\/ICASSP40776.2020.9053334"},{"issue":"3","key":"2896_CR36","doi-asserted-by":"publisher","first-page":"1455","DOI":"10.1121\/1.426686","volume":"105","author":"S Lee","year":"1999","unstructured":"S. Lee, A. Potamianos, S. Narayanan, Acoustics of children\u2019s speech: developmental changes of temporal and spectral parameters. J. Acoust. Soc. Am. 105(3), 1455\u20131468 (1999)","journal-title":"J. Acoust. Soc. Am."},{"key":"2896_CR37","doi-asserted-by":"crossref","unstructured":"B. Li, S.V. Chang, T.N. Sainath, R. Pang, Y. He, T. Strohman, Y. Wu, Towards fast and accurate streaming end-to-end ASR. in Proc. ICASSP (2020), pp. 6069\u20136073","DOI":"10.1109\/ICASSP40776.2020.9054715"},{"key":"2896_CR38","doi-asserted-by":"crossref","unstructured":"B. Li, A. Gulati, J. Yu, T.N. Sainath, C.C. Chiu, A. Narayanan, S.Y. Chang, R. Pang, Y. He, J. Qin, et\u00a0al., A better and faster end-to-end model for streaming ASR. in Proc. ICASSP (2021), pp. 5634\u20135638","DOI":"10.1109\/ICASSP39728.2021.9413899"},{"key":"2896_CR39","unstructured":"R. Lu, M. Shahin, B. Ahmed, Improving children\u2019s speech recognition by fine-tuning self-supervised adult speech representations. arXiv preprint arXiv:2211.07769 (2022)"},{"issue":"12","key":"2896_CR40","first-page":"3265","volume":"90","author":"M Morise","year":"2007","unstructured":"M. Morise, T. Takahashi, H. Kawahara, T. Irino, Power spectrum estimation method for periodic signals virtually irrespective to time window position. Trans. IEICE 90(12), 3265\u20133267 (2007)","journal-title":"Trans. IEICE"},{"key":"2896_CR41","doi-asserted-by":"crossref","unstructured":"R. Nisimura, A. Lee, H. Saruwatari, K. Shikano, Public speech-oriented guidance system with adult and child discrimination capability. in Proc. ICASSP, vol.\u00a01 (2004), pp. 433\u2013436","DOI":"10.1109\/ICASSP.2004.1326015"},{"key":"2896_CR42","doi-asserted-by":"crossref","unstructured":"V. Panayotov, G. Chen, D. Povey, S. Khudanpur, Librispeech: an ASR corpus based on public domain audio books. in 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (IEEE, 2015), pp. 5206\u20135210","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"2896_CR43","unstructured":"R. Patterson, I. Nimmo-Smith, J. Holdsworth, P. Rice, An efficient auditory filterbank based on the gammatone function (1987)"},{"key":"2896_CR44","doi-asserted-by":"crossref","unstructured":"V. Peddinti, D. Povey, S. Khudanpur, A time delay neural network architecture for efficient modeling of long temporal contexts. in Proc. INTERSPEECH (2015)","DOI":"10.21437\/Interspeech.2015-647"},{"key":"2896_CR45","unstructured":"D. Povey, A. Ghoshal, G. Boulianne, L. Burget, O. Glembek, N. Goel, M. Hannemann, P. Motlicek, Y. Qian, P. Schwarz, J. Silovsky, G. Stemmer, K. Vesely, The Kaldi Speech Recognition Toolkit. in Proc. ASRU (2011)"},{"key":"2896_CR46","doi-asserted-by":"crossref","unstructured":"D. Povey, V. Peddinti, D. Galvez, P. Ghahremani, V. Manohar, X. Na, Y. Wang, S. Khudanpur, Purely sequence-trained neural networks for ASR based on lattice-free MMI. in Proc. INTERSPEECH (2016), pp. 2751\u20132755","DOI":"10.21437\/Interspeech.2016-595"},{"issue":"10","key":"2896_CR47","doi-asserted-by":"publisher","first-page":"6228","DOI":"10.1007\/s00034-023-02399-y","volume":"42","author":"K Radha","year":"2023","unstructured":"K. Radha, M. Bansal, Feature fusion and ablation analysis in gender identification of preschool children from spontaneous speech. Circuits Syst. Sign. Process. 42(10), 6228\u20136252 (2023)","journal-title":"Circuits Syst. Sign. Process."},{"key":"2896_CR48","doi-asserted-by":"crossref","unstructured":"T. Robinson, J. Fransen, D. Pye, J. Foote, S. Renals, WSJCAM0: A British English speech corpus for large vocabulary continuous speech recognition. in Proc. ICASSP, vol.\u00a01 (1995), pp. 81\u201384","DOI":"10.1109\/ICASSP.1995.479278"},{"key":"2896_CR49","unstructured":"A. Rousseau, P. Del\u00e9glise, Y. Est\u00e8ve, TED-LIUM: an automatic speech recognition dedicated corpus. in Proceedings of the Eighth International Conference on Language Resources and Evaluation (LREC\u201912) (2012), pp. 125\u2013129"},{"key":"2896_CR50","doi-asserted-by":"crossref","unstructured":"L. Rumberg, H. Ehlert, U. L\u00fcdtke, J. Ostermann, Age-Invariant Training for End-to-End Child Speech Recognition Using Adversarial Multi-Task Learning. in Proc. INTERSPEECH (2021), 3850\u20133854","DOI":"10.21437\/Interspeech.2021-1241"},{"key":"2896_CR51","doi-asserted-by":"crossref","unstructured":"M. Russell, S. D\u2019Arcy, Challenges for computer recognition of children\u2019s speech. in Proc. Speech and Language Technologies in Education (SLaTE) (2007)","DOI":"10.21437\/SLaTE.2007-26"},{"key":"2896_CR52","doi-asserted-by":"crossref","unstructured":"T.N. Sainath, Y. He, A. Narayanan, R. Botros, R. Pang, D. Rybach, C. Allauzen, E. Variani, J. Qin, Q.N. Le-The, S.Y. Chang, B. Li, A. Gulati, J. Yu, C.C. Chiu, D. Caseiro, W. Li, Q. Liang, P. Rondon, An efficient streaming non-recurrent on-device end-to-end model with improvements to rare-word modeling. in Proc. INTERSPEECH (2021), pp. 1777\u20131781","DOI":"10.21437\/Interspeech.2021-206"},{"key":"2896_CR53","doi-asserted-by":"crossref","unstructured":"T.N. Sainath, O. Vinyals, A. Senior, H. Sak, Convolutional, long short-term memory, fully connected deep neural networks. in Proc. ICASSP (2015), pp 4580\u20134584","DOI":"10.1109\/ICASSP.2015.7178838"},{"key":"2896_CR54","doi-asserted-by":"crossref","unstructured":"J. Schalkwyk, D. Beeferman, F. Beaufays, B. Byrne, C. Chelba, M. Cohen, M. Kamvar, B. Strope, Your word is my command: google search by voice: a case study. in Advances in Speech Recognition: Mobile Environments, Call Centers and Clinics, chap.\u00a04 (2010), pp. 61\u201390","DOI":"10.1007\/978-1-4419-5951-5_4"},{"issue":"2","key":"2896_CR55","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1109\/JSTSP.2019.2959393","volume":"14","author":"M Shahin","year":"2019","unstructured":"M. Shahin, U. Zafar, B. Ahmed, The automatic detection of speech disorders in children: challenges, opportunities, and preliminary results. IEEE J. Select. Top. Sign. Process. 14(2), 400\u2013412 (2019)","journal-title":"IEEE J. Select. Top. Sign. Process."},{"key":"2896_CR56","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1016\/j.dsp.2018.05.003","volume":"79","author":"S Shahnawazuddin","year":"2018","unstructured":"S. Shahnawazuddin, N. Adiga, H.K. Kathania, G. Pradhan, R. Sinha, Studying the role of pitch-adaptive spectral estimation and speaking-rate normalization in automatic speech recognition. Digit. Sign. Process. 79, 142\u2013151 (2018)","journal-title":"Digit. Sign. Process."},{"key":"2896_CR57","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1016\/j.patrec.2019.12.019","volume":"131","author":"S Shahnawazuddin","year":"2020","unstructured":"S. Shahnawazuddin, N. Adiga, H.K. Kathania, B.T. Sai, Creating speaker independent ASR system through prosody modification based data augmentation. Patt. Recogn. Lett. 131, 213\u2013218 (2020)","journal-title":"Patt. Recogn. Lett."},{"key":"2896_CR58","doi-asserted-by":"crossref","unstructured":"S. Shahnawazuddin, N. Adiga, K. Kumar, A. Poddar, W. Ahmad, Voice conversion based data augmentation to improve children\u2019s speech recognition in limited data scenario. in Proc. INTERSPEECH (2020), pp. 4382\u20134386","DOI":"10.21437\/Interspeech.2020-1112"},{"key":"2896_CR59","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1016\/j.dsp.2019.06.015","volume":"93","author":"S Shahnawazuddin","year":"2019","unstructured":"S. Shahnawazuddin, N. Adiga, B.T. Sai, W. Ahmad, H.K. Kathania, Developing speaker independent ASR system using limited data through prosody modification based on fuzzy classification of spectral bins. Digit. Sign. Process. 93, 34\u201342 (2019)","journal-title":"Digit. Sign. Process."},{"key":"2896_CR60","doi-asserted-by":"crossref","unstructured":"S. Shahnawazuddin, Ankita, A. Kumar, H.K. Kathania, Gammatone-filterbank based pitch-normalized cepstral coefficients for zero-resource children\u2019s ASR. in Proc. SPECOM (2023), pp. 494\u2013505","DOI":"10.1007\/978-3-031-48309-7_40"},{"key":"2896_CR61","doi-asserted-by":"publisher","first-page":"101289","DOI":"10.1016\/j.csl.2021.101289","volume":"72","author":"PG Shivakumar","year":"2022","unstructured":"P.G. Shivakumar, S. Narayanan, End-to-end neural systems for automatic children speech recognition: an empirical study. Comput. Speech Lang. 72, 101289 (2022)","journal-title":"Comput. Speech Lang."},{"key":"2896_CR62","unstructured":"K. Shobaki, J.P. Hosom, R. Cole, Cslu: Kids\u2019 speech version 1.1. Linguistic Data Consortium (2007)"},{"key":"2896_CR63","unstructured":"Z. Shuyang, M. Singh, A. Woubie, R. Karhila, Data augmentation for children ASR and child-adult speaker classification using voice conversion methods. in Proc. INTERSPEECH (2023)"},{"key":"2896_CR64","unstructured":"M. Slaney, An efficient implementation of the Patterson\u2013Holdsworth auditory filter bank (2000)"},{"key":"2896_CR65","doi-asserted-by":"crossref","unstructured":"D.V. Smith, A. Sneddon, L. Ward, A. Duenser, J. Freyne, D. Silvera-Tawil, A. Morgan, Improving child speech disorder assessment by incorporating out-of-domain adult speech. in Proc. INTERSPEECH (2017), pp. 2690\u20132694","DOI":"10.21437\/Interspeech.2017-455"},{"key":"2896_CR66","doi-asserted-by":"crossref","unstructured":"A. Waibel, T. Hanazawa, G. Hinton, K. Shikano, K. Lang, Phoneme recognition using time-delay neural networks. IEEE Trans. Acoust. Speech Sign. Process. 328\u2013339 (1989)","DOI":"10.1109\/29.21701"},{"key":"2896_CR67","doi-asserted-by":"crossref","unstructured":"S. Watanabe, T. Hori, S. Karita, T. Hayashi, J. Nishitoba, Y. Unno, N.E.Y. Soplin, J. Heymann, M. Wiesner, N. Chen, et\u00a0al., Espnet: end-to-end speech processing toolkit. arXiv preprint arXiv:1804.00015 (2018)","DOI":"10.21437\/Interspeech.2018-1456"},{"issue":"8","key":"2896_CR68","doi-asserted-by":"publisher","first-page":"1240","DOI":"10.1109\/JSTSP.2017.2763455","volume":"11","author":"S Watanabe","year":"2017","unstructured":"S. Watanabe, T. Hori, S. Kim, J.R. Hershey, T. Hayashi, Hybrid CTC\/attention architecture for end-to-end speech recognition. IEEE J. Select. Top. Sign. Process. 11(8), 1240\u20131253 (2017)","journal-title":"IEEE J. Select. Top. Sign. Process."},{"key":"2896_CR69","doi-asserted-by":"crossref","unstructured":"L. Ye, G. Cheng, R. Yang, Z. Yang, S. Tian, P. Zhang, Y. Yan, Improving Recognition of Out-of-vocabulary Words in E2E Code-switching ASR by Fusing Speech Generation Methods. in Proc. INTERSPEECH (2022), pp. 3163\u20133167","DOI":"10.21437\/Interspeech.2022-719"}],"container-title":["Circuits, Systems, and Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-024-02896-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00034-024-02896-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-024-02896-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,17]],"date-time":"2025-03-17T19:27:42Z","timestamp":1742239662000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00034-024-02896-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,2]]},"references-count":69,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,3]]}},"alternative-id":["2896"],"URL":"https:\/\/doi.org\/10.1007\/s00034-024-02896-8","relation":{},"ISSN":["0278-081X","1531-5878"],"issn-type":[{"type":"print","value":"0278-081X"},{"type":"electronic","value":"1531-5878"}],"subject":[],"published":{"date-parts":[[2024,11,2]]},"assertion":[{"value":"13 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 October 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 October 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 November 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The work presented in the uploaded manuscript is an original one and the manuscript is not currently under consideration for publication elsewhere.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"It is hereby confirmed that the manuscript has been read and approved for submission by all the named authors. It is therefore requested, to consider the submitted manuscript for publication in the esteemed journal.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}]}}