{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T18:01:35Z","timestamp":1774720895429,"version":"3.50.1"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T00:00:00Z","timestamp":1747958400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T00:00:00Z","timestamp":1747958400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Univ Access Inf Soc"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s10209-025-01225-3","type":"journal-article","created":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T14:36:34Z","timestamp":1748010994000},"page":"2741-2756","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Enhancing child-machine interaction for Indian children speaking English as non-native language using hybrid CNN and customized dictionary"],"prefix":"10.1007","volume":"24","author":[{"given":"Neha","family":"Kasture","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pooja","family":"Jain","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,23]]},"reference":[{"key":"1225_CR1","doi-asserted-by":"publisher","unstructured":"Terzopoulos, G., Satratzemi, M.: Voice assistants and smart speakers in everyday life and in education. Informatics in Education. pp. 473\u2013490 (2020). https:\/\/doi.org\/10.15388\/infedu.2020.21","DOI":"10.15388\/infedu.2020.21"},{"key":"1225_CR2","doi-asserted-by":"publisher","unstructured":"Purington, A., Taft, J.G., Sannon, S., Bazarova, N.N., Taylor, S.H.: \u201dalexa is my new bff\u201d: Social roles, user satisfaction, and personification of the amazon echo. In: Proceedings of the 2017 CHI Conference Extended Abstracts on Human Factors in Computing Systems. CHI EA \u201917. Association for Computing Machinery, New York. (2017). pp. 2853\u20132859 https:\/\/doi.org\/10.1145\/3027063.3053246","DOI":"10.1145\/3027063.3053246"},{"key":"1225_CR3","doi-asserted-by":"crossref","unstructured":"Potamianos, A., Narayanan, S., Lee, S.: Automatic speech recognition for children. (1997)","DOI":"10.21437\/Eurospeech.1997-623"},{"issue":"3","key":"1225_CR4","doi-asserted-by":"publisher","first-page":"1455","DOI":"10.1121\/1.426686","volume":"105","author":"S Lee","year":"1999","unstructured":"Lee, S., Potamianos, A., Narayanan, S.S.: Acoustics of children\u2019s speech: developmental changes of temporal and spectral parameters. J Acoust Soc Am 105(3), 1455\u201368 (1999)","journal-title":"J Acoust Soc Am"},{"key":"1225_CR5","first-page":"1","volume":"257","author":"S Eguchi","year":"1969","unstructured":"Eguchi, S., Hirsh, I.: Development of speech sounds in children. Acta Otolaryngol. Suppl. 257, 1\u201351 (1969)","journal-title":"Acta Otolaryngol. Suppl."},{"key":"1225_CR6","unstructured":"Shivakumar, P.G., Potamianos, A., Lee, S., Narayanan, S.S.: Improving speech recognition for children using acoustic adaptation and pronunciation modeling. In: WOCCI (2014)"},{"key":"1225_CR7","doi-asserted-by":"crossref","unstructured":"Liao, H., Pundak, G., Siohan, O., Carroll, M.K., Coccaro, N., Jiang, Q., Sainath, T., Senior, A., Beaufays, F., Bacchiani, M.: Large vocabulary automatic speech recognition for children. In: INTERSPEECH (2015)","DOI":"10.21437\/Interspeech.2015-373"},{"key":"1225_CR8","doi-asserted-by":"crossref","unstructured":"Palaz, D., Collobert, R., Magimai-Doss, M.: Estimating phoneme class conditional probabilities from raw speech signal using convolutional neural networks. CoRR abs\/1304.1018 (2013) arXiv:1304.1018","DOI":"10.21437\/Interspeech.2013-438"},{"key":"1225_CR9","doi-asserted-by":"crossref","unstructured":"Golik, P., T\u00fcske, Z., Schl\u00fcter, R., Ney, H.: Convolutional neural networks for acoustic modeling of raw time signal in lvcsr. (2015)","DOI":"10.21437\/Interspeech.2015-6"},{"key":"1225_CR10","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1250\/ast.33.215","volume":"33","author":"R Mugitani","year":"2012","unstructured":"Mugitani, R., Hiroya, S.: Development of vocal tract and acoustic features in children. Acoust. Sci. Technol. 33, 215\u2013220 (2012). https:\/\/doi.org\/10.1250\/ast.33.215","journal-title":"Acoust. Sci. Technol."},{"key":"1225_CR11","doi-asserted-by":"publisher","unstructured":"Gerosa, M., Lee, S., Giuliani, D., Narayanan, S.: Analyzing children\u2019s speech: An acoustic study of consonants and consonant-vowel transition, vol. 1. (2006). https:\/\/doi.org\/10.1109\/ICASSP.2006.1660040","DOI":"10.1109\/ICASSP.2006.1660040"},{"key":"1225_CR12","doi-asserted-by":"publisher","first-page":"521","DOI":"10.1093\/ajcn\/72.2.521S","volume":"72","author":"A Rogol","year":"2000","unstructured":"Rogol, A., Clark, P., Roemmich, J.: Growth and pubertal development in children and adolescents: Effects of diet and physical activity. Am. J. Clin. Nutr. 72, 521\u20138 (2000). https:\/\/doi.org\/10.1093\/ajcn\/72.2.521S","journal-title":"Am. J. Clin. Nutr."},{"key":"1225_CR13","doi-asserted-by":"publisher","unstructured":"Wilpon, J.G., Jacobsen, C.N.: A study of speech recognition for children and the elderly. In: 1996 IEEE International Conference on Acoustics, Speech, and Signal Processing Conference Proceedings, vol. 1, pp. 349\u20133521 (1996). https:\/\/doi.org\/10.1109\/ICASSP.1996.541104","DOI":"10.1109\/ICASSP.1996.541104"},{"issue":"2","key":"1225_CR14","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1109\/89.985544","volume":"10","author":"S Narayanan","year":"2002","unstructured":"Narayanan, S., Potamianos, A.: Creating conversational interfaces for children. IEEE Trans Speech Audio Process 10(2), 65\u201378 (2002). https:\/\/doi.org\/10.1109\/89.985544","journal-title":"IEEE Trans Speech Audio Process"},{"key":"1225_CR15","doi-asserted-by":"crossref","unstructured":"Li, Q., Russell, M.: An analysis of the causes of increased error rates in children$$^{2}$$s speech recognition. (2002)","DOI":"10.21437\/ICSLP.2002-221"},{"key":"1225_CR16","doi-asserted-by":"publisher","unstructured":"Gerosa, M., Giuliani, D., Narayanan, S., Potamianos, A.: A review of asr technologies for children\u2019s speech. Proceedings of the 2nd Workshop on Child, Computer and Interaction, WOCCI \u201909 (2009). https:\/\/doi.org\/10.1145\/1640377.1640384","DOI":"10.1145\/1640377.1640384"},{"key":"1225_CR17","unstructured":"Gray, S., Willett, D., Lu, J., Pinto, J., Maergner, P., Bodenstab, N.: Child automatic speech recognition for us english: child interaction with living-room-electronic-devices. In: WOCCI (2014)"},{"key":"1225_CR18","doi-asserted-by":"publisher","unstructured":"Steidl, S., Stemmer, G., Hacker, C., Noeth, E., Niemann, H.: Improving children\u2019s speech recognition by hmm interpolation with an adults\u2019 speech recognizer 2781. pp. 600\u2013607 (2003). https:\/\/doi.org\/10.1007\/978-3-540-45243-0_76","DOI":"10.1007\/978-3-540-45243-0_76"},{"key":"1225_CR19","unstructured":"Booth, E., Carns, J., Kennington, C., Rafla, N.: Evaluating and improving child-directed automatic speech recognition. In: Proceedings of the 12th Language Resources and Evaluation Conference. European Language Resources Association, Marseille, France. pp. 6340\u20136345 (2020) https:\/\/aclanthology.org\/2020.lrec-1.778"},{"key":"1225_CR20","unstructured":"Amodei, D., Anubhai, R., Battenberg, E., Case, C., Casper, J., Catanzaro, B., Chen, J., Chrzanowski, M., Coates, A., Diamos, G., Elsen, E., Engel, J.H., Fan, L., Fougner, C., Han, T., Hannun, A.Y., Jun, B., LeGresley, P., Lin, L., Narang, S., Ng, A.Y., Ozair, S., Prenger, R., Raiman, J., Satheesh, S., Seetapun, D., Sengupta, S., Wang, Y., Wang, Z., Wang, C., Xiao, B., Yogatama, D., Zhan, J., Zhu, Z.: Deep speech 2: End-to-end speech recognition in english and mandarin. CoRR abs\/1512.02595 (2015) arXiv:1512.02595"},{"key":"1225_CR21","doi-asserted-by":"crossref","unstructured":"Matassoni, M., Gretter, R., Falavigna, D., Giuliani, D.: Non-native children speech recognition through transfer learning. CoRR abs\/1809.09658 (2018) arXiv:1809.09658","DOI":"10.1109\/ICASSP.2018.8462059"},{"key":"1225_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2020.101077","volume":"63","author":"P Gurunath Shivakumar","year":"2020","unstructured":"Gurunath Shivakumar, P., Georgiou, P.: Transfer learning from adult to children for speech recognition: evaluation, analysis and recommendations. Comput Speech Lang 63, 101077 (2020). https:\/\/doi.org\/10.1016\/j.csl.2020.101077","journal-title":"Comput Speech Lang"},{"key":"1225_CR23","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11042-022-13435-5","volume":"82","author":"V Kadyan","year":"2022","unstructured":"Kadyan, V., Hasija, T., Singh, A.: Prosody features based low resource punjabi children asr and t-nt classifier using data augmentation. Multimed Tools Appl 82, 1\u201322 (2022). https:\/\/doi.org\/10.1007\/s11042-022-13435-5","journal-title":"Multimed Tools Appl"},{"key":"1225_CR24","doi-asserted-by":"publisher","unstructured":"Shahnawazuddin, S., Dey, A., Sinha, R.: Pitch-adaptive front-end features for robust children\u2019s asr, pp. 3459\u20133463 (2016). https:\/\/doi.org\/10.21437\/Interspeech.2016-1020","DOI":"10.21437\/Interspeech.2016-1020"},{"key":"1225_CR25","unstructured":"Cole, R., Pellom, B., John-Paul, Hosom: University of Colorado Prompted and Read Children\u2019s Speech Corpus. Technical Report TR-CSLR2006-02, University of Colorado,2006 (2006)"},{"key":"1225_CR26","unstructured":"Cole, R., Pellom, B.: University of Colorado Read and Summarized stories Corpus. Technical Report TR-CSLR2006-03, Center for Spoken Language Research, University of Colorado,2006 (2006)"},{"key":"1225_CR27","doi-asserted-by":"publisher","DOI":"10.3390\/su14020614","author":"T Hasija","year":"2022","unstructured":"Hasija, T., Kadyan, V., Guleria, K., Alharbi, A., Alyami, H., Goyal, N.: Prosodic feature-based discriminatively trained low resource speech recognition system. Sustainability (2022). https:\/\/doi.org\/10.3390\/su14020614","journal-title":"Sustainability"},{"key":"1225_CR28","doi-asserted-by":"crossref","unstructured":"Lopes, C., Perdig\u00e3o, F.: Timit acoustic-phonetic continuous speech corpus. (2012)","DOI":"10.1186\/1687-6180-2012-158"},{"key":"1225_CR29","unstructured":"Batliner, A., Blomberg, M., Elenius, D., Giuliani, D., Gerosa, M., Hacker, C., Russell, M., Steidl, S., Wong, M.: The PF STAR Children\u2019s Speech Corpus"},{"key":"1225_CR30","doi-asserted-by":"publisher","unstructured":"Shivakumar, P.G., Georgiou, P.: Transfer Learning from Adult to Children for Speech Recognition: Evaluation, Analysis and Recommendations. arXiv (2018). https:\/\/doi.org\/10.48550\/ARXIV.1805.03322. https:\/\/arxiv.org\/abs\/1805.03322","DOI":"10.48550\/ARXIV.1805.03322"},{"key":"1225_CR31","doi-asserted-by":"crossref","unstructured":"Shobaki, K., Hosom, J.-p., Cole, R.A.: The ogi kids\u2019 speech corpus and recognizers. In: In ICSLP (2000)","DOI":"10.21437\/ICSLP.2000-800"},{"key":"1225_CR32","doi-asserted-by":"publisher","unstructured":"Tulsiani, H., Swarup, P., Rao, P.: Acoustic and language modeling for children\u2019s read speech assessment. In: 2017 Twenty-third National Conference on Communications (NCC), pp. 1\u20136 (2017). https:\/\/doi.org\/10.1109\/NCC.2017.8077101","DOI":"10.1109\/NCC.2017.8077101"},{"key":"1225_CR33","doi-asserted-by":"crossref","unstructured":"Samudravijaya, K., Rao, P., Agrawal, S.: Hindi speech database., pp. 456\u2013459 (2000)","DOI":"10.21437\/ICSLP.2000-847"},{"issue":"2","key":"1225_CR34","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1006\/csla.2000.0139","volume":"14","author":"M Russell","year":"2000","unstructured":"Russell, M., Series, R.W., Wallace, J.L., Brown, C., Skilling, A.: The star system: an interactive pronunciation tutor for young children. Comput Speech Lang 14(2), 161\u2013175 (2000). https:\/\/doi.org\/10.1006\/csla.2000.0139","journal-title":"Comput Speech Lang"},{"key":"1225_CR35","doi-asserted-by":"crossref","unstructured":"Mostow, J., Roth, S., Hauptmann, A., Kane, M.: A prototype reading coach that listens. In: AAAI (1994)","DOI":"10.1145\/215585.215665"},{"key":"1225_CR36","unstructured":"Serizel, R., Giuliani, D.: Deep neural network adaptation for children\u2019s and adults\u2019 speech recognition. (2014)"},{"key":"1225_CR37","doi-asserted-by":"publisher","unstructured":"Giuliani, D., BabaAli, B.: Large vocabulary children\u2019s speech recognition with DNN-HMM and SGMM acoustic modeling. In: Proc. Interspeech 2015, pp. 1635\u20131639 (2015). https:\/\/doi.org\/10.21437\/Interspeech.2015-378","DOI":"10.21437\/Interspeech.2015-378"},{"key":"1225_CR38","doi-asserted-by":"publisher","unstructured":"Abdel-Hamid, O., Mohamed, A.-r., Jiang, H., Penn, G.: Applying convolutional neural networks concepts to hybrid nn-hmm model for speech recognition. In: 2012 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4277\u20134280 (2012). https:\/\/doi.org\/10.1109\/ICASSP.2012.6288864","DOI":"10.1109\/ICASSP.2012.6288864"},{"issue":"2","key":"1225_CR39","doi-asserted-by":"publisher","first-page":"348","DOI":"10.1109\/TASL.2010.2047812","volume":"19","author":"J Tepperman","year":"2011","unstructured":"Tepperman, J., Lee, S., Narayanan, S., Alwan, A.: A generative student model for scoring word reading skills. IEEE Trans. Audio Speech Lang. Process. 19(2), 348\u2013360 (2011). https:\/\/doi.org\/10.1109\/TASL.2010.2047812","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"1225_CR40","doi-asserted-by":"publisher","unstructured":"Lovato, S.B., Piper, A.M., Wartella, E.A.: Hey google, do unicorns exist? conversational agents as a path to answers to children\u2019s questions. In: Proceedings of the 18th ACM International Conference on Interaction Design and Children. IDC \u201919, pp. 301\u2013313. Association for Computing Machinery, New York. (2019). https:\/\/doi.org\/10.1145\/3311927.3323150. https:\/\/doi.org\/10.1145\/3311927.3323150","DOI":"10.1145\/3311927.3323150"},{"key":"1225_CR41","doi-asserted-by":"crossref","unstructured":"Ghai, S., Sinha, R.: Exploring the role of spectral smoothing in context of children\u2019s speech recognition, pp. 1607\u20131610 (2009)","DOI":"10.21437\/Interspeech.2009-209"},{"key":"1225_CR42","first-page":"1559","volume":"1","author":"N Dehak","year":"2009","unstructured":"Dehak, N., Dehak, R., Kenny, P., Brummer, N., Ouellet, P., Dumouchel, P.: Support vector machines versus fast scoring in the low-dimensional total variability space for speaker verification. Interspeech. 1, 1559\u20131562 (2009)","journal-title":"Interspeech."},{"key":"1225_CR43","unstructured":"Gold, B., Morgan, N.: Speech and audio signal processing, (1999)"},{"key":"1225_CR44","unstructured":"Meng, Y.: Speech recognition on dsp: Algorithm optimization and performance analysis (2019)"},{"key":"1225_CR45","unstructured":"Bogert, B.P.: The quefrency analysis of time series for echoes : cepstrum, pseudo-autocovariance, cross-cepstrum and saphe cracking. (1963)"},{"key":"1225_CR46","doi-asserted-by":"publisher","first-page":"788","DOI":"10.1109\/TASL.2010.2064307","volume":"19","author":"N Dehak","year":"2011","unstructured":"Dehak, N., Kenny, P., Dehak, R., Dumouchel, P., Ouellet, P.: Front-end factor analysis for speaker verification. IEEE Trans Audio Speech Lang Process 19, 788\u2013798 (2011). https:\/\/doi.org\/10.1109\/TASL.2010.2064307","journal-title":"IEEE Trans Audio Speech Lang Process"},{"issue":"10","key":"1225_CR47","doi-asserted-by":"publisher","first-page":"1533","DOI":"10.1109\/TASLP.2014.2339736","volume":"22","author":"O Abdel-Hamid","year":"2014","unstructured":"Abdel-Hamid, O., Mohamed, A.-R., Jiang, H., Deng, L., Penn, G., Yu, D.: Convolutional neural networks for speech recognition. IEEE\/ACM Trans Audio Speech Lang Process 22(10), 1533\u20131545 (2014). https:\/\/doi.org\/10.1109\/TASLP.2014.2339736","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"1225_CR48","unstructured":"Goodfellow, I., Bengio, Y., Courville, A.: Deep Learning. MIT Press, (2016). http:\/\/www.deeplearningbook.org"},{"key":"1225_CR49","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv 1409.1556 (2014)"},{"key":"1225_CR50","unstructured":"Radford, A., Kim, J.W., Xu, T., Brockman, G., McLeavey, C., Sutskever, I.: Robust Speech Recognition via Large-Scale Weak Supervision (2022). https:\/\/arxiv.org\/abs\/2212.04356"},{"key":"1225_CR51","doi-asserted-by":"publisher","unstructured":"Wills, S., Bai, Y., Tejedor-Garc\u00ed\u00ada, C., Cucchiarini, C., Strik, H.: Automatic Speech Recognition of Non-Native Child Speech for Language Learning Applications. https:\/\/doi.org\/10.48550\/arXiv.2306.16710","DOI":"10.48550\/arXiv.2306.16710"},{"key":"1225_CR52","doi-asserted-by":"publisher","DOI":"10.3390\/e24101490","author":"K Radha","year":"2022","unstructured":"Radha, K., Bansal, M.: Audio augmentation for non-native children\u2019s speech recognition through discriminative learning. Entropy (2022). https:\/\/doi.org\/10.3390\/e24101490","journal-title":"Entropy"},{"key":"1225_CR53","doi-asserted-by":"crossref","unstructured":"Kathania, H., Singh, M., Gr\u00f3sz, T., Kurimo, M.: Data augmentation using prosody and false starts to recognize non-native children\u2019s speech (2020). https:\/\/arxiv.org\/abs\/2008.12914","DOI":"10.21437\/Interspeech.2020-2199"},{"key":"1225_CR54","doi-asserted-by":"publisher","unstructured":"Knill, K., Wang, L., Wang, Y., Wu, X., Gales, M.: Non-native children\u2019s automatic speech recognition: The interspeech 2020 shared task alta systems (2020). https:\/\/doi.org\/10.17863\/CAM.55812","DOI":"10.17863\/CAM.55812"}],"container-title":["Universal Access in the Information Society"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10209-025-01225-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10209-025-01225-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10209-025-01225-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,5]],"date-time":"2025-12-05T11:34:48Z","timestamp":1764934488000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10209-025-01225-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,23]]},"references-count":54,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["1225"],"URL":"https:\/\/doi.org\/10.1007\/s10209-025-01225-3","relation":{},"ISSN":["1615-5289","1615-5297"],"issn-type":[{"value":"1615-5289","type":"print"},{"value":"1615-5297","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,23]]},"assertion":[{"value":"24 April 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 May 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}