{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T21:35:44Z","timestamp":1784064944754,"version":"3.55.0"},"reference-count":57,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101954","type":"journal-article","created":{"date-parts":[[2026,2,12]],"date-time":"2026-02-12T16:44:06Z","timestamp":1770914646000},"page":"101954","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["Modeling the temporal envelope of sub-band signals for improving the performance of children\u2019s speech recognition system in zero-resource scenario"],"prefix":"10.1016","volume":"100","author":[{"given":"Kaustav","family":"Das","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Biswaranjan","family":"Pattanayak","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gayadhar","family":"Pradhan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"3","key":"10.1016\/j.csl.2026.101954_b1","doi-asserted-by":"crossref","first-page":"75","DOI":"10.1109\/MSP.2009.932166","article-title":"Developments and directions in speech recognition and understanding, part 1 [DSP Education]","volume":"26","author":"Baker","year":"2009","journal-title":"IEEE Signal Process. Mag."},{"key":"10.1016\/j.csl.2026.101954_b2","doi-asserted-by":"crossref","unstructured":"Batliner, A., Blomberg, M., D\u2019Arcy, S., Elenius, D., Giuliani, D., Gerosa, M., Hacker, C., Russell, M., Steidl, S., Wong, M., 2005. The PF_STAR children\u2019s speech corpus. In: Proc. European Conference on Speech Communication and Technology.","DOI":"10.21437\/Interspeech.2005-705"},{"key":"10.1016\/j.csl.2026.101954_b3","unstructured":"Bridle, J.S., 1973. An efficient elastic-template method for detecting given words in running speech. In: Proc. Brit. Acoust. Soc. Meeting. pp. 1\u20134."},{"issue":"1","key":"10.1016\/j.csl.2026.101954_b4","doi-asserted-by":"crossref","first-page":"9","DOI":"10.1023\/A:1013670312989","article-title":"Phonetic searching vs. LVCSR: How to find what you really want in audio archives","volume":"5","author":"Cardillo","year":"2002","journal-title":"Int. J. Speech Technol."},{"key":"10.1016\/j.csl.2026.101954_b5","unstructured":"Clements, M., Cardillo, P., Miller, M., 2001. Phonetic searching of digital audio. In: Proc. Broadcast Engineering Conference. pp. 131\u2013140."},{"issue":"4","key":"10.1016\/j.csl.2026.101954_b6","doi-asserted-by":"crossref","first-page":"357","DOI":"10.1109\/TASSP.1980.1163420","article-title":"Comparison of parametric representations for monosyllabic word recognition in continuously spoken sentences","volume":"28","author":"Davis","year":"1980","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"issue":"5","key":"10.1016\/j.csl.2026.101954_b7","doi-asserted-by":"crossref","first-page":"357","DOI":"10.1109\/89.466659","article-title":"Speaker adaptation using constrained estimation of Gaussian mixtures","volume":"3","author":"Digalakis","year":"1995","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"10.1016\/j.csl.2026.101954_b8","doi-asserted-by":"crossref","unstructured":"Do, C.T., Stylianou, Y., 2017. Improved Automatic Speech Recognition Using Subband Temporal Envelope Features and Time-Delay Neural Network Denoising Autoencoder. In: Proc. Interspeech. pp. 3832\u20133836.","DOI":"10.21437\/Interspeech.2017-1096"},{"key":"10.1016\/j.csl.2026.101954_b9","doi-asserted-by":"crossref","unstructured":"Dubois, C., Charlet, D., 2008. Using textual information from LVCSR transcripts for phonetic-based spoken term detection. In: Proc. ICASSP. pp. 4961\u20134964.","DOI":"10.1109\/ICASSP.2008.4518771"},{"key":"10.1016\/j.csl.2026.101954_b10","unstructured":"Fisher, W.M., 1986. Ther DARPA speech recognition research database: specifications and status. In: Proc. DARPA Workshop on Speech Recognition. pp. 93\u201399."},{"key":"10.1016\/j.csl.2026.101954_b11","unstructured":"Fousek, P., Hermansky, H., 2006. Towards ASR based on hierarchical posterior-based keyword recognition. In: Proc. ICASSP. I\u2013I."},{"key":"10.1016\/j.csl.2026.101954_b12","first-page":"1","article-title":"An approach for reducing pitch induced mismatches to detect keywords in children\u2019s speech","author":"Garnaik","year":"2022","journal-title":"Multimedia Tools Appl."},{"key":"10.1016\/j.csl.2026.101954_b13","doi-asserted-by":"crossref","unstructured":"Ghai, S., Sinha, R., 2010. Analyzing pitch robustness of PMVDR and MFCC features for children\u2019s speech recognition. In: Proc. SPCOM. pp. 1\u20135.","DOI":"10.1109\/SPCOM.2010.5560549"},{"key":"10.1016\/j.csl.2026.101954_b14","doi-asserted-by":"crossref","unstructured":"Ghai, S., Sinha, R., 2011. A study on the effect of pitch on LPCC and PLPC features for children\u2019s ASR in comparison to MFCC. In: Proc. Interspeech. pp. 2589\u20132592.","DOI":"10.21437\/Interspeech.2011-662"},{"key":"10.1016\/j.csl.2026.101954_b15","doi-asserted-by":"crossref","unstructured":"He, L., Dellwo, V., 2016. A Praat-based algorithm to extract the amplitude envelope and temporal fine structure using the Hilbert transform. In: Proc. Interspeech. pp. 530\u2013534.","DOI":"10.21437\/Interspeech.2016-1447"},{"issue":"4","key":"10.1016\/j.csl.2026.101954_b16","doi-asserted-by":"crossref","first-page":"1738","DOI":"10.1121\/1.399423","article-title":"Perceptual linear predictive (PLP) analysis of speech","volume":"87","author":"Hermansky","year":"1990","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101954_b17","first-page":"289","article-title":"Temporal patterns (TRAPs) in ASR of noisy speech","volume":"vol. 1","author":"Hermansky","year":"1999"},{"key":"10.1016\/j.csl.2026.101954_b18","series-title":"Basic Principles of a General Theory of Linear Integrals Equations","author":"Hilbert","year":"1912"},{"issue":"6","key":"10.1016\/j.csl.2026.101954_b19","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","article-title":"Deep neural networks for acoustic modeling in speech recognition","volume":"29","author":"Hinton","year":"2012","journal-title":"IEEE Signal Process. Mag."},{"key":"10.1016\/j.csl.2026.101954_b20","doi-asserted-by":"crossref","first-page":"98","DOI":"10.1016\/j.specom.2021.11.003","article-title":"A formant modification method for improved ASR of children\u2019s speech","volume":"136","author":"Kathania","year":"2022","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101954_b21","doi-asserted-by":"crossref","first-page":"1853","DOI":"10.1109\/LSP.2021.3108509","article-title":"Temporal envelope and fine structure cues for dysarthric speech detection using CNNs","volume":"28","author":"Kodrasi","year":"2021","journal-title":"IEEE Signal Process. Lett."},{"issue":"1","key":"10.1016\/j.csl.2026.101954_b22","doi-asserted-by":"crossref","first-page":"49","DOI":"10.1109\/89.650310","article-title":"A frequency warping approach to speaker normalization","volume":"6","author":"Lee","year":"1998","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"10.1016\/j.csl.2026.101954_b23","doi-asserted-by":"crossref","first-page":"1892","DOI":"10.1007\/s00034-020-01565-w","article-title":"A pitch and noise robust keyword spotting system using SMAC features with prosody modification","volume":"40","author":"Maity","year":"2021","journal-title":"Circuits Systems Signal Process."},{"key":"10.1016\/j.csl.2026.101954_b24","doi-asserted-by":"crossref","first-page":"183","DOI":"10.1007\/s10772-013-9217-1","article-title":"Recent developments in spoken term detection: a survey","volume":"17","author":"Mandal","year":"2014","journal-title":"Int. J. Speech Technol."},{"key":"10.1016\/j.csl.2026.101954_b25","doi-asserted-by":"crossref","unstructured":"Mishne, G., Carmel, D., Hoory, R., Roytman, A., Soffer, A., 2005. Automatic analysis of call-center conversations. In: Proc. ACM International Conference on Information and Knowledge Management. pp. 453\u2013459.","DOI":"10.1145\/1099554.1099684"},{"key":"10.1016\/j.csl.2026.101954_b26","series-title":"Phonetic Search Methods for Large Speech Databases","author":"Moyal","year":"2013"},{"key":"10.1016\/j.csl.2026.101954_b27","doi-asserted-by":"crossref","first-page":"1602","DOI":"10.1109\/TASL.2008.2004526","article-title":"Epoch extraction from speech signals","volume":"16","author":"Murthy","year":"2008","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.101954_b28","series-title":"Open keyword search 2013 evaluation (openkws13) plan","author":"National Institute of Standards and Technology (NIST)","year":"2013"},{"key":"10.1016\/j.csl.2026.101954_b29","series-title":"Proc. Interspeech","first-page":"2070","article-title":"An empirical analysis of word error rate and keyword error rate","author":"Park","year":"2008"},{"key":"10.1016\/j.csl.2026.101954_b30","doi-asserted-by":"crossref","first-page":"183","DOI":"10.1016\/j.patrec.2021.07.015","article-title":"Pitch-robust acoustic feature using single frequency filtering for children\u2019s KWS","volume":"150","author":"Pattanayak","year":"2021","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.csl.2026.101954_b31","doi-asserted-by":"crossref","unstructured":"Pattanayak, B., Pradhan, G., 2022. Significance of single frequency filter for the development of children\u2019s KWS system. In: Proc. Interspeech. pp. 3183\u20133187.","DOI":"10.21437\/Interspeech.2022-980"},{"issue":"5","key":"10.1016\/j.csl.2026.101954_b32","doi-asserted-by":"crossref","first-page":"544","DOI":"10.1049\/iet-spr.2019.0027","article-title":"Adaptive spectral smoothening for development of robust keyword spotting system","volume":"13","author":"Pattanayak","year":"2019","journal-title":"IET Signal Process."},{"key":"10.1016\/j.csl.2026.101954_b33","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., et al., 2011. The Kaldi speech recognition toolkit. In: Proc. ASRU."},{"key":"10.1016\/j.csl.2026.101954_b34","doi-asserted-by":"crossref","unstructured":"Prasanna, S., Govind, D., Rao, K.S., Yegnanarayana, B., 2010. Fast prosody modification using instants of significant excitation. In: Proc. Speech Prosody.","DOI":"10.21437\/SpeechProsody.2010-126"},{"issue":"8","key":"10.1016\/j.csl.2026.101954_b35","doi-asserted-by":"crossref","first-page":"2552","DOI":"10.1109\/TASL.2011.2155061","article-title":"Significance of vowel-like regions for speaker verification under degraded condition","volume":"19","author":"Prasanna","year":"2011","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"4","key":"10.1016\/j.csl.2026.101954_b36","doi-asserted-by":"crossref","first-page":"556","DOI":"10.1109\/TASL.2008.2010884","article-title":"Vowel onset point detection using source, spectral peaks, and modulation spectrum energies","volume":"17","author":"Prasanna","year":"2009","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.101954_b37","doi-asserted-by":"crossref","unstructured":"Robinson, T., Fransen, J., Pye, D., Foote, J., Renals, S., 1995. WSJCAMO: a British English speech corpus for large vocabulary continuous speech recognition. In: Proc. ICASSP. pp. 81\u201384.","DOI":"10.1109\/ICASSP.1995.479278"},{"key":"10.1016\/j.csl.2026.101954_b38","doi-asserted-by":"crossref","unstructured":"Rose, R.C., Paul, D.B., 1990. A hidden Markov model based keyword recognition system. In: Proc. ICASSP. pp. 129\u2013132.","DOI":"10.1109\/ICASSP.1990.115555"},{"key":"10.1016\/j.csl.2026.101954_b39","first-page":"1","article-title":"Data-adaptive single-pole filtering of magnitude spectra for robust keyword spotting","author":"Rout","year":"2022","journal-title":"Circuits Systems Signal Process."},{"key":"10.1016\/j.csl.2026.101954_b40","doi-asserted-by":"crossref","first-page":"101","DOI":"10.1016\/j.specom.2022.09.004","article-title":"Enhancement of formant regions in magnitude spectra to develop children\u2019s KWS system in zero resource scenario","volume":"144","author":"Rout","year":"2022","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101954_b41","doi-asserted-by":"crossref","first-page":"61","DOI":"10.1007\/978-1-4419-5951-5_4","article-title":"\u201cYour word is my command\u201d: Google search by voice: A case study","author":"Schalkwyk","year":"2010","journal-title":"Adv. Speech Recognit.: Mob. Environ. Call Centers Clin."},{"key":"10.1016\/j.csl.2026.101954_b42","doi-asserted-by":"crossref","first-page":"11","DOI":"10.1016\/j.dsp.2018.12.011","article-title":"Improving the performance of keyword spotting system for children\u2019s speech through prosody modification","volume":"86","author":"Shahnawazuddin","year":"2019","journal-title":"Digit. Signal Process."},{"key":"10.1016\/j.csl.2026.101954_b43","doi-asserted-by":"crossref","first-page":"103","DOI":"10.1016\/j.csl.2017.10.007","article-title":"Assessment of pitch-adaptive front-end signal processing for children\u2019s speech recognition","volume":"48","author":"Sinha","year":"2018","journal-title":"Comput. Speech Lang.","ISSN":"https:\/\/id.crossref.org\/issn\/0885-2308","issn-type":"print"},{"key":"10.1016\/j.csl.2026.101954_b44","doi-asserted-by":"crossref","unstructured":"Sm\u00eddl, L., Psutka, J.V., 2006. Comparison of keyword spotting methods for searching in speech. In: Proc. Ninth International Conference on Spoken Language Processing.","DOI":"10.21437\/Interspeech.2006-521"},{"key":"10.1016\/j.csl.2026.101954_b45","series-title":"Musan: A music, speech, and noise corpus","author":"Snyder","year":"2015"},{"key":"10.1016\/j.csl.2026.101954_b46","series-title":"Acoustic Keyword Spotting in Speech with Applications to Data Mining","author":"Thambiratnam","year":"2005"},{"key":"10.1016\/j.csl.2026.101954_b47","doi-asserted-by":"crossref","unstructured":"Tsao, Y., Li, J., Lee, C.-H., 2009. Ensemble speaker and speaking environment modeling approach with advanced online estimation process. In: Proc. ICASSP. pp. 3833\u20133836.","DOI":"10.1109\/ICASSP.2009.4960463"},{"issue":"3","key":"10.1016\/j.csl.2026.101954_b48","doi-asserted-by":"crossref","first-page":"247","DOI":"10.1016\/0167-6393(93)90095-3","article-title":"Assessment for automatic speech recognition: II. NOISEX-92: A database and an experiment to study the effect of additive noise on speech recognition systems","volume":"12","author":"Varga","year":"1993","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101954_b49","doi-asserted-by":"crossref","unstructured":"Wallace, R., Vogt, R., Sridharan, S., 2007. A phonetic search approach to the 2006 NIST spoken term detection evaluation. In: Proc. Interspeech. pp. 2385\u20132388.","DOI":"10.21437\/Interspeech.2007-180"},{"key":"10.1016\/j.csl.2026.101954_b50","doi-asserted-by":"crossref","unstructured":"Wang, Y., Hansen, J., Allu, G.K., Kumaresan, R., 2003. Average instantaneous frequency (AIF) and average log-envelopes (ALE) for ASR with the AURORA 2 Database. In: Proc. European Conference on Speech Communication and Technology.","DOI":"10.21437\/Eurospeech.2003-7"},{"key":"10.1016\/j.csl.2026.101954_b51","doi-asserted-by":"crossref","unstructured":"Wegmann, S., Faria, A., Janin, A., Riedhammer, K., Morgan, N., 2013. The tao of ATWV: Probing the mysteries of keyword search performance. In: Proc. ASRU. pp. 192\u2013197.","DOI":"10.1109\/ASRU.2013.6707728"},{"key":"10.1016\/j.csl.2026.101954_b52","doi-asserted-by":"crossref","unstructured":"Witbrock, M.J., Hauptmann, A.G., 1997. Using words and phonetic strings for efficient information retrieval from imperfectly transcribed spoken documents. In: Proc. ACM International Conference on Digital Libraries. pp. 30\u201335.","DOI":"10.1145\/263690.263779"},{"issue":"12","key":"10.1016\/j.csl.2026.101954_b53","doi-asserted-by":"crossref","first-page":"1822","DOI":"10.1109\/LSP.2019.2950763","article-title":"Significance of pitch-based spectral normalization for children\u2019s speech recognition","volume":"26","author":"Yadav","year":"2019","journal-title":"IEEE Signal Process. Lett."},{"key":"10.1016\/j.csl.2026.101954_b54","doi-asserted-by":"crossref","first-page":"pp. 102922","DOI":"10.1016\/j.dsp.2020.102922","article-title":"Pitch and noise normalized acoustic feature for children\u2019s ASR","volume":"109","author":"Yadav","year":"2021","journal-title":"Digit. Signal Process."},{"key":"10.1016\/j.csl.2026.101954_b55","doi-asserted-by":"crossref","unstructured":"Yadav, I.C., Shahnawazuddin, S., Govind, D., Pradhan, G., 2018. Spectral Smoothing by Variational mode Decomposition and its Effect on Noise and Pitch Robustness of ASR System. In: Proc. ICASSP. pp. 5629\u20135633.","DOI":"10.1109\/ICASSP.2018.8462133"},{"key":"10.1016\/j.csl.2026.101954_b56","doi-asserted-by":"crossref","first-page":"55","DOI":"10.1016\/j.dsp.2018.12.013","article-title":"Addressing noise and pitch sensitivity of speech recognition system through variational mode decomposition based spectral smoothing","volume":"86","author":"Yadav","year":"2019","journal-title":"Digit. Signal Process."},{"issue":"5","key":"10.1016\/j.csl.2026.101954_b57","doi-asserted-by":"crossref","first-page":"pp. 45","DOI":"10.1109\/79.536824","article-title":"A review of large-vocabulary continuous-speech","volume":"13","author":"Young","year":"1996","journal-title":"IEEE Signal Process. Mag."}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000173?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000173?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:10:57Z","timestamp":1779225057000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000173"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":57,"alternative-id":["S0885230826000173"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101954","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Modeling the temporal envelope of sub-band signals for improving the performance of children\u2019s speech recognition system in zero-resource scenario","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101954","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"101954"}}