{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T05:38:04Z","timestamp":1776922684952,"version":"3.51.2"},"reference-count":54,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Speech Communication"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1016\/j.specom.2026.103392","type":"journal-article","created":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T15:54:08Z","timestamp":1774972448000},"page":"103392","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["A study on the layer-wise transferability of self-supervised learning features for children\u2019s speech processing tasks"],"prefix":"10.1016","volume":"180","author":[{"given":"Abhijit","family":"Sinha","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6367-5203","authenticated-orcid":false,"given":"Hemant Kumar","family":"Kathania","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mikko","family":"Kurimo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.specom.2026.103392_b1","doi-asserted-by":"crossref","DOI":"10.1016\/j.dsp.2024.104385","article-title":"Developing children\u2019s ASR system under low-resource conditions using end-to-end architecture","volume":"146","author":"Ankita","year":"2024","journal-title":"Digit. Signal Process."},{"key":"10.1016\/j.specom.2026.103392_b2","series-title":"International Conference on Machine Learning","first-page":"1298","article-title":"Data2vec: A general framework for self-supervised learning in speech, vision and language","author":"Baevski","year":"2022"},{"key":"10.1016\/j.specom.2026.103392_b3","first-page":"12449","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.specom.2026.103392_b4","series-title":"The PF_STAR children\u2019s speech corpus","author":"Batliner","year":"2005"},{"key":"10.1016\/j.specom.2026.103392_b5","series-title":"Proc. INTERSPEECH","first-page":"2761","article-title":"The PF_STAR children\u2019s speech corpus","author":"Batliner","year":"2005"},{"issue":"6","key":"10.1016\/j.specom.2026.103392_b6","doi-asserted-by":"crossref","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","article-title":"Wavlm: Large-scale self-supervised pre-training for full stack speech processing","volume":"16","author":"Chen","year":"2022","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.specom.2026.103392_b7","series-title":"Interspeech","first-page":"2410","article-title":"A survey about databases of children\u2019s speech","author":"Claus","year":"2013"},{"key":"10.1016\/j.specom.2026.103392_b8","article-title":"The CMU kids corpus","volume":"11","author":"Eskenazi","year":"1997","journal-title":"Linguist. Data Consort."},{"key":"10.1016\/j.specom.2026.103392_b9","series-title":"Interspeech","article-title":"DRAFT: A novel framework to reduce domain shifting in self-supervised learning and its application to children\u2019s ASR","author":"Fan","year":"2022"},{"key":"10.1016\/j.specom.2026.103392_b10","series-title":"Interspeech 2024","first-page":"5173","article-title":"Benchmarking children\u2019s ASR with supervised and self-supervised speech foundation models","author":"Fan","year":"2024"},{"key":"10.1016\/j.specom.2026.103392_b11","series-title":"ICASSP","first-page":"1","article-title":"Using modified adult speech as data augmentation for child speech recognition","author":"Fan","year":"2023"},{"issue":"6","key":"10.1016\/j.specom.2026.103392_b12","doi-asserted-by":"crossref","first-page":"1242","DOI":"10.1109\/JSTSP.2022.3200910","article-title":"Towards better domain adaptation for self-supervised models: A case study of child ASR","volume":"16","author":"Fan","year":"2022","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.specom.2026.103392_b13","doi-asserted-by":"crossref","DOI":"10.1016\/j.csl.2023.101567","article-title":"Towards inclusive automatic speech recognition","volume":"84","author":"Feng","year":"2024","journal-title":"Comput. Speech Lang."},{"issue":"10\u201311","key":"10.1016\/j.specom.2026.103392_b14","doi-asserted-by":"crossref","first-page":"847","DOI":"10.1016\/j.specom.2007.01.002","article-title":"Acoustic variability and automatic recognition of children\u2019s speech","volume":"49","author":"Gerosa","year":"2007","journal-title":"Speech Commun."},{"key":"10.1016\/j.specom.2026.103392_b15","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103392_b16","series-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Exploring self-supervised pre-trained asr models for dysarthric and elderly speech recognition","author":"Hu","year":"2023"},{"key":"10.1016\/j.specom.2026.103392_b17","doi-asserted-by":"crossref","first-page":"46938","DOI":"10.1109\/ACCESS.2023.3275106","article-title":"A wav2vec2-based experimental study on self-supervised learning methods to improve child speech recognition","volume":"11","author":"Jain","year":"2023","journal-title":"IEEE Access"},{"key":"10.1016\/j.specom.2026.103392_b18","doi-asserted-by":"crossref","unstructured":"Johnson, A., Fan, R., Morris, R., Alwan, A., 2022. LPC Augment: an LPC-based ASR Data Augmentation Algorithm for Low and Zero-Resource Children\u2019s Dialects. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 8577\u20138581.","DOI":"10.1109\/ICASSP43922.2022.9746281"},{"key":"10.1016\/j.specom.2026.103392_b19","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7669","article-title":"Libri-light: A benchmark for asr with limited or no supervision","author":"Kahn","year":"2020"},{"key":"10.1016\/j.specom.2026.103392_b20","unstructured":"Kathania, H.K., Kadiri, S.R., Alku, P., Kurimo, M., 2020. Study of Formant Modification for Children ASR. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP."},{"key":"10.1016\/j.specom.2026.103392_b21","doi-asserted-by":"crossref","DOI":"10.1016\/j.specom.2021.11.003","article-title":"A formant modification method for improved ASR of children\u2019s speech","author":"Kathania","year":"2022","journal-title":"Speech Commun."},{"issue":"5","key":"10.1016\/j.specom.2026.103392_b22","doi-asserted-by":"crossref","first-page":"3158","DOI":"10.1121\/1.2981639","article-title":"Speech production variability in fricatives of children and adults: Results of functional data analysis","volume":"124","author":"Koenig","year":"2008","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.specom.2026.103392_b23","doi-asserted-by":"crossref","DOI":"10.1016\/j.patrec.2025.08.010","article-title":"Zero-shot KWS for children\u2019s speech using layer-wise features from SSL models","author":"Kutum","year":"2025","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.specom.2026.103392_b24","doi-asserted-by":"crossref","DOI":"10.1121\/1.426686","article-title":"Acoustics of children\u2019s speech: Developmental changes of temporal and spectral parameters","author":"Lee","year":"1999","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.specom.2026.103392_b25","series-title":"Interspeech 2023","first-page":"1035","article-title":"Towards robust family-infant audio analysis based on unsupervised pretraining of wav2vec 2.0 on large-scale unlabeled family audio","author":"Li","year":"2023"},{"key":"10.1016\/j.specom.2026.103392_b26","doi-asserted-by":"crossref","unstructured":"Li, J., Hasegawa-Johnson, M., McElwain, N.L., 2024a. Analysis of Self-Supervised Speech Models on Children\u2019s Speech and Infant Vocalizations. In: 2024 IEEE International Conference on Acoustics, Speech, and Signal Processing Workshops. ICASSPW, pp. 550\u2013554.","DOI":"10.1109\/ICASSPW62465.2024.10626416"},{"key":"10.1016\/j.specom.2026.103392_b27","first-page":"550","article-title":"Analysis of self-supervised speech models on children\u2019s speech and infant vocalizations","author":"Li","year":"2024","journal-title":"IEEE Int. Conf. Acoust. Speech Signal Process. Work. (ICASSP)"},{"key":"10.1016\/j.specom.2026.103392_b28","doi-asserted-by":"crossref","first-page":"2267","DOI":"10.1109\/TASLP.2021.3091805","article-title":"Recent progress in the CUHK dysarthric speech recognition system","volume":"29","author":"Liu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103392_b29","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5206","article-title":"Librispeech: an asr corpus based on public domain audio books","author":"Panayotov","year":"2015"},{"key":"10.1016\/j.specom.2026.103392_b30","series-title":"2021 IEEE Automatic Speech Recognition and Understanding Workshop","first-page":"914","article-title":"Layer-wise analysis of a self-supervised speech representation model","author":"Pasad","year":"2021"},{"key":"10.1016\/j.specom.2026.103392_b31","series-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Comparative layer-wise analysis of self-supervised speech models","author":"Pasad","year":"2023"},{"key":"10.1016\/j.specom.2026.103392_b32","doi-asserted-by":"crossref","DOI":"10.3390\/app14062353","article-title":"Improving end-to-end models for children\u2019s speech recognition","author":"Patel","year":"2024","journal-title":"Appl. Sci."},{"key":"10.1016\/j.specom.2026.103392_b33","series-title":"Interspeech 2021","first-page":"3400","article-title":"Emotion recognition from speech using wav2vec 2.0 embeddings","author":"Pepino","year":"2021"},{"issue":"6","key":"10.1016\/j.specom.2026.103392_b34","doi-asserted-by":"crossref","first-page":"603","DOI":"10.1109\/TSA.2003.818026","article-title":"Robust recognition of children\u2019s speech","volume":"11","author":"Potamianos","year":"2003","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"10.1016\/j.specom.2026.103392_b35","series-title":"IEEE 2011 Workshop on Automatic Speech Recognition and Understanding","article-title":"The Kaldi speech recognition toolkit","author":"Povey","year":"2011"},{"key":"10.1016\/j.specom.2026.103392_b36","series-title":"Proceedings of the 40th International Conference on Machine Learning","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume":"vol. 202","author":"Radford","year":"2023"},{"key":"10.1016\/j.specom.2026.103392_b37","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2023.107661","article-title":"Speech and speaker recognition using raw waveform modeling for adult and children\u2019s speech: A comprehensive review","volume":"131","author":"Radha","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.specom.2026.103392_b38","doi-asserted-by":"crossref","unstructured":"Robinson, T., Fransen, J., Pye, D., Foote, J., Renals, S., 1995. WSJCAM0: A British English Speech Corpus For Large Vocabulary Continuous Speech Recognition. In: Proc. ICASSP. Vol. 1, pp. 81\u201384.","DOI":"10.1109\/ICASSP.1995.479278"},{"key":"10.1016\/j.specom.2026.103392_b39","doi-asserted-by":"crossref","unstructured":"Rolland, T., Abad, A., 2024. Exploring Adapters with Conformers for Children\u2019s Automatic Speech Recognition. In: ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 12747\u201312751.","DOI":"10.1109\/ICASSP48485.2024.10447091"},{"key":"10.1016\/j.specom.2026.103392_b40","unstructured":"Rolland, T., Abad, A., Cucchiarini, C., Strik, H., 2022. Multilingual Transfer Learning for Children Automatic Speech Recognition. In: International Conference on Language Resources and Evaluation."},{"key":"10.1016\/j.specom.2026.103392_b41","doi-asserted-by":"crossref","DOI":"10.1016\/j.patrec.2019.12.019","article-title":"Creating speaker independent ASR system through prosody modification based data augmentation","author":"Shahnawazuddin","year":"2020","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.specom.2026.103392_b42","article-title":"Transfer learning from adult to children for speech recognition: Evaluation, analysis and recommendations","volume":"63","author":"Shivakumar","year":"2020","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.specom.2026.103392_b43","article-title":"End-to-end neural systems for automatic children speech recognition: An empirical study","volume":"72","author":"Shivakumar","year":"2022","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.specom.2026.103392_b44","doi-asserted-by":"crossref","first-page":"3759","DOI":"10.1109\/LSP.2025.3602636","article-title":"Can layer-wise SSL features improve zero-shot ASR performance for children\u2019s speech?","volume":"32","author":"Sinha","year":"2025","journal-title":"IEEE Signal Process. Lett."},{"key":"10.1016\/j.specom.2026.103392_b45","series-title":"INTERSPEECH","article-title":"Beyond traditional speech modifications : Utilizing self supervised features for enhanced zero-shot children ASR","author":"Sinha","year":"2025"},{"key":"10.1016\/j.specom.2026.103392_b46","doi-asserted-by":"crossref","unstructured":"Sinha, A., Kumar, H., Joshi, M., Kathania, H.K., Narayanan, S., Kadiri, S.R., 2025c. Layer-Wise Analysis of Self-Supervised Representations for Age and Gender Classification in Children\u2019s Speech. In: Proc. WOCCI 2025. pp. 46\u201350.","DOI":"10.21437\/WOCCI.2025-10"},{"key":"10.1016\/j.specom.2026.103392_b47","series-title":"International Conference on Signal Processing and Communications","first-page":"1","article-title":"Effect of speech modification on Wav2Vec2 models for children speech recognition","author":"Sinha","year":"2024"},{"key":"10.1016\/j.specom.2026.103392_b48","series-title":"Interspeech","article-title":"Children\u2019s speech recognition through discrete token enhancement","author":"Sukhadia","year":"2024"},{"key":"10.1016\/j.specom.2026.103392_b49","series-title":"Interspeech","article-title":"Transfer learning for robust low-resource children\u2019s speech ASR with transformers and source-filter warping","author":"Thienpondt","year":"2022"},{"key":"10.1016\/j.specom.2026.103392_b50","series-title":"Interspeech","article-title":"Analysis of disfluency in children\u2019s speech","author":"Tran","year":"2020"},{"key":"10.1016\/j.specom.2026.103392_b51","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7967","article-title":"Fine-tuning wav2vec2 for speaker recognition","author":"Vaessen","year":"2022"},{"key":"10.1016\/j.specom.2026.103392_b52","doi-asserted-by":"crossref","first-page":"1510","DOI":"10.1044\/1092-4388(2007\/104)","article-title":"Vowel acoustic space development in children: a synthesis of acoustic and anatomic data","volume":"50 6","author":"Vorperian","year":"2007","journal-title":"J. Speech Lang. Hear. Res. : JSLHR"},{"key":"10.1016\/j.specom.2026.103392_b53","series-title":"Interspeech","first-page":"1194","article-title":"SUPERB: Speech processing universal PERformance benchmark","author":"wen Yang","year":"2021"},{"key":"10.1016\/j.specom.2026.103392_b54","series-title":"Interspeech","article-title":"On the difficulties of automatic speech recognition for kindergarten-aged children","author":"Yeung","year":"2018"}],"container-title":["Speech Communication"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000403?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000403?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T04:45:33Z","timestamp":1776919533000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167639326000403"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":54,"alternative-id":["S0167639326000403"],"URL":"https:\/\/doi.org\/10.1016\/j.specom.2026.103392","relation":{},"ISSN":["0167-6393"],"issn-type":[{"value":"0167-6393","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A study on the layer-wise transferability of self-supervised learning features for children\u2019s speech processing tasks","name":"articletitle","label":"Article Title"},{"value":"Speech Communication","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.specom.2026.103392","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"103392"}}