{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T04:08:12Z","timestamp":1743048492538,"version":"3.40.3"},"publisher-location":"Cham","reference-count":24,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031483080"},{"type":"electronic","value":"9783031483097"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-48309-7_40","type":"book-chapter","created":{"date-parts":[[2023,11,21]],"date-time":"2023-11-21T20:03:21Z","timestamp":1700597001000},"page":"494-505","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Gammatone-Filterbank Based Pitch-Normalized Cepstral Coefficients for\u00a0Zero-Resource Children\u2019s ASR"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3916-9693","authenticated-orcid":false,"given":"Syed","family":"Shahnawazuddin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6688-0872","authenticated-orcid":false,"family":"Ankita","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Avinash","family":"Kumar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6367-5203","authenticated-orcid":false,"given":"Hemant Kumar","family":"Kathania","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,22]]},"reference":[{"key":"40_CR1","doi-asserted-by":"crossref","unstructured":"Batliner, A., et al.: The PF_STAR children\u2019s speech corpus. In: Proceedings of INTERSPEECH, pp. 2761\u20132764 (2005)","DOI":"10.21437\/Interspeech.2005-705"},{"issue":"12","key":"40_CR2","doi-asserted-by":"publisher","first-page":"1293","DOI":"10.3390\/app7121293","volume":"7","author":"EP Damsk\u00e4gg","year":"2017","unstructured":"Damsk\u00e4gg, E.P., V\u00e4lim\u00e4ki, V.: Audio time stretching using fuzzy classification of spectral bins. Appl. Sci. 7(12), 1293 (2017). https:\/\/doi.org\/10.3390\/app7121293","journal-title":"Appl. Sci."},{"key":"40_CR3","doi-asserted-by":"crossref","unstructured":"D\u2019Arcy, S., Russell, M.: A comparison of human and computer recognition accuracy for children\u2019s speech. In: Proceedings of INTERSPEECH, pp. 2197\u20132200 (2005)","DOI":"10.21437\/Interspeech.2005-697"},{"issue":"3","key":"40_CR4","doi-asserted-by":"publisher","first-page":"531","DOI":"10.1109\/TSP.2013.2288675","volume":"62","author":"K Dragomiretskiy","year":"2013","unstructured":"Dragomiretskiy, K., Zosso, D.: Variational mode decomposition. IEEE Trans. Signal Process. 62(3), 531\u2013544 (2013)","journal-title":"IEEE Trans. Signal Process."},{"key":"40_CR5","doi-asserted-by":"publisher","unstructured":"Gerosa, M., Giuliani, D., Narayanan, S., Potamianos, A.: A review of ASR technologies for children\u2019s speech. In: Proceedings of Workshop on Child, Computer and Interaction, pp. 7:1\u20137:8 (2009). https:\/\/doi.org\/10.1145\/1640377.1640384","DOI":"10.1145\/1640377.1640384"},{"key":"40_CR6","doi-asserted-by":"publisher","unstructured":"Gold, B., Morgan, N., Ellis, D., O\u2019Shaughnessy, D.: Speech and audio signal processing: Processing and perception of speech and music, second edition. J. Acoust. Soc. Am. 132, 1861 (2012). https:\/\/doi.org\/10.1121\/1.4742973","DOI":"10.1121\/1.4742973"},{"issue":"4","key":"40_CR7","doi-asserted-by":"publisher","first-page":"2205","DOI":"10.1007\/s00034-021-01885-5","volume":"41","author":"V Kumar","year":"2021","unstructured":"Kumar, V., Kumar, A., Shahnawazuddin, S.: Creating robust children\u2019s ASR system in zero-resource condition through out-of-domain data augmentation. Circ. Syst. Signal Process. 41(4), 2205\u20132220 (2021). https:\/\/doi.org\/10.1007\/s00034-021-01885-5","journal-title":"Circ. Syst. Signal Process."},{"key":"40_CR8","doi-asserted-by":"publisher","unstructured":"Kumar Kathania, H., Reddy Kadiri, S., Alku, P., Kurimo, M.: Study of formant modification for children ASR. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7429\u20137433 (2020). https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9053334","DOI":"10.1109\/ICASSP40776.2020.9053334"},{"issue":"3","key":"40_CR9","doi-asserted-by":"publisher","first-page":"1455","DOI":"10.1121\/1.426686","volume":"105","author":"S Lee","year":"1999","unstructured":"Lee, S., Potamianos, A., Narayanan, S.S.: Acoustics of children\u2019s speech: developmental changes of temporal and spectral parameters. J. Acoust. Soc. Am. 105(3), 1455\u20131468 (1999). https:\/\/doi.org\/10.1121\/1.426686","journal-title":"J. Acoust. Soc. Am."},{"issue":"4","key":"40_CR10","doi-asserted-by":"publisher","first-page":"561","DOI":"10.1109\/PROC.1975.9792","volume":"63","author":"J Makhoul","year":"1975","unstructured":"Makhoul, J.: Linear prediction: a tutorial review. Proc. IEEE 63(4), 561\u2013580 (1975). https:\/\/doi.org\/10.1109\/PROC.1975.9792","journal-title":"Proc. IEEE"},{"key":"40_CR11","unstructured":"Patterson, R., Nimmo-Smith, I., Holdsworth, J., Rice, P.: An efficient auditory filterbank based on the gammatone function (1987)"},{"key":"40_CR12","doi-asserted-by":"crossref","unstructured":"Peddinti, V., Povey, D., Khudanpur, S.: A time delay neural network architecture for efficient modeling of long temporal contexts. In: Proceedings of INTERSPEECH (2015)","DOI":"10.21437\/Interspeech.2015-647"},{"issue":"6","key":"40_CR13","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1109\/TSA.2003.818026","volume":"11","author":"A Potaminaos","year":"2003","unstructured":"Potaminaos, A., Narayanan, S.: Robust recognition of children speech. IEEE Trans. Speech Audio Process. 11(6), 603\u2013616 (2003). https:\/\/doi.org\/10.1109\/TSA.2003.818026","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"40_CR14","unstructured":"Povey, D., et al.: The Kaldi Speech recognition toolkit. In: Proceedings of ASRU (2011)"},{"key":"40_CR15","doi-asserted-by":"crossref","unstructured":"Povey, D., et al.: Purely sequence-trained neural networks for ASR based on lattice-free MMI. In: Proceedings of INTERSPEECH, pp. 2751\u20132755 (2016)","DOI":"10.21437\/Interspeech.2016-595"},{"key":"40_CR16","doi-asserted-by":"publisher","unstructured":"Robinson, T., Fransen, J., Pye, D., Foote, J., Renals, S.: WSJCAM0: a British English speech corpus for large vocabulary continuous speech recognition. In: Proceedings of ICASSP, vol. 1, pp. 81\u201384 (1995). https:\/\/doi.org\/10.1109\/ICASSP.1995.479278","DOI":"10.1109\/ICASSP.1995.479278"},{"key":"40_CR17","doi-asserted-by":"crossref","unstructured":"Russell, M., D\u2019Arcy, S.: Challenges for computer recognition of children\u2019s speech. In: Proceedings of Speech and Language Technologies in Education (SLaTE) (2007)","DOI":"10.21437\/SLaTE.2007-26"},{"issue":"3","key":"40_CR18","doi-asserted-by":"publisher","first-page":"325","DOI":"10.1017\/S135132491600005X","volume":"23","author":"R Serizel","year":"2017","unstructured":"Serizel, R., Giuliani, D.: Deep-neural network approaches for speech recognition with heterogeneous groups of speakers including children. Nat. Lang. Eng. 23(3), 325\u2013350 (2017). https:\/\/doi.org\/10.1017\/S135132491600005X","journal-title":"Nat. Lang. Eng."},{"key":"40_CR19","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1016\/j.patrec.2019.12.019","volume":"131","author":"S Shahnawazuddin","year":"2020","unstructured":"Shahnawazuddin, S., Adiga, N., Kathania, H.K., Sai, B.T.: Creating speaker independent ASR system through prosody modification based data augmentation. Pattern Recogn. Lett. 131, 213\u2013218 (2020). https:\/\/doi.org\/10.1016\/j.patrec.2019.12.019","journal-title":"Pattern Recogn. Lett."},{"key":"40_CR20","doi-asserted-by":"publisher","unstructured":"Shahnawazuddin, S., Adiga, N., Kumar, K., Poddar, A., Ahmad, W.: Voice conversion based data augmentation to improve children\u2019s speech recognition in limited data scenario. In: Proceedings of INTERSPEECH, pp. 4382\u20134386 (2020). https:\/\/doi.org\/10.21437\/Interspeech.2020-1112","DOI":"10.21437\/Interspeech.2020-1112"},{"key":"40_CR21","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1016\/j.dsp.2019.06.015","volume":"93","author":"S Shahnawazuddin","year":"2019","unstructured":"Shahnawazuddin, S., Adiga, N., Sai, B.T., Ahmad, W., Kathania, H.K.: Developing speaker independent ASR system using limited data through prosody modification based on fuzzy classification of spectral bins. Digital Signal Process. 93, 34\u201342 (2019). https:\/\/doi.org\/10.1016\/j.dsp.2019.06.015","journal-title":"Digital Signal Process."},{"key":"40_CR22","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1016\/j.csl.2017.10.007","volume":"48","author":"R Sinha","year":"2018","unstructured":"Sinha, R., Shahnawazuddin, S.: Assessment of pitch-adaptive front-end signal processing for children\u2019s speech recognition. Comput. Speech Lang. 48, 103\u2013121 (2018). https:\/\/doi.org\/10.1016\/j.csl.2017.10.007","journal-title":"Comput. Speech Lang."},{"key":"40_CR23","unstructured":"Slaney, M., et al.: An efficient implementation of the Patterson-Holdsworth auditory filter bank. Apple Computer, Perception Group, Technical Report, vol. 35, no. 8 (1993)"},{"issue":"3","key":"40_CR24","doi-asserted-by":"publisher","first-page":"328","DOI":"10.1109\/29.21701","volume":"37","author":"A Waibel","year":"1989","unstructured":"Waibel, A., Hanazawa, T., Hinton, G., Shikano, K., Lang, K.: Phoneme recognition using time-delay neural networks. IEEE Trans. Acoust. Speech Signal Process. 37(3), 328\u2013339 (1989). https:\/\/doi.org\/10.1109\/29.21701","journal-title":"IEEE Trans. Acoust. Speech Signal Process."}],"container-title":["Lecture Notes in Computer Science","Speech and Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-48309-7_40","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T14:50:21Z","timestamp":1730559021000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-48309-7_40"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031483080","9783031483097"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-48309-7_40","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"22 November 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SPECOM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Speech and Computer","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Dharwad","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 November 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 December 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"specom2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.iitdh.ac.in\/specom-2023\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"174","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"94","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"54% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}