{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T16:57:36Z","timestamp":1780937856447,"version":"3.54.1"},"reference-count":46,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100008845","name":"Xinjiang University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100008845","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023B01005"],"award-info":[{"award-number":["2023B01005"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132262","type":"journal-article","created":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T15:43:52Z","timestamp":1774885432000},"page":"132262","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Hierarchical analysis and efficient fine-tuning of weakly supervised speech models"],"prefix":"10.1016","volume":"321","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-7129-104X","authenticated-orcid":false,"given":"Jian","family":"Peng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7959-5761","authenticated-orcid":false,"given":"Lixu","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongchao","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yineng","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4361-5368","authenticated-orcid":false,"given":"Nurmemet","family":"Yolwas","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wushour","family":"Silamu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132262_bib0001","unstructured":"Ardila, R., Branson, M., Davis, K., Henretty, M., Kohler, M., Meyer, J., Morais, R., Saunders, L., Tyers, F. M., & Weber, G. (2019). Common voice: A massively-multilingual speech corpus. arXiv preprint arXiv: 1912.06670."},{"key":"10.1016\/j.eswa.2026.132262_bib0002","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"8","key":"10.1016\/j.eswa.2026.132262_bib0003","doi-asserted-by":"crossref","first-page":"1798","DOI":"10.1109\/TPAMI.2013.50","article-title":"Representation learning: A review and new perspectives","volume":"35","author":"Bengio","year":"2013","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"11","key":"10.1016\/j.eswa.2026.132262_bib0004","doi-asserted-by":"crossref","first-page":"1595","DOI":"10.1016\/S0165-1684(02)00304-3","article-title":"Hybrid representations for audiophonic signal encoding","volume":"82","author":"Daudet","year":"2002","journal-title":"Signal Processing"},{"key":"10.1016\/j.eswa.2026.132262_bib0005","series-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: Human language technologies, volume 1 (long and short papers)","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.eswa.2026.132262_bib0006","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv: 2010.11929."},{"key":"10.1016\/j.eswa.2026.132262_bib0007","series-title":"22nd Annual conference of the international speech communication association, interspeech 2021, Brno, Czechia, August 30 - September 3, 2021","first-page":"896","article-title":"SynthASR: Unlocking synthetic data for speech recognition","author":"Fazel","year":"2021"},{"key":"10.1016\/j.eswa.2026.132262_bib0008","series-title":"International conference on machine learning","first-page":"1126","article-title":"Model-agnostic meta-learning for fast adaptation of deep networks","author":"Finn","year":"2017"},{"issue":"12","key":"10.1016\/j.eswa.2026.132262_bib0009","doi-asserted-by":"crossref","first-page":"2639","DOI":"10.1162\/0899766042321814","article-title":"Canonical correlation analysis: an overview with application to learning methods","volume":"16","author":"Hardoon","year":"2004","journal-title":"Neural Computation"},{"issue":"5786","key":"10.1016\/j.eswa.2026.132262_bib0010","doi-asserted-by":"crossref","first-page":"504","DOI":"10.1126\/science.1127647","article-title":"Reducing the dimensionality of data with neural networks","volume":"313","author":"Hinton","year":"2006","journal-title":"Science"},{"key":"10.1016\/j.eswa.2026.132262_bib0011","series-title":"International conference on machine learning","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","author":"Houlsby","year":"2019"},{"key":"10.1016\/j.eswa.2026.132262_bib0012","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"issue":"2","key":"10.1016\/j.eswa.2026.132262_bib0013","first-page":"3","article-title":"Lora: Low-rank adaptation of large language models","volume":"1","author":"Hu","year":"2022","journal-title":"ICLR"},{"issue":"1","key":"10.1016\/j.eswa.2026.132262_bib0014","doi-asserted-by":"crossref","first-page":"106","DOI":"10.1113\/jphysiol.1962.sp006837","article-title":"Receptive fields, binocular interaction and functional architecture in the cat\u2019s visual cortex","volume":"160","author":"Hubel","year":"1962","journal-title":"The Journal of Physiology"},{"key":"10.1016\/j.eswa.2026.132262_bib0015","series-title":"ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP)","first-page":"8229","article-title":"Europarl-st: A multilingual corpus for speech translation of parliamentary debates","author":"Iranzo-S\u00e1nchez","year":"2020"},{"key":"10.1016\/j.eswa.2026.132262_bib0016","series-title":"24th Annual conference of the international speech communication association, interspeech 2023, Dublin, Ireland, August 20-24, 2023","first-page":"5242","article-title":"Adaptation of whisper models to child speech recognition","author":"Jain","year":"2023"},{"key":"10.1016\/j.eswa.2026.132262_bib0017","doi-asserted-by":"crossref","first-page":"46938","DOI":"10.1109\/ACCESS.2023.3275106","article-title":"A wav2vec2-based experimental study on self-supervised learning methods to improve child speech recognition","volume":"11","author":"Jain","year":"2023","journal-title":"IEEE Access"},{"key":"10.1016\/j.eswa.2026.132262_bib0018","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"3390","article-title":"Measuring catastrophic forgetting in neural networks","author":"Kemker","year":"2018"},{"issue":"3","key":"10.1016\/j.eswa.2026.132262_bib0019","doi-asserted-by":"crossref","first-page":"323","DOI":"10.1109\/89.759041","article-title":"Improved phase vocoder time-scale modification of audio","volume":"7","author":"Laroche","year":"1999","journal-title":"IEEE Transactions on Speech and Audio processing"},{"issue":"7553","key":"10.1016\/j.eswa.2026.132262_bib0020","doi-asserted-by":"crossref","first-page":"436","DOI":"10.1038\/nature14539","article-title":"Deep learning","volume":"521","author":"LeCun","year":"2015","journal-title":"Nature"},{"issue":"11","key":"10.1016\/j.eswa.2026.132262_bib0021","doi-asserted-by":"crossref","first-page":"2278","DOI":"10.1109\/5.726791","article-title":"Gradient-based learning applied to document recognition","volume":"86","author":"LeCun","year":"2002","journal-title":"Proceedings of the IEEE"},{"key":"10.1016\/j.eswa.2026.132262_bib0022","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5051","article-title":"Learning to learn from noisy labeled data","author":"Li","year":"2019"},{"issue":"12","key":"10.1016\/j.eswa.2026.132262_bib0023","doi-asserted-by":"crossref","first-page":"3197","DOI":"10.1007\/s10115-022-01756-8","article-title":"Interpretable deep learning: Interpretation, interpretability, trustworthiness, and beyond","volume":"64","author":"Li","year":"2022","journal-title":"Knowledge and Information Systems"},{"key":"10.1016\/j.eswa.2026.132262_bib0024","doi-asserted-by":"crossref","DOI":"10.7717\/peerj-cs.1973","article-title":"Real-time multilingual speech recognition and speaker diarization system based on whisper segmentation","volume":"10","author":"Lyu","year":"2024","journal-title":"PeerJ Computer Science"},{"key":"10.1016\/j.eswa.2026.132262_bib0025","series-title":"Interspeech","first-page":"498","article-title":"Montreal forced aligner: Trainable text-speech alignment using kaldi","volume":"vol. 2017","author":"McAuliffe","year":"2017"},{"key":"10.1016\/j.eswa.2026.132262_bib0026","series-title":"Proceedings of the 32nd international conference on neural information processing systems","first-page":"5732","article-title":"Insights on representational similarity in neural networks with canonical correlation","author":"Morcos","year":"2018"},{"key":"10.1016\/j.eswa.2026.132262_bib0027","series-title":"2015 IEEE international conference on acoustics, speech and signal processing (ICASSP)","first-page":"5206","article-title":"Librispeech: An ASR corpus based on public domain audio books","author":"Panayotov","year":"2015"},{"key":"10.1016\/j.eswa.2026.132262_bib0028","doi-asserted-by":"crossref","first-page":"372","DOI":"10.1162\/tacl_a_00656","article-title":"What do self-supervised speech models know about words?","volume":"12","author":"Pasad","year":"2024","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"10.1016\/j.eswa.2026.132262_bib0029","series-title":"2021 IEEE automatic speech recognition and understanding workshop (ASRU)","first-page":"914","article-title":"Layer-wise analysis of a self-supervised speech representation model","author":"Pasad","year":"2021"},{"key":"10.1016\/j.eswa.2026.132262_bib0030","series-title":"ICASSP 2023-2023 IEEE international conference on acoustics, speech and signal processing (ICASSP)","first-page":"1","article-title":"Comparative layer-wise analysis of self-supervised speech models","author":"Pasad","year":"2023"},{"key":"10.1016\/j.eswa.2026.132262_bib0031","series-title":"Interspeech 2020","first-page":"2757","article-title":"MLS: A large-scale multilingual dataset for speech research","author":"Pratap","year":"2020"},{"key":"10.1016\/j.eswa.2026.132262_bib0032","series-title":"International conference on machine learning","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2023"},{"issue":"333","key":"10.1016\/j.eswa.2026.132262_bib0033","article-title":"Open-source conversational AI with speechbrain 1.0","volume":"25","author":"Ravanelli","year":"2024","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.eswa.2026.132262_bib0034","series-title":"Lrec","first-page":"125","article-title":"Ted-lium: An automatic speech recognition dedicated corpus","author":"Rousseau","year":"2012"},{"key":"10.1016\/j.eswa.2026.132262_bib0035","unstructured":"Sanh, V., Debut, L., Chaumond, J., & Wolf, T. (2019). DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter. arXiv preprint arXiv: 1910.01108."},{"issue":"8","key":"10.1016\/j.eswa.2026.132262_bib0036","doi-asserted-by":"crossref","first-page":"340","DOI":"10.1016\/S1364-6613(00)01704-6","article-title":"On the role of space and time in auditory processing","volume":"5","author":"Shamma","year":"2001","journal-title":"Trends in Cognitive Sciences"},{"key":"10.1016\/j.eswa.2026.132262_bib0037","series-title":"Proceedings of the 2022 conference on empirical methods in natural language processing: EMNLP 2022 - industry track, Abu Dhabi, UAE, December 7 - 11, 2022","first-page":"285","article-title":"Speechnet: Weakly supervised, end-to-end speech recognition at industrial scale","author":"Tang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132262_bib0038","first-page":"5998","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132262_bib0039","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"322","article-title":"Symmetric cross entropy for robust learning with noisy labels","author":"Wang","year":"2019"},{"key":"10.1016\/j.eswa.2026.132262_bib0040","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13876","article-title":"Regularizing class-wise predictions via self-knowledge distillation","author":"Yun","year":"2020"},{"key":"10.1016\/j.eswa.2026.132262_bib0041","series-title":"2024 IEEE international conference on multimedia and expo (ICME)","first-page":"1","article-title":"A study on incorporating whisper for robust speech assessment","author":"Zezario","year":"2024"},{"issue":"1","key":"10.1016\/j.eswa.2026.132262_bib0042","doi-asserted-by":"crossref","first-page":"27","DOI":"10.1631\/FITEE.1700808","article-title":"Visual interpretability for deep learning: A survey","volume":"19","author":"Zhang","year":"2018","journal-title":"Frontiers of Information Technology & Electronic Engineering"},{"issue":"5","key":"10.1016\/j.eswa.2026.132262_bib0043","doi-asserted-by":"crossref","first-page":"726","DOI":"10.1109\/TETCI.2021.3100641","article-title":"A survey on neural network interpretability","volume":"5","author":"Zhang","year":"2021","journal-title":"IEEE Transactions on Emerging Topics in Computational Intelligence"},{"key":"10.1016\/j.eswa.2026.132262_bib0044","first-page":"8792","article-title":"Generalized cross entropy loss for training deep neural networks with noisy labels","volume":"31","author":"Zhang","year":"2018","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"6","key":"10.1016\/j.eswa.2026.132262_bib0045","doi-asserted-by":"crossref","first-page":"1227","DOI":"10.1109\/JSTSP.2022.3184480","article-title":"Improving automatic speech recognition performance for low-resource languages with self-supervised models","volume":"16","author":"Zhao","year":"2022","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"issue":"1","key":"10.1016\/j.eswa.2026.132262_bib0046","doi-asserted-by":"crossref","first-page":"44","DOI":"10.1093\/nsr\/nwx106","article-title":"A brief introduction to weakly supervised learning","volume":"5","author":"Zhou","year":"2018","journal-title":"National Science Review"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426011759?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426011759?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T15:57:37Z","timestamp":1780934257000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426011759"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":46,"alternative-id":["S0957417426011759"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132262","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Hierarchical analysis and efficient fine-tuning of weakly supervised speech models","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132262","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132262"}}