{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T06:16:26Z","timestamp":1783318586329,"version":"3.54.6"},"reference-count":61,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T00:00:00Z","timestamp":1776988800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T00:00:00Z","timestamp":1776988800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s13042-026-03084-7","type":"journal-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T06:42:10Z","timestamp":1777012930000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing continuous speech recognition with CapsNet and WaveRNN: a transfer learning approach"],"prefix":"10.1007","volume":"17","author":[{"given":"Emna","family":"Bouhajeb","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chiraz","family":"Jlassi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Najet","family":"Arous","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,24]]},"reference":[{"key":"3084_CR1","doi-asserted-by":"crossref","unstructured":"Li J, et al (2022) Recent advances in end-to-end automatic speech recognition. APSIPA Trans Signal Inf Process 11(1)","DOI":"10.1561\/116.00000050"},{"key":"3084_CR2","doi-asserted-by":"publisher","first-page":"131858","DOI":"10.1109\/ACCESS.2021.3112535","volume":"9","author":"S Alharbi","year":"2021","unstructured":"Alharbi S, Alrazgan M, Alrashed A, Alnomasi T, Almojel R, Alharbi R, Alharbi S, Alturki S, Alshehri F, Almojil M (2021) Automatic speech recognition: systematic literature review. Ieee Access 9:131858\u2013131876","journal-title":"Ieee Access"},{"key":"3084_CR3","doi-asserted-by":"publisher","first-page":"225","DOI":"10.1016\/S0925-2312(02)00618-5","volume":"51","author":"N Arous","year":"2003","unstructured":"Arous N, Ellouze N (2003) Cooperative supervised and unsupervised learning algorithm for phoneme recognition in continuous speech and speaker-independent context. Neurocomputing 51:225\u2013235","journal-title":"Neurocomputing"},{"key":"3084_CR4","first-page":"12449","volume":"33","author":"A Baevski","year":"2020","unstructured":"Baevski A, Zhou Y, Mohamed A, Auli M (2020) Wav2vec 2.0: a framework for self-supervised learning of speech representations. Adv Neural Inf Process Syst 33:12449\u201312460","journal-title":"Adv Neural Inf Process Syst"},{"key":"3084_CR5","unstructured":"Radford A, Kim JW, Xu T, Brockman G, McLeavey C, Sutskever I (2023) Robust speech recognition via large-scale weak supervision. In: International Conference on Machine Learning, pp. 28492\u201328518. PMLR"},{"key":"3084_CR6","doi-asserted-by":"crossref","unstructured":"Puvvada KC, \u017belasko P, Huang H, Hrinchuk O, Koluguri NR, Dhawan K, Majumdar S, Rastorgueva E, Chen Z, Lavrukhin V et al (2024) Less is more: Accurate speech recognition & translation without web-scale data. arXiv preprint arXiv:2406.19674","DOI":"10.21437\/Interspeech.2024-2294"},{"key":"3084_CR7","unstructured":"Kalchbrenner N, Elsen E, Simonyan K, Noury S, Casagrande N, Lockhart E, Stimberg F, Oord A, Dieleman S, Kavukcuoglu K (2018) Efficient neural audio synthesis. In: International Conference on Machine Learning, pp. 2410\u20132419. PMLR"},{"key":"3084_CR8","doi-asserted-by":"publisher","first-page":"101228","DOI":"10.1016\/j.csl.2021.101228","volume":"70","author":"K Lee","year":"2021","unstructured":"Lee K, Joe H, Lim H, Kim K, Kim S, Han CW, Kim H-G (2021) Sequential routing framework: fully capsule network-based speech recognition. Comput Speech Language 70:101228","journal-title":"Comput Speech Language"},{"key":"3084_CR9","doi-asserted-by":"crossref","unstructured":"Zhang Y, Zhu G, Duan Z (2022) A probabilistic fusion framework for spoofing aware speaker verification. arXiv preprint arXiv:2202.05253","DOI":"10.21437\/Odyssey.2022-11"},{"key":"3084_CR10","unstructured":"Ardila R, Branson M, Davis K, Kohler M, Meyer J, Henretty M, Morais R, Saunders L, Tyers F, Weber G (2020) Common voice: A massively-multilingual speech corpus. In: Proceedings of the Twelfth Language Resources and Evaluation Conference, pp. 4218\u20134222"},{"key":"3084_CR11","doi-asserted-by":"crossref","unstructured":"Chai L, Du J, Liu D-Y, Tu Y-H, Lee C-H (2021) Acoustic modeling for multi-array conversational speech recognition in the chime-6 challenge. In: 2021 IEEE Spoken Language Technology Workshop (SLT), pp. 912\u2013918. IEEE","DOI":"10.1109\/SLT48900.2021.9383628"},{"issue":"47","key":"3084_CR12","first-page":"39","volume":"177","author":"H Vani","year":"2020","unstructured":"Vani H, Anusuya M (2020) Fuzzy speech recognition: a review. Int J Comput Appl 177(47):39\u201354","journal-title":"Int J Comput Appl"},{"key":"3084_CR13","doi-asserted-by":"publisher","first-page":"101815","DOI":"10.1016\/j.csl.2025.101815","volume":"95","author":"P Tutt\u00f6s\u00ed","year":"2026","unstructured":"Tutt\u00f6s\u00ed P, Dhillon M, Sang L, Eastwood S, Bhatia P, Dinh QM, Kapoor A, Jin Y, Lim A (2026) Bersting at the screams: a benchmark for distanced, emotional and shouted speech recognition. Comput Speech Language 95:101815","journal-title":"Comput Speech Language"},{"key":"3084_CR14","unstructured":"Matthew B, Karkala S, Hossain S, Krishnapatnam M, Aggarwal A, Zahir Z, Pandhare HV, Shah V (2025) Quantization-aware training for speech recognition models on edge devices"},{"key":"3084_CR15","doi-asserted-by":"publisher","first-page":"101869","DOI":"10.1016\/j.inffus.2023.101869","volume":"99","author":"A Mehrish","year":"2023","unstructured":"Mehrish A, Majumder N, Bharadwaj R, Mihalcea R, Poria S (2023) A review of deep learning techniques for speech processing. Information Fusion 99:101869","journal-title":"Information Fusion"},{"key":"3084_CR16","unstructured":"Mohammed A, Sunar MS, Salam MS (2021) Speech recognition toolkits: A review. In: The 2ndNational Conference for Ummah Network 2021 (INTER-UMMAH 2021) And the 3rd International Conference on Universal Wellbeing 2021 (ICUW 2021)\u201cEDU sandbox: competency development and innovative strategies for a new normal agenda\u201d(Volume 2) ISBN 978-616-7773-37-7, p. 228"},{"key":"3084_CR17","doi-asserted-by":"publisher","first-page":"4117","DOI":"10.1016\/j.matpr.2021.02.640","volume":"46","author":"JR Koya","year":"2021","unstructured":"Koya JR, Rao SVM (2021) Deep bidirectional neural networks for robust speech recognition under heavy background noise. Mater Today Proc 46:4117\u20134121","journal-title":"Mater Today Proc"},{"key":"3084_CR18","unstructured":"Amodei D, Ananthanarayanan S, Anubhai R, Bai J, Battenberg E, Case C, Casper J, Catanzaro B, Cheng Q, Chen G et al (2016) Deep speech 2: End-to-end speech recognition in english and mandarin. In: International Conference on Machine Learning, pp. 173\u2013182. PMLR"},{"key":"3084_CR19","doi-asserted-by":"publisher","first-page":"105234","DOI":"10.1016\/j.dsp.2025.105234","volume":"163","author":"R Akter","year":"2025","unstructured":"Akter R, Islam MR, Debnath SK, Sarker PK, Uddin MK (2025) A hybrid cnn-lstm model for environmental sound classification: leveraging feature engineering and transfer learning. Digital Signal Processing 163:105234","journal-title":"Digital Signal Processing"},{"key":"3084_CR20","doi-asserted-by":"crossref","unstructured":"Panayotov V, Chen G, Povey D, Khudanpur S (2015) Librispeech: an asr corpus based on public domain audio books. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5206\u20135210. IEEE","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"3084_CR21","doi-asserted-by":"crossref","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (long and Short Papers), pp. 4171\u20134186","DOI":"10.18653\/v1\/N19-1423"},{"key":"3084_CR22","doi-asserted-by":"publisher","first-page":"122807","DOI":"10.1016\/j.eswa.2023.122807","volume":"242","author":"Z Zhao","year":"2024","unstructured":"Zhao Z, Alzubaidi L, Zhang J, Duan Y, Gu Y (2024) A comparison review of transfer learning and self-supervised learning: definitions, applications, advantages and limitations. Expert Syst Appl 242:122807","journal-title":"Expert Syst Appl"},{"key":"3084_CR23","doi-asserted-by":"crossref","unstructured":"Xu M, Jin A, Wang S, Su M, Ng T, Mason H, Han S, Lei Z, Deng Y, Huang Z, et al (2024) Conformer-based speech recognition on extreme edge-computing devices. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 6: Industry Track), pp. 131\u2013139","DOI":"10.18653\/v1\/2024.naacl-industry.12"},{"key":"3084_CR24","doi-asserted-by":"crossref","unstructured":"Coleman EN, Quarantiello L, Liu Z, Yang Q, Mukherjee S, Hurtado J, Lomonaco V (2025) Parameter-efficient continual fine-tuning: a survey. arXiv preprint arXiv:2504.13822","DOI":"10.2139\/ssrn.5510257"},{"key":"3084_CR25","unstructured":"Andreyev A (2025) Quantization for openai\u2019s whisper models: a comparative analysis. arXiv preprint arXiv:2503.09905"},{"key":"3084_CR26","doi-asserted-by":"crossref","unstructured":"Jiang H, Zhang LL, Li Y, Wu Y, Cao S, Cao T, Yang Y, Li J, Yang M, Qiu L (2023) Accurate and structured pruning for efficient automatic speech recognition. arXiv preprint arXiv:2305.19549","DOI":"10.21437\/Interspeech.2023-809"},{"issue":"3","key":"3084_CR27","doi-asserted-by":"publisher","first-page":"168","DOI":"10.1007\/s11063-024-11614-z","volume":"56","author":"J Zhao","year":"2024","unstructured":"Zhao J, Li R, Tian M, An W (2024) Multi-view self-supervised learning and multi-scale feature fusion for automatic speech recognition. Neural Process Lett 56(3):168","journal-title":"Neural Process Lett"},{"key":"3084_CR28","unstructured":"Momo\u00a0Ziazet, J (2025) Energy-aware optimization and machine learning frameworks for sustainable cognitive networks. PhD thesis, Concordia University"},{"key":"3084_CR29","doi-asserted-by":"crossref","unstructured":"Zhou W, Berger S, Schl\u00fcter R, Ney H (2021) Phoneme based neural transducer for large vocabulary speech recognition. In: ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5644\u20135648. IEEE","DOI":"10.1109\/ICASSP39728.2021.9413648"},{"key":"3084_CR30","doi-asserted-by":"crossref","unstructured":"Tarek B, Najet A, Noureddine E (2014) Hierarchical speech recognition system using mfcc feature extraction and dynamic spiking rsom. In: 15th IEEE\/ACIS International Conference on Software Engineering, Artificial Intelligence, Networking and Parallel\/Distributed Computing (SNPD), pp. 1\u20136. IEEE","DOI":"10.1109\/SNPD.2014.6888680"},{"issue":"1","key":"3084_CR31","first-page":"11","volume":"2","author":"J Chiraz","year":"2015","unstructured":"Chiraz J, Arous N, Ellouze N (2015) Cooperativegrowing hierarchical recurrent self organizing model for phoneme recogni-tion. Int J Comput Neural Eng 2(1):11\u201315","journal-title":"Int J Comput Neural Eng"},{"issue":"3","key":"3084_CR32","doi-asserted-by":"publisher","first-page":"891","DOI":"10.3390\/make5030047","volume":"5","author":"MU Haq","year":"2023","unstructured":"Haq MU, Sethi MAJ, Rehman AU (2023) Capsule network with its limitation, modification, and applications\u2013a survey. Mach Learn Knowledge Extraction 5(3):891\u2013921","journal-title":"Mach Learn Knowledge Extraction"},{"key":"3084_CR33","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1016\/j.neunet.2021.06.018","volume":"143","author":"Y Li","year":"2021","unstructured":"Li Y, Zhao W, Cambria E, Wang S, Eger S (2021) Graph routing between capsules. Neural Netw 143:345\u2013354","journal-title":"Neural Netw"},{"key":"3084_CR34","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1162\/tacl_a_00353","volume":"9","author":"A Roy","year":"2021","unstructured":"Roy A, Saffar M, Vaswani A, Grangier D (2021) Efficient content-based sparse attention with routing transformers. Trans Assoc Comput Linguistics 9:53\u201368","journal-title":"Trans Assoc Comput Linguistics"},{"issue":"4","key":"3084_CR35","first-page":"10907","volume":"46","author":"V Parisae","year":"2024","unstructured":"Parisae V, Nagakishore Bhavanam S (2024) Multi scale encoder-decoder network with time frequency attention and s-tcn for single channel speech enhancement. J Intell Fuzzy Syst 46(4):10907\u201310907","journal-title":"J Intell Fuzzy Syst"},{"issue":"01","key":"3084_CR36","doi-asserted-by":"publisher","first-page":"2550067","DOI":"10.1142\/S0219467825500676","volume":"26","author":"V Parisae","year":"2026","unstructured":"Parisae V, Nagakishore Bhavanam S (2026) Stacked u-net with time-frequency attention and deep connection net for single channel speech enhancement. Int J Image Graph 26(01):2550067","journal-title":"Int J Image Graph"},{"key":"3084_CR37","doi-asserted-by":"crossref","unstructured":"Parisae V, Bhavanam SN, Devi MV (2024) Progressive learning framework for speech enhancement using multi-scale convolution and s-tcn. In: 2024 8th International Conference on Inventive Systems and Control (ICISC), pp. 83\u201389. IEEE","DOI":"10.1109\/ICISC62624.2024.00021"},{"issue":"2","key":"3084_CR38","doi-asserted-by":"publisher","first-page":"831","DOI":"10.1007\/s11042-024-19076-0","volume":"84","author":"V Parisae","year":"2025","unstructured":"Parisae V, Bhavanam SN (2025) Adaptive attention mechanism for single channel speech enhancement. Multimed Tools Appl 84(2):831\u2013856","journal-title":"Multimed Tools Appl"},{"issue":"11","key":"3084_CR39","doi-asserted-by":"publisher","first-page":"16654","DOI":"10.1007\/s11227-024-06098-6","volume":"80","author":"BVL Pereira","year":"2024","unstructured":"Pereira BVL, Carvalho MB, Alves PAADSDAN, Ribeiro PRDA, Oliveira ACM, Almeida Neto A (2024) Automatic phoneme recognition by deep neural networks. J Supercomput 80(11):16654\u201316678","journal-title":"J Supercomput"},{"key":"3084_CR40","doi-asserted-by":"crossref","unstructured":"Song W, Deng I (2025) A hybrid architecture combining cnn, lstm, and attention mechanisms for automatic speech recognition. In: 2025 11th international conference on computing and artificial intelligence (ICCAI), pp. 285\u2013292. IEEE","DOI":"10.1109\/ICCAI66501.2025.00052"},{"key":"3084_CR41","doi-asserted-by":"crossref","unstructured":"Wang T, Zhou L, Zhang Z, Wu Y, Liu S, Gaur Y, Chen Z, Li J, Wei F (2023) Viola: Unified codec language models for speech recognition, synthesis, and translation. arXiv preprint arXiv:2305.16107","DOI":"10.1109\/TASLP.2024.3434425"},{"key":"3084_CR42","unstructured":"Oord Avd, Dieleman S, Zen H, Simonyan K, Vinyals O, Graves A, Kalchbrenner N, Senior A, Kavukcuoglu K (2016) Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499"},{"issue":"2","key":"3084_CR43","doi-asserted-by":"publisher","first-page":"203","DOI":"10.3390\/fractalfract7020203","volume":"7","author":"PL Seabe","year":"2023","unstructured":"Seabe PL, Moutsinga CRB, Pindza E (2023) Forecasting cryptocurrency prices using lstm, gru, and bi-directional lstm: a deep learning approach. Fractal Fract 7(2):203","journal-title":"Fractal Fract"},{"key":"3084_CR44","doi-asserted-by":"crossref","unstructured":"Guan B, Cao J, Wang X, Wang Z, Sui M, Wang Z (2024) Integrated method of deep learning and large language model in speech recognition. In: 2024 IEEE 7th international conference on electronic information and communication technology (ICEICT), pp. 487\u2013490. IEEE","DOI":"10.1109\/ICEICT61637.2024.10671048"},{"issue":"4","key":"3084_CR45","doi-asserted-by":"publisher","first-page":"1205","DOI":"10.3390\/s21041205","volume":"21","author":"M Algabri","year":"2021","unstructured":"Algabri M, Mathkour H, Alsulaiman MM, Bencherif MA (2021) Deep learning-based detection of articulatory features in arabic and english speech. Sensors 21(4):1205","journal-title":"Sensors"},{"issue":"1","key":"3084_CR46","doi-asserted-by":"publisher","first-page":"433","DOI":"10.1007\/s11325-020-02102-4","volume":"25","author":"M Wei","year":"2021","unstructured":"Wei M, Du J, Wang X, Lu H, Wang W, Lin P (2021) Voice disorders in severe obstructive sleep apnea patients and comparison of two acoustic analysis software programs: Mdvp and praat. Sleep Breathing 25(1):433\u2013439","journal-title":"Sleep Breathing"},{"key":"3084_CR47","unstructured":"Lou H, Paik H, Hu W, Yao L (2024) Aligner-guided training paradigm: Advancing text-to-speech models with aligner guided duration. arXiv preprint arXiv:2412.08112"},{"key":"3084_CR48","doi-asserted-by":"crossref","unstructured":"Yamamoto K (2021) Learnable companding quantization for accurate low-bit neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5029\u20135038","DOI":"10.1109\/CVPR46437.2021.00499"},{"key":"3084_CR49","doi-asserted-by":"crossref","unstructured":"Cui Y, Jia M, Lin T-Y, Song Y, Belongie S (2019) Class-balanced loss based on effective number of samples. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9268\u20139277","DOI":"10.1109\/CVPR.2019.00949"},{"key":"3084_CR50","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101"},{"issue":"1","key":"3084_CR51","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1109\/JPROC.2020.3004555","volume":"109","author":"F Zhuang","year":"2020","unstructured":"Zhuang F, Qi Z, Duan K, Xi D, Zhu Y, Zhu H, Xiong H, He Q (2020) A comprehensive survey on transfer learning. Proc IEEE 109(1):43\u201376","journal-title":"Proc IEEE"},{"issue":"3","key":"3084_CR52","doi-asserted-by":"publisher","first-page":"379","DOI":"10.1002\/j.1538-7305.1948.tb01338.x","volume":"27","author":"CE Shannon","year":"1948","unstructured":"Shannon CE (1948) A mathematical theory of communication. Bell Syst Tech J 27(3):379\u2013423","journal-title":"Bell Syst Tech J"},{"issue":"9","key":"3084_CR53","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3477140","volume":"54","author":"J Mena","year":"2021","unstructured":"Mena J, Pujol O, Vitri\u00e0 J (2021) A survey on uncertainty estimation in deep learning classification systems from a bayesian perspective. ACM Comput Surv (CSUR) 54(9):1\u201335","journal-title":"ACM Comput Surv (CSUR)"},{"key":"3084_CR54","first-page":"498","volume":"2017","author":"M McAuliffe","year":"2017","unstructured":"McAuliffe M, Socolof M, Mihuc S, Wagner M, Sonderegger M (2017) Montreal forced aligner: trainable text-speech alignment using kaldi. Interspeech 2017:498\u2013502","journal-title":"Interspeech"},{"key":"3084_CR55","unstructured":"Wang R, Sun K (2024) Timit speaker profiling: a comparison of multi-task learning and single-task learning approaches. arXiv preprint arXiv:2404.12077"},{"key":"3084_CR56","doi-asserted-by":"crossref","unstructured":"Bousmina A, Jlassi C, Arous N (2016) Combining ensemble methods of bagging, subagging and random subspace for phoneme recognition. In: 2016 2nd International Conference on Advanced Technologies for Signal and Image Processing (ATSIP), pp. 677\u2013682. IEEE","DOI":"10.1109\/ATSIP.2016.7523165"},{"key":"3084_CR57","doi-asserted-by":"crossref","unstructured":"Cheng M, Wang W, Zhang Y, Qin X, Li M (2023) Target-speaker voice activity detection via sequence-to-sequence prediction. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135. IEEE","DOI":"10.1109\/ICASSP49357.2023.10094752"},{"key":"3084_CR58","doi-asserted-by":"crossref","unstructured":"Guan L (2023) Weight prediction boosts the convergence of adamw. In: Pacific-Asia Conference on Knowledge Discovery and Data Mining, pp. 329\u2013340. Springer","DOI":"10.1007\/978-3-031-33374-3_26"},{"key":"3084_CR59","unstructured":"Chorowski JK, Bahdanau D, Serdyuk D, Cho K, Bengio Y (2015)Attention-based models for speech recognition. Adv Neural Inf Process Syst 28"},{"key":"3084_CR60","doi-asserted-by":"crossref","unstructured":"Schneider S, Baevski A, Collobert R, Auli M (2019) wav2vec: Unsupervised pre-training for speech recognition. arXiv preprint arXiv:1904.05862","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"3084_CR61","unstructured":"Baevski A, Schneider S, Auli M (2019) vq-wav2vec: Self-supervised learning of discrete speech representations. arXiv preprint arXiv:1910.05453"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-026-03084-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-026-03084-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-026-03084-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T05:54:13Z","timestamp":1783317253000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-026-03084-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,24]]},"references-count":61,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["3084"],"URL":"https:\/\/doi.org\/10.1007\/s13042-026-03084-7","relation":{},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"value":"1868-8071","type":"print"},{"value":"1868-808X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,24]]},"assertion":[{"value":"24 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable. This study uses only publicly available benchmark datasets (TIMIT and CHiME-6) with pre-existing ethical approvals.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}],"article-number":"255"}}