{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T10:18:54Z","timestamp":1783246734643,"version":"3.54.6"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T00:00:00Z","timestamp":1731110400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T00:00:00Z","timestamp":1731110400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Circuits Syst Signal Process"],"published-print":{"date-parts":[[2025,3]]},"DOI":"10.1007\/s00034-024-02915-8","type":"journal-article","created":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T10:24:57Z","timestamp":1731147897000},"page":"2020-2040","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Dhivehi Speech Recognition: A Multimodal Approach for Dhivehi Language in Resource-Constrained Settings"],"prefix":"10.1007","volume":"44","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6397-6049","authenticated-orcid":false,"given":"Sunakshi","family":"Mehra","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2046-8642","authenticated-orcid":false,"given":"Virender","family":"Ranga","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0420-1892","authenticated-orcid":false,"given":"Ritu","family":"Agarwal","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,9]]},"reference":[{"issue":"1","key":"2915_CR1","doi-asserted-by":"publisher","first-page":"27","DOI":"10.21608\/ejle.2020.47685.1015","volume":"8","author":"ER Abdelmaksoud","year":"2021","unstructured":"E.R. Abdelmaksoud, A. Hassen, N. Hassan, M. Hesham, Convolutional neural network for arabic speech recognition. Egypt. J. Language Eng. 8(1), 27\u201338 (2021). https:\/\/doi.org\/10.21608\/ejle.2020.47685.1015","journal-title":"Egypt. J. Language Eng."},{"key":"2915_CR2","doi-asserted-by":"publisher","unstructured":"D. Amodei, S. Ananthanarayanan, R. Anubhai, J. Bai, E. Battenberg, C. Case, et al.. Deep speech 2: End-to-end speech recognition in english and mandarin. In\u00a0International conference on machine learning\u00a0(pp. 173\u2013182). PMLR. (2016) https:\/\/doi.org\/10.5555\/3045390.3045410","DOI":"10.5555\/3045390.3045410"},{"key":"2915_CR3","doi-asserted-by":"publisher","unstructured":"M. Artetxe, G. Labaka, E. Agirre, K. Cho, K. Unsupervised neural machine translation.\u00a0arXiv preprint arXiv:1710.11041. (2017) https:\/\/doi.org\/10.48550\/arXiv.1710.11041","DOI":"10.48550\/arXiv.1710.11041"},{"issue":"3","key":"2915_CR4","doi-asserted-by":"publisher","first-page":"264","DOI":"10.1109\/89.906000","volume":"9","author":"E Bocchieri","year":"2001","unstructured":"E. Bocchieri, B.W. Mak, Subspace distribution clustering hidden Markov model. IEEE Trans. Speech Audio Process. 9(3), 264\u2013275 (2001). https:\/\/doi.org\/10.1109\/89.906000","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"2915_CR5","doi-asserted-by":"publisher","unstructured":"L. Burget, P. Schwarz, M. Agarwal, P. Akyazi, K. Feng, A. Ghoshal et al. Multilingual acoustic modeling for speech recognition based on subspace Gaussian mixture models. In\u00a02010 IEEE international conference on acoustics, speech and signal processing\u00a0(pp. 4334\u20134337). IEEE. (2010) https:\/\/doi.org\/10.1109\/ICASSP.2014.6854671","DOI":"10.1109\/ICASSP.2014.6854671"},{"key":"2915_CR6","doi-asserted-by":"publisher","unstructured":"W. Byrne, P. Beyerlein, J.M. Huerta, S. Khudanpur, B. Marthi, J. Morgan, et al. Towards language independent acoustic modeling. In\u00a02000 IEEE International Conference on Acoustics, Speech, and Signal Processing. Proceedings (Cat. No. 00CH37100)\u00a0(Vol. 2, pp. II1029-II1032). IEEE. (2000) https:\/\/doi.org\/10.1109\/ICASSP.2000.859138","DOI":"10.1109\/ICASSP.2000.859138"},{"key":"2915_CR7","doi-asserted-by":"publisher","unstructured":"J. Cho, M.K. Baskar, R. Li, M. Wiesner, S.H. Mallidi, N. Yalta,et al. Multilingual sequence-to-sequence speech recognition: architecture, transfer learning, and language modeling. In\u00a02018 IEEE spoken language technology workshop (SLT)\u00a0(pp. 521\u2013527). IEEE. (2018) https:\/\/doi.org\/10.48550\/arXiv.1810.03459","DOI":"10.48550\/arXiv.1810.03459"},{"key":"2915_CR8","doi-asserted-by":"publisher","unstructured":"P. Cohen, S. Dharanipragada, J. Gros, M. Monkowski, C. Neti, S. Roukos, T. Ward. Towards a universal speech recognizer for multiple languages. In\u00a01997 IEEE workshop on automatic speech recognition and understanding proceedings\u00a0(pp. 591\u2013598). IEEE. (1997) https:\/\/doi.org\/10.48550\/arXiv.1711.02207","DOI":"10.48550\/arXiv.1711.02207"},{"key":"2915_CR9","doi-asserted-by":"publisher","unstructured":"X. Cui, V. Goel, B. Kingsbury. Data augmentation for deep convolutional neural network acoustic modeling. In\u00a02015 IEEE international conference on acoustics, speech and signal processing (ICASSP)\u00a0(pp. 4545\u20134549). IEEE. (2015) https:\/\/doi.org\/10.1109\/TASLP.2015.2438544","DOI":"10.1109\/TASLP.2015.2438544"},{"issue":"9","key":"2915_CR10","doi-asserted-by":"publisher","first-page":"1469","DOI":"10.1109\/ICASSP.2014.6854671","volume":"23","author":"X Cui","year":"2015","unstructured":"X. Cui, V. Goel, B. Kingsbury, Data augmentation for deep neural network acoustic modeling. IEEE\/ACM Trans. Audio Speech Language Process. 23(9), 1469\u20131477 (2015). https:\/\/doi.org\/10.1109\/ICASSP.2014.6854671","journal-title":"IEEE\/ACM Trans. Audio Speech Language Process."},{"key":"2915_CR11","doi-asserted-by":"publisher","unstructured":"S. Dalmia, R. Sanabria, F. Metze, A.W. Black. Sequence-based multi-lingual low resource speech recognition. In\u00a02018 IEEE international conference on acoustics, speech and signal processing (ICASSP)\u00a0(pp. 4909\u20134913). IEEE. (2018) https:\/\/doi.org\/10.48550\/arXiv.1802.07420","DOI":"10.48550\/arXiv.1802.07420"},{"issue":"6","key":"2915_CR12","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3584861","volume":"22","author":"R Das","year":"2023","unstructured":"R. Das, T.D. Singh, Image-text multimodal sentiment analysis framework of assamese news articles using late fusion. ACM Trans. Asian Low-Resource Language Inf. Process. 22(6), 1\u201330 (2023). https:\/\/doi.org\/10.1145\/3584861","journal-title":"ACM Trans. Asian Low-Resource Language Inf. Process."},{"key":"2915_CR13","doi-asserted-by":"publisher","unstructured":"M.J. Gales, A. Ragni, H. AlDamarki, C. Gautier. Support vector machines for noise robust ASR. In\u00a02009 IEEE workshop on automatic speech recognition & understanding\u00a0(pp. 205\u2013210). IEEE. (2009) https:\/\/doi.org\/10.1109\/ASRU.2009.5372913","DOI":"10.1109\/ASRU.2009.5372913"},{"key":"2915_CR14","doi-asserted-by":"publisher","unstructured":"M.J. Gales, K. Yu. Canonical state models for automatic speech recognition. In\u00a0Eleventh annual conference of the international speech communication association. (2010) https:\/\/doi.org\/10.21437\/Interspeech.2010-11","DOI":"10.21437\/Interspeech.2010-11"},{"key":"2915_CR15","doi-asserted-by":"publisher","unstructured":"A.E. Gnanadesikan. Dhivehi: The language of the Maldives\u00a0(Vol. 3). Walter de Gruyter GmbH & Co KG. (2016) https:\/\/doi.org\/10.1515\/9781614512349","DOI":"10.1515\/9781614512349"},{"key":"2915_CR16","doi-asserted-by":"publisher","unstructured":"F.P. Gomez, R. Sanabria, Y.H. Sung, D. Cer, S. Dalmia, G.H. Abrego. Transforming LLMs into cross-modal and cross-lingual retrieval systems. arXiv preprint arXiv:2404.01616. (2024) https:\/\/doi.org\/10.48550\/arXiv.2404.01616","DOI":"10.48550\/arXiv.2404.01616"},{"key":"2915_CR17","doi-asserted-by":"publisher","unstructured":"A. Graves, S. Fern\u00e1ndez, F. Gomez, J. Schmidhuber. Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In Proceedings of the 23rd international conference on machine learnin (2006), pp. 369\u2013376. https:\/\/doi.org\/10.1145\/1143844.1143891","DOI":"10.1145\/1143844.1143891"},{"key":"2915_CR18","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1016\/j.procs.2016.04.024","volume":"81","author":"F Gr\u00e9zl","year":"2016","unstructured":"F. Gr\u00e9zl, E. Egorova, M. Karafi\u00e1t, Study of large data resources for multilingual training and system porting. Proc. Comput. Sci. 81, 15\u201322 (2016). https:\/\/doi.org\/10.1016\/j.procs.2016.04.024","journal-title":"Proc. Comput. Sci."},{"key":"2915_CR19","doi-asserted-by":"publisher","unstructured":"T.J. Hazen. Automatic alignment and error correction of human generated transcripts for long speech recordings. In Ninth international conference on spoken language processing. (2006) https:\/\/doi.org\/10.21437\/Interspeech.2006-449","DOI":"10.21437\/Interspeech.2006-449"},{"key":"2915_CR20","doi-asserted-by":"publisher","unstructured":"M. Harper. The automatic speech recogition in reverberant environments (ASpIRE) challenge. In\u00a02015 IEEE workshop on automatic speech recognition and understanding (ASRU)\u00a0(pp. 547\u2013554). IEEE. (2015) https:\/\/doi.org\/10.1109\/ASRU.2015.7404843","DOI":"10.1109\/ASRU.2015.7404843"},{"issue":"8","key":"2915_CR21","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"S. Hochreiter, J. Schmidhuber, Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997). https:\/\/doi.org\/10.1162\/neco.1997.9.8.1735","journal-title":"Neural Comput."},{"key":"2915_CR22","doi-asserted-by":"publisher","unstructured":"R. Hsiao, J. Ma, W. Hartmann, M. Karafi\u00e1t, F. Gr\u00e9zl, L. Burget. et al. Robust speech recognition in unknown reverberant and noisy conditions. In\u00a02015 IEEE workshop on automatic speech recognition and understanding (ASRU)\u00a0(pp. 533\u2013538). IEEE (2015). https:\/\/doi.org\/10.1109\/ASRU.2015.7404841","DOI":"10.1109\/ASRU.2015.7404841"},{"key":"2915_CR23","unstructured":"https:\/\/www.uib.no\/en\/discipline\/linguistics"},{"key":"2915_CR24","doi-asserted-by":"publisher","unstructured":"J.T. Huang, J. Li, D. Yu, L. Deng, Y. Gong. Cross-language knowledge transfer using multilingual deep neural network with shared hidden layers. In\u00a02013 IEEE international conference on acoustics, speech and signal processing\u00a0(pp. 7304\u20137308). IEEE. (2013) https:\/\/doi.org\/10.1109\/ICASSP.2013.6639081","DOI":"10.1109\/ICASSP.2013.6639081"},{"issue":"3","key":"2915_CR25","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1016\/0885-2308(89)90020-X","volume":"3","author":"XD Huang","year":"1989","unstructured":"X.D. Huang, M.A. Jack, Semi-continuous hidden Markov models for speech signals. Comput. Speech Lang. 3(3), 239\u2013251 (1989). https:\/\/doi.org\/10.1016\/0885-2308(89)90020-X","journal-title":"Comput. Speech Lang."},{"key":"2915_CR26","unstructured":"N. Jaitly, G.E. Hinton. Vocal tract length perturbation (VTLP) improves speech recognition. In Proceedings of ICML workshop on deep learning for audio, speech and language\u00a0(Vol. 117, p. 21) (2013)."},{"key":"2915_CR27","doi-asserted-by":"crossref","unstructured":"A. Jensson, K. Iwano, S. Furui. Development of a speech recognition system for Icelandic using machine translated text. In\u00a0Spoken languages technologies for under-resourced languages. (2008) https:\/\/aclanthology.org\/www.mt-archive.info\/05\/SLTU-2008-Jensson.pdf","DOI":"10.1155\/2008\/573832"},{"key":"2915_CR28","doi-asserted-by":"publisher","unstructured":"J. Kohler. Multi-lingual phoneme recognition exploiting acoustic-phonetic similarities of sounds. In\u00a0Proceeding of fourth international conference on spoken language processing. iCSLP'96\u00a0(Vol. 4, pp. 2195\u20132198). IEEE (1996) https:\/\/doi.org\/10.1109\/ICSLP.1996.607240","DOI":"10.1109\/ICSLP.1996.607240"},{"key":"2915_CR29","doi-asserted-by":"publisher","unstructured":"J. Kohler. Language adaptation of multilingual phone models for vocabulary independent speech recognition tasks. In\u00a0Proceedings of the 1998 IEEE international conference on acoustics, speech and signal processing, ICASSP'98 (Cat. No. 98CH36181)\u00a0(Vol. 1, pp. 417\u2013420). IEEE. (1998) https:\/\/doi.org\/10.1109\/ICASSP.1998.674456","DOI":"10.1109\/ICASSP.1998.674456"},{"issue":"8","key":"2915_CR30","doi-asserted-by":"publisher","first-page":"1471","DOI":"10.1109\/TASL.2009.2021723","volume":"17","author":"VB Le","year":"2009","unstructured":"V.B. Le, L. Besacier, Automatic speech recognition for under-resourced languages: application to Vietnamese language. IEEE Trans. Audio Speech Lang. Process. 17(8), 1471\u20131482 (2009). https:\/\/doi.org\/10.1109\/TASL.2009.2021723","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"2915_CR31","doi-asserted-by":"publisher","unstructured":"H. Lin, L. Deng, D. Yu, Y.F. Gong, A. Acero, C.H. Lee. A study on multilingual acoustic modeling for large vocabulary ASR. In\u00a02009 IEEE international conference on acoustics, speech and signal processing\u00a0(pp. 4333\u20134336). IEEE (2009). https:\/\/doi.org\/10.1109\/ICASSP.2009.4960588","DOI":"10.1109\/ICASSP.2009.4960588"},{"issue":"25","key":"2915_CR32","doi-asserted-by":"publisher","first-page":"38667","DOI":"10.1007\/s11042-023-15118-1","volume":"82","author":"S Mehra","year":"2023","unstructured":"S. Mehra, S. Susan, Deep fusion framework for speech command recognition using acoustic and linguistic features. Multimed. Tools Appl. 82(25), 38667\u201338691 (2023). https:\/\/doi.org\/10.1007\/s11042-023-15118-1","journal-title":"Multimed. Tools Appl."},{"key":"2915_CR33","doi-asserted-by":"publisher","DOI":"10.1007\/s11227-024-06015-x","author":"S Mehra","year":"2024","unstructured":"S. Mehra, V. Ranga, R. Agarwal, A deep learning approach to dysarthric utterance classification with BiLSTM-GRU, speech cue filtering, and log mel spectrograms. J. Supercomput. (2024). https:\/\/doi.org\/10.1007\/s11227-024-06015-x","journal-title":"J. Supercomput."},{"issue":"2","key":"2915_CR34","doi-asserted-by":"publisher","first-page":"1365","DOI":"10.1007\/s11760-023-02845-z","volume":"18","author":"S Mehra","year":"2024","unstructured":"S. Mehra, V. Ranga, R. Agarwal, Improving speech command recognition through decision-level fusion of deep filtered speech cues. SIViP 18(2), 1365\u20131373 (2024). https:\/\/doi.org\/10.1007\/s11760-023-02845-z","journal-title":"SIViP"},{"key":"2915_CR35","doi-asserted-by":"publisher","unstructured":"N. Mohamed. Transcending linguistic and cultural boundaries: a case study of four young maldivians' translanguaging practices. English-Medium Instruction and Translanguaging. https:\/\/doi.org\/10.21832\/9781788927338-010 (2021)","DOI":"10.21832\/9781788927338-010"},{"issue":"10","key":"2915_CR36","doi-asserted-by":"publisher","first-page":"1345","DOI":"10.1109\/TKDE.2009.191","volume":"22","author":"SJ Pan","year":"2009","unstructured":"S.J. Pan, Q. Yang, A survey on transfer learning. IEEE Trans. Knowl. Data Eng. 22(10), 1345\u20131359 (2009). https:\/\/doi.org\/10.1109\/TKDE.2009.191","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"2915_CR37","doi-asserted-by":"publisher","unstructured":"V. Peddinti, G. Chen, V. Manohar, T. Ko, D. Povey, S. Khudanpur. Jhu aspire system: Robust lvcsr with tdnns, ivector adaptation and rnn-lms. In\u00a02015 IEEE workshop on automatic speech recognition and understanding (ASRU)\u00a0(pp. 539\u2013546). IEEE. (2015) https:\/\/doi.org\/10.1109\/ASRU.2015.7404842","DOI":"10.1109\/ASRU.2015.7404842"},{"key":"2915_CR38","doi-asserted-by":"publisher","unstructured":"D. Povey, L. Burget, M. Agarwal, P. Akyazi, K. Feng, A. Ghoshal, et al. Subspace Gaussian mixture models for speech recognition. In\u00a02010 IEEE international conference on acoustics, speech and signal processing\u00a0(pp. 4330\u20134333). IEEE. (2010) https:\/\/doi.org\/10.1109\/ICASSP.2010.5495662","DOI":"10.1109\/ICASSP.2010.5495662"},{"issue":"1","key":"2915_CR39","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1109\/TASL.2011.2129911","volume":"20","author":"G Saon","year":"2011","unstructured":"G. Saon, J.T. Chien, Bayesian sensing hidden Markov models. IEEE Trans. Audio Speech Lang. Process. 20(1), 43\u201354 (2011). https:\/\/doi.org\/10.1109\/TASL.2011.2129911","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"2915_CR40","doi-asserted-by":"publisher","unstructured":"A. Sehr, C. Hofmann, R. Maas, W. Kellermann. Multi-style training of HMMs with stereo data for reverberation-robust speech recognition. In\u00a02011 joint workshop on hands-free speech communication and microphone arrays\u00a0(pp. 196\u2013200). IEEE. (2011) https:\/\/doi.org\/10.1109\/HSCMA.2011.5942396","DOI":"10.1109\/HSCMA.2011.5942396"},{"key":"2915_CR41","doi-asserted-by":"publisher","first-page":"107945","DOI":"10.1016\/j.cnsns.2024.107945","volume":"132","author":"X Song","year":"2024","unstructured":"X. Song, Z. Peng, S. Song, V. Stojanovic, Anti-disturbance state estimation for PDT-switched RDNNs utilizing time-sampling and space-splitting measurements. Commun. Nonlinear Sci. Numer. Simul. 132, 107945 (2024). https:\/\/doi.org\/10.1016\/j.cnsns.2024.107945","journal-title":"Commun. Nonlinear Sci. Numer. Simul."},{"issue":"21","key":"2915_CR42","doi-asserted-by":"publisher","first-page":"15429","DOI":"10.1007\/s00521-023-08361-y","volume":"35","author":"X Song","year":"2023","unstructured":"X. Song, P. Sun, S. Song, V. Stojanovic, Quantized neural adaptive finite-time preassigned performance control for interconnected nonlinear systems. Neural Comput. Appl. 35(21), 15429\u201315446 (2023). https:\/\/doi.org\/10.1007\/s00521-023-08361-y","journal-title":"Neural Comput. Appl."},{"key":"2915_CR43","doi-asserted-by":"publisher","first-page":"126498","DOI":"10.1016\/j.neucom.2023.126498","volume":"550","author":"X Song","year":"2023","unstructured":"X. Song, N. Wu, S. Song, Y. Zhang, V. Stojanovic, Bipartite synchronization for cooperative-competitive neural networks with reaction\u2013diffusion terms via dual event-triggered mechanism. Neurocomputing 550, 126498 (2023). https:\/\/doi.org\/10.1016\/j.neucom.2023.126498","journal-title":"Neurocomputing"},{"key":"2915_CR44","doi-asserted-by":"publisher","unstructured":"S. Takahashi, S. Sagayama. Four-level tied-structure for efficient representation of acoustic modeling. In\u00a01995 International conference on acoustics, speech, and signal processing\u00a0(Vol. 1, pp. 520\u2013523). IEEE. https:\/\/doi.org\/10.1109\/ICASSP.1995.479643(1995)","DOI":"10.1109\/ICASSP.1995.479643"},{"key":"2915_CR45","doi-asserted-by":"publisher","unstructured":"S. Tong, P.N. Garner, H. Bourlard. An investigation of deep neural networks for multilingual speech recognition training and adaptation\u00a0(No. CONF, pp. 714\u2013718). https:\/\/doi.org\/10.21437\/Interspeech.2017-1242(2017)","DOI":"10.21437\/Interspeech.2017-1242"},{"key":"2915_CR46","doi-asserted-by":"publisher","unstructured":"G. Van Driem. Glimpses of the ethnolinguistic prehistory of northeastern India. In Origins and migrations in the extended Eastern Himalayas (pp. 187\u2013211). Brill. https:\/\/doi.org\/10.1163\/9789004228368_011(2012)","DOI":"10.1163\/9789004228368_011"},{"key":"2915_CR47","doi-asserted-by":"publisher","unstructured":"N.T. Vu, D. Imseng, D. Povey, P. Motlicek, T. Schultz, H. Bourlard. Multilingual deep neural network based acoustic modeling for rapid language adaptation. In\u00a02014 IEEE international conference on acoustics, speech and signal processing (ICASSP)\u00a0(pp. 7639\u20137643). IEEE. https:\/\/doi.org\/10.1109\/ICASSP.2014.6855086 (2014)","DOI":"10.1109\/ICASSP.2014.6855086"},{"key":"2915_CR48","unstructured":"N.T. Vu, F. Metze, T. Schultz. Multilingual bottle-neck features and its application for under-resourced languages. In\u00a0Spoken language technologies for under-resourced languages. https:\/\/www.isca-archive.org\/sltu_2012\/vu12_sltu.pdf (2012)"},{"key":"2915_CR49","doi-asserted-by":"publisher","unstructured":"C. Wang, Y. Tang, X. Ma, A. Wu, S. Popuri, D. Okhonko, J. Pino. Fairseq S2T: Fast speech-to-text modeling with fairseq. arXiv preprint arXiv:2010.05171. https:\/\/doi.org\/10.48550\/arXiv.2010.05171(2020)","DOI":"10.48550\/arXiv.2010.05171(2020)"},{"key":"2915_CR50","doi-asserted-by":"publisher","unstructured":"A.S.M.B. Wazir, J.H. Chuah. Spoken Arabic digits recognition using deep learning. In\u00a02019 IEEE international conference on automatic control and intelligent systems (I2CACIS)\u00a0(pp. 339\u2013344). IEEE. https:\/\/doi.org\/10.1109\/I2CACIS.2019.8825004 (2019)","DOI":"10.1109\/I2CACIS.2019.8825004"},{"key":"2915_CR51","doi-asserted-by":"publisher","unstructured":"D. Yu, M.L. Seltzer, J. Li, J.T. Huang, F. Seide. Feature learning in deep neural networks-studies on speech recognition tasks. arXiv preprint arXiv:1301.3605. https:\/\/doi.org\/10.48550\/arXiv.1301.3605 (2013)","DOI":"10.48550\/arXiv.1301.3605"},{"key":"2915_CR52","doi-asserted-by":"publisher","unstructured":"L. Zhang, D. Karakos, W. Hartmann, R. Hsiao, R. Schwartz, S. Tsakalidis. Enhancing low resource keyword spotting with automatically retrieved web documents. In\u00a0Sixteenth annual conference of the international speech communication association. https:\/\/doi.org\/10.21437\/Interspeech.2015-262 (2025)","DOI":"10.21437\/Interspeech.2015-262"},{"issue":"4","key":"2915_CR53","doi-asserted-by":"publisher","first-page":"673","DOI":"10.26599\/TST.2022.9010038","volume":"28","author":"Q Zhang","year":"2023","unstructured":"Q. Zhang, H. Zhang, K. Zhou, L. Zhang, Developing a physiological signal-based, mean threshold and decision-level fusion algorithm (PMD) for emotion recognition. Tsinghua Sci. Technol. 28(4), 673\u2013685 (2023). https:\/\/doi.org\/10.26599\/TST.2022.9010038","journal-title":"Tsinghua Sci. Technol."},{"key":"2915_CR54","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1007\/s10772-018-09573-7","volume":"22","author":"T Zia","year":"2019","unstructured":"T. Zia, U. Zahid, Long short-term memory recurrent neural network architectures for Urdu acoustic modeling. Int. J. Speech Technol. 22, 21\u201330 (2019). https:\/\/doi.org\/10.1007\/s10772-018-09573-7","journal-title":"Int. J. Speech Technol."},{"key":"2915_CR55","doi-asserted-by":"publisher","first-page":"111","DOI":"10.1016\/j.inffus.2022.09.012","volume":"90","author":"J Zhu","year":"2023","unstructured":"J. Zhu, C. Huang, P. De Meo, DFMKE: A dual fusion multi-modal knowledge graph embedding framework for entity alignment. Inf. Fusion 90, 111\u2013119 (2023). https:\/\/doi.org\/10.1016\/j.inffus.2022.09.012","journal-title":"Inf. Fusion"}],"container-title":["Circuits, Systems, and Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-024-02915-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00034-024-02915-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-024-02915-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,17]],"date-time":"2025-03-17T19:27:27Z","timestamp":1742239647000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00034-024-02915-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,9]]},"references-count":55,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,3]]}},"alternative-id":["2915"],"URL":"https:\/\/doi.org\/10.1007\/s00034-024-02915-8","relation":{},"ISSN":["0278-081X","1531-5878"],"issn-type":[{"value":"0278-081X","type":"print"},{"value":"1531-5878","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,9]]},"assertion":[{"value":"24 March 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 October 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 October 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 November 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors state that they have no known financial or personal interests that could have impacted the conclusions described in this work.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The article does not include any studies or investigations that involve human or animal participants.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Human and animal ethics"}},{"value":"All authors have approved the manuscript and given their full consent for its publication.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}]}}