{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T14:58:20Z","timestamp":1784645900456,"version":"3.55.0"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,4,12]],"date-time":"2025-04-12T00:00:00Z","timestamp":1744416000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,4,12]],"date-time":"2025-04-12T00:00:00Z","timestamp":1744416000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"name":"Humanities and Social Sciences Research Planning Fund of the Ministry of Education of China","award":["21A10011003"],"award-info":[{"award-number":["21A10011003"]}]},{"name":"CCF-Zhipu AI Large Model of China Computer Federation","award":["202211"],"award-info":[{"award-number":["202211"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J AUDIO SPEECH MUSIC PROC."],"DOI":"10.1186\/s13636-025-00403-8","type":"journal-article","created":{"date-parts":[[2025,4,12]],"date-time":"2025-04-12T01:54:51Z","timestamp":1744422891000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Silent speech recognition using visual cascading fusion of tongue-lip movements based on pre-trained and fine-tuned model"],"prefix":"10.1186","volume":"2025","author":[{"given":"Chongchong","family":"Yu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuening","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0743-0700","authenticated-orcid":false,"given":"Zhaopeng","family":"Qian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,4,12]]},"reference":[{"issue":"5","key":"403_CR1","doi-asserted-by":"publisher","first-page":"3646","DOI":"10.1121\/1.4987881","volume":"141","author":"B Denby","year":"2017","unstructured":"B. Denby, S. Chen, Y. Zheng, K. Xu, Y. Yang, C. Leboullenger, P. Roussel, Recent results in silent speech interfaces. J. Acoust. Soc. Am. 141(5), 3646\u20133646 (2017). https:\/\/doi.org\/10.1121\/1.4987881","journal-title":"J. Acoust. Soc. Am."},{"issue":"4","key":"403_CR2","doi-asserted-by":"publisher","first-page":"270","DOI":"10.1016\/j.specom.2009.08.002","volume":"52","author":"B Denby","year":"2010","unstructured":"B. Denby, T. Schultz, K. Honda, T. Hueber, J.M. Gibert, J.S. Brumberg, Silent speech interfaces. Speech. Commun. 52(4), 270\u2013287 (2010). https:\/\/doi.org\/10.1016\/j.specom.2009.08.002","journal-title":"Speech. Commun."},{"issue":"12","key":"403_CR3","doi-asserted-by":"publisher","first-page":"2386","DOI":"10.1109\/TASLP.2017.2740000","volume":"25","author":"GS Meltzner","year":"2017","unstructured":"G.S. Meltzner, J.T. Heaton, Y. Deng, G.D. Luca, S.H. Roy, J.C. Kline, Silent speech recognition as an alternative communication device for persons with laryngectomy. IEEE\/ACM Trans. Audio. Speech. Lang. Process. 25(12), 2386\u20132398 (2017). https:\/\/doi.org\/10.1109\/TASLP.2017.2740000","journal-title":"IEEE\/ACM Trans. Audio. Speech. Lang. Process."},{"key":"403_CR4","doi-asserted-by":"publisher","unstructured":"X. Menendez-Pidal, J. B. Polikoff, S. M. Peters, J. E. Leonzio, H. T. Bunnell, The Nemours database of dysarthric speech, in: Proceeding of Fourth International Conference on Spoken Language Processing, ICSLP'96, IEEE, 1996, Oct 03\u201306, Philadelphia, PA, USA, pp. 1962\u20131965. https:\/\/doi.org\/10.1109\/ICSLP.1996.608020","DOI":"10.1109\/ICSLP.1996.608020"},{"issue":"4","key":"403_CR5","doi-asserted-by":"publisher","first-page":"367","DOI":"10.1016\/j.specom.2010.01.001","volume":"52","author":"JS Brumberg","year":"2010","unstructured":"J.S. Brumberg, A. Nieto-Castanon, P.R. Kennedy, F.H. Guenther, Brain\u2013computer interfaces for speech communication. Speech. Commun. 52(4), 367\u2013379 (2010). https:\/\/doi.org\/10.1016\/j.specom.2010.01.001","journal-title":"Speech. Commun."},{"key":"403_CR6","doi-asserted-by":"publisher","first-page":"217","DOI":"10.3389\/fnins.2015.00217","volume":"9","author":"C Herff","year":"2015","unstructured":"C. Herff, D. Heger, A. de Pesters, D. Telaar, P. Brunner, G. Schalk, T. Schultz, Brain-to-text: decoding spoken phrases from phone representations in the brain. Front. Neurosci. 9, 217 (2015). https:\/\/doi.org\/10.3389\/fnins.2015.00217","journal-title":"Front. Neurosci."},{"key":"403_CR7","doi-asserted-by":"publisher","unstructured":"K. Brigham, B. V. K. V. Kumar. Imagined speech classification with EEG signals for silent communication: a preliminary investigation into synthetic telepathy, in: 2010 4th International Conference on Bioinformatics and Biomedical Engineering, IEEE, 2010, June 18\u201320, pp. 1\u20134. https:\/\/doi.org\/10.1109\/ICBBE.2010.5515807.","DOI":"10.1109\/ICBBE.2010.5515807"},{"key":"403_CR8","doi-asserted-by":"publisher","unstructured":"S. C. Jou, T. Schultz, M. Walliczek, F. Kraft, A. Waibel, Towards continuous speech recognition using surface electromyography, in: Ninth International Conference on Spoken Language Processing (Interspeech2006, ICSLP), 2006, Sep 17\u201321, Pittsburgh, Pennsylvania, pp. 573\u2013576. https:\/\/doi.org\/10.21437\/Interspeech.2006-212","DOI":"10.21437\/Interspeech.2006-212"},{"issue":"4","key":"403_CR9","doi-asserted-by":"publisher","first-page":"341","DOI":"10.1016\/j.specom.2009.12.002","volume":"52","author":"T Schultz","year":"2010","unstructured":"T. Schultz, M. Wand, Modeling coarticulation in EMG-based continuous speech recognition. Speech. Commun. 52(4), 341\u2013353 (2010). https:\/\/doi.org\/10.1016\/j.specom.2009.12.002","journal-title":"Speech. Commun."},{"key":"403_CR10","doi-asserted-by":"publisher","unstructured":"M. Janke, M. Wand, K. Nakamura, T. Schultz, Further investigations on EMG-to-speech conversion, in: 2012 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, 2012, Mar 25\u201330, Kyoto, Japan, pp. 365\u2013368. https:\/\/doi.org\/10.1109\/ICASSP.2012.6287892","DOI":"10.1109\/ICASSP.2012.6287892"},{"issue":"10","key":"403_CR11","doi-asserted-by":"publisher","first-page":"2515","DOI":"10.1109\/TBME.2014.2319000","volume":"61","author":"M Wand","year":"2014","unstructured":"M. Wand, M. Janke, T. Schultz, Tackling speaking mode varieties in EMG-based speech recognition. IEEE Trans. Biomed. Eng. 61(10), 2515\u20132526 (2014). https:\/\/doi.org\/10.1109\/TBME.2014.2319000","journal-title":"IEEE Trans. Biomed. Eng."},{"key":"403_CR12","doi-asserted-by":"publisher","unstructured":"M. Zahner, M. Janke, M. Wand, T. Schultz, Conversion from facial myoelectric signals to speech: a unit selection approach, in: Fifteenth Annual Conference of the International Speech Communication Association (Interspeech2014), 2014, Sep 14\u201318, Singapore, pp. 1184\u20131188. https:\/\/doi.org\/10.21437\/Interspeech.2014-300","DOI":"10.21437\/Interspeech.2014-300"},{"key":"403_CR13","doi-asserted-by":"publisher","unstructured":"Y. Deng, J. T. Heaton, G. S. Meltzner, Towards a practical silent speech recognition system, in: Fifteenth Annual Conference of the International Speech Communication Association (Interspeech2014), 2014, Sep 14\u201318, Singapore, pp. 1164\u20131168. https:\/\/doi.org\/10.21437\/Interspeech.2014-296","DOI":"10.21437\/Interspeech.2014-296"},{"key":"403_CR14","doi-asserted-by":"publisher","unstructured":"X. Tan, X. Jiang, Z. Lin, X. Liu, C. Dai, W. Chen, Extracting spatial muscle activation patterns in facial and neck muscles for silent speech recognition using high-density sEMG, IEEE Transactions on Instrumentation and Measurement. 72 (2023) 1\u201313. No. 4006313. https:\/\/doi.org\/10.1109\/TIM.2023.3277930","DOI":"10.1109\/TIM.2023.3277930"},{"issue":"3","key":"403_CR15","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1016\/j.specom.2007.09.001","volume":"50","author":"T Toda","year":"2008","unstructured":"T. Toda, A.W. Black, K. Tokuda, Statistical mapping between articulatory movements and acoustic spectrum using a Gaussian mixture model. Speech Commun. 50(3), 215\u2013227 (2008). https:\/\/doi.org\/10.1016\/j.specom.2007.09.001","journal-title":"Speech Commun."},{"key":"403_CR16","doi-asserted-by":"publisher","unstructured":"A. Toutios, S. S. Narayanan, Articulatory synthesis of French connected speech from EMA data, in: INTERSPEECH2013. 2013, Aug 25\u201329, Lyon, France, pp. 2738\u20132742. https:\/\/doi.org\/10.21437\/Interspeech.2013-628","DOI":"10.21437\/Interspeech.2013-628"},{"key":"403_CR17","doi-asserted-by":"publisher","unstructured":"J. Wang, S. Hahm, T. Mau, Determining an optimal set of flesh points on tongue, lips, and jaw for continuous silent speech recognition, in: Proceedings of SLPAT 2015: 6th Workshop on Speech and Language Processing for Assistive Technologies, 2015, Sep 11, Dresden, Germany, pp. 79\u201385. https:\/\/doi.org\/10.18653\/v1\/w15-5114","DOI":"10.18653\/v1\/w15-5114"},{"issue":"3","key":"403_CR18","doi-asserted-by":"publisher","first-page":"533","DOI":"10.1006\/jpho.2002.0166","volume":"30","author":"P Badin","year":"2002","unstructured":"P. Badin, G. Bailly, L. Reveret, M. Baciu, C. Segebarth, C. Savariaux, Three-dimensional linear articulatory modeling of tongue, lips and face, based on MRI and video images. J. Phon. 30(3), 533\u2013553 (2002). https:\/\/doi.org\/10.1006\/jpho.2002.0166","journal-title":"J. Phon."},{"key":"403_CR19","first-page":"2597","volume-title":"15th International Congress of Phonetic Sciences (ICPhS)","author":"P Birkholz","year":"2003","unstructured":"P. Birkholz, D. Jackel, A three-dimensional model of the vocal tract for speech synthesis, in 15th International Congress of Phonetic Sciences (ICPhS). (Barcelona, Spain, 2003), pp.2597\u20132600"},{"key":"403_CR20","doi-asserted-by":"publisher","unstructured":"E. Petajan, B. Bischoff, D. Bodoff, N. M. Brooke, An improved automatic lipreading system to enhance speech recognition, Proceedings of the SIGCHI Conference on Human Factors in Computing Systems, 1988, May 01, pp. 19\u201325. https:\/\/doi.org\/10.1145\/57167.57170","DOI":"10.1145\/57167.57170"},{"key":"403_CR21","doi-asserted-by":"publisher","unstructured":"Z. Zhou, G. Zhao, M. Pietik\u00e4inen, Towards a practical lipreading system, CVPR 2011, IEEE, 2011, June 20\u201325, Colorado Springs, CO, USA, pp. 137\u2013144. https:\/\/doi.org\/10.1109\/CVPR. 2011.5995345.","DOI":"10.1109\/CVPR"},{"issue":"2","key":"403_CR22","doi-asserted-by":"publisher","first-page":"198","DOI":"10.1109\/34.982900","volume":"24","author":"I Matthews","year":"2002","unstructured":"I. Matthews, T.F. Cootes, J.A. Bangham, S. Cox, R. Harvey, Extraction of visual features for lipreading. IEEE Trans. Pattern Anal. Mach. Intell. 24(2), 198\u2013213 (2002). https:\/\/doi.org\/10.1109\/34.982900","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"403_CR23","doi-asserted-by":"publisher","first-page":"24","DOI":"10.1016\/j.specom.2021.02.001","volume":"128","author":"MS Ribeiro","year":"2021","unstructured":"M.S. Ribeiro, J. Cleland, A. Eshky, K. Richmond, S. Renals, Exploiting ultrasound tongue imaging for the automatic detection of speech articulation errors. Speech. Commun. 128, 24\u201334 (2021). https:\/\/doi.org\/10.1016\/j.specom.2021.02.001","journal-title":"Speech. Commun."},{"issue":"5","key":"403_CR24","doi-asserted-by":"publisher","first-page":"3195","DOI":"10.1121\/10.0002486","volume":"148","author":"M Tabain","year":"2020","unstructured":"M. Tabain, A. Kochetov, R. Beare, An ultrasound and formant study of manner contrasts at four coronal places of articulation. J. Acoust. Soc. Am. 148(5), 3195\u20133217 (2020). https:\/\/doi.org\/10.1121\/10.0002486","journal-title":"J. Acoust. Soc. Am."},{"key":"403_CR25","doi-asserted-by":"publisher","unstructured":"D. R. Mohapatra, P. Saha, Y. Liu, B. Gick, S. Fels, Vocal tract area function extraction using ultrasound for articulatory speech synthesis, in: Proc. 11th ISCA Speech Synthesis Workshop (SSW 11), 2021, Aug 26\u201328, Budapest, Hungary, pp. 90\u201395. https:\/\/doi.org\/10.21437\/SSW.2021-16","DOI":"10.21437\/SSW.2021-16"},{"key":"403_CR26","doi-asserted-by":"publisher","unstructured":"T. Gr\u00f3sz, G. Gosztolya, L. T\u00f3th, T. G. Csap\u00f3, A. Mark\u00f3, F0 estimation for DNN-based ultrasound silent speech interfaces, 2018 IEEE International Conference on Acoustics, in: Speech and Signal Processing (ICASSP), IEEE, 2018, April 15\u201320, Calgary, AB, Canada, pp. 291\u2013295. https:\/\/doi.org\/10.1109\/ICASSP.2018.8461732","DOI":"10.1109\/ICASSP.2018.8461732"},{"issue":"4","key":"403_CR27","doi-asserted-by":"publisher","first-page":"288","DOI":"10.1016\/j.specom.2009.11.004","volume":"52","author":"T Hueber","year":"2010","unstructured":"T. Hueber, E. Benaroya, G. Chollet, B. Denby, G. Dreyfus, M. Stone, Development of a silent speech interface driven by ultrasound and optical images of the tongue and lips. Speech Commun. 52(4), 288\u2013300 (2010). https:\/\/doi.org\/10.1016\/j.specom.2009.11.004","journal-title":"Speech Commun."},{"key":"403_CR28","unstructured":"N. Kimura, Z. Su, T. Saeki, J. Rekimoto, SSR7000: a synchronized corpus of ultrasound tongue imaging for end-to-end silent speech recognition, in: Proceedings of the Thirteenth Language Resources and Evaluation Conference (LREC 2022), 2022, June 20\u201325, Marseille, France, pp. 6866\u20136873. https:\/\/aclanthology.org\/2022.lrec-1.741\/"},{"key":"403_CR29","doi-asserted-by":"publisher","unstructured":"K. J. Teplansky, A. Wisler, B. Cao, W. Liang, Tongue and lip motion patterns in alaryngeal speech, in: INTERSPEECH2020. 2020, Oct 25\u201329, Shanghai, China, pp. 4576\u20134580. https:\/\/doi.org\/10.21437\/Interspeech.2020-2854","DOI":"10.21437\/Interspeech.2020-2854"},{"key":"403_CR30","doi-asserted-by":"publisher","unstructured":"M. S. Ribeiro, J. Sanger, J. X. Zhang, A. Eshky, A. Wrench, K. Richmond, S. Renals, TaL: a synchronised multi-speaker corpus of ultrasound tongue imaging, audio, and lip videos, in: 2021 IEEE Spoken Language Technology Workshop (SLT), IEEE, 2021, Jan 19\u201322. Shenzhen, China, pp. 1109\u20131116. https:\/\/doi.org\/10.1109\/SLT48900.2021.9383619","DOI":"10.1109\/SLT48900.2021.9383619"},{"issue":"2","key":"403_CR31","doi-asserted-by":"publisher","first-page":"L1","DOI":"10.1088\/0266-5611\/21\/2\/L01","volume":"21","author":"G Dassios","year":"2005","unstructured":"G. Dassios, A.S. Fokas, F. Kariotou, On the non-uniqueness of the inverse MEG problem. Inverse Prob. 21(2), L1 (2005). https:\/\/doi.org\/10.1088\/0266-5611\/21\/2\/L01","journal-title":"Inverse Prob."},{"issue":"1","key":"403_CR32","doi-asserted-by":"publisher","first-page":"263","DOI":"10.1007\/s13311-022-01190-2","volume":"19","author":"S Luo","year":"2023","unstructured":"S. Luo, Q. Rabbani, N.E. Crone, Brain-computer interface: applications to speech decoding and synthesis to augment communication. Neurotherapeutics 19(1), 263\u2013273 (2023). https:\/\/doi.org\/10.1007\/s13311-022-01190-2","journal-title":"Neurotherapeutics"},{"issue":"4","key":"403_CR33","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1016\/j.specom.2009.11.003","volume":"52","author":"C Jorgensen","year":"2010","unstructured":"C. Jorgensen, S. Dusan, Speech interfaces based upon surface electromyography. Speech Communication. 52(4), 354\u2013366 (2010). https:\/\/doi.org\/10.1016\/j.specom.2009.11.003","journal-title":"Speech Communication."},{"key":"403_CR34","doi-asserted-by":"publisher","first-page":"578504","DOI":"10.3389\/fneur.2020.578504","volume":"11","author":"F Felici","year":"2020","unstructured":"F. Felici, A. Del Vecchio, Surface electromyography: what limits its use in exercise and sport physiology. Front. Neurol. 11, 578504 (2020). https:\/\/doi.org\/10.3389\/fneur.2020.578504","journal-title":"Front. Neurol."},{"issue":"12","key":"403_CR35","doi-asserted-by":"publisher","first-page":"2375","DOI":"10.1109\/TASLP.2017.2738568","volume":"25","author":"M Janke","year":"2017","unstructured":"M. Janke, L. Diener, EMG-to-speech: direct generation of speech from facial electromyographic signals. IEEE\/ACM Trans. Audio. Speech. Lang. Process. 25(12), 2375\u20132385 (2017). https:\/\/doi.org\/10.1109\/TASLP.2017.2738568","journal-title":"IEEE\/ACM Trans. Audio. Speech. Lang. Process."},{"key":"403_CR36","doi-asserted-by":"publisher","unstructured":"B. Cao, N. Sebkhi, A. Bhavsar, O. T. Inan, R. Samlan, T. Mau, J. Wang, Investigating speech reconstruction for laryngectomees for silent speech interfaces, in: Interspeech2021, 2021, Aug 30 - Sep 3, Brno, Czechi, pp. 651\u2013655. https:\/\/doi.org\/10.21437\/Interspeech.2021-1842","DOI":"10.21437\/Interspeech.2021-1842"},{"key":"403_CR37","doi-asserted-by":"publisher","unstructured":"Y. Zhao, R. Xu, X. Wang, P. Hou, H. Tang, M. Song, Hearing lips: improving lip reading by distilling speech recognizers, in: Proceedings of the AAAI Conference on Artificial Intelligence, 2020, 34(04), Feb 7\u201312, New York Hilton Midtown, New York, New York, USA, pp. 6917\u20136924. https:\/\/doi.org\/10.1609\/aaai.v34i04.6174","DOI":"10.1609\/aaai.v34i04.6174"},{"issue":"S1","key":"403_CR38","doi-asserted-by":"publisher","first-page":"S39","DOI":"10.1121\/1.2022317","volume":"77","author":"YC Ching","year":"1985","unstructured":"Y.C. Ching, Lipreading Cantonese with voice pitch. J Acoust Soc Am 77(S1), S39\u2013S40 (1985). https:\/\/doi.org\/10.1121\/1.2022317","journal-title":"J Acoust Soc Am"},{"key":"403_CR39","unstructured":"F. Hu, Tonal effect on vowel articulation in a tone language, in: International Symposium on Tonal Aspects of Languages: With Emphasis on Tone Languages, 2004, Mar 28\u201331, Beijing China, pp. 97\u2013100. https:\/\/api.semanticscholar.org\/CorpusID:6157053"},{"key":"403_CR40","unstructured":"D. Erickson, R. Iwata, M. Endo, A. Fujino, Effect of tone height on jaw and tongue articulation in Mandarin Chinese, in: International Symposium on Tonal Aspects of Languages: With Emphasis on Tone Languages, 2004, Mar 28\u201331, Beijing, China, pp. 53\u201356. https:\/\/api.semanticscholar.org\/CorpusID:226202813"},{"key":"403_CR41","doi-asserted-by":"publisher","unstructured":"J. J. Ohala, Production of tone, Tone, Academic Press, 1978, pp.5\u201339. https:\/\/doi.org\/10.1016\/B978-0-12-267350-4.50006-6","DOI":"10.1016\/B978-0-12-267350-4.50006-6"},{"key":"403_CR42","doi-asserted-by":"publisher","unstructured":"J. Cai, B. Denby, P. Roussel-Ragot, G. Dreyfus, Recognition and real time performances of a lightweight ultrasound based silent speech interface employing a language model, in: Twelfth Annual Conference of the International Speech Communication Association (Interspeech2011), 2011, Aug 28\u201331, Florence, Italy, pp. 1005\u20131008. \/https:\/\/doi.org\/10.21437\/Interspeech.2011-410","DOI":"10.21437\/Interspeech.2011-410"},{"key":"403_CR43","doi-asserted-by":"publisher","first-page":"42","DOI":"10.1016\/j.specom.2018.02.002","volume":"98","author":"Y Ji","year":"2018","unstructured":"Y. Ji, L. Liu, H. Wang, Z. Liu, Z. Niu, B. Denby, Updating the silent speech challenge benchmark with deep learning. Speech Commun. 98, 42\u201350 (2018). https:\/\/doi.org\/10.1016\/j.specom.2018.02.002","journal-title":"Speech Commun."},{"key":"403_CR44","unstructured":"N. Kimura, Z. Su, T. Saeki, End-to-end deep learning speech recognition model for silent speech challenge, in: Interspeech2020, 2020, Oct 25\u201329, Shanghai, China, pp. 1025\u20131026. https:\/\/api.semanticscholar.org\/CorpusID:226202813"},{"issue":"3","key":"403_CR45","doi-asserted-by":"publisher","first-page":"1867","DOI":"10.1121\/10.0017356","volume":"153","author":"A Serrurier","year":"2023","unstructured":"A. Serrurier, C. Neuschaefer-Rube, Morphological and acoustic modeling of the vocal tract. J Acoust Soc Am 153(3), 1867\u20131886 (2023). https:\/\/doi.org\/10.1121\/10.0017356","journal-title":"J Acoust Soc Am"},{"issue":"3","key":"403_CR46","doi-asserted-by":"publisher","first-page":"1452","DOI":"10.1121\/10.0017362","volume":"153","author":"LM Svensson","year":"2023","unstructured":"L.M. Svensson, Rapid movements at segment boundaries. J Acoust Soc Am 153(3), 1452\u20131467 (2023). https:\/\/doi.org\/10.1121\/10.0017362","journal-title":"J Acoust Soc Am"},{"issue":"7441","key":"403_CR47","doi-asserted-by":"publisher","first-page":"327","DOI":"10.1038\/nature11911","volume":"495","author":"KE Bouchard","year":"2013","unstructured":"K.E. Bouchard, N. Mesgarani, K. Johnson, E.F. Chang, Functional organization of human sensorimotor cortex for speech articulation. Nature 495(7441), 327\u2013332 (2013). https:\/\/doi.org\/10.1038\/nature11911","journal-title":"Nature"},{"key":"403_CR48","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1016\/j.inffus.2019.07.011","volume":"54","author":"Y Zhang","year":"2020","unstructured":"Y. Zhang, Y. Liu, P. Sun, H. Yan, X. Zhao, L. Zhang, IFCNN: a general image fusion framework based on convolutional neural network. Inf. Fusion 54, 99\u2013118 (2020)","journal-title":"Inf. Fusion"},{"key":"403_CR49","doi-asserted-by":"publisher","unstructured":"V. Kazemi, J. Sullivan, One millisecond face alignment with an ensemble of regression trees, in: 2014 IEEE Conference on Computer Vision and Pattern Recognition (CVPR2014), 2014, June 24\u201327, Columbus, OH, USA, pp. 1867\u20131874, https:\/\/doi.org\/10.1109\/CVPR.2014.241","DOI":"10.1109\/CVPR.2014.241"},{"key":"403_CR50","unstructured":"C. Tseng, An acoustic phonetic study on tones in Mandarin Chinese, Brown University, 1981. http:\/\/ir.sinica.edu.tw\/handle\/201000000A\/6232"}],"container-title":["EURASIP Journal on Audio, Speech, and Music Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00403-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s13636-025-00403-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00403-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,12]],"date-time":"2025-04-12T01:54:54Z","timestamp":1744422894000},"score":1,"resource":{"primary":{"URL":"https:\/\/asmp-eurasipjournals.springeropen.com\/articles\/10.1186\/s13636-025-00403-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,12]]},"references-count":50,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["403"],"URL":"https:\/\/doi.org\/10.1186\/s13636-025-00403-8","relation":{},"ISSN":["1687-4722"],"issn-type":[{"value":"1687-4722","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,4,12]]},"assertion":[{"value":"12 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 March 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 April 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"16"}}