{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,27]],"date-time":"2026-05-27T00:04:20Z","timestamp":1779840260485,"version":"3.53.1"},"publisher-location":"Cham","reference-count":19,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032090362","type":"print"},{"value":"9783032090379","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,10,24]],"date-time":"2025-10-24T00:00:00Z","timestamp":1761264000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,24]],"date-time":"2025-10-24T00:00:00Z","timestamp":1761264000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-09037-9_5","type":"book-chapter","created":{"date-parts":[[2025,10,23]],"date-time":"2025-10-23T04:37:22Z","timestamp":1761194242000},"page":"55-65","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Mel-Spectrogram Reconstruction from\u00a0Video Lip Sequences Using an\u00a0Autoencoder Architecture"],"prefix":"10.1007","author":[{"given":"Daphne Sof\u00eda","family":"Gonz\u00e1lez-Cano","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daniel","family":"S\u00e1nchez-Ruiz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jes\u00fas","family":"Garc\u00eda-Ram\u00edrez","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,24]]},"reference":[{"key":"5_CR1","unstructured":"Afouras, T., Chung, J.S., Zisserman, A.: Lrs3-ted: a large-scale dataset for visual speech recognition. In: Interspeech (2018)"},{"key":"5_CR2","unstructured":"Afouras, T., Chung, J.S., Zisserman, A.: Self-supervised pre-training for lip-to-speech generation. In: Interspeech, pp. 2942\u20132946 (2020)"},{"key":"5_CR3","doi-asserted-by":"crossref","unstructured":"Akbari, H., et al.: Lip2audspec: speech reconstruction from silent lip movements video. In: ICASSP, pp. 2516\u20132520 (2018)","DOI":"10.1109\/ICASSP.2018.8461856"},{"issue":"1","key":"5_CR4","doi-asserted-by":"publisher","first-page":"874","DOI":"10.1038\/s41598-020-57649-7","volume":"10","author":"H Akbari","year":"2020","unstructured":"Akbari, H., Khalighinejad, B., Herrero, J.L., Mehta, A.D., Mesgarani, N.: Towards reconstructing intelligible speech from the human auditory cortex. Sci. Rep. 10(1), 874 (2020). https:\/\/doi.org\/10.1038\/s41598-020-57649-7","journal-title":"Sci. Rep."},{"key":"5_CR5","doi-asserted-by":"crossref","unstructured":"Babu, P.A., Nagaraju, V.S., Vallabhuni, R.R.: Speech emotion recognition system with librosa. In: 2021 10th IEEE International Conference on Communication Systems and Network Technologies (CSNT), pp. 421\u2013424. IEEE (2021)","DOI":"10.1109\/CSNT51715.2021.9509714"},{"key":"5_CR6","doi-asserted-by":"crossref","unstructured":"Chung, J.S., Senior, A., Vinyals, O., Zisserman, A.: Lip reading in the wild. arXiv preprint arXiv:1611.05358 (2016)","DOI":"10.1007\/978-3-319-54184-6_6"},{"issue":"5","key":"5_CR7","doi-asserted-by":"publisher","first-page":"2421","DOI":"10.1121\/1.2229005","volume":"120","author":"M Cooke","year":"2006","unstructured":"Cooke, M., Barker, J., Cunningham, S., Shao, X.: An audio-visual corpus for speech perception and automatic speech recognition. J. Acoust. Soc. Am. 120(5), 2421\u20132424 (2006)","journal-title":"J. Acoust. Soc. Am."},{"issue":"2","key":"5_CR8","doi-asserted-by":"publisher","first-page":"798","DOI":"10.3390\/app14020798","volume":"14","author":"Z Dong","year":"2024","unstructured":"Dong, Z., Xu, Y., Abel, A., Wang, D.: Lip2speech: lightweight multi-speaker speech reconstruction with gabor features. Appl. Sci. 14(2), 798 (2024). https:\/\/doi.org\/10.3390\/app14020798","journal-title":"Appl. Sci."},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"Ephrat, A., Peleg, S.: Vid2speech: speech reconstruction from silent video. In: ICASSP (2017)","DOI":"10.1109\/ICASSP.2017.7953127"},{"key":"5_CR10","first-page":"1175","volume":"28","author":"J Gonzalez","year":"2020","unstructured":"Gonzalez, J., Wang, Z.D., Liu, C.Y.: Synthesis of speech from articulatory motion. IEEE\/ACM Trans. Audio Speech Lang. Process. 28, 1175\u20131189 (2020)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"5_CR11","unstructured":"Harte, N., et\u00a0al.: Ndutavsc: a large-scale audio-visual corpus in German for speech recognition. In: AVSP (2015)"},{"key":"5_CR12","unstructured":"Lugaresi, C., et\u00a0al.: Mediapipe: a framework for perceiving and processing reality. In: Third Workshop on Computer Vision for AR\/VR at IEEE Computer Vision and Pattern Recognition (CVPR), vol.\u00a02019 (2019)"},{"key":"5_CR13","unstructured":"Paszke, A.: Pytorch: an imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703 (2019)"},{"key":"5_CR14","unstructured":"Prajwal, K.R., Mukhopadhyay, R., Namboodiri, V.P., Jawahar, C.V.: Lip2wav: learning speech reconstruction from lip movements. In: CVPR, pp. 1868\u20131877 (2020)"},{"key":"5_CR15","doi-asserted-by":"crossref","unstructured":"Qu, L., Weber, C., Wermter, S.: Lipsound: neural mel-spectrogram reconstruction for lip reading. In: INTERSPEECH, pp. 2768\u20132772 (2019)","DOI":"10.21437\/Interspeech.2019-1393"},{"issue":"2","key":"5_CR16","doi-asserted-by":"publisher","first-page":"2772","DOI":"10.1109\/TNNLS.2022.3191677","volume":"35","author":"L Qu","year":"2024","unstructured":"Qu, L., Weber, C., Wermter, S.: Lipsound2: self-supervised pre-training for lip-to-speech reconstruction and lip reading. IEEE Trans. Neural Netw. Learn. Syst. 35(2), 2772\u20132782 (2024). https:\/\/doi.org\/10.1109\/TNNLS.2022.3191677","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"10","key":"5_CR17","doi-asserted-by":"publisher","first-page":"2448","DOI":"10.1109\/TBME.2010.2053369","volume":"57","author":"HR Sharifzadeh","year":"2010","unstructured":"Sharifzadeh, H.R., McLoughlin, I.V., Ahmadi, F.: Reconstruction of normal sounding speech for laryngectomy patients through a modified celp codec. IEEE Trans. Biomed. Eng. 57(10), 2448\u20132458 (2010)","journal-title":"IEEE Trans. Biomed. Eng."},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Yang, H., Zhang, Y., Wu, Z., Lian, S., Lin, W.: Lrw-1000: a naturally-distributed large-scale benchmark for lip reading in the wild. In: ACM MM (2019)","DOI":"10.1109\/FG.2019.8756582"},{"key":"5_CR19","doi-asserted-by":"crossref","first-page":"352","DOI":"10.1016\/j.neucom.2021.10.119","volume":"489","author":"Y Zhou","year":"2022","unstructured":"Zhou, Y., Xu, J., Li, Y.: A comprehensive survey on lip reading: datasets, methods and challenges. Neurocomputing 489, 352\u2013370 (2022)","journal-title":"Neurocomputing"}],"container-title":["Lecture Notes in Computer Science","Advances in Soft Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-09037-9_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,26]],"date-time":"2026-05-26T23:34:02Z","timestamp":1779838442000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-09037-9_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,24]]},"ISBN":["9783032090362","9783032090379"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-09037-9_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,24]]},"assertion":[{"value":"24 October 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MICAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Mexican International Conference on Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Guanajuato","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Mexico","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 November 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 November 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"micai2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/micai.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}