{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T14:36:20Z","timestamp":1743086180171,"version":"3.40.3"},"publisher-location":"Cham","reference-count":44,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031200588"},{"type":"electronic","value":"9783031200595"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-20059-5_26","type":"book-chapter","created":{"date-parts":[[2022,10,28]],"date-time":"2022-10-28T16:02:50Z","timestamp":1666972970000},"page":"452-468","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["VisageSynTalk: Unseen Speaker Video-to-Speech Synthesis via\u00a0Speech-Visage Feature Selection"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4182-1000","authenticated-orcid":false,"given":"Joanna","family":"Hong","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6514-0018","authenticated-orcid":false,"given":"Minsu","family":"Kim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5306-6853","authenticated-orcid":false,"given":"Yong Man","family":"Ro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,10,29]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Akbari, H., Arora, H., Cao, L., Mesgarani, N.: Lip2Audspec: speech reconstruction from silent lip movements video. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 2516\u20132520. IEEE (2018)","key":"26_CR1","DOI":"10.1109\/ICASSP.2018.8461856"},{"unstructured":"Assael, Y.M., Shillingford, B., Whiteson, S., De Freitas, N.: LipNet: end-to-end sentence-level lipreading. arXiv preprint arXiv:1611.01599 (2016)","key":"26_CR2"},{"unstructured":"Burnham, D., Campbell, R., Away, G., Dodd, B.: Hearing Eye II: The Psychology of Speechreading and Auditory-Visual Speech. Psychology Press (2013)","key":"26_CR3"},{"issue":"1","key":"26_CR4","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1109\/79.911195","volume":"18","author":"T Chen","year":"2001","unstructured":"Chen, T.: Audiovisual speech processing. IEEE Sig. Process. Mag. 18(1), 9\u201321 (2001)","journal-title":"IEEE Sig. Process. Mag."},{"doi-asserted-by":"crossref","unstructured":"Chen, Y.H., Wu, D.Y., Wu, T.H., Lee, H.Y.: Again-VC: a one-shot voice conversion using activation guidance and adaptive instance normalization. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5954\u20135958. IEEE (2021)","key":"26_CR5","DOI":"10.1109\/ICASSP39728.2021.9414257"},{"key":"26_CR6","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1007\/978-3-319-54184-6_6","volume-title":"Computer Vision \u2013 ACCV 2016","author":"JS Chung","year":"2017","unstructured":"Chung, J.S., Zisserman, A.: Lip reading in the wild. In: Lai, S.-H., Lepetit, V., Nishino, K., Sato, Y. (eds.) ACCV 2016. LNCS, vol. 10112, pp. 87\u2013103. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-54184-6_6"},{"issue":"5","key":"26_CR7","doi-asserted-by":"publisher","first-page":"2421","DOI":"10.1121\/1.2229005","volume":"120","author":"M Cooke","year":"2006","unstructured":"Cooke, M., Barker, J., Cunningham, S., Shao, X.: An audio-visual corpus for speech perception and automatic speech recognition. J. Acoust. Soc. Am. 120(5), 2421\u20132424 (2006)","journal-title":"J. Acoust. Soc. Am."},{"doi-asserted-by":"crossref","unstructured":"Ephrat, A., Halperin, T., Peleg, S.: Improved speech reconstruction from silent video. In: Proceedings of the IEEE International Conference on Computer Vision Workshops, pp. 455\u2013462 (2017)","key":"26_CR8","DOI":"10.1109\/ICCVW.2017.61"},{"doi-asserted-by":"crossref","unstructured":"Ephrat, A., Peleg, S.: Vid2Speech: speech reconstruction from silent video. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5095\u20135099. IEEE (2017)","key":"26_CR9","DOI":"10.1109\/ICASSP.2017.7953127"},{"unstructured":"Ganin, Y., Lempitsky, V.: Unsupervised domain adaptation by backpropagation. In: International Conference on Machine Learning, pp. 1180\u20131189. PMLR (2015)","key":"26_CR10"},{"doi-asserted-by":"crossref","unstructured":"Gatys, L.A., Ecker, A.S., Bethge, M.: Image style transfer using convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2414\u20132423 (2016)","key":"26_CR11","DOI":"10.1109\/CVPR.2016.265"},{"unstructured":"Goodfellow, I., et al.: Generative adversarial nets. In: Advances in Neural Information Processing Systems, vol. 27 (2014)","key":"26_CR12"},{"issue":"2","key":"26_CR13","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D Griffin","year":"1984","unstructured":"Griffin, D., Lim, J.: Signal estimation from modified short-time Fourier transform. IEEE Trans. Acoust. Speech Sig. Process. 32(2), 236\u2013243 (1984)","journal-title":"IEEE Trans. Acoust. Speech Sig. Process."},{"doi-asserted-by":"crossref","unstructured":"Gui, N., Ge, D., Hu, Z.: AFS: an attention-based mechanism for supervised feature selection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 3705\u20133713 (2019)","key":"26_CR14","DOI":"10.1609\/aaai.v33i01.33013705"},{"key":"26_CR15","first-page":"1157","volume":"3","author":"I Guyon","year":"2003","unstructured":"Guyon, I., Elisseeff, A.: An introduction to variable and feature selection. J. Mach. Learn. Res. 3, 1157\u20131182 (2003)","journal-title":"J. Mach. Learn. Res."},{"issue":"5","key":"26_CR16","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1109\/TMM.2015.2407694","volume":"17","author":"N Harte","year":"2015","unstructured":"Harte, N., Gillen, E.: TCD-TIMIT: an audio-visual corpus of continuous speech. IEEE Trans. Multimedia 17(5), 603\u2013615 (2015)","journal-title":"IEEE Trans. Multimedia"},{"key":"26_CR17","doi-asserted-by":"publisher","first-page":"3654","DOI":"10.1109\/TASLP.2021.3126925","volume":"29","author":"J Hong","year":"2021","unstructured":"Hong, J., Kim, M., Park, S.J., Ro, Y.M.: Speech reconstruction with reminiscent sound via visual voice memory. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 3654\u20133667 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Huang, X., Belongie, S.: Arbitrary style transfer in real-time with adaptive instance normalization. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1501\u20131510 (2017)","key":"26_CR18","DOI":"10.1109\/ICCV.2017.167"},{"key":"26_CR19","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"179","DOI":"10.1007\/978-3-030-01219-9_11","volume-title":"Computer Vision \u2013 ECCV 2018","author":"X Huang","year":"2018","unstructured":"Huang, X., Liu, M.-Y., Belongie, S., Kautz, J.: Multimodal unsupervised image-to-image translation. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11207, pp. 179\u2013196. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01219-9_11"},{"issue":"11","key":"26_CR20","doi-asserted-by":"publisher","first-page":"2009","DOI":"10.1109\/TASLP.2016.2585878","volume":"24","author":"J Jensen","year":"2016","unstructured":"Jensen, J., Taal, C.H.: An algorithm for predicting the intelligibility of speech masked by modulated noise maskers. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(11), 2009\u20132022 (2016)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Karras, T., Laine, S., Aila, T.: A style-based generator architecture for generative adversarial networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4401\u20134410 (2019)","key":"26_CR21","DOI":"10.1109\/CVPR.2019.00453"},{"doi-asserted-by":"crossref","unstructured":"Kim, M., Hong, J., Park, S.J., Ro, Y.M.: Multi-modality associative bridging through memory: speech sound recollected from face video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 296\u2013306 (2021)","key":"26_CR22","DOI":"10.1109\/ICCV48922.2021.00036"},{"unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)","key":"26_CR23"},{"issue":"6","key":"26_CR24","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3136625","volume":"50","author":"J Li","year":"2017","unstructured":"Li, J., et al.: Feature selection: a data perspective. ACM Comput. Surv. (CSUR) 50(6), 1\u201345 (2017)","journal-title":"ACM Comput. Surv. (CSUR)"},{"issue":"5","key":"26_CR25","doi-asserted-by":"publisher","first-page":"322","DOI":"10.1089\/cmb.2015.0189","volume":"23","author":"Y Li","year":"2016","unstructured":"Li, Y., Chen, C.Y., Wasserman, W.W.: Deep feature selection: theory and application to identify enhancers and promoters. J. Comput. Biol. 23(5), 322\u2013336 (2016)","journal-title":"J. Comput. Biol."},{"doi-asserted-by":"crossref","unstructured":"Liao, Y., Latty, R., Yang, B.: Feature selection using batch-wise attenuation and feature mask normalization. In: 2021 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20139. IEEE (2021)","key":"26_CR26","DOI":"10.1109\/IJCNN52387.2021.9533531"},{"issue":"11","key":"26_CR27","first-page":"2578","volume":"9","author":"L Van der Maaten","year":"2008","unstructured":"Van der Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9(11), 2578\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."},{"doi-asserted-by":"crossref","unstructured":"Michelsanti, D., Slizovskaia, O., Haro, G., G\u00f3mez, E., Tan, Z.H., Jensen, J.: Vocoder-based speech synthesis from silent videos. In: Interspeech 2020, pp. 3530\u20133534 (2020)","key":"26_CR28","DOI":"10.21437\/Interspeech.2020-1026"},{"unstructured":"Milner, B., Le Cornu, T.: Reconstructing intelligible audio speech from visual speech features. In: Interspeech 2015 (2015)","key":"26_CR29"},{"doi-asserted-by":"crossref","unstructured":"Mira, R., Vougioukas, K., Ma, P., Petridis, S., Schuller, B.W., Pantic, M.: End-to-end video-to-speech synthesis using generative adversarial networks. arXiv preprint arXiv:2104.13332 (2021)","key":"26_CR30","DOI":"10.1109\/TCYB.2022.3162495"},{"unstructured":"Mirza, M., Osindero, S.: Conditional generative adversarial nets. arXiv preprint arXiv:1411.1784 (2014)","key":"26_CR31"},{"doi-asserted-by":"crossref","unstructured":"Nagrani, A., Chung, J.S., Zisserman, A.: VoxCeleb: a large-scale speaker identification dataset. arXiv preprint arXiv:1706.08612 (2017)","key":"26_CR32","DOI":"10.21437\/Interspeech.2017-950"},{"doi-asserted-by":"crossref","unstructured":"Prajwal, K., Mukhopadhyay, R., Namboodiri, V.P., Jawahar, C.: Learning individual speaking styles for accurate lip to speech synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13796\u201313805 (2020)","key":"26_CR33","DOI":"10.1109\/CVPR42600.2020.01381"},{"doi-asserted-by":"publisher","unstructured":"Rix, A., Beerends, J., Hollier, M., Hekstra, A.: Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs. In: 2001 IEEE International Conference on Acoustics, Speech, and Signal Processing. Proceedings (Cat. No. 01CH37221), vol. 2, pp. 749\u2013752 (2001). https:\/\/doi.org\/10.1109\/ICASSP.2001.941023","key":"26_CR34","DOI":"10.1109\/ICASSP.2001.941023"},{"doi-asserted-by":"crossref","unstructured":"Roy, D., Murty, K.S.R., Mohan, C.K.: Feature selection using deep neural networks. In: 2015 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20136. IEEE (2015)","key":"26_CR35","DOI":"10.1109\/IJCNN.2015.7280626"},{"doi-asserted-by":"crossref","unstructured":"Selvaraju, R.R., Cogswell, M., Das, A., Vedantam, R., Parikh, D., Batra, D.: Grad-CAM: visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 618\u2013626 (2017)","key":"26_CR36","DOI":"10.1109\/ICCV.2017.74"},{"doi-asserted-by":"crossref","unstructured":"Stafylakis, T., Tzimiropoulos, G.: Combining residual networks with LSTMs for lipreading. arXiv preprint arXiv:1703.04105 (2017)","key":"26_CR37","DOI":"10.21437\/Interspeech.2017-85"},{"doi-asserted-by":"crossref","unstructured":"Taal, C.H., Hendriks, R.C., Heusdens, R., Jensen, J.: A short-time objective intelligibility measure for time-frequency weighted noisy speech. In: 2010 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 4214\u20134217. IEEE (2010)","key":"26_CR38","DOI":"10.1109\/ICASSP.2010.5495701"},{"unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, pp. 5998\u20136008 (2017)","key":"26_CR39"},{"doi-asserted-by":"crossref","unstructured":"Vougioukas, K., Ma, P., Petridis, S., Pantic, M.: Video-driven speech reconstruction using generative adversarial networks. arXiv preprint arXiv:1906.06301 (2019)","key":"26_CR40","DOI":"10.21437\/Interspeech.2019-1445"},{"doi-asserted-by":"crossref","unstructured":"Vougioukas, K., Petridis, S., Pantic, M.: End-to-end speech-driven facial animation with temporal GANs. arXiv preprint arXiv:1805.09313 (2018)","key":"26_CR41","DOI":"10.1007\/s11263-019-01251-8"},{"doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Tacotron: towards end-to-end speech synthesis. arXiv preprint arXiv:1703.10135 (2017)","key":"26_CR42","DOI":"10.21437\/Interspeech.2017-1452"},{"doi-asserted-by":"crossref","unstructured":"Yadav, R., Sardana, A., Namboodiri, V.P., Hegde, R.M.: Speech prediction in silent videos using variational autoencoders. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7048\u20137052. IEEE (2021)","key":"26_CR43","DOI":"10.1109\/ICASSP39728.2021.9414040"},{"doi-asserted-by":"crossref","unstructured":"Zhang, S., Zhu, X., Lei, Z., Shi, H., Wang, X., Li, S.Z.: S3FD: single shot scale-invariant face detector. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 192\u2013201 (2017)","key":"26_CR44","DOI":"10.1109\/ICCV.2017.30"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-20059-5_26","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,28]],"date-time":"2022-10-28T16:12:11Z","timestamp":1666973531000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-20059-5_26"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031200588","9783031200595"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-20059-5_26","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"29 October 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}