{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T16:46:34Z","timestamp":1765039594208,"version":"3.41.0"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030780944"},{"type":"electronic","value":"9783030780951"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-78095-1_21","type":"book-chapter","created":{"date-parts":[[2021,7,2]],"date-time":"2021-07-02T23:20:19Z","timestamp":1625268019000},"page":"277-290","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Multimodal Fusion and Sequence Learning for Cued Speech Recognition from Videos"],"prefix":"10.1007","author":[{"given":"Katerina","family":"Papadimitriou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Maria","family":"Parelli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Galini","family":"Sapountzaki","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Georgios","family":"Pavlakos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Petros","family":"Maragos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gerasimos","family":"Potamianos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,7,3]]},"reference":[{"issue":"1","key":"21_CR1","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1016\/j.specom.2004.10.013","volume":"44","author":"V Attina","year":"2004","unstructured":"Attina, V., Beautemps, D., Cathiard, M., Odisio, M.: A pilot study of temporal organization in cued speech production of French syllables: rules for a cued speech synthesizer. Speech Commun. 44(1), 197\u2013214 (2004)","journal-title":"Speech Commun."},{"unstructured":"Bheda, V., Radpour, D.: Using deep convolutional networks for gesture recognition in American Sign Language. CoRR abs\/1710.06836 (2017)","key":"21_CR2"},{"doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo Vadis, action recognition? A new model and the Kinetics dataset. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4724\u20134733 (2017)","key":"21_CR3","DOI":"10.1109\/CVPR.2017.502"},{"doi-asserted-by":"crossref","unstructured":"Cho, K., Merrienboer, B.V., G\u00fcl\u00e7ehre, C., Bougares, F., Schwenk, H., Bengio, Y.: Learning phrase representations using RNN encoder-decoder for statistical machine translation. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1724\u20131734 (2014)","key":"21_CR4","DOI":"10.3115\/v1\/D14-1179"},{"unstructured":"Community, B.O.: Blender - a 3D modelling and rendering package. Blender Foundation, Stichting Blender Foundation, Amsterdam (2018). http:\/\/www.blender.org","key":"21_CR5"},{"issue":"1","key":"21_CR6","first-page":"3","volume":"112","author":"RO Cornett","year":"1967","unstructured":"Cornett, R.O.: Cued speech. Am. Ann. Deaf 112(1), 3\u201313 (1967)","journal-title":"Am. Ann. Deaf"},{"doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 248\u2013255 (2009)","key":"21_CR7","DOI":"10.1109\/CVPR.2009.5206848"},{"doi-asserted-by":"crossref","unstructured":"Freitag, M., Al-Onaizan, Y.: Beam search strategies for neural machine translation. CoRR abs\/1702.01806 (2017)","key":"21_CR8","DOI":"10.18653\/v1\/W17-3207"},{"unstructured":"Fuse, M.: Mixamo: Quality 3D Character Animation in Minutes (2015). https:\/\/www.mixamo.com","key":"21_CR9"},{"doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the International Conference on Machine Learning (ICML) (2006)","key":"21_CR10","DOI":"10.1145\/1143844.1143891"},{"doi-asserted-by":"crossref","unstructured":"Hannun, A., Lee, A., Xu, Q., Collobert, R.: Sequence-to-sequence speech recognition with time-depth separable convolutions. CoRR abs\/1904.02619 (2019)","key":"21_CR11","DOI":"10.21437\/Interspeech.2019-2460"},{"doi-asserted-by":"crossref","unstructured":"Hara, K., Kataoka, H., Satoh, Y.: Learning spatio-temporal features with 3D residual networks for action recognition. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3154\u20133160 (2017)","key":"21_CR12","DOI":"10.1109\/ICCVW.2017.373"},{"doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Delving deep into rectifiers: surpassing human-level performance on ImageNet classification. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp. 1026\u20131034 (2015)","key":"21_CR13","DOI":"10.1109\/ICCV.2015.123"},{"doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016)","key":"21_CR14","DOI":"10.1109\/CVPR.2016.90"},{"issue":"6","key":"21_CR15","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1016\/j.specom.2010.03.001","volume":"52","author":"P Heracleous","year":"2010","unstructured":"Heracleous, P., Beautemps, D., Aboutabit, N.: Cued speech automatic recognition in normal-hearing and deaf subjects. Speech Commun. 52(6), 504\u2013512 (2010)","journal-title":"Speech Commun."},{"unstructured":"Heracleous, P., Beautemps, D., Hagita, N.: Continuous phoneme recognition in cued speech for French. In: Proceedings of the European Signal Processing Conference (EUSIPCO), pp. 2090\u20132093 (2012)","key":"21_CR16"},{"issue":"8","key":"21_CR17","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. CoRR abs\/1412.6980 (2014)","key":"21_CR18"},{"issue":"4","key":"21_CR19","doi-asserted-by":"publisher","first-page":"496","DOI":"10.1353\/aad.2019.0031","volume":"164","author":"L Liu","year":"2019","unstructured":"Liu, L., Feng, G.: A pilot study on Mandarin Chinese cued speech. Am. Ann. Deaf 164(4), 496\u2013518 (2019)","journal-title":"Am. Ann. Deaf"},{"doi-asserted-by":"crossref","unstructured":"Liu, L., Feng, G., Beautemps, D.: Automatic temporal segmentation of hand movements for hand positions recognition in French cued speech. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 3061\u20133065 (2018)","key":"21_CR20","DOI":"10.1109\/ICASSP.2018.8462090"},{"doi-asserted-by":"crossref","unstructured":"Liu, L., Hueber, T., Feng, G., Beautemps, D.: Visual recognition of continuous cued speech using a tandem CNN-HMM approach. In: Proceedings of Interspeech, pp. 2643\u20132647 (2018)","key":"21_CR21","DOI":"10.21437\/Interspeech.2018-2434"},{"doi-asserted-by":"crossref","unstructured":"Liu, L., Li, J., Feng, G., Zhang, X.: Automatic detection of the temporal segmentation of hand movements in British English cued speech. In: Proceedings of Interspeech, pp. 2285\u20132289 (2019)","key":"21_CR22","DOI":"10.21437\/Interspeech.2019-2353"},{"key":"21_CR23","doi-asserted-by":"publisher","first-page":"292","DOI":"10.1109\/TMM.2020.2976493","volume":"23","author":"L Liu","year":"2021","unstructured":"Liu, L., Feng, G., Beautemps, D., Zhang, X.P.: Re-synchronization using the hand preceding model for multi-modal fusion in automatic continuous cued speech recognition. IEEE Trans. Multimedia 23, 292\u2013305 (2021)","journal-title":"IEEE Trans. Multimedia"},{"doi-asserted-by":"crossref","unstructured":"Martinez, J., Hossain, R., Romero, J., Little, J.J.: A simple yet effective baseline for 3D human pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp. 2659\u20132668 (2017)","key":"21_CR24","DOI":"10.1109\/ICCV.2017.288"},{"doi-asserted-by":"crossref","unstructured":"Papadimitriou, K., Potamianos, G.: A fully convolutional sequence learning approach for cued speech recognition from videos. In: Proceedings of the European Signal Processing Conference (EUSIPCO), pp. 326\u2013330 (2021)","key":"21_CR25","DOI":"10.23919\/Eusipco47968.2020.9287365"},{"doi-asserted-by":"crossref","unstructured":"Parelli, M., Papadimitriou, K., Potamianos, G., Pavlakos, G., Maragos, P.: Exploiting 3D hand pose estimation in deep learning-based sign language recognition from RGB videos. In: Proceedings of the ECCV 2020 Workshops, pp. 249\u2013263 (2020)","key":"21_CR26","DOI":"10.1007\/978-3-030-66096-3_18"},{"unstructured":"Paszke, A., et al.: Automatic differentiation in PyTorch. In: Proceedings of the NIPS-W (2017)","key":"21_CR27"},{"doi-asserted-by":"crossref","unstructured":"Potamianos, G., et al.: Audio and visual modality combination in speech processing applications. In: Oviatt, S., Schuller, B., Cohen, P., Sonntag, D., Potamianos, G., Kr\u00fcger, A. (eds.) The Handbook of Multimodal-Multisensor Interfaces, Volume 1: Foundations, User Modeling, and Common Modality Combinations, pp. 489\u2013543. Morgan-Claypool (2017)","key":"21_CR28","DOI":"10.1145\/3015783.3015797"},{"doi-asserted-by":"crossref","unstructured":"Rao, G.A., Syamala, K., Kishore, P.V.V., Sastry, A.S.C.S.: Deep convolutional neural networks for sign language recognition. In: Proceedings of the Signal Processing and Communication Engineering Systems (SPACES), pp. 194\u2013197 (2018)","key":"21_CR29","DOI":"10.1109\/SPACES.2018.8316344"},{"doi-asserted-by":"crossref","unstructured":"Simon, T., Joo, H., Matthews, I., Sheikh, Y.: Hand keypoint detection in single images using multiview bootstrapping. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4645\u20134653 (2017)","key":"21_CR30","DOI":"10.1109\/CVPR.2017.494"},{"doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2818\u20132826 (2016)","key":"21_CR31","DOI":"10.1109\/CVPR.2016.308"},{"unstructured":"Vaswani, A., et al.: Attention is all you need. In: Proceedings of the Conference on Neural Information Processing Systems (NeurIPS), pp. 5998\u20136008 (2017)","key":"21_CR32"},{"doi-asserted-by":"crossref","unstructured":"Zimmermann, C., Brox, T.: Learning to estimate 3D hand pose from single RGB images. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp. 4913\u20134921 (2017)","key":"21_CR33","DOI":"10.1109\/ICCV.2017.525"}],"container-title":["Lecture Notes in Computer Science","Universal Access in Human-Computer Interaction. Access to Media, Learning and Assistive Environments"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-78095-1_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,2]],"date-time":"2025-07-02T22:38:26Z","timestamp":1751495906000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-78095-1_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030780944","9783030780951"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-78095-1_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"3 July 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"HCII","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Human-Computer Interaction","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 July 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 July 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"hcii2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2021.hci.international\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}