{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,14]],"date-time":"2026-01-14T16:22:41Z","timestamp":1768407761193,"version":"3.49.0"},"publisher-location":"Cham","reference-count":44,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030878016","type":"print"},{"value":"9783030878023","type":"electronic"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-87802-3_27","type":"book-chapter","created":{"date-parts":[[2021,9,21]],"date-time":"2021-09-21T23:36:52Z","timestamp":1632267412000},"page":"291-302","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Speaker-Dependent Visual Command Recognition in Vehicle Cabin: Methodology and Evaluation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0412-7765","authenticated-orcid":false,"given":"Denis","family":"Ivanko","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7935-0569","authenticated-orcid":false,"given":"Dmitry","family":"Ryumin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7479-2851","authenticated-orcid":false,"given":"Alexandr","family":"Axyonov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6503-1447","authenticated-orcid":false,"given":"Alexey","family":"Kashevnik","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,9,22]]},"reference":[{"key":"27_CR1","unstructured":"Schroeder, P., Meyers, M., Kostuniuk, L.: National survey on distracted driving attitudes and behaviors, Report No. DOT HS 811 729. National Highway Traffic Safety Administration, Washington, DC (2019)"},{"key":"27_CR2","doi-asserted-by":"publisher","first-page":"746","DOI":"10.1038\/264746a0","volume":"264","author":"H McGurk","year":"1976","unstructured":"McGurk, H., MacDonald, J.: Hearing lips and seeing voices. Nature 264, 746\u2013748 (1976)","journal-title":"Nature"},{"key":"27_CR3","unstructured":"Shillingford, B., Assael, Y., Hoffman, M., et al.: Large-Scale Visual Speech Recognition. In: arXiv eprint 1807.05162, pp.\u00a01\u201321 (2018)"},{"key":"27_CR4","doi-asserted-by":"publisher","first-page":"981","DOI":"10.1007\/s11760-019-01630-1","volume":"14","author":"X Chen","year":"2020","unstructured":"Chen, X., Du, J., Zhang, H.: Lipreading with DenseNet and resBi-LSTM. Signal Image Video Process 14, 981\u2013989 (2020)","journal-title":"Signal Image Video Process"},{"key":"27_CR5","doi-asserted-by":"crossref","unstructured":"Afouras, T., Chung, J.C., Senior, A., et al.: Deep Audio-visual Speech Recognition. In: IEEE Transactions on Pattern Analysis and Machine Intelligence, pp.\u00a01\u201313 (2018)","DOI":"10.1109\/TPAMI.2018.2889052"},{"key":"27_CR6","doi-asserted-by":"publisher","first-page":"34986","DOI":"10.1109\/ACCESS.2021.3062752","volume":"9","author":"A Kashevnik","year":"2021","unstructured":"Kashevnik, A., et al.: Multimodal corpus design for audio-visual speech recognition in vehicle cabin. IEEE Access 9, 34986\u201335003 (2021)","journal-title":"IEEE Access"},{"issue":"3","key":"27_CR7","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1109\/TASL.2008.2011515","volume":"17","author":"G Papandreou","year":"2009","unstructured":"Papandreou, G., Katsamanis, A., Pitsikalis, V., Maragos, P.: Adaptive multimodal fusion by uncertainty compensation with application to audiovisual speech recognition. IEEE Trans. Audio Speech Lang. Process. 17(3), 423\u2013435 (2009)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"27_CR8","doi-asserted-by":"publisher","first-page":"4765","DOI":"10.1109\/TSP.2009.2026513","volume":"57","author":"M Gurban","year":"2009","unstructured":"Gurban, M., Thiran, J.P.: Information theoretic feature extraction for audio-visual speech recognition. IEEE Trans. Signal Process. 57, 4765\u20134776 (2009)","journal-title":"IEEE Trans. Signal Process."},{"key":"27_CR9","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"757","DOI":"10.1007\/978-3-319-66429-3_76","volume-title":"Speech and Computer","author":"D Ivanko","year":"2017","unstructured":"Ivanko, D., et al.: Using a high-speed video camera for robust audio-visual speech recognition in acoustically noisy conditions. In: Karpov, A., Potapova, R., Mporas, I. (eds.) SPECOM 2017. LNCS (LNAI), vol. 10458, pp. 757\u2013766. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-66429-3_76"},{"issue":"7","key":"27_CR10","doi-asserted-by":"publisher","first-page":"1254","DOI":"10.1109\/TMM.2009.2030637","volume":"11","author":"G Zhao","year":"2009","unstructured":"Zhao, G., Barnard, M., Pietikainen, M.: Lipreading with local spatiotemporal descriptors. IEEE Trans. Multimedia 11(7), 1254\u20131265 (2009)","journal-title":"IEEE Trans. Multimedia"},{"key":"27_CR11","doi-asserted-by":"crossref","unstructured":"Potamianos, G., Graf, H.P., Cosatto, E.: An image transform approach for hmm based automatic lipreading. In: IEEE Conference on Image Processing, pp.\u00a0173\u2013177 (1998)","DOI":"10.1109\/ICIP.1998.999008"},{"issue":"4","key":"27_CR12","doi-asserted-by":"publisher","first-page":"319","DOI":"10.1007\/s12193-018-0267-1","volume":"12","author":"D Ivanko","year":"2018","unstructured":"Ivanko, D., et al.: Multimodal speech recognition: increasing accuracy using high speed video data. J. Multimodal User Interfaces 12(4), 319\u2013328 (2018). https:\/\/doi.org\/10.1007\/s12193-018-0267-1","journal-title":"J. Multimodal User Interfaces"},{"key":"27_CR13","doi-asserted-by":"publisher","first-page":"590","DOI":"10.1016\/j.imavis.2014.06.004","volume":"32","author":"Z Zhou","year":"2014","unstructured":"Zhou, Z., Zhao, G., Hong, X., Pietikainen, M.: A review of recent advances in visual speech decoding. Image Vis. Comput. 32, 590\u2013605 (2014)","journal-title":"Image Vis. Comput."},{"key":"27_CR14","unstructured":"Assael, Y., Shillingford, B., Whiteson, S., Freitas, N.: LipNet: end-to-end sentence-level lipreading. In: GPU Technology Conference, pp.\u00a01\u201314 (2017)"},{"issue":"6","key":"27_CR15","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton, G., et al.: Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process. Mag. 29(6), 82\u201397 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"27_CR16","doi-asserted-by":"crossref","unstructured":"Chung, J.S., Zisserman, A.: Lip reading in the wild. In: Asian Conference on Computer Vision, pp.\u00a087\u2013103 (2016)","DOI":"10.1007\/978-3-319-54184-6_6"},{"key":"27_CR17","first-page":"3982","volume":"2017","author":"M Wand","year":"2017","unstructured":"Wand, M., Schmidhuber, J.: Improving speaker-independent lipreading with domain adversarial training. Interspeech 2017, 3982\u20133987 (2017)","journal-title":"Interspeech"},{"key":"27_CR18","doi-asserted-by":"crossref","unstructured":"Kagirov, I., Ryumin, D., Axyonov, A.: Method for multimodal recognition of one-handed sign language gestures through 3D convolution and LSTM neural networks. In: International Conference on Speech and Computer, pp.\u00a0191\u2013200 (2019)","DOI":"10.1007\/978-3-030-26061-3_20"},{"key":"27_CR19","doi-asserted-by":"crossref","unstructured":"Stafylakis, T., Tzimiropoulos, G.: Combining residual networks with LSTMs for lipreading. In: Interspeech, pp.\u00a03652\u20133656, ISCA (2017)","DOI":"10.21437\/Interspeech.2017-85"},{"key":"27_CR20","doi-asserted-by":"crossref","unstructured":"Wand, M., Koutnik, K., Schmidhuber, J.: Lipreading with long short-term memory. International Conference on Acoustics, Speech, and Signal Processing, pp.\u00a06115\u20136119 (2016)","DOI":"10.1109\/ICASSP.2016.7472852"},{"key":"27_CR21","doi-asserted-by":"crossref","unstructured":"Petridis, S., Wang, Y., Li, Z., Pantic, M.: End-to-end multi-view lipreading. In: British Machine Vision Conference, pp.\u00a01\u201314 (2017)","DOI":"10.5244\/C.31.161"},{"key":"27_CR22","doi-asserted-by":"crossref","unstructured":"Sui, S., Bennamoun, M., Togneri, R.: Listening with your eyes: towards a practical visual speech recognition system using deep Boltzmann machines. In: ICCV, pp.\u00a0154\u2013162 (2015)","DOI":"10.1109\/ICCV.2015.26"},{"key":"27_CR23","doi-asserted-by":"crossref","unstructured":"Noda, K., Yamaguchi, Y., Nakadai, K., et al.: Lipreading using convolutional neural network. In: Interspeech, pp.\u00a01149\u20131153 (2014)","DOI":"10.21437\/Interspeech.2014-293"},{"key":"27_CR24","doi-asserted-by":"crossref","unstructured":"Almajai, I., Cox, S., Harvey, R., Lan, Y.: Improved speaker independent lip reading using speaker adaptive training and deep neural networks. In: IEEE International Conference on Acoustics, Speech, and Signal Processing, pp.\u00a02722\u20132726 (2016)","DOI":"10.1109\/ICASSP.2016.7472172"},{"key":"27_CR25","doi-asserted-by":"crossref","unstructured":"Newman, J., Cox, S.: Language identification using visual features. In: Proc. IEEE Audio Speech and Language Processing, vol.\u00a020, no.\u00a07, pp.\u00a01936\u20131947 (2012)","DOI":"10.1109\/TASL.2012.2191956"},{"key":"27_CR26","doi-asserted-by":"crossref","unstructured":"Lan, Y., Theobald, B., Harvey, R.: View independent computer lip-reading. In: Proc. International Conference Multimedia Expo (ICME), pp.\u00a0432\u2013437 (2012)","DOI":"10.1109\/ICME.2012.192"},{"key":"27_CR27","doi-asserted-by":"crossref","unstructured":"Estellers, V., Thiran, J.: Multi-pose lipreading and audio-visual speech recognition. In: EURALISP Journal Advanced Signal Processing, vol.\u00a051 (2012)","DOI":"10.1186\/1687-6180-2012-51"},{"key":"27_CR28","unstructured":"Huang, Z., Zeng, Z., Liu, B., et al.: Pixel-BERT: Aligning Image Pixels with Text by Deep Multi-Modal Transformers. In: arXiv: 2004.00849, pp.\u00a01\u201317 (2020)"},{"key":"27_CR29","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1007\/978-3-319-99579-3_26","volume-title":"Speech and Computer","author":"D Ivanko","year":"2018","unstructured":"Ivanko, D., Ryumin, D., Axyonov, A., \u017delezn\u00fd, M.: Designing advanced geometric features for automatic Russian visual speech recognition. In: Karpov, A., Jokisch, O., Potapova, R. (eds.) SPECOM 2018. LNCS (LNAI), vol. 11096, pp. 245\u2013254. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-319-99579-3_26"},{"key":"27_CR30","doi-asserted-by":"publisher","first-page":"135383","DOI":"10.1109\/ACCESS.2020.3011502","volume":"8","author":"A Rajagopal","year":"2020","unstructured":"Rajagopal, A., et al.: A deep learning model based on multi-objective particle swarm optimization for scene classification in unmanned aerial vehicles. IEEE Access 8, 135383\u2013135393 (2020)","journal-title":"IEEE Access"},{"key":"27_CR31","first-page":"477","volume-title":"Lip-Reading Using Pixel-Based and Geometry-Based Features for Multimodal Human-Robot Interfaces. Smart Innovation, Systems and Technologies","author":"D Ivanko","year":"2020","unstructured":"Ivanko, D., Ryumin, D., Kipyatkova, I., et al.: Lip-Reading Using Pixel-Based and Geometry-Based Features for Multimodal Human-Robot Interfaces. Smart Innovation, Systems and Technologies, vol. 154, pp. 477\u2013486. Springer, Singapore (2020)"},{"key":"27_CR32","doi-asserted-by":"crossref","unstructured":"Xu, K., Li, D., Cassimatis, N., Wang, X. LCANet: End-to-end lipreading with cascaded attention-ctc. arXiv preprint arXiv:1803.04988, pp.\u00a01\u201310 (2018)","DOI":"10.1109\/FG.2018.00088"},{"key":"27_CR33","first-page":"1","volume":"2744","author":"E Ryumina","year":"2020","unstructured":"Ryumina, E., Karpov, A.: Facial expression recognition using distance importance scores between facial landmarks. CEUR Workshop Proceedings 2744, 1\u201310 (2020)","journal-title":"CEUR Workshop Proceedings"},{"key":"27_CR34","doi-asserted-by":"crossref","unstructured":"Thanda, A., Venkatesan, S.M.: Audio visual speech recognition using deep recurrent neural networks. In: Multimodal Pattern Recognition of Social Signals in Human-Computer-Interaction, pp.\u00a098\u2013109 (2017)","DOI":"10.1007\/978-3-319-59259-6_9"},{"key":"27_CR35","doi-asserted-by":"publisher","first-page":"177","DOI":"10.5194\/isprs-archives-XLIV-2-W1-2021-177-2021","volume":"XLIV-2\/W1-2021","author":"E Ryumina","year":"2021","unstructured":"Ryumina, E., Ryumin, D., Ivanko, D., Karpov, A.: A Novel Method for Protective Face Mask Detection Using Convolutional Neural Networks and Image Histograms. ISPRS-International Archives of the Photogrammetry, Remote Sensing and Spatial Information Sciences XLIV-2\/W1-2021, 177\u2013182 (2021)","journal-title":"ISPRS-International Archives of the Photogrammetry, Remote Sensing and Spatial Information Sciences"},{"key":"27_CR36","doi-asserted-by":"crossref","unstructured":"Lee, B., et al.: AVICAR: Audio-Visual Speech Corpus in a Car Environment. In: 8th International Conference on Spoken Language Processing, ICSLP 2004, pp.\u00a01\u20135 (2004)","DOI":"10.21437\/Interspeech.2004-424"},{"key":"27_CR37","unstructured":"Ortega, A. et al.: AV@CAR: A Spanish multichannel multimodal corpus for in-vehicle automatic audio-visual speech recognition. In: LREC, pp.\u00a01354\u20131359 (2004)"},{"key":"27_CR38","unstructured":"Milo\u0161, M., Milo\u0161\u017eelezny, M., C\u00edsa\u0159, P.: Czech Audio-Visual Speech Corpus of a Car Driver for In-Vehicle Audio-Visual Speech Recognition. In: International Conference on Audio-Visual Speech Processing, AVSP, pp.\u00a01\u20135 (2003)"},{"key":"27_CR39","doi-asserted-by":"crossref","unstructured":"Kawasaki, T., et al.: An audio-visual in-car corpus \u201cCENSREC-2-AV\u201d for robust bimodal speech recognition. In: Vehicle Systems and Driver Modelling, pp.\u00a0181\u2013190 (2017)","DOI":"10.1515\/9781501504129-014"},{"key":"27_CR40","unstructured":"Vosk offiline speech recognition API Kaldi based, [online] Available: https:\/\/alphacephei.com\/vosk\/"},{"key":"27_CR41","unstructured":"Kartynnik, Y., Ablavatski, A., Grishchenko, I., Grundmann, M.: Real-time Facial Surface Geometry from Monocular Video on Mobile GPUs. In: CVPR Workshop on Computer Vision for Augmented and Virtual Reality 2019, IEEE, pp.\u00a01\u20134 (2019)"},{"key":"27_CR42","unstructured":"Howard, A., Zhu, M., Chen, B., et al.: MobileNets: Efficient Convolutional Neural Networks for Mobile Vision Applications. In arXiv:1704.04861, pp.\u00a01\u20139 (2017)"},{"key":"27_CR43","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., et al.: Densely Connected Convolutional Networks. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.\u00a02261\u20132269 (2018)","DOI":"10.1109\/CVPR.2017.243"},{"key":"27_CR44","unstructured":"Barret, Z., Vijay, V., Jonathon, S., Quoc, L.: Learning Transferable Architectures for Scalable Image Recognition. In: Conference on Computer Vision and Pattern Recognition (CVPR), pp.\u00a08697\u20138710 (2018)"}],"container-title":["Lecture Notes in Computer Science","Speech and Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-87802-3_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,8]],"date-time":"2024-09-08T17:39:45Z","timestamp":1725817185000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-87802-3_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030878016","9783030878023"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-87802-3_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"22 September 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SPECOM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Speech and Computer","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"St Petersburg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Russia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 September 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"specom2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/specom.nw.ru\/2021\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"163","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"74","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"45% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.5","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5.5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held online due to the COVID-19 pandemic.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}