{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T05:10:51Z","timestamp":1743138651301,"version":"3.40.3"},"publisher-location":"Cham","reference-count":20,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030209834"},{"type":"electronic","value":"9783030209841"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-20984-1_7","type":"book-chapter","created":{"date-parts":[[2019,5,27]],"date-time":"2019-05-27T23:03:25Z","timestamp":1558998205000},"page":"71-83","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Improving Audio-Visual Speech Recognition Using Gabor Recurrent Neural Networks"],"prefix":"10.1007","author":[{"given":"Ali S.","family":"Saudi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mahmoud I.","family":"Khalil","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hazem M.","family":"Abbas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,5,15]]},"reference":[{"key":"7_CR1","doi-asserted-by":"crossref","unstructured":"Chang, S.Y., Morgan, N.: Robust CNN-based speech recognition with Gabor filter kernels. In: Fifteenth Annual Conference of the International Speech Communication Association (2014)","DOI":"10.21437\/Interspeech.2014-226"},{"key":"7_CR2","doi-asserted-by":"crossref","unstructured":"Feng, W., Guan, N., Li, Y., Zhang, X., Luo, Z.: Audio visual speech recognition with multimodal recurrent neural networks. In: IEEE International Joint Conference on Neural Networks (IJCNN), pp. 681\u2013688. IEEE (2017)","DOI":"10.1109\/IJCNN.2017.7965918"},{"issue":"26","key":"7_CR3","first-page":"429","volume":"93","author":"D Gabor","year":"1946","unstructured":"Gabor, D.: Theory of communication. Part 1: the analysis of information. J. Inst. Electr. Eng. Part III Radio Commun. Eng. 93(26), 429\u2013441 (1946)","journal-title":"J. Inst. Electr. Eng. Part III Radio Commun. Eng."},{"key":"7_CR4","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning, pp. 369\u2013376. ACM (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"7_CR5","unstructured":"Graves, A., Jaitly, N.: Towards end-to-end speech recognition with recurrent neural networks. In: International Conference on Machine Learning, pp. 1764\u20131772 (2014)"},{"key":"7_CR6","unstructured":"Hannun, A., et al.: Deep speech: scaling up end-to-end speech recognition. arXiv:1412.5567 (2014)"},{"issue":"8","key":"7_CR7","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"7_CR8","doi-asserted-by":"publisher","first-page":"4357","DOI":"10.1109\/TIP.2018.2835143","volume":"27","author":"S Luan","year":"2018","unstructured":"Luan, S., Chen, C., Zhang, B., Han, J., Liu, J.: Gabor convolutional networks. IEEE Trans. Image Process. 27, 4357\u20134366 (2018)","journal-title":"IEEE Trans. Image Process."},{"issue":"5588","key":"7_CR9","doi-asserted-by":"publisher","first-page":"746","DOI":"10.1038\/264746a0","volume":"264","author":"H McGurk","year":"1976","unstructured":"McGurk, H., MacDonald, J.: Hearing lips and seeing voices. Nature 264(5588), 746\u2013748 (1976)","journal-title":"Nature"},{"key":"7_CR10","doi-asserted-by":"crossref","unstructured":"Mesgarani, N., David, S., Shamma, S.: Representation of phonemes in primary auditory cortex: how the brain analyzes speech. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), vol. 4, pp. 765\u2013768. IEEE (2007)","DOI":"10.1109\/ICASSP.2007.367025"},{"issue":"3","key":"7_CR11","doi-asserted-by":"publisher","first-page":"727","DOI":"10.1016\/j.compeleceng.2012.12.011","volume":"39","author":"S Meshgini","year":"2013","unstructured":"Meshgini, S., Aghagolzadeh, A., Seyedarabi, H.: Face recognition using Gabor-based direct linear discriminant analysis and support vector machine. Comput. Electr. Eng. 39(3), 727\u2013745 (2013)","journal-title":"Comput. Electr. Eng."},{"issue":"4","key":"7_CR12","doi-asserted-by":"publisher","first-page":"854","DOI":"10.1109\/TNN.2002.1021886","volume":"13","author":"S Nakamura","year":"2002","unstructured":"Nakamura, S.: Statistical multimodal integration for audio-visual speech processing. IEEE Trans. Neural Netw. 13(4), 854\u2013866 (2002)","journal-title":"IEEE Trans. Neural Netw."},{"issue":"4","key":"7_CR13","doi-asserted-by":"publisher","first-page":"722","DOI":"10.1007\/s10489-014-0629-7","volume":"42","author":"K Noda","year":"2015","unstructured":"Noda, K., Yamaguchi, Y., Nakadai, K., Okuno, H.G., Ogata, T.: Audio-visual speech recognition using deep learning. Appl. Intell. 42(4), 722\u2013737 (2015)","journal-title":"Appl. Intell."},{"key":"7_CR14","doi-asserted-by":"crossref","unstructured":"Patterson, E.K., Gurbuz, S., Tufekci, Z., Gowdy, J.N.: CUAVE: a new audio-visual database for multimodal human-computer interface research. In: Proceedings of IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), Orlando, Florida, USA, vol. 2, pp. II-2017\u2013II-2020. IEEE (2002)","DOI":"10.1109\/ICASSP.2002.1006168"},{"key":"7_CR15","doi-asserted-by":"crossref","unstructured":"Petridis, S., Li, Z., Pantic, M.: End-to-end visual speech recognition with LSTMs. arXiv:1701.05847 (2017)","DOI":"10.1109\/ICASSP.2017.7952625"},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"Petridis, S., Pantic, M.: Deep complementary bottleneck features for visual speech recognition. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 2304\u20132308. IEEE (2016)","DOI":"10.1109\/ICASSP.2016.7472088"},{"issue":"5","key":"7_CR17","doi-asserted-by":"publisher","first-page":"553","DOI":"10.1016\/j.imavis.2006.05.002","volume":"25","author":"L Shen","year":"2007","unstructured":"Shen, L., Bai, L., Fairhurst, M.: Gabor wavelets and general discriminant analysis for face identification and verification. Image Vis. Comput. 25(5), 553\u2013563 (2007)","journal-title":"Image Vis. Comput."},{"key":"7_CR18","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1007\/978-3-319-59259-6_9","volume-title":"Multimodal Pattern Recognition of Social Signals in Human-Computer-Interaction","author":"A Thanda","year":"2017","unstructured":"Thanda, A., Venkatesan, S.M.: Audio visual speech recognition using deep recurrent neural networks. In: Schwenker, F., Scherer, S. (eds.) MPRSS 2016. LNCS (LNAI), vol. 10183, pp. 98\u2013109. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-59259-6_9"},{"issue":"2","key":"7_CR19","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1023\/B:VISI.0000013087.49260.fb","volume":"57","author":"P Viola","year":"2004","unstructured":"Viola, P., Jones, M.J.: Robust real-time face detection. Int. J. Comput. Vis. 57(2), 137\u2013154 (2004)","journal-title":"Int. J. Comput. Vis."},{"issue":"2","key":"7_CR20","doi-asserted-by":"publisher","first-page":"533","DOI":"10.1109\/TIP.2009.2035882","volume":"19","author":"B Zhang","year":"2010","unstructured":"Zhang, B., Gao, Y., Zhao, S., Liu, J.: Local derivative pattern versus local binary pattern: face recognition with high-order local pattern descriptor. IEEE Trans. Image Process. 19(2), 533\u2013544 (2010)","journal-title":"IEEE Trans. Image Process."}],"container-title":["Lecture Notes in Computer Science","Multimodal Pattern Recognition of Social Signals in Human-Computer-Interaction"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-20984-1_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,28]],"date-time":"2023-05-28T00:02:48Z","timestamp":1685232168000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-20984-1_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030209834","9783030209841"],"references-count":20,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-20984-1_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"15 May 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MPRSS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"IAPR Workshop on Multimodal Pattern Recognition of Social Signals in Human-Computer Interaction","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Beijing","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2018","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 August 2018","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 August 2018","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mprss2018","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/neuro.informatik.uni-ulm.de\/MPRSS2018\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"12","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"9","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"75% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}