{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,11]],"date-time":"2024-09-11T05:01:30Z","timestamp":1726030890471},"publisher-location":"Cham","reference-count":22,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030205171"},{"type":"electronic","value":"9783030205188"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-20518-8_54","type":"book-chapter","created":{"date-parts":[[2019,6,4]],"date-time":"2019-06-04T23:02:40Z","timestamp":1559689360000},"page":"655-666","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Toward Robust Mispronunciation Detection via Audio-Visual Speech Recognition"],"prefix":"10.1007","author":[{"given":"Mahdie","family":"Karbasi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Steffen","family":"Zeiler","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jan","family":"Freiwald","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dorothea","family":"Kolossa","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,5,16]]},"reference":[{"issue":"3","key":"54_CR1","doi-asserted-by":"publisher","first-page":"475","DOI":"10.1109\/TASLP.2017.2783545","volume":"26","author":"AH Abdelaziz","year":"2018","unstructured":"Abdelaziz, A.H.: Comparing fusion models for DNN-based audiovisual continuous speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. (TASLP) 26(3), 475\u2013484 (2018)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process. (TASLP)"},{"issue":"5","key":"54_CR2","first-page":"863","volume":"23","author":"AH Abdelaziz","year":"2015","unstructured":"Abdelaziz, A.H., Zeiler, S., Kolossa, D.: Learning dynamic stream weights for coupled-HMM-based audio-visual speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 23(5), 863\u2013876 (2015)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"5","key":"54_CR3","doi-asserted-by":"publisher","first-page":"2421","DOI":"10.1121\/1.2229005","volume":"120","author":"M Cooke","year":"2006","unstructured":"Cooke, M., Barker, J., Cunningham, S., Shao, X.: An audio-visual corpus for speech perception and automatic speech recognition. J. Acoust. Soc. Am. 120(5), 2421\u20132424 (2006). https:\/\/doi.org\/10.1121\/1.2229005","journal-title":"J. Acoust. Soc. Am."},{"unstructured":"Freiwald, J., et al.: Utilizing slow feature analysis for lipreading. In: Proceedings of ITG, November 2018","key":"54_CR4"},{"doi-asserted-by":"crossref","unstructured":"Gergen, S., Zeiler, S., Hussen Abdelaziz, A., Nickel, R., Kolossa, D.: Dynamic stream weighting for turbo-decoding-based audiovisual ASR. In: Proceedings of ITG, pp. 2135\u20132139, September 2016","key":"54_CR5","DOI":"10.21437\/Interspeech.2016-166"},{"unstructured":"Graham, C.R., Lonsdale, D., Kennington, C., Johnson, A., McGhee, J.: Elicited imitation as an oral proficiency measure with ASR scoring. In: LREC (2008)","key":"54_CR6"},{"key":"54_CR7","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1016\/j.specom.2014.12.008","volume":"67","author":"W Hu","year":"2015","unstructured":"Hu, W., Qian, Y., Soong, F.K., Wang, Y.: Improved mispronunciation detection with deep neural network trained acoustic models and transfer learning based logistic regression classifiers. Speech Commun. 67, 154\u2013166 (2015)","journal-title":"Speech Commun."},{"doi-asserted-by":"crossref","unstructured":"Kjellstr\u00f6m, H., Engwall, O., Abdou, S.M., B\u00e4lter, O.: Audio-visual phoneme classification for pronunciation training applications. In: Proceedings of the Eighth Annual Conference of the International Speech Communication Association (2007)","key":"54_CR8","DOI":"10.21437\/Interspeech.2007-294"},{"doi-asserted-by":"crossref","unstructured":"Lee, A., Glass, J.: A comparison-based approach to mispronunciation detection. In: Proceedings of Spoken Language Technology Workshop (SLT), pp. 382\u2013387 (2012)","key":"54_CR9","DOI":"10.1109\/SLT.2012.6424254"},{"doi-asserted-by":"crossref","unstructured":"Lee, A., Zhang, Y., Glass, J.: Mispronunciation detection via dynamic time warping on deep belief network-based posteriorgrams. In: Proceedings of ICASSP, pp. 8227\u20138231 (2013)","key":"54_CR10","DOI":"10.1109\/ICASSP.2013.6639269"},{"issue":"1","key":"54_CR11","doi-asserted-by":"publisher","first-page":"193","DOI":"10.1109\/TASLP.2016.2621675","volume":"25","author":"K Li","year":"2017","unstructured":"Li, K., Qian, X., Meng, H.: Mispronunciation detection and diagnosis in L2 English speech using multidistribution deep neural networks. IEEE\/ACM Trans. Audio Speech Lang. Process. 25(1), 193\u2013207 (2017)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Li, W., Chen, N., Siniscalchi, M., Lee, C.H.: Improving mispronunciation detection for non-native learners with multisource information and LSTM-based deep models. In: Proceedings of Interspeech, pp. 2759\u20132763, September 2017","key":"54_CR12","DOI":"10.21437\/Interspeech.2017-464"},{"issue":"11","key":"54_CR13","doi-asserted-by":"publisher","first-page":"783042","DOI":"10.1155\/S1110865702206083","volume":"2002","author":"AV Nefian","year":"2002","unstructured":"Nefian, A.V., Liang, L., Pi, X., Liu, X., Murphy, K.: Dynamic Bayesian networks for audio-visual speech recognition. EURASIP J. Adv. Signal Process. 2002(11), 783042 (2002)","journal-title":"EURASIP J. Adv. Signal Process."},{"unstructured":"Picard, S., Ananthakrishnan, G., Wik, P., Engwall, O., Abdou, S.: Detection of specific mispronunciations using audiovisual features. In: Proceedings of Auditory-Visual Speech Processing (2010)","key":"54_CR14"},{"issue":"5","key":"54_CR15","doi-asserted-by":"publisher","first-page":"846","DOI":"10.1109\/TASLP.2016.2520364","volume":"24","author":"S Receveur","year":"2016","unstructured":"Receveur, S., Weiss, R., Fingscheidt, T.: Turbo automatic speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(5), 846\u2013862 (2016)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"2\u20133","key":"54_CR16","doi-asserted-by":"publisher","first-page":"511","DOI":"10.1016\/S0167-6393(03)00031-1","volume":"41","author":"M Richardson","year":"2003","unstructured":"Richardson, M., Bilmes, J., Diorio, C.: Hidden-articulator Markov models for speech recognition. Speech Commun. 41(2\u20133), 511\u2013529 (2003)","journal-title":"Speech Commun."},{"doi-asserted-by":"crossref","unstructured":"Ronen, O., Neumeyer, L., Franco, H.: Automatic detection of mispronunciation for language instruction. In: Proceedings of the Fifth Eurospeech (1997)","key":"54_CR17","DOI":"10.21437\/Eurospeech.1997-231"},{"issue":"1","key":"54_CR18","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1109\/TASL.2007.909330","volume":"16","author":"J Tepperman","year":"2008","unstructured":"Tepperman, J., Narayanan, S.: Using articulatory representations to detect segmental errors in nonnative pronunciation. IEEE\/ACM Trans. Audio Speech Lang. Process. 16(1), 8\u201322 (2008)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"unstructured":"Truong, K., Neri, A., Cucchiarini, C., Strik, H.: Automatic pronunciation error detection: an acoustic-phonetic approach. In: Proceedings of InSTIL\/ICALL Symposium (2004)","key":"54_CR19"},{"doi-asserted-by":"crossref","unstructured":"Wang, Y.B., Lee, L.S.: Improved approaches of modeling and detecting error patterns with empirical analysis for computer-aided pronunciation training. In: Proceedings of ICASSP, pp. 5049\u20135052 (2012)","key":"54_CR20","DOI":"10.1109\/ICASSP.2012.6289055"},{"issue":"3","key":"54_CR21","doi-asserted-by":"publisher","first-page":"564","DOI":"10.1109\/TASLP.2014.2387413","volume":"23","author":"YB Wang","year":"2015","unstructured":"Wang, Y.B., Lee, L.S.: Supervised detection and unsupervised discovery of pronunciation error patterns for computer-assisted language learning. IEEE\/ACM Trans. Audio Speech Lang. Process. (TASLP) 23(3), 564\u2013579 (2015)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process. (TASLP)"},{"doi-asserted-by":"crossref","unstructured":"Zeiler, S., Nickel, R., Ma, N., Brown, G., Kolossa, D.: Robust audiovisual speech recognition using noise-adaptive linear discriminant analysis. In: Proceedings of ICASSP (2016)","key":"54_CR22","DOI":"10.1109\/ICASSP.2016.7472187"}],"container-title":["Lecture Notes in Computer Science","Advances in Computational Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-20518-8_54","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,19]],"date-time":"2022-09-19T13:35:56Z","timestamp":1663594556000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-20518-8_54"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030205171","9783030205188"],"references-count":22,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-20518-8_54","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"16 May 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"IWANN","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Work-Conference on Artificial Neural Networks","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Gran Canaria","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Spain","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 June 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 June 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iwann2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iwann.uma.es\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"210","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"150","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"71% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"2,9","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"2,5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}}]}}