{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T21:22:42Z","timestamp":1743110562486,"version":"3.40.3"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030898199"},{"type":"electronic","value":"9783030898205"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-89820-5_4","type":"book-chapter","created":{"date-parts":[[2021,10,20]],"date-time":"2021-10-20T20:35:26Z","timestamp":1634762126000},"page":"46-58","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Improving a Conversational Speech Recognition System Using Phonetic and\u00a0Neural Transcript Correction"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1916-2902","authenticated-orcid":false,"given":"Mario","family":"Campos-Soberanis","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1832-2011","authenticated-orcid":false,"given":"Diego","family":"Campos-Sobrino","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4104-0765","authenticated-orcid":false,"given":"Rafael","family":"Viana-C\u00e1mara","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,10,21]]},"reference":[{"key":"4_CR1","doi-asserted-by":"publisher","unstructured":"Bassil, Y., Alwani, M.: Post-editing ERRO correction algorithm for speech recognition using bing spelling suggestion. Int. J. Adv. Comput. Sci. Appl. 3 (2012). https:\/\/doi.org\/10.14569\/IJACSA.2012.030217","DOI":"10.14569\/IJACSA.2012.030217"},{"key":"4_CR2","unstructured":"Berg, M.: Modelling of natural dialogues in the context of speech-based information and control systems. Ph.D. thesis, July 2014"},{"issue":"1","key":"4_CR3","doi-asserted-by":"publisher","first-page":"57","DOI":"10.13053\/rcs-148-1-6","volume":"148","author":"D Campos-Sobrino","year":"2019","unstructured":"Campos-Sobrino, D., Campos-Soberanis, M., Mart\u00ednez-Chin, I., Uc-Cetina, V.: Correcci\u00f3n de errores del reconocedor de voz de google usando m\u00e9tricas de distancia fon\u00e9tica. Res. Comput. Sci. 148(1), 57\u201370 (2019)","journal-title":"Res. Comput. Sci."},{"key":"4_CR4","doi-asserted-by":"crossref","unstructured":"Chan, W., Jaitly, N., Le, Q.V., Vinyals, O.: Listen, attend and spell: A neural network for large vocabulary conversational speech recognition. In: ICASSP (2016). http:\/\/williamchan.ca\/papers\/wchan-icassp-2016.pdf","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"4_CR5","unstructured":"Chorowski, J., Bahdanau, D., Cho, K., Bengio, Y.: End-to-end continuous speech recognition using attention-based recurrent NN: first results. In: NIPS 2014 Workshop on Deep Learning, December 2014 (2014)"},{"key":"4_CR6","doi-asserted-by":"publisher","unstructured":"Errattahi, R., Hannani, A.E., Ouahmane, H.: Automatic speech recognition errors detection and correction: a review. Procedia Comput. Sci. 128, 32\u201337 (2018). https:\/\/doi.org\/10.1016\/j.procs.2018.03.005, http:\/\/www.sciencedirect.com\/science\/article\/pii\/S1877050918302187, 1st International Conference on Natural Language and Speech Processing","DOI":"10.1016\/j.procs.2018.03.005"},{"key":"4_CR7","doi-asserted-by":"publisher","unstructured":"Fang, A., Filice, S., Limsopatham, N., Rokhlenko, O.: Using Phoneme Representations to Build Predictive Models Robust to ASR Errors, pp. 699\u2013708. ACM, July 2020. https:\/\/doi.org\/10.1145\/3397271.3401050","DOI":"10.1145\/3397271.3401050"},{"key":"4_CR8","unstructured":"Ghannay, S., Caubri\u00e8re, A., Est\u00e8ve, Y., Laurent, A., Morin, E.: End-to-end named entity extraction from speech. In: EEE Spoken Language Technology Workshop (2018)"},{"key":"4_CR9","unstructured":"Graves, A., Jaitly, N.: Towards end-to-end speech recognition with recurrent neural networks. In: 31st International Conference on Machine Learning (ICML 2014), vol. 5, pp. 1764\u20131772, January 2014"},{"key":"4_CR10","doi-asserted-by":"crossref","unstructured":"Graves, A.: Sequence transduction with recurrent neural networks (2012)","DOI":"10.1007\/978-3-642-24797-2"},{"key":"4_CR11","doi-asserted-by":"publisher","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural \u2019networks. In: Proceedings of the 23rd International Con-ference on Machine Learning, vol. 2006, pp. 369\u2013376, January 2006. https:\/\/doi.org\/10.1145\/1143844.1143891","DOI":"10.1145\/1143844.1143891"},{"key":"4_CR12","doi-asserted-by":"crossref","unstructured":"Haghani, P., et al.: From audio to semantics: approaches to end-to-end spoken language understanding. In: Spoken Language Technology Workshop (2018)","DOI":"10.1109\/SLT.2018.8639043"},{"key":"4_CR13","unstructured":"Liao, J., et al.: Improving readability for automatic speech recognition transcription (2020)"},{"key":"4_CR14","doi-asserted-by":"publisher","unstructured":"Limsopatham, N., Rokhlenko, O., Carmel, D.: Research challenges in building a voice-based artificial personal shopper - position paper. In: Proceedings of the 2018 EMNLP Workshop SCAI: The 2nd International Workshop on Search-Oriented Conversational AIpp, 40\u201345, January 2018. https:\/\/doi.org\/10.18653\/v1\/W18-5706","DOI":"10.18653\/v1\/W18-5706"},{"key":"4_CR15","doi-asserted-by":"crossref","unstructured":"Lugosch, L., Ravanelli, M., Ignoto, P., Tomar, V.S., Bengio, Y.: Speech model pre-training for end-to-end spoken language understanding (2019)","DOI":"10.21437\/Interspeech.2019-2396"},{"key":"4_CR16","doi-asserted-by":"publisher","unstructured":"Ogawa, A., Hori, T.: Error detection and accuracy estimation in automatic speech recognition using deep bidirectional recurrent neural networks. Speech Commun. 89 (2017).https:\/\/doi.org\/10.1016\/j.specom.2017.02.009","DOI":"10.1016\/j.specom.2017.02.009"},{"key":"4_CR17","doi-asserted-by":"publisher","unstructured":"Qian, Y., et al.: Exploring ASR-free end-to-end modeling to improve spoken language understanding in a cloud-based dialog system. In: 2017 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp. 569\u2013576 (2017). https:\/\/doi.org\/10.1109\/ASRU.2017.8268987","DOI":"10.1109\/ASRU.2017.8268987"},{"key":"4_CR18","doi-asserted-by":"publisher","unstructured":"Schumann, R., Angkititrakul, P.: Incorporating asr errors with attention-based, jointly trained RNN for intent detection and slot filling. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6059\u20136063, April 2018. https:\/\/doi.org\/10.1109\/ICASSP.2018.8461598","DOI":"10.1109\/ICASSP.2018.8461598"},{"key":"4_CR19","unstructured":"Shivakumar, P.G., Li, H., Knight, K., Georgiou, P.G.: Learning from past mistakes: improving automatic speech recognition output via noisy-clean phrase context modeling. CoRR abs\/1802.02607 (2018), http:\/\/arxiv.org\/abs\/1802.02607"},{"key":"4_CR20","doi-asserted-by":"publisher","unstructured":"Song, S., Zhang, N., Huang, H.: Named entity recognition based on conditional random fields. Clust. Comput. 22, 1\u201312 (2019). https:\/\/doi.org\/10.1007\/s10586-017-1146-3","DOI":"10.1007\/s10586-017-1146-3"},{"key":"4_CR21","doi-asserted-by":"crossref","unstructured":"Twiefel, J., Baumann, T., Heinrich, S., Wermter, S.: Improving domain-independent cloud-based speech recognition with domain-dependent phonetic post-processing. In: Proceedings of the Twenty-Eighth AAAI Conference on Artificial Intelligencevol. vol. 2, pp. 1529\u20131535, July 2014","DOI":"10.1609\/aaai.v28i1.8929"},{"issue":"8","key":"4_CR22","first-page":"1163","volume":"149","author":"R Viana-C\u00e1mara","year":"2020","unstructured":"Viana-C\u00e1mara, R., Campos-Soberanis, M., Campos-Sobrino, D.: Modelo h\u0131brido fon\u00e9tico-neural para correcci\u00f3n en sistemas de reconocimiento del habla. Res. Comput. Sci. 149(8), 1163\u20131177 (2020)","journal-title":"Res. Comput. Sci."},{"key":"4_CR23","doi-asserted-by":"publisher","unstructured":"Vorontsov, I., Kulakovskiy, I., Makeev, V.: Jaccard index based similarity measure to compare transcription factor binding site models. Algorith. Mol. Biol. AMB 8, 23 (2013). https:\/\/doi.org\/10.1186\/1748-7188-8-23","DOI":"10.1186\/1748-7188-8-23"},{"key":"4_CR24","doi-asserted-by":"publisher","unstructured":"Vtyurina, A., Fourney, A., Morris, M., Findlater, L., White, R.: Bridging screen readers and voice assistants for enhanced eyes-free web search. In: he World Wide Web Conference, pp. 3590\u20133594, May 2019). https:\/\/doi.org\/10.1145\/3308558.3314136","DOI":"10.1145\/3308558.3314136"},{"key":"4_CR25","doi-asserted-by":"publisher","unstructured":"Zobel, J., Dart, P.: Phonetic string matching: lessons from information retrieval, July 2002. https:\/\/doi.org\/10.1145\/243199.243258","DOI":"10.1145\/243199.243258"}],"container-title":["Lecture Notes in Computer Science","Advances in Soft Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-89820-5_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,13]],"date-time":"2023-01-13T01:09:19Z","timestamp":1673572159000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-89820-5_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030898199","9783030898205"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-89820-5_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"21 October 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MICAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Mexican International Conference on Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 October 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 October 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"micai2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.micai.org\/2021\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"129","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"58","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"45% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}