{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T07:15:44Z","timestamp":1761808544050,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":35,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789811955372"},{"type":"electronic","value":"9789811955389"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-981-19-5538-9_8","type":"book-chapter","created":{"date-parts":[[2022,10,31]],"date-time":"2022-10-31T11:02:45Z","timestamp":1667214165000},"page":"123-131","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Comparison of Automatic Speech Recognition Systems"],"prefix":"10.1007","author":[{"given":"Joshua Y.","family":"Kim","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunfeng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rafael A.","family":"Calvo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kathryn","family":"McCabe","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Silas C. R.","family":"Taylor","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bj\u00f6rn W.","family":"Schuller","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kaihang","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,11,1]]},"reference":[{"key":"8_CR1","unstructured":"Belambert: Asr-evaluation. https:\/\/github.com\/belambert\/asr-evaluation"},{"issue":"2","key":"8_CR2","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1007\/s10579-007-9040-x","volume":"41","author":"J Carletta","year":"2007","unstructured":"Carletta J (2007) Unleashing the killer corpus: experiences in creating the multi-everything ami meeting corpus. Lang Resour Eval 41(2):181\u2013190","journal-title":"Lang Resour Eval"},{"key":"8_CR3","doi-asserted-by":"crossref","unstructured":"Chiu CC, Sainath TN, Wu Y, Prabhavalkar R, Nguyen P, Chen Z, Kannan A, Weiss RJ, Rao K, Gonina E, et\u00a0al (2018) State-of-the-art speech recognition with sequence-to-sequence models. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 4774\u20134778","DOI":"10.1109\/ICASSP.2018.8462105"},{"issue":"3","key":"8_CR4","first-page":"16","volume":"10","author":"SK Gaikwad","year":"2010","unstructured":"Gaikwad SK, Gawali BW, Yannawar P (2010) A review on speech recognition technique. Int J Comput Appl 10(3):16\u201324","journal-title":"Int J Comput Appl"},{"key":"8_CR5","doi-asserted-by":"crossref","unstructured":"Garofolo JS, Lamel LF, Fisher WM, Fiscus JG, Pallett DS (1993) Darpa timit acoustic-phonetic continuous speech corpus cd-rom. nist speech disc 1-1.1. NASA STI\/Recon technical report n 93, 27403","DOI":"10.6028\/NIST.IR.4930"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Gillick L, Cox SJ (1989) Some statistical issues in the comparison of speech recognition algorithms. In: International conference on acoustics, speech, and signal processing. IEEE, pp 532\u2013535","DOI":"10.1109\/ICASSP.1989.266481"},{"key":"8_CR7","doi-asserted-by":"crossref","unstructured":"Gopal RK, Solanki P, Bokhour B, Skorohod N, Hernandez-Lujan D, Gordon H (2021) Provider, staff, and patient perspectives on medical visits using clinical video telehealth: a foundation for educational initiatives to improve medical care in telehealth. J Nurse Practit","DOI":"10.1016\/j.nurpra.2021.02.020"},{"issue":"6","key":"8_CR8","doi-asserted-by":"publisher","first-page":"1751","DOI":"10.1007\/s11606-020-05673-w","volume":"35","author":"HS Gordon","year":"2020","unstructured":"Gordon HS, Solanki P, Bokhour BG, Gopal RK (2020) \u201ci\u2019m not feeling like i\u2019m part of the conversation\u2019\u2019 patients\u2019 perspectives on communicating in clinical video telehealth visits. J Gen Intern Med 35(6):1751\u20131758","journal-title":"J Gen Intern Med"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Hazarika D, Poria S, Mihalcea R, Cambria E, Zimmermann R (2018) Icon: interactive conversational memory network for multimodal emotion detection. In: Proceedings of the 2018 conference on empirical methods in natural language processing, pp 2594\u20132604","DOI":"10.18653\/v1\/D18-1280"},{"key":"8_CR10","doi-asserted-by":"crossref","unstructured":"Hazarika D, Poria S, Zadeh A, Cambria E, Morency LP, Zimmermann R (2018) Conversational memory network for emotion recognition in dyadic dialogue videos. In: Proceedings of the conference. Association for computational linguistics. North American Chapter. Meeting, vol\u00a02018, p\u00a02122. NIH Public Access","DOI":"10.18653\/v1\/N18-1193"},{"key":"8_CR11","doi-asserted-by":"crossref","unstructured":"Henton C (2005) Bitter pills to swallow. asr and tts have drug problems. Int J Speech Technol 8(3), 247\u2013257","DOI":"10.1007\/s10772-006-5889-0"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"James G, Witten D, Hastie T, Tibshirani R (2013) An introduction to statistical learning, vol\u00a0112. Springer","DOI":"10.1007\/978-1-4614-7138-7"},{"issue":"03","key":"8_CR13","first-page":"20","volume":"7","author":"V K\u00ebpuska","year":"2017","unstructured":"K\u00ebpuska V, Bohouta G (2017) Comparing speech recognition systems (microsoft api, google api and cmu sphinx). Int J Eng Res Appl 7(03):20\u201324","journal-title":"Int J Eng Res Appl"},{"key":"8_CR14","unstructured":"Kim JY, Calvo RA, Yacef K, Enfield N (2019) A review on dyadic conversation visualizations-purposes, data, lens of analysis. arXiv:1905.00653"},{"key":"8_CR15","doi-asserted-by":"crossref","unstructured":"Kim JY, Kim GY, Yacef K (2019) Detecting depression in dyadic conversations with multimodal narratives and visualizations. In: Australasian joint conference on artificial intelligence. Springer, pp 303\u2013314","DOI":"10.1007\/978-3-030-35288-2_25"},{"key":"8_CR16","doi-asserted-by":"crossref","unstructured":"Kim JY, Yacef K, Kim G, Liu C, Calvo R, Taylor S (2021) Monah: multi-modal narratives for humans to analyze conversations. In: Proceedings of the 16th conference of the European chapter of the association for computational linguistics: main volume, pp 466\u2013479","DOI":"10.18653\/v1\/2021.eacl-main.37"},{"issue":"10","key":"8_CR17","first-page":"1995","volume":"3361","author":"Y LeCun","year":"1995","unstructured":"LeCun Y, Bengio Y et al (1995) Convolutional networks for images, speech, and time series. Handbook of Brain Theory and Neural Netw 3361(10):1995","journal-title":"Handbook of Brain Theory and Neural Netw"},{"key":"8_CR18","doi-asserted-by":"crossref","unstructured":"Li J, Zhao R, Chen Z, Liu C, Xiao X, Ye G, Gong Y (2018) Developing far-field speaker system via teacher-student learning. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5699\u20135703","DOI":"10.1109\/ICASSP.2018.8462209"},{"key":"8_CR19","doi-asserted-by":"crossref","unstructured":"Liu C, Lim RL, McCabe KL, Taylor S, Calvo RA (2016) A web-based telehealth training platform incorporating automated nonverbal behavior feedback for teaching communication skills to medical students: a randomized crossover study. J Med Internet Res 18(9):e246","DOI":"10.2196\/jmir.6299"},{"issue":"1","key":"8_CR20","doi-asserted-by":"publisher","first-page":"31801","DOI":"10.3402\/meo.v21.31801","volume":"21","author":"C Liu","year":"2016","unstructured":"Liu C, Scott KM, Lim RL, Taylor S, Calvo RA (2016) Eqclinic: a platform for learning communication skills in clinical consultations. Med Educ Online 21(1):31801","journal-title":"Med Educ Online"},{"key":"8_CR21","doi-asserted-by":"crossref","unstructured":"Majumder N, Poria S, Hazarika D, Mihalcea R, Gelbukh A, Cambria E (2019) Dialoguernn: An attentive rnn for emotion detection in conversations. In: Proceedings of the AAAI conference on artificial intelligence, vol\u00a033, pp 6818\u20136825","DOI":"10.1609\/aaai.v33i01.33016818"},{"key":"8_CR22","doi-asserted-by":"crossref","unstructured":"Mani A, Palaskar S, Konam S (2020) Towards understanding asr error correction for medical conversations. In: Proceedings of the first workshop on natural language processing for medical conversations, pp 7\u201311","DOI":"10.18653\/v1\/2020.nlpmc-1.2"},{"key":"8_CR23","doi-asserted-by":"crossref","unstructured":"Miao K, Biermann O, Miao Z, Leung S, Wang J, Gai k (2020) integrated parallel system for audio conferencing voice transcription and speaker identification. In: 2020 international conference on high performance big data and intelligent systems (HPBD &IS). IEEE, pp\u00a01\u20138","DOI":"10.1109\/HPBDIS49115.2020.9130598"},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Mittal T, Bhattacharya U, Chandra R, Bera A, Manocha D (2020) M3er: multiplicative multimodal emotion recognition using facial, textual, and speech cues. In: AAAI, pp 1359\u20131367","DOI":"10.1609\/aaai.v34i02.5492"},{"issue":"7\u20138","key":"8_CR25","doi-asserted-by":"publisher","first-page":"1053","DOI":"10.1111\/jocn.15178","volume":"29","author":"C Nielsen","year":"2020","unstructured":"Nielsen C, Agerskov H, Bistrup C, Clemensen J (2020) Evaluation of a telehealth solution developed to improve follow-up after kidney transplantation. J Clin Nurs 29(7\u20138):1053\u20131063","journal-title":"J Clin Nurs"},{"key":"8_CR26","doi-asserted-by":"crossref","unstructured":"Renals S, Swietojanski P (2017) Distant speech recognition experiments using the AMI corpus. New Era for robust speech recognition, pp 355\u2013368","DOI":"10.1007\/978-3-319-64680-0_16"},{"key":"8_CR27","doi-asserted-by":"crossref","unstructured":"Roy BC, Roy DK, Vosoughi S (2010) Automatic estimation of transcription accuracy and difficulty","DOI":"10.21437\/Interspeech.2010-548"},{"key":"8_CR28","doi-asserted-by":"crossref","unstructured":"Saon G, Kuo HKJ, Rennie S, Picheny M (2015) The IBM 2015 english conversational telephone speech recognition system. arXiv:1505.05899","DOI":"10.21437\/Interspeech.2015-632"},{"key":"8_CR29","unstructured":"Siohan O, Ramabhadran B, Kingsbury B (2005) Constructing ensembles of asr systems using randomized decision trees. In: Proceedings.(ICASSP\u201905). IEEE international conference on acoustics, speech, and signal processing, 2005. vol\u00a01. IEEE, pp I\u2013197"},{"issue":"9","key":"8_CR30","doi-asserted-by":"publisher","first-page":"1120","DOI":"10.1109\/LSP.2014.2325781","volume":"21","author":"P Swietojanski","year":"2014","unstructured":"Swietojanski P, Ghoshal A, Renals S (2014) Convolutional neural networks for distant speech recognition. IEEE Signal Process Lett 21(9):1120\u20131124","journal-title":"IEEE Signal Process Lett"},{"key":"8_CR31","doi-asserted-by":"crossref","unstructured":"Tang Z, Meng HY, Manocha D (2020) Low-frequency compensated synthetic impulse responses for improved far-field speech recognition. In: ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6974\u20136978","DOI":"10.1109\/ICASSP40776.2020.9054454"},{"key":"8_CR32","doi-asserted-by":"crossref","unstructured":"Xiong W, Droppo J, Huang X, Seide F, Seltzer M, Stolcke A, Yu D, Zweig G (2016) Achieving human parity in conversational speech recognition. arXiv:1610.05256","DOI":"10.1109\/TASLP.2017.2756440"},{"key":"8_CR33","doi-asserted-by":"crossref","unstructured":"Xiong W, Wu L, Alleva F, Droppo J, Huang X, Stolcke A (2018) The microsoft 2017 conversational speech recognition system. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5934\u20135938","DOI":"10.1109\/ICASSP.2018.8461870"},{"key":"8_CR34","doi-asserted-by":"crossref","unstructured":"Zadeh A, Liang PP, Mazumder N, Poria S, Cambria E, Morency LP (2018) Memory fusion network for multi-view sequential learning. In: Proceedings of the AAAI conference on artificial intelligence, vol\u00a032","DOI":"10.1609\/aaai.v32i1.12021"},{"key":"8_CR35","doi-asserted-by":"crossref","unstructured":"Zhao T, Zhao Y, Wang S, Han M (2021) Unet++-based multi-channel speech dereverberation and distant speech recognition. In: 2021 12th international symposium on Chinese spoken language processing (ISCSLP). IEEE, pp\u00a01\u20135","DOI":"10.1109\/ISCSLP49672.2021.9362064"}],"container-title":["Lecture Notes in Electrical Engineering","Conversational AI for Natural Human-Centric Interaction"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-19-5538-9_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,31]],"date-time":"2022-10-31T11:05:25Z","timestamp":1667214325000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-19-5538-9_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9789811955372","9789811955389"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-981-19-5538-9_8","relation":{},"ISSN":["1876-1100","1876-1119"],"issn-type":[{"type":"print","value":"1876-1100"},{"type":"electronic","value":"1876-1119"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"1 November 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}