{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,13]],"date-time":"2026-01-13T14:53:05Z","timestamp":1768315985694,"version":"3.49.0"},"reference-count":34,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"3","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2024,3,1]]},"DOI":"10.1587\/transinf.2023hcp0010","type":"journal-article","created":{"date-parts":[[2024,2,29]],"date-time":"2024-02-29T22:25:16Z","timestamp":1709245516000},"page":"363-373","source":"Crossref","is-referenced-by-count":2,"title":["Simultaneous Adaptation of Acoustic and Language Models for Emotional Speech Recognition Using Tweet Data"],"prefix":"10.1587","volume":"E107.D","author":[{"given":"Tetsuo","family":"KOSAKA","sequence":"first","affiliation":[{"name":"Yamagata University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kazuya","family":"SAEKI","sequence":"additional","affiliation":[{"name":"Yamagata University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yoshitaka","family":"AIZAWA","sequence":"additional","affiliation":[{"name":"Yamagata University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Masaharu","family":"KATO","sequence":"additional","affiliation":[{"name":"Yamagata University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takashi","family":"NOSE","sequence":"additional","affiliation":[{"name":"Tohoku University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","unstructured":"[1] L. Smidl, A. Chylek, and J. Svec, \u201cA Multimodal dialogue system for air traffic control trainees based on discrete-event simulation,\u201d Proc. Interspeech2016, San Francisco, USA, pp.379-380, 2016."},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] A. Maier, J. Hough, and D. Schlangen, \u201cTowards deep end-of-turn prediction for situated spoken dialogue systems,\u201d Proc. Interspeech2017, Stockholm, Sweden, pp.1676-1680, 2017. 10.21437\/interspeech.2017-1593","DOI":"10.21437\/Interspeech.2017-1593"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] M. Li, Z. He, and J. Wu, \u201cTarget-based state and tracking algorithm for spoken dialogue system,\u201d Proc. Interspeech2016, San Francisco, USA, pp.2711-2715, 2016. 10.21437\/interspeech.2016-800","DOI":"10.21437\/Interspeech.2016-800"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] C. Liu, P. Xu, and R. Sarikaya, \u201cDeep contextual language understanding in spoken dialogue systems,\u201d Proc. Interspeech2015, Dresden, Germany, pp.120-124, 2015. 10.21437\/interspeech.2015-39","DOI":"10.21437\/Interspeech.2015-39"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] P.-H. Su, D. Vandyke, M. Gasic, D. Kim, N. Mrksic, T.-H. Wen, and S. Young, \u201cLearning from real users: rating dialogue success with neural networks for reinforcement learning in spoken dialogue systems,\u201d Proc. Interspeech2015, Dresden, Germany, pp.2007-2011, 2015. 10.21437\/interspeech.2015-456","DOI":"10.21437\/Interspeech.2015-456"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] A. Lee, K. Oura, and K. Tokuda, \u201cMMDAgent-a fully open-source toolkit for voice interaction systems,\u201d Proc. ICASSP2013, Vancouver, Canada, pp.8382-8385, 2013. 10.1109\/icassp.2013.6639300","DOI":"10.1109\/ICASSP.2013.6639300"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] K. Ohta, R. Marumoto, R. Nishimura, and N. Kitaoka, \u201cSelecting type of response for chat-like spoken dialogue systems based on acoustic features of user utterances,\u201d Proc. APSIPA-ASC2017, Kuala Lumpur, Malaysia, pp.1248-1252, 2017. 10.1109\/apsipa.2017.8282230","DOI":"10.1109\/APSIPA.2017.8282230"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] T. Kawahara, \u201cSpoken dialogue system for a human-like conversational robot ERICA,\u201d Proc. IWSDS2018, Singapore, pp.65-75, 2018. 10.1007\/978-981-13-9443-0_6","DOI":"10.1007\/978-981-13-9443-0_6"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] R. Zhang, A. Atsushi, S. Kobashikawa, and Y. Aono, \u201cInteraction and transition model for speech emotion recognition in dialogue,\u201d Proc. Interspeech2017, Stockholm, Sweden, pp.1094-1097, 2017. 10.21437\/interspeech.2017-713","DOI":"10.21437\/Interspeech.2017-713"},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] F. Burkhardt, A. Paeschke, M. Rolfes, W.F. Sendlmeier, and B. Weiss, \u201cA database of German emotional speech,\u201d Proc. Interspeech2005, Lisbon, Portugal, pp.3-6, 2005. 10.21437\/interspeech.2005-446","DOI":"10.21437\/Interspeech.2005-446"},{"key":"11","unstructured":"[11] A. Batliner, C. Hacker, S. Steidl, E. Noth, S. D&apos;Arcy, M. Russell, and M. Wong, \u201cYou stupid tin box-children interacting with the AIBO robot: A cross-linguistic emotional speech corpus,\u201d Proc. of LREC2004, Lisbon, Portugal, pp.171-174, 2004."},{"key":"12","doi-asserted-by":"publisher","unstructured":"[12] Y. Arimoto, H. Kawatsu, S. Ohno, and H. Iida, \u201cNaturalistic emotional speech collection paradigm with online game and its psychological and acoustical assessment,\u201d Acoust. Sci. Technol., vol.33, no.6, pp.359-369, 2012. 10.1250\/ast.33.359","DOI":"10.1250\/ast.33.359"},{"key":"13","doi-asserted-by":"publisher","unstructured":"[13] H. Mori, T. Satake, M. Nakamura, and H. Kasuya, \u201cConstructing a spoken dialogue corpus for studying paralinguistic information in expressive conversation and analyzing its statistical \/ acoustic characteristics,\u201d Speech Communication, vol.53, no.1, pp.36-50, 2011. 10.1016\/j.specom.2010.08.002","DOI":"10.1016\/j.specom.2010.08.002"},{"key":"14","doi-asserted-by":"crossref","unstructured":"[14] E. Takeishi, T. Nose, Y. Chiba, and A. Ito, \u201cConstruction and analysis of phonetically and prosodically balanced emotional speech database,\u201d Proc. O-COCOSDA2016, Bali, Indonesia, pp.16-21, 2016. 10.1109\/icsda.2016.7918977","DOI":"10.1109\/ICSDA.2016.7918977"},{"key":"15","unstructured":"[15] K. Maekawa, \u201cCorpus of spontaneous Japanese: Its design and evaluation,\u201d Proc. ISCA &amp; IEEE Workshop on Spontaneous Speech Processing and Recognition, Tokyo, Japan, pp.1-6, 2003."},{"key":"16","unstructured":"[16] K. Shinoda, \u201cTransfer learning in speech recognition: Speaker adaptation,\u201d Journal of the Japanese Society for Artificial Intelligence, vol.27, no.4, pp.359-364, 2012 (in Japanese)."},{"key":"17","unstructured":"[17] K. Mukaihara, S. Sakti, G. Neubig, T. Toda, and S. Nakamura, \u201cBottleneck features for emotional speech recognition,\u201d IPSJ SIG Tech. Report, vol.2015-SLP-107, no.2, Nagano, Japan, pp.1-6, 2015 (in Japanese)."},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] Y. Ijima, T. Nose, M. Tachibana, and T. Kobayashi, \u201cA Rapid Model Adaptation Technique for Emotional Speech Recognition with Style Estimation Based on Multiple-Regression HMM,\u201d IEICE Trans. Inf. &amp; Syst., vol.E93-D, no.1, pp.107-115, 2010.","DOI":"10.1587\/transinf.E93.D.107"},{"key":"19","doi-asserted-by":"publisher","unstructured":"[19] M. Sheikhan, D. Gharavian, and F. Ashoftedel, \u201cUsing DTW neural-based MFCC warping to improve emotional speech recognition,\u201d Neural Computing and Applications, vol.21, no.7, pp.1-9, 2021. 10.1007\/s00521-011-0620-8","DOI":"10.1007\/s00521-011-0620-8"},{"key":"20","doi-asserted-by":"crossref","unstructured":"[20] Y. Sun, Y. Zhou, Q. Zhao, and Y. Yan, \u201cAcoustic feature optimization for emotion affected speech recognition,\u201d Proc. ICIECS2009, Wuhan, China, pp.1-4, 2009. 10.1109\/iciecs.2009.5365821","DOI":"10.1109\/ICIECS.2009.5365821"},{"key":"21","doi-asserted-by":"publisher","unstructured":"[21] M. Bashirpour and M. Geravanchizadeh, \u201cRobust emotional speech recognition based on binaural model and emotional auditory mask in noisy environments,\u201d EURASIP Journal on Audio, Speech, and Music Processing, vol.2018, no.1, 2018. 10.1186\/s13636-018-0133-9","DOI":"10.1186\/s13636-018-0133-9"},{"key":"22","doi-asserted-by":"publisher","unstructured":"[22] M. Geravanchizadeh, E. Forouhandeh, and M. Bashirpour, \u201cFeature compensation based on the normalization of vocal tract length for the improvement of emotion-affected speech recognition,\u201d EURASIP Journal on Audio, Speech, and Music Processing, vol.2021, no.1, 2021. 10.1186\/s13636-021-00216-5","DOI":"10.1186\/s13636-021-00216-5"},{"key":"23","doi-asserted-by":"crossref","unstructured":"[23] T. Kosaka, Y. Aizawa, M. Kato, and T. Nose, \u201cAcoustic model adaptation for emotional speech recognition using Twitter-based emotional speech corpus,\u201d Proc. APSIPA-ASC2018, Honolulu, Hawaii, pp.1747-1751, 2018. 10.23919\/apsipa.2018.8659756","DOI":"10.23919\/APSIPA.2018.8659756"},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] K. Saeki, M. Kato, and T. Kosaka, \u201cPerformance Improvement of Prosody-Controlled Voice Conversion by Language Model Adaptation,\u201d Proc. IEEE GCCE2019, Osaka, Japan, pp.854-856, 2019. 10.1109\/gcce46687.2019.9015444","DOI":"10.1109\/GCCE46687.2019.9015444"},{"key":"25","unstructured":"[25] K. Saeki, M. Kato, and T. Kosaka, \u201cLanguage model adaptation for emotional speech recognition using tweet data,\u201d Proc. APSIPA-ASC2020, Auckland, New Zealand, pp.371-375, 2020."},{"key":"26","unstructured":"[26] A. Ito and M. Kohda, \u201cEvaluation of task adaptation using N-gram count mixture,\u201d IEICE Trans., vol.J83-D-II, no.11, pp.2418-2427, 2000 (in Japanese)."},{"key":"27","unstructured":"[27] T. Kudo, K. Yamamoto, and Y. Matsumoto, \u201cApplying conditional random fields to Japanese morphological analysis,\u201d Proc. EMNLP2004, Barcelona, Spain, pp.230-237, 2004."},{"key":"28","unstructured":"[28] K. Tomita, A. Takagi, M. Kato, and T. Kosaka, \u201cEvaluation of unsupervised cross adaptation using highly accurate models,\u201d Proc. ASJ2016 Autumn Meeting, Toyama, Japan, pp.95-96, 2016 (in Japanese)."},{"key":"29","unstructured":"[29] P. Daniel et al., \u201cThe Kaldi speech recognition toolkit,\u201d Proc. 2011 IEEE Workshop on Automatic Speech Recognition and Understanding, Hawaii, USA, pp.1-4, 2011."},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] M. Sakurai and T. Kosaka, \u201cEmotion recognition combining acoustic and linguistic features based on speech recognition results,\u201d Proc. IEEE GCCE2021, Kyoto, Japan, pp.889-892, 2021. 10.1109\/gcce53005.2021.9621810","DOI":"10.1109\/GCCE53005.2021.9621810"},{"key":"31","doi-asserted-by":"crossref","unstructured":"[31] T. Koseki and T. Kosaka, \u201cMultimodal spoken dialog system using state estimation by body motion,\u201d Proc. IEEE GCCE2017, Nagoya, Japan, pp.348-351, 2017. 10.1109\/gcce.2017.8229423","DOI":"10.1109\/GCCE.2017.8229423"},{"key":"32","unstructured":"[32] H. Iida and Y. Arimoto, \u201cA method for estimating the degree of emotional expressions by parameterizing acoustic and linguistic features and treating them integratedly,\u201d Studies in Pragmatics, no.8, pp.33-46, 2006 (in Japanese)."},{"key":"33","doi-asserted-by":"publisher","unstructured":"[33] H.-I. Yun and J.-S. Park, \u201cEnd-to-end emotional speech recognition using acoustic model adaptation based on knowledge distillation,\u201d Multimedia Tools and Applications, vol.82, no.15, pp.22759-22776, 2023. 10.1007\/s11042-023-14680-y","DOI":"10.1007\/s11042-023-14680-y"},{"key":"34","doi-asserted-by":"crossref","unstructured":"[34] V. Raju, K. Gurugubelli, M.S. Ganesh, and A.K. Vuppala, \u201cTowards feature-space emotional speech adaptation for TDNN based Telugu ASR systems,\u201d Proc. SMM19, Vienna, Austria, pp.16-20, 2019. 10.21437\/smm.2019-4","DOI":"10.21437\/SMM.2019-4"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E107.D\/3\/E107.D_2023HCP0010\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,13]],"date-time":"2024-11-13T16:46:07Z","timestamp":1731516367000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E107.D\/3\/E107.D_2023HCP0010\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3,1]]},"references-count":34,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2023hcp0010","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,3,1]]},"article-number":"2023HCP0010"}}