{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T14:19:46Z","timestamp":1743085186770,"version":"3.40.3"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031176173"},{"type":"electronic","value":"9783031176180"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-17618-0_6","type":"book-chapter","created":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T03:50:31Z","timestamp":1664596231000},"page":"67-75","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Applying Generative Adversarial Networks and\u00a0Vision Transformers in\u00a0Speech Emotion Recognition"],"prefix":"10.1007","author":[{"given":"Panikos","family":"Heracleous","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Satoru","family":"Fukayama","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Ogata","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yasser","family":"Mohammad","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,10,2]]},"reference":[{"key":"6_CR1","doi-asserted-by":"crossref","unstructured":"Busso, C., Bulut, M., Narayanan, S.: Toward effective automatic recognition systems of emotion in speech. In: Gratch, J., Marsella, S. (eds.) Social Emotions in Nature and Artifact: Emotions in Human and Human-Computer Interaction, pp. 110\u2013127. Oxford University Press, New York, November 2013","DOI":"10.1093\/acprof:oso\/9780195387643.003.0008"},{"key":"6_CR2","doi-asserted-by":"crossref","unstructured":"Feng, H., Uno, S., Kawahara, T.: End-to-end speech emotion recognition combined with acoustic-to-word ASR model. In: INTERSPEECH, pp. 501\u2013505 (2020)","DOI":"10.21437\/Interspeech.2020-1180"},{"key":"6_CR3","doi-asserted-by":"crossref","unstructured":"Huang, J., Tao, J., Liu, B., Lian, Z.: Learning utterance-level representations with label smoothing for speech emotion recognition. In: Proceedings of Interspeech, pp. 4079\u20134083 (2020)","DOI":"10.21437\/Interspeech.2020-1391"},{"key":"6_CR4","doi-asserted-by":"crossref","unstructured":"Jalal, M.A., Milner, R., Hain, T., Moore, R.K.: Removing bias with residual mixture of multi-view attention for speech emotion recognition. In: Proceedings of Interspeech, pp. 4084\u20134088 (2020)","DOI":"10.21437\/Interspeech.2020-3005"},{"key":"6_CR5","doi-asserted-by":"crossref","unstructured":"Jalal, M.A., Milner, R., Hain, T.: Empirical interpretation of speech emotion perception with attention based model for speech emotion recognition. In: Proceedings of Interspeech, pp. 4113\u20134117 (2020)","DOI":"10.21437\/Interspeech.2020-3007"},{"key":"6_CR6","doi-asserted-by":"crossref","unstructured":"Stuhlsatz, A., Meyer, C., Eyben, F., Zielke1, T., Meier, G., Schuller, B.: Deep neural networks for acoustic emotion recognition: raising the benchmarks. In: Proceedings of ICASSP, pp. 5688\u20135691 (2011)","DOI":"10.1109\/ICASSP.2011.5947651"},{"key":"6_CR7","doi-asserted-by":"crossref","unstructured":"Han, K., Yu, D., Tashev, I.: Speech emotion recognition using deep neural network and extreme learning machine. In: Proceedings of Interspeech, pp. 2023\u20132027 (2014)","DOI":"10.21437\/Interspeech.2014-57"},{"key":"6_CR8","doi-asserted-by":"crossref","unstructured":"Lim, W., Jang, D., Lee, T.: Speech emotion recognition using convolutional and recurrent neural networks. In: Proceedings of Signal and Information Processing Association Annual Summit and Conference (APSIPA) (2016)","DOI":"10.1109\/APSIPA.2016.7820699"},{"key":"6_CR9","doi-asserted-by":"publisher","first-page":"2352","DOI":"10.1162\/neco_a_00990","volume":"29","author":"W Rawat","year":"2017","unstructured":"Rawat, W., Wang, Z.: Deep convolutional neural networks for image classification: a comprehensive review. Neural Commun. 29, 2352\u20132449 (2017)","journal-title":"Neural Commun."},{"key":"6_CR10","series-title":"Lecture Notes in Electrical Engineering","doi-asserted-by":"publisher","first-page":"441","DOI":"10.1007\/978-981-10-0557-2_44","volume-title":"Information Science and Applications (ICISA) 2016","author":"X-P Huynh","year":"2016","unstructured":"Huynh, X.-P., Tran, T.-D., Kim, Y.-G.: Convolutional neural network models for facial expression recognition using BU-3DFE database. In: Information Science and Applications (ICISA) 2016. LNEE, vol. 376, pp. 441\u2013450. Springer, Singapore (2016). https:\/\/doi.org\/10.1007\/978-981-10-0557-2_44"},{"key":"6_CR11","doi-asserted-by":"crossref","unstructured":"Jalal, M., Milner, R., Hain, T.: Empirical interpretation of speech emotion perception with attention based model for speech emotion recognition. In: INTERSPEECH, pp. 4113\u20134117 (2020)","DOI":"10.21437\/Interspeech.2020-3007"},{"key":"6_CR12","doi-asserted-by":"crossref","unstructured":"Padi, S., Sadjadi, S.O., Sriram, R.D., Manocha, D.: Improved speech emotion recognition using transfer learning and spectrogram augmentation. In: ICMI, pp. 645\u2013652 (2021)","DOI":"10.1145\/3462244.3481003"},{"key":"6_CR13","doi-asserted-by":"crossref","unstructured":"Xu, Y., Xu, H., Zou, J.: HGEM: a hierarchical grained and feature model for acoustic emotion recognition. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6499\u20136503. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053039"},{"key":"6_CR14","unstructured":"Baevski, A., Zhou, H., Mohamed, A., Auli, M.: wav2vec 2.0: a framework for self-supervised learning of speech representations. arXiv preprint arXiv:2006.11477 (2020)"},{"key":"6_CR15","unstructured":"Wang, Y., Boumadane, A., Heba, A.: A fine-tuned Wav2vec 2.0\/Hubert Benchmark For Speech Emotion Recognition, Speaker Verification and Spoken Language Understanding. arXiv preprint arXiv:2111.02735 (2021)"},{"issue":"1","key":"6_CR16","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1016\/j.csl.2012.02.005","volume":"27","author":"B Schuller","year":"2013","unstructured":"Schuller, B., et al.: Paralinguistics in speech and languagestate-of-the-art and the challenge. Comput. Speech Lang. 27(1), 4\u201339 (2013)","journal-title":"Comput. Speech Lang."},{"key":"6_CR17","unstructured":"Ian, G., et al.: Generative adversarial nets. In: Advances in Neural Information Processing Systems (2014)"},{"key":"6_CR18","unstructured":"Dosovitskiy, A, et al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv:2010.11929v2 (2020)"},{"key":"6_CR19","doi-asserted-by":"crossref","unstructured":"Zhu, J.Y., Park, T., Isola, P., Efros, A.A.: Unpaired image-toimage translation using cycle-consistent adversarial networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2223\u20132232 (2017)","DOI":"10.1109\/ICCV.2017.244"},{"key":"6_CR20","doi-asserted-by":"crossref","unstructured":"Kaneko, T., Kameoka, H.: Parallel-data-free voice conversion using cycle-consistent adversarial networks. In: 26th European Signal Processing Conference arXiv:1711.11293, November 2017 (2018)","DOI":"10.23919\/EUSIPCO.2018.8553236"},{"key":"6_CR21","doi-asserted-by":"crossref","unstructured":"Bao, F., Neumann, M., Vu, N.T.: Cyclegan-based emotion style transfer as data augmentation for speech emotion recognition. In: Proceedings of Interspeech 2019, pp. 2828\u20132832 (2019)","DOI":"10.21437\/Interspeech.2019-2293"},{"key":"6_CR22","unstructured":"Vaswani, A., et al.: Attention is all you need. arXiv:1706.03762 (2017)"},{"key":"6_CR23","unstructured":"Livingstone, S.R., Peck, K., Russo, F.A.: RAVDESS: the ryerson audio-visual database of emotional speech and song. In: 22nd Annual Meeting of the Canadian Society for Brain, Behaviour and Cognitive Science (CSBBCS) (Kingston, ON) (2012)"},{"issue":"6","key":"6_CR24","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton, G., et al.: Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process. Mag. 29(6), 82\u201397 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"6_CR25","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: Librispeech: an ASR corpus based on public domain 382 audio books. In: IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 5206\u20135210 (2015)","DOI":"10.1109\/ICASSP.2015.7178964"}],"container-title":["Lecture Notes in Computer Science","HCI International 2022 - Late Breaking Papers. Multimodality in Advanced Interaction Environments"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-17618-0_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T03:50:56Z","timestamp":1664596256000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-17618-0_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031176173","9783031176180"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-17618-0_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"2 October 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"HCII","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Human-Computer Interaction","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 June 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 July 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"hcii2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2022.hci.international\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}