{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,24]],"date-time":"2025-08-24T00:02:30Z","timestamp":1755993750773,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,11,24]],"date-time":"2023-11-24T00:00:00Z","timestamp":1700784000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,11,24]]},"DOI":"10.1145\/3638682.3638698","type":"proceedings-article","created":{"date-parts":[[2024,5,22]],"date-time":"2024-05-22T15:32:08Z","timestamp":1716391928000},"page":"108-114","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["A multimodal emotion recognition method based on multiple fusion of audio-visual modalities"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-3861-7122","authenticated-orcid":false,"given":"Zheng","family":"Wu","sequence":"first","affiliation":[{"name":"Key Laboratory of Education Informatization for Nationalities (Yunnan Normal University), Ministry of Education, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1287-857X","authenticated-orcid":false,"given":"Jianhou","family":"Gan","sequence":"additional","affiliation":[{"name":"Key Laboratory of Education Informatization for Nationalities (Yunnan Normal University), Ministry of Education, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1731-3863","authenticated-orcid":false,"given":"Jinsheng","family":"Liu","sequence":"additional","affiliation":[{"name":"Key Laboratory of Education Informatization for Nationalities (Yunnan Normal University), Ministry of Education, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9878-4664","authenticated-orcid":false,"given":"Jun","family":"Wang","sequence":"additional","affiliation":[{"name":"Key Laboratory of Education Informatization for Nationalities (Yunnan Normal University), Ministry of Education, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,5,22]]},"reference":[{"volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 32","author":"Zhou H.","key":"e_1_3_2_1_1_1","unstructured":"Zhou, H., Huang, M., Zhang, T., Zhu, X. and Liu, B., 2018, April. Emotional chatting machine: Emotional conversation generation with internal and external memory. In Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 32, No. 1)."},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"2","author":"Huang C.","unstructured":"Huang, C., Zaiane, O.R., Trabelsi, A. and Dziri, N., 2018, June. Automatic dialogue generation with expressed emotions. In Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 2 (Short Papers) (pp. 49-54)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P14-5010"},{"volume-title":"2020 International Conference on Electronics, Information, and Communication (ICEIC) (pp. 1-3). IEEE.","author":"Abdullah M.","key":"e_1_3_2_1_4_1","unstructured":"Abdullah, M., Ahmad, M. and Han, D., 2020, January. Facial expression recognition in videos: An CNN-LSTM based model for video classification. In 2020 International Conference on Electronics, Information, and Communication (ICEIC) (pp. 1-3). IEEE."},{"volume-title":"Proceedings of Interspeech 2020 (pp. 4113-4117)","author":"Jalal M.A.","key":"e_1_3_2_1_5_1","unstructured":"Jalal, M.A., Milner, R. and Hain, T., 2020, October. Empirical interpretation of speech emotion perception with attention based model for speech emotion recognition. In Proceedings of Interspeech 2020 (pp. 4113-4117). International Speech Communication Association (ISCA)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Xia R. and Ding Z. 2019. Emotion-cause pair extraction: A new task to emotion analysis in texts. arXiv preprint arXiv:1906.01267.","DOI":"10.18653\/v1\/P19-1096"},{"volume-title":"Proceedings of the 58th annual meeting of the association for computational linguistics (pp. 3718-3727)","author":"Yu W.","key":"e_1_3_2_1_7_1","unstructured":"Yu, W., Xu, H., Meng, F., Zhu, Y., Ma, Y., Wu, J., Zou, J. and Yang, K., 2020, July. Ch-sims: A chinese multimodal sentiment analysis dataset with fine-grained annotation of modality. In Proceedings of the 58th annual meeting of the association for computational linguistics (pp. 3718-3727)."},{"key":"e_1_3_2_1_8_1","first-page":"164","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 34","author":"Mai S.","unstructured":"Mai, S., Hu, H. and Xing, S., 2020, April. Modality to modality translation: An adversarial representation learning and graph fusion network for multimodal fusion. In Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 34, No. 01, pp. 164-172)."},{"volume-title":"2019 International Conference on Multimodal Interaction (pp. 595-601)","author":"Wang Y.","key":"e_1_3_2_1_9_1","unstructured":"Wang, Y., Wu, J. and Hoashi, K., 2019, October. Multi-attention fusion network for video-based emotion recognition. In 2019 International Conference on Multimodal Interaction (pp. 595-601)."},{"key":"e_1_3_2_1_10_1","volume-title":"Association for Computational Linguistics. Meeting.","volume":"2019","author":"Tsai","year":"2019","unstructured":"Tsai, Yao-Hung Hubert, \"Multimodal transformer for unaligned multimodal language sequences.\" Proceedings of the conference. Association for Computational Linguistics. Meeting. Vol. 2019. NIH Public Access, 2019."},{"volume-title":"2018 IEEE Spoken Language Technology Workshop (SLT) (pp. 112-118)","author":"Yoon S.","key":"e_1_3_2_1_11_1","unstructured":"Yoon, S., Byun, S. and Jung, K., 2018, December. Multimodal speech emotion recognition using audio and text. In 2018 IEEE Spoken Language Technology Workshop (SLT) (pp. 112-118). IEEE."},{"volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 4652-4661)","author":"Chudasama V.","key":"e_1_3_2_1_12_1","unstructured":"Chudasama, V., Kar, P., Gudmalwar, A., Shah, N., Wasnik, P. and Onoe, N., 2022. M2FNet: multi-modal fusion network for emotion recognition in conversation. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 4652-4661)."},{"volume-title":"2022 26th International Conference on Pattern Recognition (ICPR) (pp. 2822-2828)","author":"Chumachenko K.","key":"e_1_3_2_1_13_1","unstructured":"Chumachenko, K., Iosifidis, A. and Gabbouj, M., 2022, August. Self-attention fusion for audiovisual emotion recognition with incomplete data. In 2022 26th International Conference on Pattern Recognition (ICPR) (pp. 2822-2828). IEEE."},{"volume-title":"Seventh European conference on speech communication and technology.","author":"Nogueiras A.","key":"e_1_3_2_1_14_1","unstructured":"Nogueiras, A., Moreno, A., Bonafonte, A. and Mari\u00f1o, J.B., 2001. Speech emotion recognition using hidden Markov models. In Seventh European conference on speech communication and technology."},{"volume-title":"Ninth international conference on spoken language processing.","author":"Neiberg D.","key":"e_1_3_2_1_15_1","unstructured":"Neiberg, D., Elenius, K. and Laskowski, K., 2006. Emotion recognition in spontaneous speech using GMMs. In Ninth international conference on spoken language processing."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2076804"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Han K. Yu D. and Tashev I. 2014 September. Speech emotion recognition using deep neural network and extreme learning machine. In Interspeech 2014.","DOI":"10.21437\/Interspeech.2014-57"},{"volume-title":"2017 IEEE International conference on acoustics, speech and signal processing (ICASSP) (pp. 2227-2231)","author":"Mirsamadi S.","key":"e_1_3_2_1_18_1","unstructured":"Mirsamadi, S., Barsoum, E. and Zhang, C., 2017, March. Automatic speech emotion recognition using recurrent neural networks with local attention. In 2017 IEEE International conference on acoustics, speech and signal processing (ICASSP) (pp. 2227-2231). IEEE."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1208"},{"volume-title":"2017 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 5115-5119)","author":"Bertero D.","key":"e_1_3_2_1_20_1","unstructured":"Bertero, D. and Fung, P., 2017, March. A first look into a convolutional neural network for speech emotion detection. In 2017 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 5115-5119). IEEE."},{"key":"e_1_3_2_1_21_1","unstructured":"Al-agha Lecturer Salwa A. P. H. H. Saleh and P. R. F. Ghani. \"Geometric-based feature extraction and classification for emotion expressions of 3D video film.\" Journal of Advances in Information Technology Vol 8.2 (2017)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Zadeh A. Chen M. Poria S. Cambria E. and Morency L.P. 2017. Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250.","DOI":"10.18653\/v1\/D17-1115"},{"key":"e_1_3_2_1_23_1","volume-title":"A vector quantized masked autoencoder for speech emotion recognition. arXiv preprint arXiv:2304.11117","author":"Sadok S.","year":"2023","unstructured":"Sadok, S., Leglaive, S., & S\u00e9guier, R. (2023). A vector quantized masked autoencoder for speech emotion recognition. arXiv preprint arXiv:2304.11117."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"e_1_3_2_1_25_1","first-page":"303","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 34","author":"Zhao S.","unstructured":"Zhao, S., Ma, Y., Gu, Y., Yang, J., Xing, T., Xu, P., Hu, R., Chai, H. and Keutzer, K., 2020, April. An end-to-end visual-audio attention network for emotion recognition in user-generated videos. In Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 34, No. 01, pp. 303-311)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Xu H. Zhang H. Han K. Wang Y. Peng Y. and Li X. 2019. Learning alignment for multimodal emotion recognition from speech. arXiv preprint arXiv:1909.05645.","DOI":"10.21437\/Interspeech.2019-3247"},{"key":"e_1_3_2_1_27_1","first-page":"3510","volume-title":"Proceedings of the AAAI conference on artificial intelligence (Vol. 35","author":"Zhao Z.","unstructured":"Zhao, Z., Liu, Q. and Zhou, F., 2021, May. Robust lightweight facial expression recognition network with label distribution training. In Proceedings of the AAAI conference on artificial intelligence (Vol. 35, No. 4, pp. 3510-3519)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.3390\/s21227665"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.3390\/app12010327"},{"volume-title":"2020 IEEE International Conference on Image Processing (ICIP) (pp. 251-255)","author":"Ghaleb E.","key":"e_1_3_2_1_31_1","unstructured":"Ghaleb, E., Niehues, J. and Asteriadis, S., 2020, October. Multimodal attention-mechanism for temporal emotion recognition. In 2020 IEEE International Conference on Image Processing (ICIP) (pp. 251-255). IEEE."},{"key":"e_1_3_2_1_32_1","volume-title":"SpeechFormer: A hierarchical efficient framework incorporating the characteristics of speech. arXiv preprint arXiv:2203.03812","author":"Chen W.","year":"2022","unstructured":"Chen, W., Xing, X., Xu, X., Pang, J., & Du, L. (2022). SpeechFormer: A hierarchical efficient framework incorporating the characteristics of speech. arXiv preprint arXiv:2203.03812."},{"key":"e_1_3_2_1_33_1","volume-title":"International Conference on Internet of Things and Connected Technologies (pp. 134-146)","author":"Kanani C. S.","year":"2020","unstructured":"Kanani, C. S., Gill, K. S., Behera, S., Choubey, A., Gupta, R. K., & Misra, R. (2020, July). Shallow over deep neural networks: a empirical analysis for human emotion classification using audio data. In International Conference on Internet of Things and Connected Technologies (pp. 134-146). Cham: Springer International Publishing."}],"event":{"name":"VSIP 2023: 2023 the 5th International Conference on Video, Signal and Image Processing","acronym":"VSIP 2023","location":"Harbin China"},"container-title":["Proceedings of the 2023 5th International Conference on Video, Signal and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3638682.3638698","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3638682.3638698","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T02:00:07Z","timestamp":1755914407000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3638682.3638698"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,24]]},"references-count":33,"alternative-id":["10.1145\/3638682.3638698","10.1145\/3638682"],"URL":"https:\/\/doi.org\/10.1145\/3638682.3638698","relation":{},"subject":[],"published":{"date-parts":[[2023,11,24]]},"assertion":[{"value":"2024-05-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}