{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T09:43:24Z","timestamp":1776937404576,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,21]]},"DOI":"10.1145\/3797552.3797664","type":"proceedings-article","created":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T08:35:40Z","timestamp":1776933340000},"page":"690-698","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["A Multi-modal dialogue emotion recognition framework based on D-GBA"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-6202-6090","authenticated-orcid":false,"given":"Haoqiang","family":"Liu","sequence":"first","affiliation":[{"name":"College of Computer Science, Zhengzhou University of Aeronautics, Zhengzhou, Henan, China,"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1340-8027","authenticated-orcid":false,"given":"Hongmei","family":"Wang","sequence":"additional","affiliation":[{"name":"College of Computer Science, Zhengzhou University of Aeronautics, Zhengzhou, Henan, China,"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9016-4571","authenticated-orcid":false,"given":"Yuanhao","family":"Ma","sequence":"additional","affiliation":[{"name":"College of Computer Science, Zhengzhou University of Aeronautics, Zhengzhou, Henan, China,"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4488-5378","authenticated-orcid":false,"given":"Jianhui","family":"Chen","sequence":"additional","affiliation":[{"name":"College of Computer Science, Zhengzhou University of Aeronautics, Zhengzhou, Henan, China,"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8049-207X","authenticated-orcid":false,"given":"Xingyu","family":"Liu","sequence":"additional","affiliation":[{"name":"College of Computer Science, Zhengzhou University of Aeronautics, Zhengzhou, Henan, China,"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5399-1168","authenticated-orcid":false,"given":"Jiayi","family":"Wang","sequence":"additional","affiliation":[{"name":"College of Computer Science, Zhengzhou University of Aeronautics, Zhengzhou, Henan, China,"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,4,23]]},"reference":[{"key":"e_1_3_3_1_1_2","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","volume":"32","author":"Zadeh A.","year":"2018","unstructured":"Zadeh, A., Liang, P.P., Poria, S., Vij, P., Cambria, E., Morency, L.-P.: Multiattention recurrent network for human communication comprehension. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)"},{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1145\/3380688.3380693"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.09.041"},{"key":"e_1_3_3_1_4_2","first-page":"8763","volume-title":"International Conference on Machine Learning","author":"Radford A.","year":"2021","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021). PmLR"},{"key":"e_1_3_3_1_5_2","volume-title":"Hoi","author":"Li J.","year":"2021","unstructured":"Li, J., Selvaraju, R., Gotmare, A., Joty, S., Xiong, C., Hoi, S.C.H.: Align before fuse: Vision and language representation learning with momentum distillation. Advances in neural information processing systems 34, 9694\u20139705 (2021)"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-023-14885-1"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Wang H. Yeung D.-Y.: A survey on bayesian deep learning. ACM computing surveys (csur) 53(5) 1\u201337 (2020)","DOI":"10.1145\/3409383"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1656"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2023.3259010"},{"key":"e_1_3_3_1_10_2","volume-title":"Zhou","author":"Feng X.","year":"2024","unstructured":"Feng, X., Lin, Y., He, L., Li, Y., Chang, L., Zhou, Y.: Knowledge-guided dynamic modality attention fusion framework for multimodal sentiment analysis. arXiv preprint arXiv:2410.04491 (2024)"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2023.3265653"},{"key":"e_1_3_3_1_12_2","volume-title":"Seals","author":"Albladi A.","year":"2025","unstructured":"Albladi, A., Islam, M., Seals, C.: Sentiment analysis of twitter data using nlp models: A comprehensive review. IEEE Access (2025)"},{"key":"e_1_3_3_1_13_2","first-page":"4186","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Devlin J.","year":"2019","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (long and Short Papers), pp. 4171\u20134186 (2019)"},{"key":"e_1_3_3_1_14_2","volume-title":"Stoyanov","author":"Liu Y.","year":"1907","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., Stoyanov, V.: Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)"},{"key":"e_1_3_3_1_15_2","unstructured":"Baevski A. Zhou Y. Mohamed A. Auli M.: wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems 33 12449\u201312460 (2020)"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Gulati A. Qin J. Chiu C.-C. Parmar N. Zhang Y. Yu J. Han W. Wang S. Zhang Z. Wu Y. et al.: Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"e_1_3_3_1_17_2","unstructured":"Dosovitskiy A. Beyer L. Kolesnikov A. Weissenborn D. Zhai X. Unterthiner T. Dehghani M. Minderer M. Heigold G. Gelly S. et al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"e_1_3_3_1_18_2","first-page":"6114","volume-title":"International Conference on Machine Learning","author":"Tan M.","year":"2019","unstructured":"Tan, M., Le, Q.: Efficientnet: Rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, pp. 6105\u20136114 (2019). PMLR"},{"key":"e_1_3_3_1_19_2","volume-title":"Mihalcea","author":"Poria S.","year":"2020","unstructured":"Poria, S., Hazarika, D., Majumder, N., Mihalcea, R.: Beneath the tip of the iceberg: Current challenges and new directions in sentiment analysis research. IEEE transactions on affective computing 14(1), 108\u2013132 (2020)"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.5555\/AAI28864910"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.06.072"},{"key":"e_1_3_3_1_22_2","volume-title":"Okumura","author":"Li D.","year":"2023","unstructured":"Li, D., Wang, Y., Funakoshi, K., Okumura, M.: Joyful: Joint modality fusion and graph contrastive learning for multimodal emotion recognition. arXiv preprint 22arXiv:2311.11009 (2023)"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3700748"},{"key":"e_1_3_3_1_24_2","article-title":"Eegbased multimodal emotion recognition: A machine learning perspective","author":"Liu H.","year":"2024","unstructured":"Liu, H., Lou, T., Zhang, Y., Wu, Y., Xiao, Y., Jensen, C.S., Zhang, D.: Eegbased multimodal emotion recognition: A machine learning perspective. IEEE Transactions on Instrumentation and Measurement (2024)","journal-title":"IEEE Transactions on Instrumentation and Measurement ("},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1007\/s42452-021-04897-7"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1874246"},{"key":"e_1_3_3_1_27_2","volume-title":"Bansal","author":"Tan H.","year":"1908","unstructured":"Tan, H., Bansal, M.: Lxmert: Learning cross-modality encoder representationsfrom transformers. arXiv preprint arXiv:1908.07490(2019)"},{"issue":"4","key":"e_1_3_3_1_28_2","first-page":"111","article-title":"Performance analysis of various activation functionsin generalized mlp architectures of neural networks","volume":"1","author":"Karlik B.","year":"2011","unstructured":"Karlik, B., Olgac, A.V.: Performance analysis of various activation functionsin generalized mlp architectures of neural networks. International Journal ofArtifcial Intelligence and Expert Systems 1(4),111-122(2011)","journal-title":"International Journal ofArtifcial Intelligence and Expert Systems"},{"key":"e_1_3_3_1_29_2","volume-title":"Yang","author":"Liu W.","year":"2016","unstructured":"Liu, W., Wen, Y., Yu, 2., Yang, M.: Large-margin softmax loss for convolutionalneural networks. arXiv preprint arXiv:1612.02295 (2016)"},{"key":"e_1_3_3_1_30_2","volume-title":"Mihalcea","author":"Poria","year":"1810","unstructured":"Poria, $., Hazarika, D., Majumder, N., Naik, G., Cambria, E., Mihalcea, R.: Meld:A multimodal multi-party dataset for emotion recognition in conversations. arXivpreprint arXiv:1810.02508(2018)"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","unstructured":"Zadeh A. Chen M. Poria S. Cambria E. Morency L.-P.: Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250 (2017)","DOI":"10.18653\/v1\/D17-1115"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"crossref","unstructured":"Liu Z. Shen Y. Lakshminarasimhan V.B. Liang P.P. Zadeh A. Morency L.-P.: Efficient low-rank multimodal fusion with modality-specific factors. arXiv preprint arXiv:1806.00064 (2018)","DOI":"10.18653\/v1\/P18-1209"},{"key":"e_1_3_3_1_33_2","volume-title":"Gelbukh","author":"Ghosal D.","year":"1908","unstructured":"Ghosal, D., Majumder, N., Poria, S., Chhaya, N., Gelbukh, A.: Dialoguegcn: A graph convolutional neural network for emotion recognition in conversation. arXiv preprint arXiv:1908.11540 (2019)"},{"key":"e_1_3_3_1_34_2","volume-title":"Huai","author":"Hu D.","year":"1978","unstructured":"Hu, D., Wei, L., Huai, X.: Dialoguecrn: Contextual reasoning networks for emotion recognition in conversations. arXiv preprint arXiv:2106.01978 (2021)"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747397"},{"key":"e_1_3_3_1_36_2","volume-title":"Li","author":"Hu G.","year":"2022","unstructured":"Hu, G., Lin, T.-E., Zhao, Y., Lu, G., Wu, Y., Li, Y.: Unimse: Towards unified multimodal sentiment analysis and emotion recognition. arXiv preprint arXiv:2211.11256 (2022)"},{"key":"e_1_3_3_1_37_2","volume-title":"Efros","author":"Shocher A.","year":"2023","unstructured":"Shocher, A., Dravid, A., Gandelsman, Y., Mosseri, I., Rubinstein, M., Efros, A.A.: Idempotent generative network. arXiv preprint arXiv:2311.01462 (2023)"},{"key":"e_1_3_3_1_38_2","volume-title":"Application Research of Computers","author":"Shen Xudong","year":"2024","unstructured":"Shen Xudong, Huang Xianying, Zou Shihao. A Multimodal Dialogue Emotion Recognition Model Based on Temporal - Aware DAG[J]. Application Research of Computers, 2024, 41(1)"}],"event":{"name":"ICAIE 2025: 2025 4th International Conference on Artificial Intelligence and Education","location":"Nanjing China","acronym":"ICAIE 2025"},"container-title":["Proceedings of the 2025 4th International Conference on Artificial Intelligence and Education"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3797552.3797664","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T08:50:44Z","timestamp":1776934244000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797552.3797664"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,21]]},"references-count":38,"alternative-id":["10.1145\/3797552.3797664","10.1145\/3797552"],"URL":"https:\/\/doi.org\/10.1145\/3797552.3797664","relation":{},"subject":[],"published":{"date-parts":[[2025,11,21]]},"assertion":[{"value":"2026-04-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}