{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,24]],"date-time":"2026-02-24T06:43:20Z","timestamp":1771915400063,"version":"3.50.1"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,1,27]],"date-time":"2025-01-27T00:00:00Z","timestamp":1737936000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,27]],"date-time":"2025-01-27T00:00:00Z","timestamp":1737936000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Natural Science Foundation of Sichuan, China","award":["No. 2022NSFSC0571"],"award-info":[{"award-number":["No. 2022NSFSC0571"]}]},{"name":"the Sichuan Science and Technology Program","award":["No. 2018JY0273"],"award-info":[{"award-number":["No. 2018JY0273"]}]},{"name":"the China Scholarship Council","award":["No. 201908510026"],"award-info":[{"award-number":["No. 201908510026"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1007\/s00530-024-01618-z","type":"journal-article","created":{"date-parts":[[2025,1,27]],"date-time":"2025-01-27T03:03:04Z","timestamp":1737946984000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Hierarchical heterogeneous graph network based multimodal emotion recognition in conversation"],"prefix":"10.1007","volume":"31","author":[{"given":"Junyin","family":"Peng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hong","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenbin","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,27]]},"reference":[{"key":"1618_CR1","doi-asserted-by":"crossref","unstructured":"Qin, L., Che, W., Li, Y., Ni, M., Liu, T.: Dcr-net: A deep co-interactive relation network for joint dialog act recognition and sentiment classification. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 8665\u20138672 (2020)","DOI":"10.1609\/aaai.v34i05.6391"},{"key":"1618_CR2","doi-asserted-by":"crossref","unstructured":"Song, Z., Zheng, X., Liu, L., Xu, M., Huang, X.-J.: Generating responses with a specific emotion in dialog. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 3685\u20133695 (2019)","DOI":"10.18653\/v1\/P19-1359"},{"key":"1618_CR3","doi-asserted-by":"publisher","unstructured":"Bhavan, A., Chauhan, P., Hitkul, Shah, R.R.: Bagged support vector machines for emotion recognition from speech. Knowl. Based. Syst. 184, 104886 (2019) https:\/\/doi.org\/10.1016\/j.knosys.2019.104886","DOI":"10.1016\/j.knosys.2019.104886"},{"issue":"7","key":"1618_CR4","doi-asserted-by":"publisher","first-page":"4873","DOI":"10.1007\/s10462-021-10030-2","volume":"54","author":"K Cortis","year":"2021","unstructured":"Cortis, K., Davis, B.: Over a decade of social opinion mining: a systematic review. Artif. Intell. Rev. 54(7), 4873\u20134965 (2021). https:\/\/doi.org\/10.1007\/s10462-021-10030-2","journal-title":"Artif. Intell. Rev."},{"key":"1618_CR5","doi-asserted-by":"publisher","unstructured":"Zhou, H., Huang, M., Zhang, T., Zhu, X., Liu, B.: Emotional chatting machine: Emotional conversation generation with internal and external memory. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018). https:\/\/doi.org\/10.1609\/aaai.v32i1.11325","DOI":"10.1609\/aaai.v32i1.11325"},{"issue":"4","key":"1618_CR6","doi-asserted-by":"publisher","first-page":"2124","DOI":"10.1109\/TII.2018.2867174","volume":"15","author":"RL Rosa","year":"2018","unstructured":"Rosa, R.L., Schwartz, G.M., Ruggiero, W.V., Rodr\u00edguez, D.Z.: A knowledge-based recommendation system that includes sentiment analysis and deep learning. IEEE Trans. Industr. Inf. 15(4), 2124\u20132135 (2018). https:\/\/doi.org\/10.1109\/TII.2018.2867174","journal-title":"IEEE Trans. Industr. Inf."},{"key":"1618_CR7","doi-asserted-by":"publisher","unstructured":"Sordoni, A., Bengio, Y., Vahabi, H., Lioma, C., Grue\u00a0Simonsen, J., Nie, J.-Y.: A hierarchical recurrent encoder-decoder for generative context-aware query suggestion. In: Proceedings of the 24th ACM International on Conference on Information and Knowledge Management, pp. 553\u2013562 (2015). https:\/\/doi.org\/10.1145\/2806416.2806493","DOI":"10.1145\/2806416.2806493"},{"key":"1618_CR8","doi-asserted-by":"crossref","unstructured":"Zhong, P., Wang, D., Miao, C.: Knowledge-enriched transformer for emotion detection in textual conversations. arXiv preprint arXiv:1909.10681 (2019)","DOI":"10.18653\/v1\/D19-1016"},{"key":"1618_CR9","doi-asserted-by":"publisher","unstructured":"Majumder, N., Poria, S., Hazarika, D., Mihalcea, R., Gelbukh, A., Cambria, E.: Dialoguernn: An attentive rnn for emotion detection in conversations. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 6818\u20136825 (2019). https:\/\/doi.org\/10.1609\/aaai.v33i01.33016818","DOI":"10.1609\/aaai.v33i01.33016818"},{"key":"1618_CR10","doi-asserted-by":"crossref","unstructured":"Ghosal, D., Majumder, N., Poria, S., Chhaya, N., Gelbukh, A.: Dialoguegcn: A graph convolutional neural network for emotion recognition in conversation. arXiv preprint arXiv:1908.11540 (2019)","DOI":"10.18653\/v1\/D19-1015"},{"key":"1618_CR11","doi-asserted-by":"publisher","first-page":"102218","DOI":"10.1016\/j.inffus.2023.102218","volume":"105","author":"A Geetha","year":"2024","unstructured":"Geetha, A., Mala, T., Priyanka, D., Uma, E.: Multimodal emotion recognition with deep learning: advancements, challenges, and future directions. Inf. Fus. 105, 102218 (2024). https:\/\/doi.org\/10.1016\/j.inffus.2023.102218","journal-title":"Inf. Fus."},{"key":"1618_CR12","doi-asserted-by":"publisher","first-page":"744574","DOI":"10.3389\/fcomp.2022.744574","volume":"4","author":"A Axelsson","year":"2022","unstructured":"Axelsson, A., Buschmeier, H., Skantze, G.: Modeling feedback in interaction with conversational agents-a review. Front. Comput. Sci. 4, 744574 (2022)","journal-title":"Front. Comput. Sci."},{"key":"1618_CR13","doi-asserted-by":"publisher","first-page":"424","DOI":"10.1016\/j.inffus.2022.09.025","volume":"91","author":"A Gandhi","year":"2023","unstructured":"Gandhi, A., Adhvaryu, K., Poria, S., Cambria, E., Hussain, A.: Multimodal sentiment analysis: a systematic review of history, datasets, multimodal fusion methods, applications, challenges and future directions. Inf. Fus. 91, 424\u2013444 (2023). https:\/\/doi.org\/10.1016\/j.inffus.2022.09.025","journal-title":"Inf. Fus."},{"key":"1618_CR14","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2682899","volume":"47","author":"SK D\u2019mello","year":"2015","unstructured":"D\u2019mello, S.K., Kory, J.: A review and meta-analysis of multimodal affect detection systems. ACM Comput. Surv. (CSUR) 47, 1\u201336 (2015)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"1618_CR15","doi-asserted-by":"publisher","unstructured":"Poria, S., Cambria, E., Hazarika, D., Majumder, N., Zadeh, A., Morency, L.-P.: Context-dependent sentiment analysis in user-generated videos. In: Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (volume 1: Long Papers), pp. 873\u2013883 (2017). https:\/\/doi.org\/10.18653\/v1\/P17-1081","DOI":"10.18653\/v1\/P17-1081"},{"key":"1618_CR16","doi-asserted-by":"crossref","unstructured":"Hu, D., Wei, L., Huai, X.: Dialoguecrn: Contextual reasoning networks for emotion recognition in conversations. arXiv preprint arXiv:2106.01978 (2021)","DOI":"10.18653\/v1\/2021.acl-long.547"},{"key":"1618_CR17","doi-asserted-by":"publisher","unstructured":"Zadeh, A., Liang, P.P., Mazumder, N., Poria, S., Cambria, E., Morency, L.-P.: Memory fusion network for multi-view sequential learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018). https:\/\/doi.org\/10.1609\/aaai.v32i1.12021","DOI":"10.1609\/aaai.v32i1.12021"},{"key":"1618_CR18","doi-asserted-by":"publisher","unstructured":"Hazarika, D., Poria, S., Mihalcea, R., Cambria, E., Zimmermann, R.: Icon: Interactive conversational memory network for multimodal emotion detection. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 2594\u20132604 (2018). https:\/\/doi.org\/10.18653\/v1\/D18-1280","DOI":"10.18653\/v1\/D18-1280"},{"key":"1618_CR19","doi-asserted-by":"publisher","first-page":"57","DOI":"10.1016\/j.aiopen.2021.01.001","volume":"1","author":"J Zhou","year":"2020","unstructured":"Zhou, J., Cui, G., Hu, S., Zhang, Z., Yang, C., Liu, Z., Wang, L., Li, C., Sun, M.: Graph neural networks: a review of methods and applications. AI Open 1, 57\u201381 (2020)","journal-title":"AI Open"},{"key":"1618_CR20","doi-asserted-by":"crossref","unstructured":"Hu, J., Liu, Y., Zhao, J., Jin, Q.: Mmgcn: Multimodal fusion via deep graph convolution network for emotion recognition in conversation. arXiv preprint arXiv:2107.06779 (2021)","DOI":"10.18653\/v1\/2021.acl-long.440"},{"key":"1618_CR21","doi-asserted-by":"crossref","unstructured":"Zhang, D., Wu, L., Sun, C., Li, S., Zhu, Q., Zhou, G.: Modeling both context-and speaker-sensitive dependence for emotion detection in multi-speaker conversations. In: IJCAI, pp. 5415\u20135421 (2019)","DOI":"10.24963\/ijcai.2019\/752"},{"key":"1618_CR22","doi-asserted-by":"publisher","unstructured":"Joshi, A., Bhat, A., Jain, A., Singh, A., Modi, A.: Cogmen: Contextualized gnn based multimodal emotion recognition. In: Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 4148\u20134164 (2022). https:\/\/doi.org\/10.18653\/v1\/2022.naacl-main.306","DOI":"10.18653\/v1\/2022.naacl-main.306"},{"key":"1618_CR23","doi-asserted-by":"publisher","first-page":"126427","DOI":"10.1016\/j.neucom.2023.126427","volume":"550","author":"J Li","year":"2023","unstructured":"Li, J., Wang, X., Lv, G., Zeng, Z.: Graphmft: A graph network based multimodal fusion technique for emotion recognition in conversation. Neurocomputing 550, 126427 (2023). https:\/\/doi.org\/10.1016\/j.neucom.2023.126427","journal-title":"Neurocomputing"},{"key":"1618_CR24","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1109\/TMM.2023.3260635","volume":"26","author":"J Li","year":"2023","unstructured":"Li, J., Wang, X., Lv, G., Zeng, Z.: Graphcfc: A directed graph based cross-modal feature complementation approach for multimodal conversational emotion recognition. IEEE Trans. Multimed. 26, 77\u201389 (2023). https:\/\/doi.org\/10.1109\/TMM.2023.3260635","journal-title":"IEEE Trans. Multimed."},{"key":"1618_CR25","doi-asserted-by":"crossref","unstructured":"Fu, C., Qian, F., Su, K., Su, Y., Wang, Z., Shi, J., Liu, Z., Liu, C., Ishi, C.T.: Himul-lgg: A hierarchical decision fusion-based local-global graph neural network for multimodal emotion recognition in conversation. Neural Netw., 106764 (2024)","DOI":"10.1016\/j.neunet.2024.106764"},{"key":"1618_CR26","doi-asserted-by":"crossref","unstructured":"Schlichtkrull, M., Kipf, T.N., Bloem, P., Van Den\u00a0Berg, R., Titov, I., Welling, M.: Modeling relational data with graph convolutional networks. In: The Semantic Web: 15th International Conference, ESWC 2018, Heraklion, Crete, Greece, June 3\u20137, 2018, Proceedings 15, pp. 593\u2013607 (2018). Springer","DOI":"10.1007\/978-3-319-93417-4_38"},{"key":"1618_CR27","unstructured":"Yun, S., Jeong, M., Kim, R., Kang, J., Kim, H.J.: Graph transformer networks. Adv Neural Inf Process Syst32, (2019)"},{"key":"1618_CR28","unstructured":"Hamilton, W., Ying, Z., Leskovec, J.: Inductive representation learning on large graphs. Adv Neural Inf Process Syst. 30, (2017)"},{"key":"1618_CR29","unstructured":"Jiao, W., Yang, H., King, I., Lyu, M.R.: Higru: Hierarchical gated recurrent units for utterance-level emotion recognition. arXiv preprint arXiv:1904.04446 (2019)"},{"key":"1618_CR30","doi-asserted-by":"crossref","unstructured":"Ghosal, D., Majumder, N., Gelbukh, A., Mihalcea, R., Poria, S.: Cosmic: Commonsense knowledge for emotion identification in conversations. arXiv preprint arXiv:2010.02795 (2020)","DOI":"10.18653\/v1\/2020.findings-emnlp.224"},{"key":"1618_CR31","doi-asserted-by":"crossref","unstructured":"Hazarika, D., Poria, S., Zadeh, A., Cambria, E., Morency, L.-P., Zimmermann, R.: Conversational memory network for emotion recognition in dyadic dialogue videos. In: Proceedings of the Conference. Association for Computational Linguistics. North American Chapter. Meeting, vol. 2018, p. 2122 (2018). NIH Public Access","DOI":"10.18653\/v1\/N18-1193"},{"key":"1618_CR32","doi-asserted-by":"crossref","unstructured":"Hu, D., Hou, X., Wei, L., Jiang, L., Mo, Y.: Mm-dfn: Multimodal dynamic fusion network for emotion recognition in conversations. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7037\u20137041 (2022). IEEE","DOI":"10.1109\/ICASSP43922.2022.9747397"},{"key":"1618_CR33","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1016\/j.inffus.2022.10.009","volume":"91","author":"J Wen","year":"2023","unstructured":"Wen, J., Jiang, D., Tu, G., Liu, C., Cambria, E.: Dynamic interactive multiview memory network for emotion recognition in conversation. Inf. Fus. 91, 123\u2013133 (2023)","journal-title":"Inf. Fus."},{"issue":"8","key":"1618_CR34","doi-asserted-by":"publisher","first-page":"3727","DOI":"10.1109\/TKDE.2020.3033673","volume":"34","author":"J Yu","year":"2020","unstructured":"Yu, J., Yin, H., Li, J., Gao, M., Huang, Z., Cui, L.: Enhancing social recommendation with adversarial graph convolutional networks. IEEE Trans. Knowl. Data Eng. 34(8), 3727\u20133739 (2020). https:\/\/doi.org\/10.1109\/TKDE.2020.3033673","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"1618_CR35","doi-asserted-by":"publisher","first-page":"16386","DOI":"10.1109\/JIOT.2022.3151400","volume":"9","author":"B Wu","year":"2022","unstructured":"Wu, B., Zhong, L., Yao, L., Ye, Y.: Eagcn: an efficient adaptive graph convolutional network for item recommendation in social internet of things. IEEE Internet Things J. 9, 16386\u201316401 (2022). https:\/\/doi.org\/10.1109\/JIOT.2022.3151400","journal-title":"IEEE Internet Things J."},{"key":"1618_CR36","unstructured":"Veli\u010dkovi\u0107, P., Cucurull, G., Casanova, A., Romero, A., Lio, P., Bengio, Y.: Graph attention networks. arXiv preprint arXiv:1710.10903 (2017)"},{"key":"1618_CR37","doi-asserted-by":"crossref","unstructured":"Chen, F., Shao, J., Zhu, S., Shen, H.T.: Multivariate, multi-frequency and multimodal: Rethinking graph neural networks for emotion recognition in conversation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10761\u201310770 (2023)","DOI":"10.1109\/CVPR52729.2023.01036"},{"key":"1618_CR38","doi-asserted-by":"crossref","unstructured":"Eyben, F., W\u00f6llmer, M., Schuller, B.: Opensmile: the munich versatile and fast open-source audio feature extractor. In: Proceedings of the 18th ACM International Conference on Multimedia, pp. 1459\u20131462 (2010)","DOI":"10.1145\/1873951.1874246"},{"key":"1618_CR39","doi-asserted-by":"crossref","unstructured":"Reimers, N.: Sentence-bert: Sentence embeddings using siamese bert-networks. arXiv preprint arXiv:1908.10084 (2019)","DOI":"10.18653\/v1\/D19-1410"},{"key":"1618_CR40","doi-asserted-by":"crossref","unstructured":"Baltrusaitis, T., Zadeh, A., Lim, Y.C., Morency, L.P.: Openface 2.0: Facial behavior analysis toolkit. IEEE Computer Society, 59\u201366 (2018)","DOI":"10.1109\/FG.2018.00019"},{"key":"1618_CR41","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1007\/s10579-008-9076-6","volume":"42","author":"C Busso","year":"2008","unstructured":"Busso, C., Bulut, M., Lee, C.-C., Kazemzadeh, A., Mower, E., Kim, S., Chang, J.N., Lee, S., Narayanan, S.S.: Iemocap: interactive emotional dyadic motion capture database. Lang. Resour. Eval. 42, 335\u2013359 (2008)","journal-title":"Lang. Resour. Eval."},{"key":"1618_CR42","doi-asserted-by":"crossref","unstructured":"Poria, S., Hazarika, D., Majumder, N., Naik, G., Cambria, E., Mihalcea, R.: Meld: A multimodal multi-party dataset for emotion recognition in conversations. arXiv preprint arXiv:1810.02508 (2018)","DOI":"10.18653\/v1\/P19-1050"},{"key":"1618_CR43","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chen, M., Poria, S., Cambria, E., Morency, L.-P.: Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250 (2017)","DOI":"10.18653\/v1\/D17-1115"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01618-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01618-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01618-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,21]],"date-time":"2025-04-21T19:34:02Z","timestamp":1745264042000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01618-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,27]]},"references-count":43,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["1618"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01618-z","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,27]]},"assertion":[{"value":"30 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 December 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 January 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"81"}}