{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T15:14:04Z","timestamp":1783523644136,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":44,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819224791","type":"print"},{"value":"9789819224807","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T00:00:00Z","timestamp":1783555200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T00:00:00Z","timestamp":1783555200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-2480-7_22","type":"book-chapter","created":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T14:20:52Z","timestamp":1783520452000},"page":"353-368","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Real-Time Multimodal Emotion Recognition in\u00a0Conversations with\u00a0MLLMs"],"prefix":"10.1007","author":[{"given":"Xiaolin","family":"Zhao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junyi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhedong","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qian","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,9]]},"reference":[{"key":"22_CR1","unstructured":"Baevski, A., Zhou, H., Mohamed, A., Auli, M.: wav2vec 2.0: A framework for self-supervised learning of speech representations. In: Proceedings of the 34th International Conference on Neural Information Processing Systems (2020)"},{"key":"22_CR2","doi-asserted-by":"crossref","unstructured":"Chen, F., Sun, Z., Ouyang, D., Liu, X., Shao, J.: Learning what and when to drop: adaptive multimodal and contextual dynamics for emotion recognition in conversation. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 1064\u20131073 (2021)","DOI":"10.1145\/3474085.3475661"},{"key":"22_CR3","doi-asserted-by":"crossref","unstructured":"Chen, J., Liu, Q., Sun, J., Zhang, Y.: Truelens: video fake news detection with dual level evidence gathering and consolidation. In: Proceedings of the ACM Web Conference 2026 (WWW \u201926), pp. 1\u201311 (2026)","DOI":"10.1145\/3774904.3792362"},{"key":"22_CR4","doi-asserted-by":"crossref","unstructured":"Chen, J., Wu, M., Liu, Q., Sun, J., Ding, Y., Zhang, Y.: Equal truth: rumor detection with invariant group fairness. In: Findings of the Association for Computational Linguistics: EMNLP 2025, pp. 10994\u201311007 (2025)","DOI":"10.18653\/v1\/2025.findings-emnlp.584"},{"key":"22_CR5","doi-asserted-by":"crossref","unstructured":"Chen, J., Wu, M., Liu, Q., Zhang, Y.: Explainable prediction of knowledge recombination: a synergized method with heterogeneous hypergraph learning and large language models. Inf. Process. Manag. 63(1) (2025)","DOI":"10.1016\/j.ipm.2025.104336"},{"key":"22_CR6","doi-asserted-by":"crossref","unstructured":"Cheng, Z., et al.: Emotion-llama: multimodal emotion recognition and reasoning with instruction tuning. In: Proceedings of the 38th International Conference on Neural Information Processing Systems (2024)","DOI":"10.52202\/079017-3518"},{"key":"22_CR7","doi-asserted-by":"crossref","unstructured":"Cho, K., et al.: Learning phrase representations using RNN encoder\u2013decoder for statistical machine translation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1724\u20131734 (2014)","DOI":"10.3115\/v1\/D14-1179"},{"key":"22_CR8","doi-asserted-by":"crossref","unstructured":"Chudasama, V., Kar, P., Gudmalwar, A., Shah, N., Wasnik, P., Onoe, N.: M2fnet: multi-modal fusion network for emotion recognition in conversation (2022)","DOI":"10.1109\/CVPRW56347.2022.00511"},{"key":"22_CR9","doi-asserted-by":"crossref","unstructured":"Fei, H., Zhang, H., Wang, B., Liao, L., Liu, Q., Cambria, E.: Empathyear: an open-source avatar multimodal empathetic chatbot. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 3: System Demonstrations), pp. 61\u201371. ACL (2024)","DOI":"10.18653\/v1\/2024.acl-demos.7"},{"key":"22_CR10","doi-asserted-by":"crossref","unstructured":"Ghosal, D., Majumder, N., Gelbukh, A., Mihalcea, R., Poria, S.: COSMIC: COmmonSense knowledge for eMotion identification in conversations. In: Findings of the Association for Computational Linguistics: EMNLP 2020, pp. 2470\u20132481 (2020)","DOI":"10.18653\/v1\/2020.findings-emnlp.224"},{"key":"22_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"issue":"8","key":"22_CR12","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"22_CR13","doi-asserted-by":"crossref","unstructured":"Hsu, W.N., Bolte, B., Tsai, Y.H.H., Lakhotia, K., Salakhutdinov, R., Mohamed, A.: Hubert: self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 3451\u20133460 (2021)","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"22_CR14","doi-asserted-by":"crossref","unstructured":"Hu, D., Hou, X., Wei, L., Jiang, L., Mo, Y.: MM-DFN: multimodal dynamic fusion network for emotion recognition in conversations. In: ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7037\u20137041 (2022)","DOI":"10.1109\/ICASSP43922.2022.9747397"},{"key":"22_CR15","doi-asserted-by":"crossref","unstructured":"Hu, J., Liu, Y., Zhao, J., Jin, Q.: MMGCN: multimodal fusion via deep graph convolution network for emotion recognition in conversation. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp. 5666\u20135675 (2021)","DOI":"10.18653\/v1\/2021.acl-long.440"},{"key":"22_CR16","unstructured":"Jiao, W., Yang, H., King, I., Lyu, M.R.: HiGRU: hierarchical gated recurrent units for utterance-level emotion recognition. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 397\u2013406 (2019)"},{"key":"22_CR17","unstructured":"Ju, H., Zhang, H., Zheng, Z.: Anomalylmm: bridging generative knowledge and discriminative retrieval for text-based person anomaly search. arXiv preprint arXiv:2509.04376 (2025)"},{"key":"22_CR18","doi-asserted-by":"crossref","unstructured":"Ju, X., Zhang, D., Zhu, S., Li, J., Li, S., Zhou, G.: Real-time emotion pre-recognition in conversations with contrastive multi-modal dialogue pre-training. In: Proceedings of the 32nd ACM International Conference on Information and Knowledge Management, pp. 1045\u20131055 (2023)","DOI":"10.1145\/3583780.3615024"},{"key":"22_CR19","unstructured":"Kim, T., Vossen, P.: EmoBERTa: speaker-aware emotion recognition in conversation with RoBERTa (2021)"},{"key":"22_CR20","unstructured":"Lei, S., Dong, G., Wang, X., Wang, K., Qiao, R., Wang, S.: Instructerc: reforming emotion recognition in conversation with multi-task retrieval-augmented large language models (2024)"},{"key":"22_CR21","doi-asserted-by":"crossref","unstructured":"Li, B., et al.: Revisiting disentanglement and fusion on modality and context in conversational multimodal emotion recognition. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 5923\u20135934 (2023)","DOI":"10.1145\/3581783.3612053"},{"key":"22_CR22","doi-asserted-by":"crossref","unstructured":"Liu, Q., et al.: Legal knowledge infusion for large language models: a survey. Inf. Fusion (2026)","DOI":"10.1016\/j.inffus.2025.103426"},{"key":"22_CR23","doi-asserted-by":"crossref","unstructured":"Liu, Z., Shen, Y., Lakshminarasimhan, V.B., Liang, P.P., Bagher\u00a0Zadeh, A., Morency, L.P.: Efficient low-rank multimodal fusion with modality-specific factors. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2247\u20132256 (2018)","DOI":"10.18653\/v1\/P18-1209"},{"key":"22_CR24","doi-asserted-by":"crossref","unstructured":"Lou, F., Wang, Q., Li, H., Liu, Q.: Long-context mental health assessment with large language models via knowledge compression. IEEE Trans. Affect. Comput. 1\u201314 (2026)","DOI":"10.1109\/TAFFC.2026.3669164"},{"key":"22_CR25","doi-asserted-by":"crossref","unstructured":"Lu, W., Hu, Z., Lin, J., Wang, L.: Lecm: a model leveraging emotion cause to improve real-time emotion recognition in conversations. Know.-Based Syst. 309 (2025)","DOI":"10.1016\/j.knosys.2024.112900"},{"key":"22_CR26","doi-asserted-by":"crossref","unstructured":"Luo, M., et al.: Panosent: a panoptic sextuple extraction benchmark for multimodal conversational aspect-based sentiment analysis. In: Proceedings of the 32nd ACM International Conference on Multimedia, MM, pp. 7667\u20137676 (2024)","DOI":"10.1145\/3664647.3680705"},{"key":"22_CR27","doi-asserted-by":"publisher","first-page":"776","DOI":"10.1109\/TMM.2023.3271019","volume":"26","author":"H Ma","year":"2024","unstructured":"Ma, H., Wang, J., Lin, H., Zhang, B., Zhang, Y., Xu, B.: A transformer-based model with self-distillation for multimodal emotion recognition in conversations. IEEE Trans. Multimedia 26, 776\u2013788 (2024)","journal-title":"IEEE Trans. Multimedia"},{"key":"22_CR28","doi-asserted-by":"crossref","unstructured":"Majumder, N., Poria, S., Hazarika, D., Mihalcea, R., Gelbukh, A., Cambria, E.: DialogueRNN: an attentive RNN for emotion detection in conversations. In: Proceedings of the Thirty-Third AAAI Conference on Artificial Intelligence and Thirty-First Innovative Applications of Artificial Intelligence Conference and Ninth AAAI Symposium on Educational Advances in Artificial Intelligence (2019)","DOI":"10.1609\/aaai.v33i01.33016818"},{"issue":"3","key":"22_CR29","doi-asserted-by":"publisher","first-page":"1743","DOI":"10.1109\/TAFFC.2022.3204972","volume":"14","author":"R Mao","year":"2023","unstructured":"Mao, R., Liu, Q., He, K., Li, W., Cambria, E.: The biases of pre-trained language models: an empirical study on prompt-based sentiment analysis and emotion detection. IEEE Trans. Affect. Comput. 14(3), 1743\u20131753 (2023)","journal-title":"IEEE Trans. Affect. Comput."},{"key":"22_CR30","doi-asserted-by":"crossref","unstructured":"Poria, S., Cambria, E., Hazarika, D., Majumder, N., Zadeh, A., Morency, L.P.: Context-dependent sentiment analysis in user-generated videos. In: Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 873\u2013883 (2017)","DOI":"10.18653\/v1\/P17-1081"},{"key":"22_CR31","doi-asserted-by":"crossref","unstructured":"Poria, S., Hazarika, D., Majumder, N., Naik, G., Cambria, E., Mihalcea, R.: MELD: a multimodal multi-party dataset for emotion recognition in conversations. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics (ACL), pp. 527\u2013536 (2019)","DOI":"10.18653\/v1\/P19-1050"},{"key":"22_CR32","doi-asserted-by":"publisher","first-page":"100943","DOI":"10.1109\/ACCESS.2019.2929050","volume":"7","author":"S Poria","year":"2019","unstructured":"Poria, S., Majumder, N., Mihalcea, R., Hovy, E.: Emotion recognition in conversation: research challenges, datasets, and recent advances. IEEE Access 7, 100943\u2013100953 (2019)","journal-title":"IEEE Access"},{"key":"22_CR33","doi-asserted-by":"crossref","unstructured":"Rasendrasoa, S., Pauchet, A., Saunier, J., Adam, S.: Real-time multimodal emotion recognition in conversation for multi-party interactions. In: Proceedings of the 2022 International Conference on Multimodal Interaction, pp. 395\u2013403 (2022)","DOI":"10.1145\/3536221.3556601"},{"key":"22_CR34","doi-asserted-by":"crossref","unstructured":"Shen, W., Chen, J., Quan, X., Xie, Z.: Dialogxl: all-in-one xlnet for multi-party conversation emotion recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 13789\u201313797 (2021)","DOI":"10.1609\/aaai.v35i15.17625"},{"key":"22_CR35","doi-asserted-by":"crossref","unstructured":"Shen, W., Wu, S., Yang, Y., Quan, X.: Directed acyclic graph network for conversational emotion recognition. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp. 1551\u20131560 (2021)","DOI":"10.18653\/v1\/2021.acl-long.123"},{"key":"22_CR36","unstructured":"Touvron, H., et al.: Llama: open and efficient foundation language models (2023)"},{"key":"22_CR37","doi-asserted-by":"crossref","unstructured":"Tsai, Y.H.H., Bai, S., Liang, P.P., Kolter, J.Z., Morency, L.P., Salakhutdinov, R.: Multimodal transformer for unaligned multimodal language sequences. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 6558\u20136569 (2019)","DOI":"10.18653\/v1\/P19-1656"},{"key":"22_CR38","doi-asserted-by":"crossref","unstructured":"Xu, H., et al.: When words smile: generating diverse emotional facial expressions from text. In: Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, EMNLP, pp. 27028\u201327046 (2025)","DOI":"10.18653\/v1\/2025.emnlp-main.1374"},{"key":"22_CR39","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chen, M., Poria, S., Cambria, E., Morency, L.P.: Tensor fusion network for multimodal sentiment analysis. In: Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing, pp. 1103\u20131114 (2017)","DOI":"10.18653\/v1\/D17-1115"},{"key":"22_CR40","unstructured":"Zhang, R., Zhang, H., Zheng, Z.: VL-uncertainty: detecting hallucination in large vision-language model via uncertainty estimation (2024)"},{"key":"22_CR41","unstructured":"Zhang, R., Zhou, D., Zheng, Z.: Sketchthinker-r1: towards efficient sketch-style reasoning in large multimodal models. In: International Conference on Learning Representations (ICLR) (2026)"},{"key":"22_CR42","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2025.107901","volume":"192","author":"Y Zhang","year":"2025","unstructured":"Zhang, Y., et al.: DialogueLLM: context and emotion knowledge-tuned large language models for emotion recognition in conversations. Neural Netw. 192, 107901 (2025)","journal-title":"Neural Netw."},{"key":"22_CR43","unstructured":"Zhang, Y., Wang, Y., Wu, Y., Wu, L., Zhu, L., Zheng, Z.: The coherence trap: when MLLM-crafted narratives exploit manipulated visual contexts. In: CVPR (2026)"},{"key":"22_CR44","doi-asserted-by":"crossref","unstructured":"Zhong, P., Wang, D., Miao, C.: Knowledge-enriched transformer for emotion detection in textual conversations. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP), pp. 165\u2013176 (2019)","DOI":"10.18653\/v1\/D19-1016"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Engineering for Decision Making"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-2480-7_22","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T14:20:58Z","timestamp":1783520458000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-2480-7_22"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,9]]},"ISBN":["9789819224791","9789819224807"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-2480-7_22","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,9]]},"assertion":[{"value":"9 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"FLINS-ISKE","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Systems and Knowledge Engineering","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Sydney","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iske2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2026.flins.cc","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}