{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T00:19:41Z","timestamp":1783556381673,"version":"3.55.0"},"publisher-location":"Cham","reference-count":41,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032222633","type":"print"},{"value":"9783032222640","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-22264-0_26","type":"book-chapter","created":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T00:09:14Z","timestamp":1783555754000},"page":"326-337","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Dynamic Prompting and\u00a0Cross-Modal Attention for\u00a0Context-Aware Multimodal Emotion Recognition"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-7579-518X","authenticated-orcid":false,"given":"Guangzi","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3327-1535","authenticated-orcid":false,"given":"Yupeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8387-5289","authenticated-orcid":false,"given":"Shanshan","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2930-7937","authenticated-orcid":false,"given":"Haoyu","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5996-2728","authenticated-orcid":false,"given":"Xingquan","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,2]]},"reference":[{"key":"26_CR1","doi-asserted-by":"crossref","unstructured":"Abdul-Mageed, M., Ungar, L.: EmoNet: fine-grained emotion detection with gated recurrent neural networks. In: Proceedings of the ACL, vol. 1, pp. 718\u2013728 (2017)","DOI":"10.18653\/v1\/P17-1067"},{"key":"26_CR2","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer Normalization, arXiv preprint arXiv:1607.06450 (2016)"},{"key":"26_CR3","doi-asserted-by":"crossref","unstructured":"Bhattacharya, U., Mittal, T., Chandra, R., et al.: Step: spatial temporal graph convolutional networks for emotion perception from gaits. In: Proceedings of the AAAI, vol. 34 (2020)","DOI":"10.1609\/aaai.v34i02.5490"},{"issue":"4","key":"26_CR4","first-page":"2408","volume":"14","author":"Y Chen","year":"2023","unstructured":"Chen, Y., Wang, Z., Peng, H., et al.: Multimodal emotion recognition with temporal and contextual consistency. IEEE Trans. Affect. Comput. 14(4), 2408\u20132420 (2023)","journal-title":"IEEE Trans. Affect. Comput."},{"key":"26_CR5","doi-asserted-by":"crossref","unstructured":"Cohn, J.F., Kruez, T.S., Matthews, I., et al.: Detecting depression from facial actions and vocal prosody. In: Proceedings of the ACII, pp. 1\u20138 (2009)","DOI":"10.1109\/ACII.2009.5349358"},{"key":"26_CR6","doi-asserted-by":"crossref","unstructured":"Dai, W., Li, J., Li, D., et al.: InstructBLIP: towards General-purpose Vision-Language Models with Instruction Tuning. arXiv preprint arXiv:2305.06500 (2023)","DOI":"10.52202\/075280-2142"},{"key":"26_CR7","unstructured":"Devlin, J., Chang, M.W., Lee, K., et al.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the NAACL-HLT, vol. 1, pp. 4171\u20134186 (2019)"},{"key":"26_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106764","volume":"181","author":"C Fu","year":"2025","unstructured":"Fu, C., Qian, F., Su, K., et al.: HiMul-LGG: a hierarchical decision fusion-based local\u2013global graph neural network for multimodal emotion recognition in conversation. Neural Netw. 181, 106764 (2025)","journal-title":"Neural Netw."},{"key":"26_CR9","unstructured":"Gao, Q., Zeng, H., Li, G., Tong, T.: Graph reasoning-based emotion recognition network. IEEE Access 3"},{"key":"26_CR10","unstructured":"Hu, E.J., Shen, Y., Wallis, P., et al.: LoRA: low-rank adaptation of large language models. In: Proceedings of the ICLR (2022)"},{"key":"26_CR11","doi-asserted-by":"crossref","unstructured":"Jaiswal, S., Misra, S., Nandi, G.: Attention-guided context-aware emotional state recognition. In: Proceedings of the UPCON (2020)","DOI":"10.1109\/UPCON50219.2020.9376440"},{"key":"26_CR12","unstructured":"Jeong, M., Lee, H.: LaERC-S: improving LLM-based emotion recognition in conversation with speaker characteristics, arXiv preprint arXiv:2501.02164 (2025)"},{"key":"26_CR13","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. In: Proceedings of the ICLR (2015)"},{"key":"26_CR14","unstructured":"Kipf, T.N., Welling, M.: Semi-supervised classification with graph convolutional networks. In: Proceedings of the ICLR (2017)"},{"key":"26_CR15","doi-asserted-by":"crossref","unstructured":"Kosti, R., Alvarez, J.M., Recasens, A., et al.: EMOTIC: emotions in context dataset. In: Proceedings of the CVPRW, pp. 61\u201369 (2017)","DOI":"10.1109\/CVPRW.2017.285"},{"issue":"7","key":"26_CR16","doi-asserted-by":"crossref","first-page":"1367","DOI":"10.3390\/electronics13071367","volume":"13","author":"DH Lee","year":"2024","unstructured":"Lee, D.H., Lee, G.S.: Gated cross-modal fusion mechanism for audio-video-based emotion recognition. Electronics 13(7), 1367 (2024)","journal-title":"Electronics"},{"key":"26_CR17","doi-asserted-by":"crossref","unstructured":"Lee, J., Kim, S., Kim, S., et al.: Context-aware emotion recognition networks. In: Proceedings of the ICCV, pp. 9363\u20139372 (2019)","DOI":"10.1109\/ICCV.2019.01024"},{"key":"26_CR18","doi-asserted-by":"crossref","unstructured":"Levi, G., Hassner, T.: Emotion recognition in the wild via convolutional neural networks and mapped binary patterns. In: Proceedings of the ICMI Workshops, pp. 503\u2013510 (2015)","DOI":"10.1145\/2818346.2830587"},{"key":"26_CR19","unstructured":"Li, W., Dong, X., Wang, Y.: Human emotion recognition with relational region-level analysis. IEEE Trans. Affect Comput. (2021)"},{"key":"26_CR20","unstructured":"Lin, C.Y.: ROUGE: a package for automatic evaluation of summaries. In: Proceedings of the ACL Workshop on Text Summarization, pp. 74\u201381 (2004)"},{"key":"26_CR21","unstructured":"Liu, H., Li, C., Wu, Q., et al.: Visual instruction tuning (LLaVA), arXiv preprint arXiv:2304.08485 (2023)"},{"key":"26_CR22","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the ICCV, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"26_CR23","first-page":"1186","volume":"48","author":"D McDuff","year":"2016","unstructured":"McDuff, D., Picard, R.W.: A review of remote physiological measurement using imaging methods. Behav. Res. Methods 48, 1186\u20131206 (2016)","journal-title":"Behav. Res. Methods"},{"key":"26_CR24","doi-asserted-by":"crossref","unstructured":"Mittal, T., Guhan, P., Bhattacharya, U., et al.: EmotiCon: context-aware multimodal emotion recognition using Frege\u2019s principle. In: Proceedings of the CVPR, pp. 14234\u201314243 (2020)","DOI":"10.1109\/CVPR42600.2020.01424"},{"key":"26_CR25","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., et al.: BLEU: a method for automatic evaluation of machine translation. In: Proceedings of the ACL, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"26_CR26","unstructured":"Paszke, A., Gross, S., Massa, F., et al.: PyTorch: an imperative style, high-performance deep learning library. In: Proceedings of the NeurIPS, vol. 32, pp. 8024\u20138035 (2019)"},{"key":"26_CR27","unstructured":"Radford, A., Kim, J.W., Hallacy, C., et al.: Learning transferable visual models from natural language supervision. In: Proceedings of the ICML, PMLR 139, pp. 8748\u20138763 (2021)"},{"key":"26_CR28","unstructured":"Ren, S., He, K., Girshick, R., et al.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Proceedings of the NeurIPS, vol. 28, pp. 91\u201399 (2015)"},{"key":"26_CR29","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et al.: Attention is all you need. In: Proceedings of the NeurIPS, vol. 30, pp. 5998\u20136008 (2017)"},{"key":"26_CR30","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., et al.: Show and tell: a neural image caption generator. In: Proceedings of the CVPR, pp. 3156\u20133164 (2015)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"26_CR31","unstructured":"Wang, J., Chen, D., Luo, C., et al.: Chatvideo: a tracklet-centric multimodal and versatile video understanding system (2023)"},{"key":"26_CR32","unstructured":"Wang, Z., Wang, L., Wu, X., et al.: Multimodal emotion recognition with vision-language prompting and modality dropout. arXiv preprint arXiv:2409.07078 (2024)"},{"key":"26_CR33","doi-asserted-by":"crossref","unstructured":"Xenos, A., Stafylakis, T., Patras, I., Tzimiropoulos, G.: A simple baseline for knowledge-based visual question answering. In: Proceedings of the EMNLP, pp. 14871\u201314877 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.919"},{"key":"26_CR34","unstructured":"Yang, Q., Wang, W., Wang, L., et al.: EmoVIT: revolutionizing emotion insights with visual instruction tuning. In: Proceedings of the CVPR (2024)"},{"key":"26_CR35","doi-asserted-by":"crossref","unstructured":"Yang, D., Chen, Z., Wang, Y., et al.: Context de-confounded emotion recognition. In: Proceedings of the CVPR, pp. 19005\u201319015 (2023)","DOI":"10.1109\/CVPR52729.2023.01822"},{"key":"26_CR36","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2024.106912","volume":"100","author":"P Yu","year":"2025","unstructured":"Yu, P., He, X., Li, H., et al.: FMLAN: a novel framework for cross-subject and cross-session EEG emotion recognition. Biomed. Signal Process. Control 100, 106912 (2025)","journal-title":"Biomed. Signal Process. Control"},{"key":"26_CR37","doi-asserted-by":"crossref","unstructured":"Zhao, Z., Liu, Q., Zhou, F.: Robust lightweight facial expression recognition network with label distribution training. In: Proceedings of the AAAI, vol. 35, pp. 3510\u20133519 (2021)","DOI":"10.1609\/aaai.v35i4.16465"},{"key":"26_CR38","unstructured":"Zheng, L., Tian, J., She, J., et al.: VLLMs provide better context for emotion understanding through common sense reasoning, arXiv preprint arXiv:2404.11636 (2024)"},{"key":"26_CR39","unstructured":"Zhang, H., Jiang, Y., Liu, Z., et al.: Contrastive language-image learning with augmented textual prompts for 3D\/4D FER using vision-language model, arXiv preprint arXiv:2404.02537 (2024)"},{"key":"26_CR40","doi-asserted-by":"crossref","unstructured":"Zhang, M., Liang, Y., Ma, H.: Context-aware affective graph reasoning for emotion recognition. In: Proceedings of the ICME, pp. 151\u2013156 (2019)","DOI":"10.1109\/ICME.2019.00034"},{"key":"26_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, S., Pan, Y., Wang, J.Z.: Learning emotion representations from verbal and nonverbal communication. In: Proceedings of the CVPR, pp. 18993\u201319004 (2023)","DOI":"10.1109\/CVPR52729.2023.01821"}],"container-title":["Lecture Notes in Computer Science","Advances in Computer Graphics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-22264-0_26","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T00:09:17Z","timestamp":1783555757000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-22264-0_26"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032222633","9783032222640"],"references-count":41,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-22264-0_26","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CGI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Computer Graphics International Conference","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hong Kong","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 July 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 July 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"42","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cgi2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.cgs-network.org\/cgi25","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}