{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:36:24Z","timestamp":1784219784763,"version":"3.55.0"},"reference-count":53,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100015749","name":"Communication University of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100015749","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013804","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013804","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62301510"],"award-info":[{"award-number":["62301510"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113886","type":"journal-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T06:48:21Z","timestamp":1777445301000},"page":"113886","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":2,"special_numbering":"PD","title":["Visual Label Augmentation-Driven Multimodal Emotion Recognition"],"prefix":"10.1016","volume":"179","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2710-0410","authenticated-orcid":false,"given":"Qinglan","family":"Wei","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yaqi","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junzhe","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Long","family":"Ye","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113886_b1","first-page":"14581","article-title":"CARAT: Contrastive feature reconstruction and aggregation for multi-modal multi-label emotion recognition","volume":"vol. 38","author":"Peng","year":"2024"},{"key":"10.1016\/j.patcog.2026.113886_b2","doi-asserted-by":"crossref","unstructured":"A.B. Zadeh, P.P. Liang, S. Poria, E. Cambria, L.P. Morency, Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph, in: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2018, pp. 2236\u20132246.","DOI":"10.18653\/v1\/P18-1208"},{"issue":"1","key":"10.1016\/j.patcog.2026.113886_b3","doi-asserted-by":"crossref","first-page":"10","DOI":"10.1109\/TBC.2022.3215245","article-title":"FV2ES: A fully End2End multimodal system for fast yet effective video emotion recognition inference","volume":"69","author":"Wei","year":"2023","journal-title":"IEEE Trans. Broadcast."},{"key":"10.1016\/j.patcog.2026.113886_b4","first-page":"1","article-title":"MERBench: A unified evaluation benchmark for multimodal emotion recognition","author":"Lian","year":"2026","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113886_b5","doi-asserted-by":"crossref","first-page":"94557","DOI":"10.1109\/ACCESS.2021.3092735","article-title":"Multimodal emotion recognition fusion analysis adapting BERT with heterogeneous feature unification","volume":"9","author":"Lee","year":"2021","journal-title":"IEEE Access"},{"key":"10.1016\/j.patcog.2026.113886_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.109259","article-title":"TETFN: A text enhanced transformer fusion network for multimodal sentiment analysis","volume":"136","author":"Wang","year":"2023","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113886_b7","doi-asserted-by":"crossref","unstructured":"Q. Li, Y. Gao, Y. Li, Mining High-Quality Samples from Raw Data and Majority Voting Method for Multimodal Emotion Recognition, in: Proceedings of the 31st ACM International Conference on Multimedia, 2023, pp. 9546\u20139550.","DOI":"10.1145\/3581783.3612862"},{"key":"10.1016\/j.patcog.2026.113886_b8","doi-asserted-by":"crossref","unstructured":"Z. Cheng, Y. Lin, Z. Chen, X. Li, S. Mao, F. Zhang, D. Ding, B. Zhang, X. Peng, Semi-Supervised Multimodal Emotion Recognition with Expression MAE, in: Proceedings of the 31st ACM International Conference on Multimedia, 2023, pp. 9436\u20139440.","DOI":"10.1145\/3581783.3612840"},{"key":"10.1016\/j.patcog.2026.113886_b9","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111340","article-title":"FrameERC: Framelet transform based multimodal graph neural networks for emotion recognition in conversation","volume":"161","author":"Li","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113886_b10","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112644","article-title":"Language-dominated fusion and self-distillation for multimodal sentiment analysis with incomplete modalities","volume":"172","author":"Shi","year":"2026","journal-title":"Pattern Recognit."},{"issue":"1","key":"10.1016\/j.patcog.2026.113886_b11","doi-asserted-by":"crossref","first-page":"34","DOI":"10.1111\/j.1467-9280.1992.tb00253.x","article-title":"Facial expressions of emotion: New findings, new questions","volume":"3","author":"Ekman","year":"1992","journal-title":"Psychol. Sci."},{"key":"10.1016\/j.patcog.2026.113886_b12","series-title":"Qwen2.5-VL technical report","author":"Bai","year":"2025"},{"key":"10.1016\/j.patcog.2026.113886_b13","series-title":"Janus-pro: Unified multimodal understanding and generation with data and model scaling","author":"Chen","year":"2025"},{"key":"10.1016\/j.patcog.2026.113886_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.110117","article-title":"MSA-GCN: Multiscale adaptive graph convolution network for gait emotion recognition","volume":"147","author":"Yin","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113886_b15","doi-asserted-by":"crossref","unstructured":"W. Zheng, J. Yu, R. Xia, A Unimodal Valence-Arousal Driven Contrastive Learning Framework for Multimodal Multi-Label Emotion Recognition, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 622\u2013631.","DOI":"10.1145\/3664647.3681638"},{"key":"10.1016\/j.patcog.2026.113886_b16","first-page":"9100","article-title":"Tailor versatile multi-modal learning for multi-label emotion recognition","volume":"vol. 36","author":"Zhang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113886_b17","doi-asserted-by":"crossref","unstructured":"Z. Wu, Z. Gong, L. Ai, P. Shi, K. Donbekci, J. Hirschberg, Beyond silent letters: Amplifying llms in emotion recognition with vocal nuances, in: Findings of the Association for Computational Linguistics: NAACL 2025, 2025, pp. 2202\u20132218.","DOI":"10.18653\/v1\/2025.findings-naacl.117"},{"key":"10.1016\/j.patcog.2026.113886_b18","article-title":"EmoRAct: A neuro-symbolic framework coupling acoustic tokens with prosody semantics for emotion recognition","author":"Yang","year":"2026","journal-title":"Pattern Recognit."},{"issue":"4","key":"10.1016\/j.patcog.2026.113886_b19","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2024.103724","article-title":"An empirical study of multimodal entity-based sentiment analysis with ChatGPT: Improving in-context learning via entity-aware contrastive learning","volume":"61","author":"Yang","year":"2024","journal-title":"Inf. Process. Manage."},{"key":"10.1016\/j.patcog.2026.113886_b20","doi-asserted-by":"crossref","unstructured":"A. Shvets, Emo Pillars: Knowledge Distillation to Support Fine-Grained Context-Aware and Context-Less Emotion Classification, in: Findings of the Association for Computational Linguistics: ACL 2025, 2025, pp. 174\u2013191.","DOI":"10.18653\/v1\/2025.findings-acl.10"},{"key":"10.1016\/j.patcog.2026.113886_b21","doi-asserted-by":"crossref","unstructured":"Y. Chen, J. Joo, Understanding and Mitigating Annotation Bias in Facial Expression Recognition, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, ICCV, 2021, pp. 14980\u201314991.","DOI":"10.1109\/ICCV48922.2021.01471"},{"key":"10.1016\/j.patcog.2026.113886_b22","doi-asserted-by":"crossref","unstructured":"D. Zeng, Z. Lin, X. Yan, Face2Exp: Combating Data Biases for Facial Expression Recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2022, pp. 20291\u201320300.","DOI":"10.1109\/CVPR52688.2022.01965"},{"key":"10.1016\/j.patcog.2026.113886_b23","doi-asserted-by":"crossref","unstructured":"S. Anand, N.K. Devulapally, S.D. Bhattacharjee, J. Yuan, Multi-Label Emotion Analysis in Conversation via Multimodal Knowledge Distillation, in: Proceedings of the 31st ACM International Conference on Multimedia, 2023, pp. 6090\u20136100.","DOI":"10.1145\/3581783.3612517"},{"key":"10.1016\/j.patcog.2026.113886_b24","series-title":"Deepseek-v3 technical report","author":"Liu","year":"2024"},{"key":"10.1016\/j.patcog.2026.113886_b25","series-title":"Humanomni: A large vision-speech language model for human-centric video understanding","author":"Zhao","year":"2025"},{"key":"10.1016\/j.patcog.2026.113886_b26","series-title":"Explainable multimodal emotion recognition","author":"Lian","year":"2023"},{"key":"10.1016\/j.patcog.2026.113886_b27","doi-asserted-by":"crossref","unstructured":"P. Jin, R. Takanobu, W. Zhang, Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 13700\u201313710.","DOI":"10.1109\/CVPR52733.2024.01300"},{"key":"10.1016\/j.patcog.2026.113886_b28","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume":"vol. 202","author":"Li","year":"2023"},{"key":"10.1016\/j.patcog.2026.113886_b29","doi-asserted-by":"crossref","unstructured":"Y. Fang, W. Wang, B. Xie, Q. Sun, L. Wu, X. Wang, T. Huang, X. Wang, Y. Cao, EVA: Exploring the Limits of Masked Visual Representation Learning at Scale, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2023, pp. 19358\u201319369.","DOI":"10.1109\/CVPR52729.2023.01855"},{"issue":"70","key":"10.1016\/j.patcog.2026.113886_b30","first-page":"1","article-title":"Scaling instruction-finetuned language models","volume":"25","author":"Chung","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.patcog.2026.113886_b31","series-title":"Emotion-LLaMAv2 and MMEVerse: A new framework and benchmark for multimodal emotion understanding","author":"Peng","year":"2026"},{"issue":"10","key":"10.1016\/j.patcog.2026.113886_b32","doi-asserted-by":"crossref","first-page":"1499","DOI":"10.1109\/LSP.2016.2603342","article-title":"Joint face detection and alignment using multitask cascaded convolutional networks","volume":"23","author":"Zhang","year":"2016","journal-title":"IEEE Signal Process. Lett."},{"key":"10.1016\/j.patcog.2026.113886_b33","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"10.1016\/j.patcog.2026.113886_b34","series-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"key":"10.1016\/j.patcog.2026.113886_b35","article-title":"Attention is all you need","volume":"vol. 30","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.patcog.2026.113886_b36","article-title":"Generalized cross entropy loss for training deep neural networks with noisy labels","volume":"vol. 31","author":"Zhang","year":"2018"},{"key":"10.1016\/j.patcog.2026.113886_b37","doi-asserted-by":"crossref","unstructured":"K. Ma, X. Wang, X. Yang, M. Zhang, J.M. Girard, L.P. Morency, ElderReact: A Multimodal Dataset for Recognizing Emotional Response in Aging Adults, in: 2019 International Conference on Multimodal Interaction, 2019, pp. 349\u2013357.","DOI":"10.1145\/3340555.3353747"},{"key":"10.1016\/j.patcog.2026.113886_b38","doi-asserted-by":"crossref","unstructured":"B. Nojavanasghari, T. Baltru\u0161aitis, C.E. Hughes, EmoReact: A Multimodal Approach and Dataset for Recognizing Emotional Responses in Children, in: Proceedings of the 18th ACM International Conference on Multimodal Interaction, 2016, pp. 137\u2013144.","DOI":"10.1145\/2993148.2993168"},{"key":"10.1016\/j.patcog.2026.113886_b39","doi-asserted-by":"crossref","unstructured":"Y.-H.H. Tsai, S. Bai, P.P. Liang, Multimodal Transformer for Unaligned Multimodal Language Sequences, in: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, 2019, pp. 6558\u20136569.","DOI":"10.18653\/v1\/P19-1656"},{"key":"10.1016\/j.patcog.2026.113886_b40","doi-asserted-by":"crossref","unstructured":"D. Zhang, X. Ju, J. Li, Multi-Modal Multi-Label Emotion Detection with Modality and Label Dependence, in: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing, EMNLP, 2020, pp. 3584\u20133593.","DOI":"10.18653\/v1\/2020.emnlp-main.291"},{"key":"10.1016\/j.patcog.2026.113886_b41","first-page":"14338","article-title":"Multi-modal multi-label emotion recognition with heterogeneous hierarchical message passing","volume":"vol. 35","author":"Zhang","year":"2021"},{"key":"10.1016\/j.patcog.2026.113886_b42","doi-asserted-by":"crossref","unstructured":"S. Ge, Z. Jiang, Z. Cheng, C. Wang, Y. Yin, Q. Gu, Learning Robust Multi-Modal Representation for Multi-Label Emotion Recognition via Adversarial Masking and Perturbation, in: Proceedings of the ACM Web Conference 2023, 2023, pp. 1510\u20131518.","DOI":"10.1145\/3543507.3583258"},{"key":"10.1016\/j.patcog.2026.113886_b43","doi-asserted-by":"crossref","unstructured":"J. Huang, J. Zhong, Q. Lei, J. Gao, Y. Yang, S. Wang, P. Li, K. Wei, Latent Distribution Decouple for Uncertain-Aware Multimodal Multi-Label Emotion Recognition, in: Findings of the Association for Computational Linguistics: ACL 2025, 2025, pp. 24123\u201324138.","DOI":"10.18653\/v1\/2025.findings-acl.1238"},{"key":"10.1016\/j.patcog.2026.113886_b44","doi-asserted-by":"crossref","first-page":"192","DOI":"10.1016\/j.patrec.2025.02.024","article-title":"Multi-corpus emotion recognition method based on cross-modal gated attention fusion","volume":"190","author":"Ryumina","year":"2025","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.patcog.2026.113886_b45","article-title":"LLaVA-OneVision: Easy visual task transfer","author":"Li","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.patcog.2026.113886_b46","series-title":"Internvl3: Exploring advanced training and test-time recipes for open-source multimodal models","author":"Zhu","year":"2025"},{"key":"10.1016\/j.patcog.2026.113886_b47","series-title":"Proceedings of the 38th International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume":"vol. 139","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113886_b48","article-title":"DINOv2: Learning robust visual features without supervision","author":"Oquab","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.patcog.2026.113886_b49","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105171","article-title":"EVA-02: A visual representation for neon genesis","volume":"149","author":"Fang","year":"2024","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.patcog.2026.113886_b50","doi-asserted-by":"crossref","unstructured":"D. Hazarika, R. Zimmermann, S. Poria, MISA: Modality-Invariant and -Specific Representations for Multimodal Sentiment Analysis, in: Proceedings of the 28th ACM International Conference on Multimedia, 2020, pp. 1122\u20131131.","DOI":"10.1145\/3394171.3413678"},{"key":"10.1016\/j.patcog.2026.113886_b51","doi-asserted-by":"crossref","first-page":"103932","DOI":"10.1109\/ACCESS.2022.3210183","article-title":"Across the universe: Biasing facial representations toward non-universal emotions with the face-STN","volume":"10","author":"Barros","year":"2022","journal-title":"IEEE Access"},{"key":"10.1016\/j.patcog.2026.113886_b52","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.108348","article-title":"Token-disentangling mutual transformer for multimodal emotion recognition","volume":"133","author":"Yin","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.patcog.2026.113886_b53","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.110744","article-title":"Multi-level feature decomposition and fusion model for video-based multimodal emotion recognition","volume":"152","author":"Liu","year":"2025","journal-title":"Eng. Appl. Artif. Intell."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326008514?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326008514?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T05:24:49Z","timestamp":1783056289000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326008514"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":53,"alternative-id":["S0031320326008514"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113886","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Visual Label Augmentation-Driven Multimodal Emotion Recognition","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113886","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113886"}}