{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T23:53:58Z","timestamp":1781740438669,"version":"3.54.5"},"reference-count":56,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.eswa.2026.132475","type":"journal-article","created":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T06:17:26Z","timestamp":1776233846000},"page":"132475","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Multimodal progressive fusion and enhancement network for multimodal sentiment analysis"],"prefix":"10.1016","volume":"323","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0663-199X","authenticated-orcid":false,"given":"Jinghui","family":"Qin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0268-1183","authenticated-orcid":false,"given":"Zhihao","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-0590-2590","authenticated-orcid":false,"given":"Qite","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-6952-4891","authenticated-orcid":false,"given":"Lihuang","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8336-5109","authenticated-orcid":false,"given":"Zhijing","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132475_bib0001","doi-asserted-by":"crossref","first-page":"23716","DOI":"10.52202\/068431-1723","article-title":"Flamingo: A visual language model for few-shot learning","volume":"35","author":"Alayrac","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132475_bib0002","series-title":"Icml","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"vol. 2","author":"Bertasius","year":"2021"},{"issue":"1","key":"10.1016\/j.eswa.2026.132475_bib0003","doi-asserted-by":"crossref","first-page":"2126","DOI":"10.1038\/s41598-025-85859-6","article-title":"Multimodal sentiment analysis based on multi-layer feature fusion and multi-task learning","volume":"15","author":"Cai","year":"2025","journal-title":"Scientific Reports"},{"key":"10.1016\/j.eswa.2026.132475_bib0004","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.125533","article-title":"Integrated sentiment analysis with BERT for enhanced hybrid recommendation systems","volume":"261","author":"Darraz","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132475_bib0005","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2023.101847","article-title":"Emotion recognition from unimodal to multimodal analysis: A review","volume":"99","author":"Ezzameli","year":"2023","journal-title":"Information Fusion"},{"issue":"1","key":"10.1016\/j.eswa.2026.132475_bib0006","doi-asserted-by":"crossref","first-page":"207","DOI":"10.1109\/TAFFC.2024.3423671","article-title":"Multi-level contrastive learning: Hierarchical alleviation of heterogeneity in multimodal sentiment analysis","volume":"16","author":"Fan","year":"2024","journal-title":"IEEE Transactions on Affective Computing"},{"key":"10.1016\/j.eswa.2026.132475_bib0007","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"14314","article-title":"Emoe: Modality-specific enhanced dynamic emotion experts","author":"Fang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132475_bib0008","doi-asserted-by":"crossref","unstructured":"Feng, X., Lin, Y., He, L., Li, Y., Chang, L., & Zhou, Y. (2024). Knowledge-guided dynamic modality attention fusion framework for multimodal sentiment analysis. arXiv preprint arXiv: 2410.04491.","DOI":"10.18653\/v1\/2024.findings-emnlp.865"},{"key":"10.1016\/j.eswa.2026.132475_bib0009","unstructured":"Gu, A., & Dao, T. (2023). Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv: 2312.00752."},{"key":"10.1016\/j.eswa.2026.132475_bib0010","first-page":"35971","article-title":"On the parameterization and initialization of diagonal state space models","volume":"35","author":"Gu","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132475_bib0011","unstructured":"Gu, A., Goel, K., & R\u00e9, C. (2021a). Efficiently modeling long sequences with structured state spaces. arXiv preprint arXiv: 2111.00396."},{"key":"10.1016\/j.eswa.2026.132475_bib0012","first-page":"572","article-title":"Combining recurrent, convolutional, and continuous-time models with linear state space layers","volume":"34","author":"Gu","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132475_bib0013","series-title":"Proceedings of the 28th ACM international conference on multimedia","first-page":"1122","article-title":"Misa: Modality-invariant and-specific representations for multimodal sentiment analysis","author":"Hazarika","year":"2020"},{"key":"10.1016\/j.eswa.2026.132475_bib0014","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"1309","article-title":"Msamba: Exploring multimodal sentiment analysis with state space models","volume":"39","author":"He","year":"2025"},{"key":"10.1016\/j.eswa.2026.132475_bib0015","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112220","article-title":"Tchfn: Multimodal sentiment analysis based on text-centric hierarchical fusion network","volume":"300","author":"Hou","year":"2024","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.132475_bib0016","doi-asserted-by":"crossref","unstructured":"Hu, G., Lin, T.-E., Zhao, Y., Lu, G., Wu, Y., & Li, Y. (2022). Unimse: Towards unified multimodal sentiment analysis and emotion recognition. arXiv preprint arXiv: 2211.11256.","DOI":"10.18653\/v1\/2022.emnlp-main.534"},{"key":"10.1016\/j.eswa.2026.132475_bib0017","unstructured":"Hu, G., Xin, Y., Lyu, W., Huang, H., Sun, C., Zhu, Z., Gui, L., Cai, R., Cambria, E., & Seifi, H. (2024). Recent trends of multimodal affective computing: A survey from NLP perspective. arXiv preprint arXiv: 2409.07388."},{"key":"10.1016\/j.eswa.2026.132475_bib0018","series-title":"Icassp 2020-2020 ieee international conference on acoustics, speech and signal processing (icassp)","first-page":"3507","article-title":"Multimodal transformer fusion for continuous emotion recognition","author":"Huang","year":"2020"},{"key":"10.1016\/j.eswa.2026.132475_bib0019","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"5923","article-title":"Revisiting disentanglement and fusion on modality and context in conversational multimodal emotion recognition","author":"Li","year":"2023"},{"key":"10.1016\/j.eswa.2026.132475_bib0020","doi-asserted-by":"crossref","first-page":"59808","DOI":"10.52202\/079017-1910","article-title":"Coupled mamba: Enhanced multimodal fusion with coupled state space model","volume":"37","author":"Li","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132475_bib0021","article-title":"A multi-scale representation and multi-level decision learning network for multimodal sentiment analysis","volume":"297","author":"Li","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132475_bib0022","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"6631","article-title":"Decoupled multimodal distilling for emotion recognition","author":"Li","year":"2023"},{"key":"10.1016\/j.eswa.2026.132475_bib0023","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"24774","article-title":"Alignmamba: Enhancing multimodal mamba with local and global cross-modal alignment","author":"Li","year":"2025"},{"key":"10.1016\/j.eswa.2026.132475_bib0024","series-title":"Proceedings of the 31st international conference on computational linguistics","first-page":"2834","article-title":"t-HNE: A text-guided hierarchical noise eliminator for multimodal sentiment analysis","author":"Li","year":"2025"},{"key":"10.1016\/j.eswa.2026.132475_bib0025","unstructured":"Li, Z., Pan, H., Zhang, K., Wang, Y., & Yu, F. (2024b). Mambadfuse: A mamba-based dual-phase model for multi-modality image fusion. arXiv preprint arXiv: 2404.08406."},{"key":"10.1016\/j.eswa.2026.132475_bib0026","first-page":"34892","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132475_bib0027","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., & Stoyanov, V. (2019). Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv: 1907.11692."},{"key":"10.1016\/j.eswa.2026.132475_bib0028","series-title":"Proceedings of the 56th annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"2247","article-title":"Efficient low-rank multimodal fusion with modality-specific factors","author":"Liu","year":"2018"},{"key":"10.1016\/j.eswa.2026.132475_bib0029","unstructured":"Loshchilov, I., & Hutter, F. (2017). Decoupled weight decay regularization. arXiv preprint arXiv: 1711.05101."},{"key":"10.1016\/j.eswa.2026.132475_bib0030","series-title":"Proceedings of the 33rd ACM international conference on multimedia","first-page":"1997","article-title":"Towards explainable fusion and balanced learning in multimodal sentiment analysis","author":"Luo","year":"2025"},{"key":"10.1016\/j.eswa.2026.132475_bib0031","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"2554","article-title":"Progressive modality reinforcement for human multimodal emotion recognition from unaligned multimodal sequences","author":"Lv","year":"2021"},{"issue":"3","key":"10.1016\/j.eswa.2026.132475_bib0032","doi-asserted-by":"crossref","first-page":"2074","DOI":"10.1109\/TAFFC.2025.3553149","article-title":"Injecting multimodal information into pre-trained language model for multimodal sentiment analysis","volume":"16","author":"Mai","year":"2025","journal-title":"IEEE Transactions on Affective Computing"},{"issue":"3","key":"10.1016\/j.eswa.2026.132475_bib0033","doi-asserted-by":"crossref","first-page":"2276","DOI":"10.1109\/TAFFC.2022.3172360","article-title":"Hybrid contrastive learning of tri-modal representation for multimodal sentiment analysis","volume":"14","author":"Mai","year":"2022","journal-title":"IEEE Transactions on Affective Computing"},{"key":"10.1016\/j.eswa.2026.132475_bib0034","series-title":"International conference on machine learning","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2023"},{"key":"10.1016\/j.eswa.2026.132475_bib0035","series-title":"Proceedings of the conference. association for computational linguistics. meeting","first-page":"2359","article-title":"Integrating multimodal information in large pretrained transformers","volume":"vol. 2020","author":"Rahman","year":"2020"},{"issue":"9","key":"10.1016\/j.eswa.2026.132475_bib0036","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3652149","article-title":"A survey of cutting-edge multimodal sentiment analysis","volume":"56","author":"Singh","year":"2024","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/j.eswa.2026.132475_bib0037","series-title":"Proceedings of the conference. association for computational linguistics. meeting","first-page":"6558","article-title":"Multimodal transformer for unaligned multimodal language sequences","volume":"vol. 2019","author":"Tsai","year":"2019"},{"key":"10.1016\/j.eswa.2026.132475_bib0038","unstructured":"Tsai, Y.-H. H., Liang, P. P., Zadeh, A., Morency, L.-P., & Salakhutdinov, R. (2018). Learning factorized multimodal representations. arXiv preprint arXiv: 1806.06176."},{"key":"10.1016\/j.eswa.2026.132475_bib0039","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"7","key":"10.1016\/j.eswa.2026.132475_bib0040","doi-asserted-by":"crossref","first-page":"12643","DOI":"10.1109\/TNNLS.2024.3446030","article-title":"Modality perception learning-based determinative factor discovery for multimodal fake news detection","volume":"36","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"issue":"7","key":"10.1016\/j.eswa.2026.132475_bib0041","doi-asserted-by":"crossref","first-page":"5731","DOI":"10.1007\/s10462-022-10144-1","article-title":"A survey on sentiment analysis methods, applications, and challenges","volume":"55","author":"Wankhade","year":"2022","journal-title":"Artificial Intelligence Review"},{"key":"10.1016\/j.eswa.2026.132475_bib0042","series-title":"Grand challenge and workshop on human multimodal language","first-page":"11","article-title":"Recognizing emotions in video using multimodal DNN feature fusion","author":"Williams","year":"2018"},{"key":"10.1016\/j.eswa.2026.132475_bib0043","unstructured":"Wu, S., Dai, D., Qin, Z., Liu, T., Lin, B., Cao, Y., & Sui, Z. (2023a). Denoising bottleneck with mutual information maximization for video multimodal fusion. arXiv preprint arXiv: 2305.14652."},{"key":"10.1016\/j.eswa.2026.132475_bib0044","unstructured":"Wu, Z., Gong, Z., Koo, J., & Hirschberg, J. (2023b). Multimodal multi-loss fusion network for sentiment analysis. arXiv preprint arXiv: 2308.00264."},{"key":"10.1016\/j.eswa.2026.132475_bib0045","unstructured":"Wu, Z., Zhang, Q., Miao, D., Yi, K., Fan, W., & Hu, L. (2024). HydiscGAN: A hybrid distributed cGAN for audio-visual privacy preservation in multimodal sentiment analysis. arXiv preprint arXiv: 2404.11938."},{"key":"10.1016\/j.eswa.2026.132475_bib0046","article-title":"Tri-modal grouping fusion network with optimal transport learning for unaligned multimodal sentiment analysis","volume":"300","author":"Xiong","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132475_bib0047","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"1642","article-title":"Disentangled representation learning for multimodal emotion recognition","author":"Yang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132475_bib0048","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"7617","article-title":"Confede: Contrastive feature decomposition for multimodal sentiment analysis","author":"Yang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132475_bib0049","series-title":"Findings of the association for computational linguistics: NAACL 2024","first-page":"2099","article-title":"Clgsi: a multimodal sentiment analysis framework based on contrastive learning guided by sentiment intensity","author":"Yang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132475_bib0050","unstructured":"Yang, Z., Li, L., Lin, K., Wang, J., Lin, C.-C., Liu, Z., & Wang, L. (2023b). The dawn of lmms: Preliminary explorations with gpt-4v (ision). arXiv preprint arXiv: 2309.17421."},{"key":"10.1016\/j.eswa.2026.132475_bib0051","series-title":"Proceedings of the 58th annual meeting of the association for computational linguistics","first-page":"3718","article-title":"Ch-sims: A chinese multimodal sentiment analysis dataset with fine-grained annotation of modality","author":"Yu","year":"2020"},{"key":"10.1016\/j.eswa.2026.132475_bib0052","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"10790","article-title":"Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis","volume":"vol. 35","author":"Yu","year":"2021"},{"key":"10.1016\/j.eswa.2026.132475_bib0053","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chen, M., Poria, S., Cambria, E., & Morency, L.-P. (2017). Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv: 1707.07250.","DOI":"10.18653\/v1\/D17-1115"},{"key":"10.1016\/j.eswa.2026.132475_bib0054","unstructured":"Zadeh, A., Zellers, R., Pincus, E., & Morency, L.-P. (2016). Mosi: Multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos. arXiv preprint arXiv: 1606.06259."},{"key":"10.1016\/j.eswa.2026.132475_bib0055","series-title":"Proceedings of the 56th annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"2236","article-title":"Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph","author":"Zadeh","year":"2018"},{"key":"10.1016\/j.eswa.2026.132475_bib0056","series-title":"Proceedings of the 2023 conference on empirical methods in natural language processing","first-page":"756","article-title":"Learning language-guided adaptive hyper-modality representation for multimodal sentiment analysis","author":"Zhang","year":"2023"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426013886?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426013886?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T23:31:13Z","timestamp":1781739073000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426013886"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":56,"alternative-id":["S0957417426013886"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132475","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Multimodal progressive fusion and enhancement network for multimodal sentiment analysis","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132475","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132475"}}