{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T17:18:36Z","timestamp":1783531116608,"version":"3.55.0"},"reference-count":55,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62501644"],"award-info":[{"award-number":["62501644"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100009110","name":"Natural Science Foundation of Xinjiang Uygur Autonomous Region","doi-asserted-by":"publisher","award":["2024D01B99"],"award-info":[{"award-number":["2024D01B99"]}],"id":[{"id":"10.13039\/100009110","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.eswa.2026.133145","type":"journal-article","created":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T07:30:25Z","timestamp":1781076625000},"page":"133145","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PB","title":["Mining the potential of LVLMs in multimodal sentiment detection through cross-modal pre-alignment"],"prefix":"10.1016","volume":"331","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9310-4703","authenticated-orcid":false,"given":"Yiwei","family":"Wei","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-1793-409X","authenticated-orcid":false,"given":"Zhengliang","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haitao","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chengyin","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingjing","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Longbiao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133145_bib0001","unstructured":"Bubeck, S., Chandrasekaran, V., Eldan, R., Gehrke, J., Horvitz, E., Kamar, E., Lee, P., Lee, Y. T., Li, Y., Lundberg, S. et al. (2023). Sparks of artificial general intelligence: Early experiments with GPT-4. arXiv preprint arXiv: 2303.12712."},{"key":"10.1016\/j.eswa.2026.133145_bib0002","series-title":"Proceedings of the 57th Annual meeting of the association for computational linguistics","first-page":"2506","article-title":"Multi-modal sarcasm detection in twitter with hierarchical fusion model","author":"Cai","year":"2019"},{"key":"10.1016\/j.eswa.2026.133145_bib0003","series-title":"Proceedings of the 57th Annual meeting of the association for computational linguistics","first-page":"5193","article-title":"Improving textual network embedding with global attention via optimal transport","author":"Chen","year":"2019"},{"key":"10.1016\/j.eswa.2026.133145_bib0004","series-title":"Convolutional neural network for sentence classification","author":"Chen","year":"2015"},{"key":"10.1016\/j.eswa.2026.133145_bib0005","series-title":"Proceedings of the 2024 Conference on empirical methods in natural language processing","first-page":"3536","article-title":"D2R: Dual-branch dynamic routing network for multimodal sentiment detection","author":"Chen","year":"2024"},{"key":"10.1016\/j.eswa.2026.133145_bib0006","unstructured":"Chiang, W.-L., Li, Z., Lin, Z., Sheng, Y., Wu, Z., Zhang, H., Zheng, L., Zhuang, S., Zhuang, Y., Gonzalez, J. E. et al. (2023). Vicuna: An open-source chatbot impressing GPT-4 with 90%* ChatGPT quality. See https:\/\/vicuna. lmsys. org (accessed 14 April 2023), 2 (3), 6.https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/."},{"key":"10.1016\/j.eswa.2026.133145_bib0007","series-title":"Advances in neural information processing systems (neurIPS)","first-page":"2292","article-title":"Sinkhorn distances: Lightspeed computation of optimal transport","volume":"vol. 26","author":"Cuturi","year":"2013"},{"key":"10.1016\/j.eswa.2026.133145_bib0008","first-page":"2292","article-title":"Lightspeed computation of optimal transport","volume":"26","author":"Distances","year":"2013","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133145_bib0009","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv: 2010.11929."},{"key":"10.1016\/j.eswa.2026.133145_bib0010","series-title":"Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition","first-page":"303","article-title":"OTA: Optimal transport assignment for object detection","author":"Ge","year":"2021"},{"key":"10.1016\/j.eswa.2026.133145_bib0011","series-title":"Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition","first-page":"15180","article-title":"ImageBind: One embedding space to bind them all","author":"Girdhar","year":"2023"},{"key":"10.1016\/j.eswa.2026.133145_bib0012","series-title":"The 22nd International conference on artificial intelligence and statistics","first-page":"1880","article-title":"Unsupervised alignment of embeddings with wasserstein procrustes","author":"Grave","year":"2019"},{"key":"10.1016\/j.eswa.2026.133145_bib0013","series-title":"Proceedings of the IEEE Conference on computer vision and pattern recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.eswa.2026.133145_bib0014","unstructured":"Hu, E. J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2021). LoRA: Low-rank adaptation of large language models. arXiv preprint arXiv: 2106.09685."},{"key":"10.1016\/j.eswa.2026.133145_bib0015","series-title":"Research anthology on implementing sentiment analysis across multiple disciplines","first-page":"1846","article-title":"Multimodal sentiment analysis: A survey and comparison","author":"Kaur","year":"2022"},{"key":"10.1016\/j.eswa.2026.133145_bib0016","series-title":"Proceedings of NAACL-HLT","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Kenton","year":"2019"},{"key":"10.1016\/j.eswa.2026.133145_bib0017","series-title":"ICASSP 2020-2020 IEEE International conference on acoustics, speech and signal processing (ICASSP)","first-page":"4477","article-title":"Gated mechanism for attention based multi modal sentiment analysis","author":"Kumar","year":"2020"},{"key":"10.1016\/j.eswa.2026.133145_bib0018","series-title":"Findings of the association for computational linguistics: NAACL 2022","first-page":"2282","article-title":"CLMLF: A contrastive learning and multi-layer fusion method for multimodal sentiment detection","author":"Li","year":"2022"},{"key":"10.1016\/j.eswa.2026.133145_bib0019","series-title":"Proceedings of the 60th Annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"1767","article-title":"Multi-modal sarcasm detection via cross-modal graph convolutional network","author":"Liang","year":"2022"},{"key":"10.1016\/j.eswa.2026.133145_bib0020","series-title":"Advances in neural information processing systems","article-title":"Visual instruction tuning","volume":"vol. 36","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133145_bib0021","doi-asserted-by":"crossref","unstructured":"Liu, H., Wang, W., & Li, H. (2022). Towards multi-modal sarcasm detection via hierarchical congruity modeling with knowledge enhancement. arXiv preprint arXiv: 2210.03501.","DOI":"10.18653\/v1\/2022.emnlp-main.333"},{"key":"10.1016\/j.eswa.2026.133145_bib0022","doi-asserted-by":"crossref","unstructured":"Liu, X., Li, P., Huang, H., Li, Z., Cui, X., Liang, J., Qin, L., Deng, W., & He, Z. (2024b). FakeNewsGPT4: Advancing multimodal fake news detection through knowledge-augmented LVLMs. arXiv preprint arXiv: 2403.01988.","DOI":"10.1145\/3664647.3681089"},{"key":"10.1016\/j.eswa.2026.133145_bib0023","series-title":"Proceedings of the 56th Annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"1990","article-title":"Visual attention model for name tagging in multimodal social media","author":"Lu","year":"2018"},{"issue":"4","key":"10.1016\/j.eswa.2026.133145_bib0024","doi-asserted-by":"crossref","first-page":"1093","DOI":"10.1016\/j.asej.2014.04.011","article-title":"Sentiment analysis algorithms and applications: A survey","volume":"5","author":"Medhat","year":"2014","journal-title":"Ain Shams Engineering Journal"},{"key":"10.1016\/j.eswa.2026.133145_bib0025","series-title":"Multimedia modeling: 22nd international conference, MMM 2016, Miami, FL, USA, January 4\u20136, 2016, proceedings, Part II 22","first-page":"15","article-title":"Sentiment analysis on multi-view social data","author":"Niu","year":"2016"},{"key":"10.1016\/j.eswa.2026.133145_bib0026","series-title":"Findings of the association for computational linguistics: EMNLP 2020","first-page":"1383","article-title":"Modeling intra and inter-modality incongruity for multi-modal sarcasm detection","author":"Pan","year":"2020"},{"key":"10.1016\/j.eswa.2026.133145_bib0027","series-title":"Proceedings of the IEEE\/CVF Winter conference on applications of computer vision","first-page":"3930","article-title":"Multimodal learning using optimal transport for sarcasm and humor detection","author":"Pramanick","year":"2022"},{"key":"10.1016\/j.eswa.2026.133145_bib0028","doi-asserted-by":"crossref","unstructured":"Qin, L., Huang, S., Chen, Q., Cai, C., Zhang, Y., Liang, B., Che, W., & Xu, R. (2023). MMSD2. 0: Towards a reliable multi-modal sarcasm detection system. arXiv preprint arXiv: 2307.07135.","DOI":"10.18653\/v1\/2023.findings-acl.689"},{"key":"10.1016\/j.eswa.2026.133145_bib0029","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.133145_bib0030","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1016\/j.imavis.2017.08.003","article-title":"A survey of multimodal sentiment analysis","volume":"65","author":"Soleymani","year":"2017","journal-title":"Image and Vision Computing"},{"key":"10.1016\/j.eswa.2026.133145_bib0031","series-title":"Proceedings of the 1st Workshop on taming large language models: Controllability in the era of interactive assistants!","first-page":"11","article-title":"PandaGPT: One model to instruction-follow them all","author":"Su","year":"2023"},{"key":"10.1016\/j.eswa.2026.133145_bib0032","series-title":"Proceedings of the 61th Annual meeting of the association for computational linguistics","first-page":"2468","article-title":"Dynamic routing transformer network for multimodal sarcasm detection","author":"Tian","year":"2023"},{"key":"10.1016\/j.eswa.2026.133145_bib0033","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., Lachaux, M.-A., Lacroix, T., Rozi\u00e8re, B., Goyal, N., Hambro, E., Azhar, F. et al. (2023). LLaMA: Open and efficient foundation language models. arXiv preprint arXiv: 2302.13971."},{"key":"10.1016\/j.eswa.2026.133145_bib0034","series-title":"Advances in neural information processing systems","article-title":"Attention is all you need","volume":"vol. 30","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.eswa.2026.133145_bib0035","doi-asserted-by":"crossref","unstructured":"Wang, W., Ding, L., Shen, L., Luo, Y., Hu, H., & Tao, D. (2024). Wisdom: Improving multimodal sentiment analysis by fusing contextual world knowledge. arXiv preprint arXiv: 2401.06659.","DOI":"10.1145\/3664647.3681403"},{"issue":"7","key":"10.1016\/j.eswa.2026.133145_bib0036","doi-asserted-by":"crossref","first-page":"5731","DOI":"10.1007\/s10462-022-10144-1","article-title":"A survey on sentiment analysis methods, applications, and challenges","volume":"55","author":"Wankhade","year":"2022","journal-title":"Artificial Intelligence Review"},{"key":"10.1016\/j.eswa.2026.133145_bib0037","series-title":"Proceedings of the 61st Annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"5240","article-title":"Tackling modality heterogeneity with multi-view calibration network for multimodal sentiment detection","author":"Wei","year":"2023"},{"key":"10.1016\/j.eswa.2026.133145_bib0038","unstructured":"Wu, D., Yang, D., Shen, H., Ma, C., & Zhou, Y. (2024). Resolving sentiment discrepancy for multimodal sentiment detection via semantics completion and decomposition. arXiv preprint arXiv: 2407.07026."},{"key":"10.1016\/j.eswa.2026.133145_bib0039","series-title":"Proceedings of the 59th Annual meeting of the association for computational linguistics and the 11th International joint conference on natural language processing (volume 1: Long papers)","article-title":"Vocabulary learning via optimal transport for neural machine translation","author":"Xu","year":"2021"},{"key":"10.1016\/j.eswa.2026.133145_bib0040","series-title":"2017 IEEE International conference on intelligence and security informatics (ISI)","first-page":"152","article-title":"Analyzing multimodal public sentiment based on hierarchical semantic attentional network","author":"Xu","year":"2017"},{"key":"10.1016\/j.eswa.2026.133145_bib0041","series-title":"Proceedings of the 2017 ACM on conference on information and knowledge management","first-page":"2399","article-title":"MultiSentiNet: A deep semantic network for multimodal sentiment analysis","author":"Xu","year":"2017"},{"key":"10.1016\/j.eswa.2026.133145_bib0042","series-title":"The 41st International ACM SIGIR Conference on research & development in information retrieval","first-page":"929","article-title":"A co-memory network for multimodal sentiment analysis","author":"Xu","year":"2018"},{"key":"10.1016\/j.eswa.2026.133145_bib0043","series-title":"Proceedings of the AAAI Conference on artificial intelligence","first-page":"371","article-title":"Multi-interactive memory network for aspect based multimodal sentiment analysis","volume":"vol. 33","author":"Xu","year":"2019"},{"key":"10.1016\/j.eswa.2026.133145_bib0044","doi-asserted-by":"crossref","first-page":"4014","DOI":"10.1109\/TMM.2020.3035277","article-title":"Image-text multimodal emotion classification via multi-view attentional network","volume":"23","author":"Yang","year":"2020","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133145_bib0045","series-title":"Proceedings of the 59th Annual meeting of the association for computational linguistics and the 11th International joint conference on natural language processing (volume 1: Long papers)","first-page":"328","article-title":"Multimodal sentiment detection based on multi-channel graph neural networks","author":"Yang","year":"2021"},{"key":"10.1016\/j.eswa.2026.133145_bib0046","series-title":"Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition","first-page":"13040","article-title":"mPLUG-Owl2: Revolutionizing multi-modal large language model with modality collaboration","author":"Ye","year":"2024"},{"key":"10.1016\/j.eswa.2026.133145_bib0047","doi-asserted-by":"crossref","unstructured":"Ye, Q., Xu, H., Ye, J., Yan, M., Liu, H., Qian, Q., Zhang, J., Huang, F., & Zhou, J. (2023). mPLUG-Owl2: Revolutionizing multi-modal large language model with modality collaboration. arXiv preprint arXiv: 2311.04257.","DOI":"10.1109\/CVPR52733.2024.01239"},{"key":"10.1016\/j.eswa.2026.133145_bib0048","doi-asserted-by":"crossref","first-page":"429","DOI":"10.1109\/TASLP.2019.2957872","article-title":"Entity-sensitive attention and fusion network for entity-level multimodal sentiment classification","volume":"28","author":"Yu","year":"2019","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10.1016\/j.eswa.2026.133145_bib0049","series-title":"Proceedings of the 63rd Annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"14499","article-title":"Incongruity-aware tension field network for multi-modal sarcasm detection","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133145_bib0050","series-title":"Proceedings of the AAAI Conference on artificial intelligence","first-page":"4884","article-title":"TOT: Topology-aware optimal transport for multimodal hate detection","volume":"vol. 37","author":"Zhang","year":"2023"},{"issue":"4","key":"10.1016\/j.eswa.2026.133145_bib0051","article-title":"Deep learning for sentiment analysis: A survey","volume":"8","author":"Zhang","year":"2018","journal-title":"Wiley Interdisciplinary Reviews: Data Mining and Knowledge Discovery"},{"key":"10.1016\/j.eswa.2026.133145_bib0052","series-title":"Proceedings of the 54th Annual meeting of the association for computational linguistics (volume 2: Short papers)","first-page":"207","article-title":"Attention-based bidirectional long short-term memory networks for relation classification","author":"Zhou","year":"2016"},{"key":"10.1016\/j.eswa.2026.133145_bib0053","series-title":"Findings of the association for computational linguistics: ACL 2023","first-page":"8184","article-title":"Aom: Detecting aspect-oriented information for multimodal aspect-based sentiment analysis","author":"Zhou","year":"2023"},{"key":"10.1016\/j.eswa.2026.133145_bib0054","series-title":"The twelfth international conference on learning representations","article-title":"MiniGPT-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","year":"2023"},{"key":"10.1016\/j.eswa.2026.133145_bib0055","series-title":"Proceedings of the 2024 Conference on empirical methods in natural language processing","first-page":"2900","article-title":"DGLF: A dual graph-based learning framework for multi-modal sarcasm detection","author":"Zhu","year":"2024"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426020555?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426020555?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T16:44:22Z","timestamp":1783529062000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426020555"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":55,"alternative-id":["S0957417426020555"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133145","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Mining the potential of LVLMs in multimodal sentiment detection through cross-modal pre-alignment","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133145","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133145"}}