{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T18:16:41Z","timestamp":1778782601292,"version":"3.51.4"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100004001","name":"Guizhou Provincial Science and Technology Department","doi-asserted-by":"publisher","award":["ZK[2023]YB449"],"award-info":[{"award-number":["ZK[2023]YB449"]}],"id":[{"id":"10.13039\/501100004001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62362048"],"award-info":[{"award-number":["62362048"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.eswa.2026.131648","type":"journal-article","created":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T00:19:39Z","timestamp":1771028379000},"page":"131648","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Toward multimodal sentiment analysis with a self-supervised knowledge-augmented network"],"prefix":"10.1016","volume":"314","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-4061-5540","authenticated-orcid":false,"given":"Yun","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7489-536X","authenticated-orcid":false,"given":"Xiaoming","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5645-6184","authenticated-orcid":false,"given":"Tianhao","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ke","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9603-9713","authenticated-orcid":false,"given":"Zhoujun","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.131648_bib0001","doi-asserted-by":"crossref","first-page":"165","DOI":"10.58496\/ADSA\/2024\/013","article-title":"Lexicon annotation in sentiment analysis for dialectal arabic: Consensus expert standardized criteria","volume":"2024","author":"sherif","year":"2024","journal-title":"Applied Data Science and Analysis"},{"key":"10.1016\/j.eswa.2026.131648_bib0002","unstructured":"Bai, J., Bai, S., Yang, S., Wang, S., Tan, S., Wang, P., Lin, J., Zhou, C., & Zhou, J. (2023). Qwen-VL: A versatile vision-language model for understanding, localization, text reading, and beyond. arXiv: 2308.12966."},{"issue":"2","key":"10.1016\/j.eswa.2026.131648_bib0003","doi-asserted-by":"crossref","first-page":"423","DOI":"10.1109\/TPAMI.2018.2798607","article-title":"Multimodal machine learning: A survey and taxonomy","volume":"41","author":"Baltru\u0161aitis","year":"2018","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.131648_bib0004","series-title":"Proceedings of the ACM web conference 2024","first-page":"1816","article-title":"A novel dual-pipeline based attention mechanism for multimodal social sentiment analysis","author":"Braytee","year":"2024"},{"key":"10.1016\/j.eswa.2026.131648_bib0005","series-title":"Proceedings of the 57th annual meeting of the association for computational linguistics","first-page":"2506","article-title":"Multi-modal sarcasm detection in twitter with hierarchical fusion model","author":"Cai","year":"2019"},{"key":"10.1016\/j.eswa.2026.131648_bib0006","unstructured":"Devlin, J. (2018). Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv: 1810.04805."},{"key":"10.1016\/j.eswa.2026.131648_bib0007","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv: 2010.11929."},{"key":"10.1016\/j.eswa.2026.131648_bib0008","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2021.108107","article-title":"Gated attention fusion network for multimodal sentiment classification","volume":"240","author":"Du","year":"2022","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.131648_bib0009","doi-asserted-by":"crossref","first-page":"12","DOI":"10.70470\/EDRAAK\/2025\/003","article-title":"Enhancing point-of-interest recommendation systems through multi-modal data integration in location-based social networks: Challenges and future directions","volume":"2025","author":"Dutta","year":"2025","journal-title":"EDRAAK"},{"key":"10.1016\/j.eswa.2026.131648_bib0010","first-page":"2296","article-title":"Are you talking to a machine? Dataset and methods for multilingual image question","volume":"28","author":"Gao","year":"2015","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.131648_bib0011","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"3394","article-title":"Dynamically adjust word representations using unaligned multimodal information","author":"Guo","year":"2022"},{"key":"10.1016\/j.eswa.2026.131648_bib0012","series-title":"Proceedings of the 28th ACM international conference on multimedia","first-page":"1122","article-title":"MISA: Modality-invariant and-specific representations for multimodal sentiment analysis","author":"Hazarika","year":"2020"},{"issue":"4","key":"10.1016\/j.eswa.2026.131648_bib0013","doi-asserted-by":"crossref","first-page":"927","DOI":"10.1109\/TMM.2017.2760101","article-title":"Twitter100k: A real-world dataset for weakly supervised cross-media retrieval","volume":"20","author":"Hu","year":"2017","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.131648_bib0014","series-title":"Proceedings of the international AAAI conference on web and social media","first-page":"216","article-title":"Vader: A parsimonious rule-based model for sentiment analysis of social media text","volume":"vol. 8","author":"Hutto","year":"2014"},{"key":"10.1016\/j.eswa.2026.131648_bib0015","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"9972","article-title":"Hierarchical conditional relation networks for video question answering","author":"Le","year":"2020"},{"key":"10.1016\/j.eswa.2026.131648_bib0016","series-title":"Findings of the association for computational linguistics","first-page":"2282","article-title":"CLMLF: A contrastive learning and multi-layer fusion method for multimodal sentiment detection","author":"Li","year":"2022"},{"key":"10.1016\/j.eswa.2026.131648_bib0017","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"26296","article-title":"Improved baselines with visual instruction tuning","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.131648_bib0018","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2023.110467","article-title":"Scanning, attention, and reasoning multimodal content for sentiment analysis","volume":"268","author":"Liu","year":"2023","journal-title":"Knowledge-Based Systems"},{"issue":"3","key":"10.1016\/j.eswa.2026.131648_bib0019","doi-asserted-by":"crossref","first-page":"2276","DOI":"10.1109\/TAFFC.2022.3172360","article-title":"Hybrid contrastive learning of tri-modal representation for multimodal sentiment analysis","volume":"14","author":"Mai","year":"2023","journal-title":"IEEE Transactions on Affective Computing"},{"key":"10.1016\/j.eswa.2026.131648_bib0020","series-title":"Multimedia modeling: 22nd international conference, MMM 2016, Miami, FL, USA, January 4-6, 2016, proceedings, part II 22","first-page":"15","article-title":"Sentiment analysis on multi-view social data","author":"Niu","year":"2016"},{"key":"10.1016\/j.eswa.2026.131648_bib0021","series-title":"Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP)","first-page":"1532","article-title":"Glove: Global vectors for word representation","author":"Pennington","year":"2014"},{"key":"10.1016\/j.eswa.2026.131648_bib0022","series-title":"Proceedings of the 2015 conference on empirical methods in natural language processing","first-page":"2539","article-title":"Deep convolutional neural network textual features and multiple kernel learning for utterance-level multimodal sentiment analysis","author":"Poria","year":"2015"},{"key":"10.1016\/j.eswa.2026.131648_bib0023","series-title":"Proceedings of the 55th annual meeting of the association for computational linguistics","first-page":"873","article-title":"Context-dependent sentiment analysis in user-generated videos","author":"Poria","year":"2017"},{"key":"10.1016\/j.eswa.2026.131648_bib0024","series-title":"Proceedings of the 24th ACM international conference on multimedia","first-page":"1136","article-title":"Detecting sarcasm in multimodal social platforms","author":"Schifanella","year":"2016"},{"key":"10.1016\/j.eswa.2026.131648_bib0025","series-title":"Advances in neural information processing systems","first-page":"568","article-title":"Two-stream convolutional networks for action recognition in videos","author":"Simonyan","year":"2014"},{"key":"10.1016\/j.eswa.2026.131648_bib0026","series-title":"International conference on learning representations","article-title":"Vl-bert: Pre-training of generic visual-linguistic representations","author":"Su","year":"2020"},{"key":"10.1016\/j.eswa.2026.131648_bib0027","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"8992","article-title":"Learning relationships between text, audio, and video via deep canonical correlation for multimodal language analysis","volume":"vol. 34","author":"Sun","year":"2020"},{"key":"10.1016\/j.eswa.2026.131648_bib0028","series-title":"Proceedings of the conference. Association for computational linguistics. meeting","first-page":"6558","article-title":"Multimodal transformer for unaligned multimodal language sequences","volume":"vol. 2019","author":"Tsai","year":"2019"},{"key":"10.1016\/j.eswa.2026.131648_bib0029","series-title":"International symposium on experimental robotics","first-page":"465","article-title":"Deep multispectral semantic scene understanding of forested environments using multimodal fusion","author":"Valada","year":"2016"},{"key":"10.1016\/j.eswa.2026.131648_bib0030","doi-asserted-by":"crossref","first-page":"4909","DOI":"10.1109\/TMM.2022.3183830","article-title":"Cross-modal enhancement network for multimodal sentiment analysis","volume":"25","author":"Wang","year":"2022","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.131648_bib0031","doi-asserted-by":"crossref","first-page":"208","DOI":"10.1016\/j.ins.2023.01.116","article-title":"Learning speaker-independent multimodal representation for sentiment analysis","volume":"628","author":"Wang","year":"2023","journal-title":"Information Sciences"},{"key":"10.1016\/j.eswa.2026.131648_bib0032","first-page":"17117","article-title":"Incomplete multimodality-diffused emotion recognition","volume":"36","author":"Wang","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.131648_bib0033","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics","first-page":"5240","article-title":"Tackling modality heterogeneity with multi-view calibration network for multimodal sentiment detection","author":"Wei","year":"2023"},{"key":"10.1016\/j.eswa.2026.131648_bib0034","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"8997","article-title":"Differential networks for visual question answering","volume":"vol. 33","author":"Wu","year":"2019"},{"issue":"8","key":"10.1016\/j.eswa.2026.131648_bib0035","doi-asserted-by":"crossref","first-page":"1583","DOI":"10.1109\/TPAMI.2016.2537340","article-title":"Deep dynamic neural networks for multimodal gesture segmentation and recognition","volume":"38","author":"Wu","year":"2016","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.131648_bib0036","series-title":"Proceedings of the 32nd ACM international conference on multimedia","first-page":"5780","article-title":"Robust multimodal sentiment analysis of image-text pairs by distribution-based feature recovery and fusion","author":"Wu","year":"2024"},{"key":"10.1016\/j.eswa.2026.131648_bib0037","series-title":"Proceedings of the 4th international conference on machine learning and soft computing","first-page":"34","article-title":"Multimodal sentiment analysis based on multi-head attention mechanism","author":"Xi","year":"2020"},{"issue":"4","key":"10.1016\/j.eswa.2026.131648_bib0038","doi-asserted-by":"crossref","first-page":"2974","DOI":"10.1109\/TII.2020.3005405","article-title":"Social image sentiment analysis by exploiting multimodal content and heterogeneous relations","volume":"17","author":"Xu","year":"2020","journal-title":"IEEE Transactions on Industrial Informatics"},{"key":"10.1016\/j.eswa.2026.131648_bib0039","series-title":"2017 IEEE international conference on intelligence and security informatics (ISI)","first-page":"152","article-title":"Analyzing multimodal public sentiment based on hierarchical semantic attentional network","author":"Xu","year":"2017"},{"key":"10.1016\/j.eswa.2026.131648_bib0040","series-title":"Proceedings of the 2017 ACM on conference on information and knowledge management","first-page":"2399","article-title":"MultiSentiNet: A deep semantic network for multimodal sentiment analysis","author":"Xu","year":"2017"},{"key":"10.1016\/j.eswa.2026.131648_bib0041","series-title":"The 41st international ACM SIGIR conference on research & development in information retrieval","first-page":"929","article-title":"A co-memory network for multimodal sentiment analysis","author":"Xu","year":"2018"},{"key":"10.1016\/j.eswa.2026.131648_bib0042","series-title":"Proceedings of the 58th annual meeting of the association for computational linguistics","first-page":"3777","article-title":"Reasoning with multimodal sarcastic tweets via modeling cross-modality contrast and semantic association","author":"Xu","year":"2020"},{"key":"10.1016\/j.eswa.2026.131648_bib0043","series-title":"Proceedings of the 28th ACM international conference on multimedia","first-page":"521","article-title":"CM-Bert: Cross-modal bert for text-audio sentiment analysis","author":"Yang","year":"2020"},{"key":"10.1016\/j.eswa.2026.131648_bib0044","series-title":"Proceedings of the 59th annual meeting of the association for computational linguistics and the 11th international joint conference on natural language processing (volume 1: Long papers)","first-page":"328","article-title":"Multimodal sentiment detection based on multi-channel graph neural networks","author":"Yang","year":"2021"},{"key":"10.1016\/j.eswa.2026.131648_bib0045","series-title":"Proceedings of the 58th annual meeting of the association for computational linguistics","first-page":"3718","article-title":"CH-SIMS: A chinese multimodal sentiment analysis dataset with fine-grained annotation of modality","author":"Yu","year":"2020"},{"key":"10.1016\/j.eswa.2026.131648_bib0046","series-title":"Proceedings of the 29th ACM international conference on multimedia","first-page":"4400","article-title":"Transformer-based feature reconstruction network for robust multimodal sentiment analysis","author":"Yuan","year":"2021"},{"key":"10.1016\/j.eswa.2026.131648_bib0047","series-title":"Proceedings of the 2017 conference on empirical methods in natural language processing","first-page":"1103","article-title":"Tensor fusion network for multimodal sentiment analysis","author":"Zadeh","year":"2017"},{"key":"10.1016\/j.eswa.2026.131648_bib0048","doi-asserted-by":"crossref","first-page":"9949","DOI":"10.1109\/TMM.2024.3405662","article-title":"Crossmodal translation based meta weight adaption for robust image-text sentiment analysis","volume":"26","author":"Zhang","year":"2024","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.131648_bib0049","series-title":"Proceedings of the 27th ACM international conference on multimedia","first-page":"148","article-title":"Effective sentiment-relevant word selection for multi-modal sentiment analysis in spoken language","author":"Zhang","year":"2019"},{"key":"10.1016\/j.eswa.2026.131648_bib0050","unstructured":"Zhao, H., Cai, Z., Si, S., Ma, X., An, K., Chen, L., Liu, Z., Wang, S., Han, W., & Chang, B. (2023). MMICL: Empowering vision-language model with multi-modal in-context learning. arXiv: 2309.07915."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426005610?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426005610?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T17:49:06Z","timestamp":1778780946000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426005610"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":50,"alternative-id":["S0957417426005610"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.131648","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Toward multimodal sentiment analysis with a self-supervised knowledge-augmented network","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.131648","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"131648"}}