{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T09:07:15Z","timestamp":1784279235794,"version":"3.55.0"},"reference-count":39,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62166006"],"award-info":[{"award-number":["62166006"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113803","type":"journal-article","created":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T22:58:41Z","timestamp":1776985121000},"page":"113803","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["Token-level refined region-based framework for multimodal sentiment analysis"],"prefix":"10.1016","volume":"179","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-2657-3701","authenticated-orcid":false,"given":"Yanbo","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1801-9901","authenticated-orcid":false,"given":"Qing","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6149-6031","authenticated-orcid":false,"given":"Mengrong","family":"Lv","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-8895-6871","authenticated-orcid":false,"given":"Qingfeng","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7928-0636","authenticated-orcid":false,"given":"Ting","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"4","key":"10.1016\/j.patcog.2026.113803_b1","article-title":"Deep learning for sentiment analysis: A survey","volume":"8","author":"Zhang","year":"2018","journal-title":"Wiley Interdiscip. Rev.: Data Min. Knowl. Discov."},{"issue":"1","key":"10.1016\/j.patcog.2026.113803_b2","doi-asserted-by":"crossref","first-page":"145","DOI":"10.1109\/TAFFC.2024.3418415","article-title":"Image encoding and fusion of multi-modal data enhance depression diagnosis in parkinson\u2019s disease patients","volume":"16","author":"Li","year":"2024","journal-title":"IEEE Trans. Affect. Comput."},{"key":"10.1016\/j.patcog.2026.113803_b3","doi-asserted-by":"crossref","first-page":"209","DOI":"10.1016\/j.inffus.2019.06.019","article-title":"A snapshot research and implementation of multimodal information fusion for data-driven emotion recognition","volume":"53","author":"Jiang","year":"2020","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.patcog.2026.113803_b4","series-title":"2024 4th International Conference on Electronic Information Engineering and Computer Communication","first-page":"1294","article-title":"Human-computer interaction in smart devices: Leveraging sentiment analysis and knowledge graphs for personalized user experiences","author":"Duan","year":"2024"},{"key":"10.1016\/j.patcog.2026.113803_b5","series-title":"International Work-Conference on the Interplay Between Natural and Artificial Computation","first-page":"431","article-title":"Human-computer interaction approach with empathic conversational agent and computer vision","author":"Pereira","year":"2024"},{"key":"10.1016\/j.patcog.2026.113803_b6","unstructured":"K. Sun, M. Tian, Sequential fusion of text-close and text-far representations for multimodal sentiment analysis, in: Proceedings of the 31st International Conference on Computational Linguistics, 2025, pp. 40\u201349."},{"key":"10.1016\/j.patcog.2026.113803_b7","doi-asserted-by":"crossref","unstructured":"A. Zhu, M. Hu, X. Wang, J. Yang, Y. Tang, F. Ren, KEBR: Knowledge Enhanced Self-Supervised Balanced Representation for Multimodal Sentiment Analysis, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 5732\u20135741.","DOI":"10.1145\/3664647.3681163"},{"key":"10.1016\/j.patcog.2026.113803_b8","doi-asserted-by":"crossref","DOI":"10.1145\/3744648","article-title":"A multimodal semantic fusion network with cross-modal alignment for multimodal sentiment analysis","author":"Zhang","year":"2025","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"issue":"21","key":"10.1016\/j.patcog.2026.113803_b9","doi-asserted-by":"crossref","first-page":"60171","DOI":"10.1007\/s11042-023-17685-9","article-title":"Multi-layer cross-modality attention fusion network for multimodal sentiment analysis","volume":"83","author":"Yin","year":"2024","journal-title":"Multimedia Tools Appl."},{"key":"10.1016\/j.patcog.2026.113803_b10","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2023.127181","article-title":"Multimodal transformer with adaptive modality weighting for multimodal sentiment analysis","volume":"572","author":"Wang","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.patcog.2026.113803_b11","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107222","article-title":"TF-BERT: Tensor-based fusion BERT for multimodal sentiment analysis","volume":"185","author":"Hou","year":"2025","journal-title":"Neural Netw."},{"key":"10.1016\/j.patcog.2026.113803_b12","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2023.107335","article-title":"A feature-based restoration dynamic interaction network for multimodal sentiment analysis","volume":"127","author":"Zeng","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.patcog.2026.113803_b13","doi-asserted-by":"crossref","unstructured":"Y.-H.H. Tsai, S. Bai, P.P. Liang, J.Z. Kolter, L.-P. Morency, R. Salakhutdinov, Multimodal Transformer for Unaligned Multimodal Language Sequences, in: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, 2019, pp. 6558\u20136569.","DOI":"10.18653\/v1\/P19-1656"},{"key":"10.1016\/j.patcog.2026.113803_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.126274","article-title":"Learning fine-grained representation with token-level alignment for multimodal sentiment analysis","volume":"269","author":"Li","year":"2025","journal-title":"Expert Syst. Appl."},{"issue":"2","key":"10.1016\/j.patcog.2026.113803_b15","doi-asserted-by":"crossref","first-page":"133","DOI":"10.1007\/s40747-024-01724-5","article-title":"TMFN: a text-based multimodal fusion network with multi-scale feature extraction and unsupervised contrastive learning for multimodal sentiment analysis","volume":"11","author":"Fu","year":"2025","journal-title":"Complex Intell. Syst."},{"key":"10.1016\/j.patcog.2026.113803_b16","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.129532","article-title":"Text-guided multi-level interaction and multi-scale spatial-memory fusion for multimodal sentiment analysis","volume":"626","author":"He","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.patcog.2026.113803_b17","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"9273","article-title":"Span-level aspect-based sentiment analysis via table filling","author":"Zhang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113803_b18","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"14340","article-title":"USSA: A unified table filling scheme for structured sentiment analysis","author":"Zhai","year":"2023"},{"key":"10.1016\/j.patcog.2026.113803_b19","first-page":"18661","article-title":"Supervised contrastive learning","volume":"33","author":"Khosla","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"14","key":"10.1016\/j.patcog.2026.113803_b20","doi-asserted-by":"crossref","first-page":"e49","DOI":"10.1093\/bioinformatics\/btl242","article-title":"Integrating structured biological data by kernel maximum mean discrepancy","volume":"22","author":"Borgwardt","year":"2006","journal-title":"Bioinformatics"},{"key":"10.1016\/j.patcog.2026.113803_b21","doi-asserted-by":"crossref","unstructured":"D. Hazarika, R. Zimmermann, S. Poria, Misa: Modality-invariant and-specific representations for multimodal sentiment analysis, in: Proceedings of the 28th ACM International Conference on Multimedia, 2020, pp. 1122\u20131131.","DOI":"10.1145\/3394171.3413678"},{"key":"10.1016\/j.patcog.2026.113803_b22","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102787","article-title":"Multimodal sentiment analysis with unimodal label generation and modality decomposition","volume":"116","author":"Zhu","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.patcog.2026.113803_b23","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102897","article-title":"Decoupled cross-attribute correlation network for multimodal sentiment analysis","volume":"117","author":"Zhao","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.patcog.2026.113803_b24","doi-asserted-by":"crossref","unstructured":"A. Zadeh, M. Chen, S. Poria, E. Cambria, L.-P. Morency, Tensor Fusion Network for Multimodal Sentiment Analysis, in: Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing, 2017, pp. 1103\u20131114.","DOI":"10.18653\/v1\/D17-1115"},{"key":"10.1016\/j.patcog.2026.113803_b25","series-title":"Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"2247","article-title":"Efficient low-rank multimodal fusion with modality-specific factors","author":"Liu","year":"2018"},{"issue":"3","key":"10.1016\/j.patcog.2026.113803_b26","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2024.103675","article-title":"A cross modal hierarchical fusion multimodal sentiment analysis method based on multi-task learning","volume":"61","author":"Wang","year":"2024","journal-title":"Inf. Process. Manage."},{"key":"10.1016\/j.patcog.2026.113803_b27","series-title":"Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021","first-page":"4730","article-title":"A text-centered shared-private framework via cross-modal prediction for multimodal sentiment analysis","author":"Wu","year":"2021"},{"key":"10.1016\/j.patcog.2026.113803_b28","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103230","article-title":"CCIN-SA: Composite cross modal interaction network with attention enhancement for multimodal sentiment analysis","author":"Yang","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.patcog.2026.113803_b29","doi-asserted-by":"crossref","unstructured":"P. Wang, Q. Zhou, Y. Wu, T. Chen, J. Hu, DLF: Disentangled-language-focused multimodal sentiment analysis, in: Proceedings of the AAAI Conference on Artificial Intelligence, 2025, pp. 21180\u201321188.","DOI":"10.1609\/aaai.v39i20.35416"},{"key":"10.1016\/j.patcog.2026.113803_b30","series-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.patcog.2026.113803_b31","series-title":"2014 Ieee International Conference on Acoustics, Speech and Signal Processing (Icassp)","first-page":"960","article-title":"COVAREP\u2014A collaborative voice analysis repository for speech technologies","author":"Degottex","year":"2014"},{"key":"10.1016\/j.patcog.2026.113803_b32","series-title":"2016 IEEE Winter Conference on Applications of Computer Vision","first-page":"1","article-title":"Openface: an open source facial behavior analysis toolkit","author":"Baltru\u0161aitis","year":"2016"},{"key":"10.1016\/j.patcog.2026.113803_b33","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113803_b34","article-title":"Improving aspect-based sentiment analysis with contrastive learning","volume":"3","author":"Xu","year":"2023","journal-title":"Nat. Lang. Process. J."},{"issue":"6","key":"10.1016\/j.patcog.2026.113803_b35","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MIS.2016.94","article-title":"Multimodal sentiment intensity analysis in videos: Facial gestures and verbal messages","volume":"31","author":"Zadeh","year":"2016","journal-title":"IEEE Intell. Syst."},{"key":"10.1016\/j.patcog.2026.113803_b36","series-title":"Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"2236","article-title":"Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph","author":"Zadeh","year":"2018"},{"key":"10.1016\/j.patcog.2026.113803_b37","doi-asserted-by":"crossref","unstructured":"W. Yu, H. Xu, F. Meng, Y. Zhu, Y. Ma, J. Wu, J. Zou, K. Yang, Ch-sims: A chinese multimodal sentiment analysis dataset with fine-grained annotation of modality, in: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, 2020, pp. 3718\u20133727.","DOI":"10.18653\/v1\/2020.acl-main.343"},{"key":"10.1016\/j.patcog.2026.113803_b38","series-title":"Fixing weight decay regularization in adam","author":"Loshchilov","year":"2017"},{"issue":"11","key":"10.1016\/j.patcog.2026.113803_b39","article-title":"Visualizing data using t-SNE","volume":"9","author":"Van der Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326007685?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326007685?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T08:28:05Z","timestamp":1784276885000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326007685"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":39,"alternative-id":["S0031320326007685"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113803","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Token-level refined region-based framework for multimodal sentiment analysis","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113803","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113803"}}