{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,8]],"date-time":"2025-03-08T05:17:32Z","timestamp":1741411052909,"version":"3.38.0"},"reference-count":58,"publisher":"Tech Science Press","issue":"1","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["CMC"],"published-print":{"date-parts":[[2025]]},"DOI":"10.32604\/cmc.2024.055943","type":"journal-article","created":{"date-parts":[[2024,12,12]],"date-time":"2024-12-12T02:31:36Z","timestamp":1733970696000},"page":"279-305","source":"Crossref","is-referenced-by-count":0,"title":["Text-Image Feature Fine-Grained Learning for Joint Multimodal Aspect-Based Sentiment Analysis"],"prefix":"10.32604","volume":"82","author":[{"given":"Tianzhi","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shunhang","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yepeng","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiankun","family":"Pi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gang","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuo","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"17807","published-online":{"date-parts":[[2025]]},"reference":[{"key":"ref1","first-page":"2723","article-title":"Cross-modal consistency with aesthetic similarity for multimodal false information detection","volume":"79","author":"Fan","year":"2024","journal-title":"Comput. Mater. Contin."},{"key":"ref2","article-title":"Multimodal consistency-specificity fusion based on information bottleneck for sentiment analysis","volume":"36","author":"Liu","year":"2024","journal-title":"J. King Saud Univ.-Comput. Inf. Sci."},{"key":"ref3","first-page":"1157","article-title":"Multimodal sentiment analysis based on a cross-modal multihead attention mechanism","volume":"78","author":"Deng","year":"2024","journal-title":"Comput. Mater. Contin."},{"key":"ref4","series-title":"Proc. 2021 Conf. Empir. Methods Nat. Lang. Process.","first-page":"4395","article-title":"Joint multi-modal aspect-sentiment analysis with auxiliary cross-modal relation detection","author":"Ju","year":"2021"},{"key":"ref5","series-title":"Proc. 60th Annu. Meet. Assoc. Comput. Linguist.","first-page":"2149","article-title":"Vision language pre-training for multimodal aspect-based sentiment analysis","author":"Ling","year":"2022"},{"key":"ref6","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2022.103038","article-title":"Cross-modal multitask transformer for end-to-end multimodal aspect-based sentiment analysis","volume":"59","author":"Yang","year":"2022","journal-title":"Inf. Process. Manag."},{"key":"ref7","doi-asserted-by":"crossref","first-page":"1305","DOI":"10.3934\/mbe.2024056","article-title":"Self-adaptive attention fusion for multimodal aspect-based sentiment analysis","volume":"21","author":"Wang","year":"2024","journal-title":"Math. Biosci. Eng."},{"key":"ref8","unstructured":"T. N. Kipf and M. Welling, \u201cSemi-supervised classification with graph convolutional networks,\u201d 2016, arXiv:1609.02907."},{"key":"ref9","series-title":"Proc. 21st ACM Int. Conf. Multimed.","first-page":"223","article-title":"Large-scale visual sentiment ontology and detectors using adjective noun pairs","author":"Borth","year":"2013"},{"key":"ref10","unstructured":"Y. Chen, \u201cConvolutional neural network for sentence classification,\u201d M.S. thesis, Univ. of Waterloo, Waterloo, ON, Canada, 2015."},{"key":"ref11","doi-asserted-by":"crossref","unstructured":"B. Shin, T. Lee, and J. D. Choi, \u201cLexicon integrated CNN models with attention for sentiment analysis,\u201d 2016, arXiv:1610.06272.","DOI":"10.18653\/v1\/W17-5220"},{"key":"ref12","article-title":"Visual sentiment analysis by attending on local image regions","volume":"31","author":"You","year":"2017","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"ref13","series-title":"Proc. 2019 Conf. Empir. Methods Nat. Lang. Process. 9th Int. Joint Conf. Nat. Lang. Process. (EMNLP-IJCNLP)","first-page":"4567","article-title":"Aspect-based sentiment classification with aspect-specific graph convolutional networks","author":"Zhang","year":"2019"},{"key":"ref14","series-title":"Proc. 2019 Conf. Empir. Methods Nat. Lang. Process. 9th Int. Joint Conf. Nat. Lang. Process. (EMNLP-IJCNLP)","first-page":"5469","article-title":"Syntax-aware aspect level sentiment classification with graph attention networks","author":"Huang","year":"2019"},{"key":"ref15","series-title":"Proc. 2019 Conf. Empir. Methods Nat. Lang. Process. 9th Int. Joint Conf. Nat. Lang. Process. (EMNLP-IJCNLP)","first-page":"5679","article-title":"Aspect-level sentiment analysis via convolution over dependency tree","author":"Sun","year":"2019"},{"key":"ref16","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"Hochreiter","year":"1997","journal-title":"Neural Comput."},{"key":"ref17","series-title":"Proc. 58th Annu. Meet. Assoc. Comput. Linguist.","first-page":"3229","article-title":"Dependency graph enhanced dual-transformer structure for aspect-based sentiment classification","author":"Tang","year":"2020"},{"key":"ref18","series-title":"Proc. 58th Annu. Meet. Assoc. Comput. Linguist.","first-page":"6578","article-title":"Relational graph attention network for aspect-based sentiment analysis","author":"Wang","year":"2020"},{"key":"ref19","doi-asserted-by":"crossref","unstructured":"A. Zadeh, M. Chen, S. Poria, E. Cambria, and L. P. Morency, \u201cTensor fusion network for multimodal sentiment analysis,\u201d 2017, arXiv:1707.07250.","DOI":"10.18653\/v1\/D17-1115"},{"key":"ref20","doi-asserted-by":"crossref","first-page":"108","DOI":"10.1109\/TAFFC.2020.3038167","article-title":"Beneath the tip of the iceberg: Current challenges and new directions in sentiment analysis research","volume":"14","author":"Poria","year":"2020","journal-title":"IEEE Trans. Affect. Comput."},{"key":"ref21","unstructured":"J. Chung, C. Gulcehre, K. Cho, and Y. Bengio, \u201cEmpirical evaluation of gated recurrent neural networks on sequence modeling,\u201d 2014, arXiv:1412.3555."},{"key":"ref22","series-title":"Proc. 2016 Conf. Empir. Methods Nat. Lang. Process.","first-page":"1042","article-title":"Real-time speech emotion and sentiment recognition for interactive dialogue systems","author":"Bertero","year":"2016"},{"key":"ref23","first-page":"6000","article-title":"Attention is all you need","author":"Vaswani","year":"2017","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"ref24","series-title":"Proc. 2015 Conf. Empir. Methods Nat. Lang. Process.","first-page":"2539","article-title":"Deep convolutional neural network textual features and multiple kernel learning for utterance-level multimodal sentiment analysis","author":"Poria","year":"2015"},{"key":"ref25","series-title":"Proc. 55th Annu. Meet. Assoc. Comput. Linguist.","first-page":"873","article-title":"Context-dependent sentiment analysis in user-generated videos","author":"Poria","year":"2017"},{"key":"ref26","doi-asserted-by":"crossref","unstructured":"P. P. Liang, Z. Liu, A. Zadeh, and L. P. Morency, \u201cMultimodal language analysis with recurrent multistage fusion,\u201d 2018, arXiv:1808.03920.","DOI":"10.18653\/v1\/D18-1014"},{"key":"ref27","series-title":"Proc. 6th Int. Conf. Multimodal Interf.","first-page":"205","article-title":"Analysis of emotion recognition using facial expressions, speech and multimodal information","author":"Busso","year":"2004"},{"key":"ref28","doi-asserted-by":"crossref","first-page":"1162","DOI":"10.1016\/j.specom.2011.06.004","article-title":"Emotion recognition using a hierarchical binary decision tree approach","volume":"53","author":"Lee","year":"2011","journal-title":"Speech Commun."},{"key":"ref29","doi-asserted-by":"crossref","unstructured":"S. Castro, D. Hazarika, V. P\u00e9rez-Rosas, R. Zimmermann, R. Mihalcea and S. Poria, \u201cTowards multimodal sarcasm detection (an _obviously_perfect paper),\u201d 2019, arXiv:1906.01815.","DOI":"10.18653\/v1\/P19-1455"},{"key":"ref30","series-title":"Proc. 57th Annu. Meet. Assoc. Comput. Linguist.","first-page":"2506","article-title":"Multi-modal sarcasm detection in twitter with hierarchical fusion model","author":"Cai","year":"2019"},{"key":"ref31","series-title":"Proc. 22nd ACM Int. Conf. Multimedia","first-page":"367","article-title":"Object-based visual sentiment concept analysis and application","author":"Chen","year":"2014"},{"key":"ref32","doi-asserted-by":"crossref","first-page":"2513","DOI":"10.1109\/TMM.2018.2803520","article-title":"Visual sentiment prediction based on automatic discovery of affective regions","volume":"20","author":"Yang","year":"2018","journal-title":"IEEE Trans. Multimed."},{"key":"ref33","series-title":"Proc. 24th ACM Int. Conf. Multimed.","first-page":"1008","article-title":"Robust visual-textual sentiment analysis: When attention meets tree-structured recursive neural networks","author":"You","year":"2016"},{"key":"ref34","series-title":"41st Int. ACM SIGIR Conf. Res. Dev. Inf. Retriev.","first-page":"929","article-title":"A co-memory network for multimodal sentiment analysis","author":"Xu","year":"2018"},{"key":"ref35","doi-asserted-by":"crossref","first-page":"24103","DOI":"10.1007\/s11042-019-7390-1","article-title":"Sentiment analysis of multimodal twitter data","volume":"78","author":"Kumar","year":"2019","journal-title":"Multimed. Tools Appl."},{"key":"ref36","series-title":"Proc. AAAI Conf. Artif. Intell.","first-page":"371","article-title":"Multi-interactive memory network for aspect based multimodal sentiment analysis","volume":"33","author":"Xu","year":"2019"},{"key":"ref37","doi-asserted-by":"crossref","first-page":"429","DOI":"10.1109\/TASLP.2019.2957872","article-title":"Entity-sensitive attention and fusion network for entity-level multimodal sentiment classification","volume":"28","author":"Yu","year":"2019","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"ref38","series-title":"Proc. Twenty-Eighth Int. Joint Conf. Artif. Intell.","first-page":"5408","article-title":"Adapting BERT for target-oriented multimodal sentiment classification","author":"Yu","year":"2019"},{"key":"ref39","unstructured":"J. Devlin, M. W. Chang, K. Lee, and K. Toutanova, \u201cBERT: Pre-training of deep bidirectional transformers for language understanding,\u201d 2018, arXiv:1810.04805."},{"key":"ref40","series-title":"Proc. 29th ACM Int. Conf. Multimed.","first-page":"3034","article-title":"Exploiting BERT for multimodal target sentiment classification through input space translation","author":"Khan","year":"2021"},{"key":"ref41","series-title":"Proc. 29th Int. Conf. Comput. Linguist.","first-page":"6784","article-title":"Learning from adjective-noun pairs: A knowledge-enhanced framework for target-oriented multimodal sentiment classification","author":"Zhao","year":"2022"},{"key":"ref42","series-title":"Proc. Nat. Lang. Process. Chin. Comput.","first-page":"145","article-title":"Multimodal aspect extraction with region-aware alignment network","author":"Wu","year":"2020"},{"key":"ref43","series-title":"Proc. 58th Annu. Meet. Assoc. Comput. Linguist.","first-page":"3342","article-title":"Improving multimodal named entity recognition via entity span detection with unified multimodal transformer","author":"Yu","year":"2020"},{"key":"ref44","series-title":"Proc. 28th ACM Int. Conf. Multimed.","first-page":"1038","article-title":"Multimodal representation with embedded visual guiding objects for named entity recognition in social media posts","author":"Wu","year":"2020"},{"key":"ref45","first-page":"8032","article-title":"MNER-QG: An end-to-end MRC framework for multimodal named entity recognition with query grounding","volume":"37","author":"Jia","year":"2023","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"ref46","series-title":"Proc. 2015 Conf. Empir. Methods Nat. Lang. Process.","first-page":"612","article-title":"Neural networks for open domain targeted sentiment","author":"Zhang","year":"2015"},{"key":"ref47","series-title":"Proc. AAAI Conf. Artif. Intell.","first-page":"6714","article-title":"A unified model for opinion target extraction and target sentiment prediction","volume":"33","author":"Li","year":"2019"},{"key":"ref48","series-title":"Proc. Conf. Assoc. Comput. Linguist.","first-page":"3685","article-title":"Relation-aware collaborative learning for unified aspect-based sentiment analysis","author":"Chen","year":"2020"},{"key":"ref49","doi-asserted-by":"crossref","unstructured":"E. F. Sang and J. Veenstra, \u201cRepresenting text chunks,\u201d 1999, arXiv:cs\/9907006.","DOI":"10.3115\/977035.977059"},{"key":"ref50","unstructured":"Y. Liu et al., \u201cRoberta: A robustly optimized BERT pretraining approach,\u201d 2019, arXiv:1907.11692."},{"key":"ref51","series-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit.","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"ref52","unstructured":"K. Simonyan and A. Zisserman, \u201cVery deep convolutional networks for large-scale image recognition,\u201d 2014, arXiv:1409.1556."},{"key":"ref53","series-title":"Proc. Conf. Assoc. Comput. Linguist.","first-page":"6558","article-title":"Multimodal transformer for unaligned multimodal language sequences","author":"Tsai","year":"2019"},{"key":"ref54","unstructured":"J. L. Ba, J. R. Kiros, and G. E. Hinton, \u201cLayer normalization,\u201d 2016, arXiv:1607.06450."},{"key":"ref55","unstructured":"T. Chen, D. Borth, T. Darrell, and S. F. Chang, \u201cDeepSentiBank: Visual sentiment concept classification with deep convolutional neural networks,\u201d 2014, arXiv:1410.8586."},{"key":"ref56","first-page":"3","article-title":"Conditional random fields: Probabilistic models for segmenting and labeling sequence data","volume":"1","author":"Lafferty","year":"2001","journal-title":"Proc. ICML"},{"key":"ref57","doi-asserted-by":"crossref","unstructured":"M. Hu, Y. Peng, Z. Huang, D. Li, and Y. Lv, \u201cOpen-domain targeted sentiment analysis via span-based extraction and classification,\u201d 2019, arXiv:1906.03820.","DOI":"10.18653\/v1\/P19-1051"},{"key":"ref58","series-title":"Proc. 28th Int. Conf. Comput. Linguist.","first-page":"272","article-title":"Joint aspect extraction and sentiment analysis with directional graph convolutional networks","author":"Chen","year":"2020"}],"container-title":["Computers, Materials &amp; Continua"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/file.techscience.com\/files\/cmc\/2025\/TSP_CMC-82-1\/TSP_CMC_55943\/TSP_CMC_55943.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,7]],"date-time":"2025-03-07T06:57:57Z","timestamp":1741330677000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.techscience.com\/cmc\/v82n1\/59211"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":58,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025]]},"published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.32604\/cmc.2024.055943","relation":{},"ISSN":["1546-2226"],"issn-type":[{"type":"electronic","value":"1546-2226"}],"subject":[],"published":{"date-parts":[[2025]]}}}