{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T05:22:56Z","timestamp":1780636976092,"version":"3.54.1"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T00:00:00Z","timestamp":1721606400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T00:00:00Z","timestamp":1721606400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s13735-024-00340-w","type":"journal-article","created":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T08:01:46Z","timestamp":1721635306000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Mual: enhancing multimodal sentiment analysis with cross-modal attention and difference loss"],"prefix":"10.1007","volume":"13","author":[{"given":"Yang","family":"Deng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yonghong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sidong","family":"Xian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Laquan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haiyang","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,7,22]]},"reference":[{"key":"340_CR1","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inform Process Syst 30"},{"key":"340_CR2","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2018) Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"340_CR3","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1016\/j.neucom.2021.10.091","volume":"471","author":"L Xiao","year":"2022","unstructured":"Xiao L, Xue Y, Wang H, Hu X, Gu D, Zhu Y (2022) Exploring fine-grained syntactic information for aspect-based sentiment classification with dual graph neural networks. Neurocomputing 471:48\u201359. https:\/\/doi.org\/10.1016\/j.neucom.2021.10.091","journal-title":"Neurocomputing"},{"key":"340_CR4","first-page":"13534","volume":"35","author":"R Mao","year":"2021","unstructured":"Mao R, Li X (2021) Bridging towers of multi-task learning with a gating mechanism for aspect-based sentiment analysis and sequential metaphor identification. Proceed AAAI Conf Artif Intell 35:13534\u201313542","journal-title":"Proceed AAAI Conf Artif Intell"},{"key":"340_CR5","doi-asserted-by":"publisher","unstructured":"Xu J, Yang S, Xiao L, Fu Z, Wu X, Ma T, He L (2022) Graph convolution over the semantic-syntactic hybrid graph enhanced by affective knowledge for aspect-level sentiment classification. In: 2022 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20138 . https:\/\/doi.org\/10.1109\/IJCNN55064.2022.9892027 . IEEE","DOI":"10.1109\/IJCNN55064.2022.9892027"},{"key":"340_CR6","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J (2021) Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 . PMLR"},{"key":"340_CR7","unstructured":"Li J, Li D, Xiong C, Hoi S (2022) Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900 . PMLR"},{"key":"340_CR8","unstructured":"Li LH, Yatskar M, Yin D, Hsieh C, Chang K (2019) Visualbert: A simple and performant baseline for vision and language. arxiv 2019. arXiv preprint arXiv:1908.03557"},{"key":"340_CR9","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2022.3204972","author":"R Mao","year":"2022","unstructured":"Mao R, Liu Q, He K, Li W, Cambria E (2022) The biases of pre-trained language models: an empirical study on prompt-based sentiment analysis and emotion detection. IEEE Trans Affect Comput. https:\/\/doi.org\/10.1109\/TAFFC.2022.3204972","journal-title":"IEEE Trans Affect Comput"},{"key":"340_CR10","doi-asserted-by":"crossref","unstructured":"Toledo GL, Marcacini RM (2022) Transfer learning with joint fine-tuning for multimodal sentiment analysis. arXiv preprint arXiv:2210.05790","DOI":"10.52591\/lxai202207173"},{"key":"340_CR11","doi-asserted-by":"crossref","unstructured":"Lai S, Xu H, Hu X, Ren Z, Liu Z (2023) Multimodal sentiment analysis: a survey. arXiv preprint arXiv:2305.07611","DOI":"10.2139\/ssrn.4487572"},{"key":"340_CR12","doi-asserted-by":"crossref","unstructured":"Morency L-P, Mihalcea R, Doshi P (2011) Towards multimodal sentiment analysis: Harvesting opinions from the web. In: Proceedings of the 13th International Conference on Multimodal Interfaces, pp. 169\u2013176","DOI":"10.1145\/2070481.2070509"},{"key":"340_CR13","doi-asserted-by":"crossref","unstructured":"Poria S, Cambria E, Gelbukh A (2015) Deep convolutional neural network textual features and multiple kernel learning for utterance-level multimodal sentiment analysis. In: Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing, pp. 2539\u20132544","DOI":"10.18653\/v1\/D15-1303"},{"key":"340_CR14","doi-asserted-by":"publisher","first-page":"101921","DOI":"10.1016\/j.inffus.2023.101921","volume":"100","author":"T Yue","year":"2023","unstructured":"Yue T, Mao R, Wang H, Hu Z, Cambria E (2023) Knowlenet: Knowledge fusion network for multimodal sarcasm detection. Inform Fusion 100:101921. https:\/\/doi.org\/10.1016\/j.inffus.2023.101921","journal-title":"Inform Fusion"},{"key":"340_CR15","doi-asserted-by":"crossref","unstructured":"Nojavanasghari B, Gopinath D, Koushik J, Baltru\u0161aitis T, Morency L-P (2016) Deep multimodal fusion for persuasiveness prediction. In: Proceedings of the 18th ACM International Conference on Multimodal Interaction, pp. 284\u2013288","DOI":"10.1145\/2993148.2993176"},{"key":"340_CR16","doi-asserted-by":"crossref","unstructured":"Mai S, Hu H, Xing S (2019) Divide, conquer and combine: Hierarchical feature fusion network with local and global perspectives for multimodal affective computing. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 481\u2013492","DOI":"10.18653\/v1\/P19-1046"},{"key":"340_CR17","doi-asserted-by":"crossref","unstructured":"Zhang H, Wang Y, Yin G, Liu K, Liu Y, Yu T (2023) Learning language-guided adaptive hyper-modality representation for multimodal sentiment analysis. arXiv preprint arXiv:2310.05804","DOI":"10.18653\/v1\/2023.emnlp-main.49"},{"key":"340_CR18","doi-asserted-by":"crossref","unstructured":"Sun T, Ni J, Wang W, Jing L, Wei Y, Nie L (2023) General debiasing for multimodal sentiment analysis. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 5861\u20135869","DOI":"10.1145\/3581783.3612051"},{"key":"340_CR19","doi-asserted-by":"publisher","first-page":"102304","DOI":"10.1016\/j.inffus.2024.102304","volume":"106","author":"L Xiao","year":"2024","unstructured":"Xiao L, Wu X, Xu J, Li W, Jin C, He L (2024) Atlantis: Aesthetic-oriented multiple granularities fusion network for joint multimodal aspect-based sentiment analysis. Inform Fusion 106:102304. https:\/\/doi.org\/10.1016\/j.inffus.2024.102304","journal-title":"Inform Fusion"},{"key":"340_CR20","unstructured":"Wang Y, Li Y, Bell P, Lai C (2023) Cross-attention is not enough: Incongruity-aware multimodal sentiment analysis and emotion recognition. arXiv preprint arXiv:2305.13583"},{"key":"340_CR21","doi-asserted-by":"crossref","unstructured":"Hazarika D, Zimmermann R, Poria S (2020) Misa: Modality-invariant and-specific representations for multimodal sentiment analysis. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 1122\u20131131","DOI":"10.1145\/3394171.3413678"},{"key":"340_CR22","doi-asserted-by":"crossref","unstructured":"Tsai Y-HH, Bai S, Liang PP, Kolter JZ, Morency L-P, Salakhutdinov R (2019) Multimodal transformer for unaligned multimodal language sequences. In: Proceedings of the Conference. Association for Computational Linguistics. Meeting, vol. 2019, p. 6558 . NIH Public Access","DOI":"10.18653\/v1\/P19-1656"},{"key":"340_CR23","unstructured":"Zhang S, Chadwick M, Ramos AGC, Bhattacharya S (2022) Cross-attention is all you need: Real-time streaming transformers for personalised speech enhancement. arXiv preprint arXiv:2211.04346"},{"key":"340_CR24","doi-asserted-by":"crossref","unstructured":"Rashed A, Elsayed S, Schmidt-Thieme L (2022) Context and attribute-aware sequential recommendation via cross-attention. In: Proceedings of the 16th ACM Conference on Recommender Systems, pp. 71\u201380","DOI":"10.1145\/3523227.3546777"},{"key":"340_CR25","doi-asserted-by":"crossref","unstructured":"Lei Y, Yang D, Li M, Wang S, Chen J, Zhang L (2023) Text-oriented modality reinforcement network for multimodal sentiment analysis from unaligned multimodal sequences. arXiv preprint arXiv:2307.13205","DOI":"10.1007\/978-981-99-9119-8_18"},{"key":"340_CR26","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems 25"},{"key":"340_CR27","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556"},{"key":"340_CR28","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"340_CR29","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"340_CR30","unstructured":"Tan M, Le Q (2019) Efficientnet: Rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, pp. 6105\u20136114 . PMLR"},{"key":"340_CR31","unstructured":"Mikolov T, Sutskever I, Chen K, Corrado GS, Dean J (2013) Distributed representations of words and phrases and their compositionality. Advances in neural information processing systems 26"},{"key":"340_CR32","doi-asserted-by":"crossref","unstructured":"Pennington J, Socher R, Manning CD (2014) Glove: Global vectors for word representation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543","DOI":"10.3115\/v1\/D14-1162"},{"key":"340_CR33","doi-asserted-by":"crossref","unstructured":"Peters ME, Neumann M, Iyyer M, Gardner M, Clark C, Lee K, Zettlemoyer L, (2018) Deep contextualized word representations. ArXiv abs\/1802.05365","DOI":"10.18653\/v1\/N18-1202"},{"key":"340_CR34","unstructured":"Sanh V, Debut L, Chaumond J, Wolf T (2019) Distilbert, a distilled version of bert: smaller, faster, cheaper and lighter. arXiv preprint arXiv:1910.01108"},{"key":"340_CR35","doi-asserted-by":"crossref","unstructured":"Niu T, Zhu S, Pang L, El\u00a0Saddik A (2016) Sentiment analysis on multi-view social data. In: MultiMedia Modeling: 22nd International Conference, MMM 2016, Miami, FL, USA, January 4-6, 2016, Proceedings, Part II 22, pp. 15\u201327 . Springer","DOI":"10.1007\/978-3-319-27674-8_2"},{"key":"340_CR36","first-page":"2611","volume":"33","author":"D Kiela","year":"2020","unstructured":"Kiela D, Firooz H, Mohan A, Goswami V, Singh A, Ringshia P, Testuggine D (2020) The hateful memes challenge: detecting hate speech in multimodal memes. Adv Neural Inform Process Syst 33:2611\u20132624","journal-title":"Adv Neural Inform Process Syst"},{"key":"340_CR37","doi-asserted-by":"crossref","unstructured":"Yu J, Jiang J, Yang L, Xia R (2020) Improving multimodal named entity recognition via entity span detection with unified multimodal transformer. Assoc Comput Linguis","DOI":"10.18653\/v1\/2020.acl-main.306"},{"key":"340_CR38","unstructured":"Paszke A, Gross S, Massa F, Lerer A, Bradbury J, Chanan G, Killeen T, Lin Z, Gimelshein N, Antiga L (2019) et al.: Pytorch: An imperative style, high-performance deep learning library. Adv Neural Inform Process 32"},{"key":"340_CR39","doi-asserted-by":"crossref","unstructured":"Wolf T, Debut L, Sanh V, Chaumond J, Delangue C, Moi A, Cistac P, Rault T, Louf R, Funtowicz M, et al (2019) Huggingface\u2019s transformers: state-of-the-art natural language processing. arXiv preprint arXiv:1910.03771","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"340_CR40","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-024-00340-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-024-00340-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-024-00340-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,6]],"date-time":"2024-09-06T12:26:35Z","timestamp":1725625595000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-024-00340-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,22]]},"references-count":40,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["340"],"URL":"https:\/\/doi.org\/10.1007\/s13735-024-00340-w","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7,22]]},"assertion":[{"value":"3 February 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 June 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 July 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 July 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The manuscript was reviewed and ethical approved for publication by all authors. The manuscript was reviewed and consents to participate by all authors.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}],"article-number":"31"}}