{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T02:32:45Z","timestamp":1783823565427,"version":"3.55.0"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T00:00:00Z","timestamp":1752192000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T00:00:00Z","timestamp":1752192000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62166039"],"award-info":[{"award-number":["62166039"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62166039"],"award-info":[{"award-number":["62166039"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62166039"],"award-info":[{"award-number":["62166039"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s00530-025-01868-5","type":"journal-article","created":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T09:22:06Z","timestamp":1752225726000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Self-attention mechanism prior to modality fusion for multimodal sentiment analysis"],"prefix":"10.1007","volume":"31","author":[{"given":"Zhongyuan","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chong","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yihan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,11]]},"reference":[{"key":"1868_CR1","first-page":"119","volume":"3","author":"A Haleem","year":"2022","unstructured":"Haleem, A., Javaid, M., Qadri, M.A., et al.: Artificial intelligence (AI) applications for marketing: a literature-based study. Int. J. Intell. Netw. 3, 119\u2013132 (2022)","journal-title":"Int. J. Intell. Netw."},{"key":"1868_CR2","doi-asserted-by":"publisher","first-page":"111149","DOI":"10.1016\/j.knosys.2023.111149","volume":"283","author":"H Shi","year":"2024","unstructured":"Shi, H., Pu, Y., Zhao, Z., et al.: Co-space Representation Interaction Network for multimodal sentiment analysis. Knowl.-Based Syst. 283, 111149 (2024)","journal-title":"Knowl.-Based Syst."},{"key":"1868_CR3","doi-asserted-by":"publisher","first-page":"102563","DOI":"10.1016\/j.displa.2023.102563","volume":"80","author":"S Lai","year":"2023","unstructured":"Lai, S., Hu, X., Xu, H., et al.: Multimodal sentiment analysis: a survey. Displays 80, 102563 (2023)","journal-title":"Displays"},{"issue":"24","key":"1868_CR4","doi-asserted-by":"publisher","first-page":"30455","DOI":"10.1007\/s10489-023-05151-w","volume":"53","author":"T Zhao","year":"2023","unstructured":"Zhao, T., Peng, J., Huang, Y., et al.: A graph convolution-based heterogeneous fusion network for multimodal sentiment analysis. Appl. Intell. 53(24), 30455\u201330468 (2023)","journal-title":"Appl. Intell."},{"issue":"1","key":"1868_CR5","first-page":"2105593","volume":"2022","author":"J He","year":"2022","unstructured":"He, J., Yanga, H., Zhang, C., et al.: Dynamic invariant-specific representation fusion network for multimodal sentiment analysis. Comput. Intell. Neurosci. 2022(1), 2105593 (2022)","journal-title":"Comput. Intell. Neurosci."},{"key":"1868_CR6","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chen, M., Poria, S., et al.: Tensor fusion network for multimodal sentiment analysis. In: EMNLP, pp. 1103\u20131114. Association for Computational Linguistics (2017)","DOI":"10.18653\/v1\/D17-1115"},{"key":"1868_CR7","doi-asserted-by":"crossref","unstructured":"Hazarika, D., Zimmermann, R., Poria, S.: Misa: modality-invariant and-specific representations for multimodal sentiment analysis. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 1122\u20131131 (2020)","DOI":"10.1145\/3394171.3413678"},{"issue":"12","key":"1868_CR8","first-page":"10790","volume":"35","author":"W Yu","year":"2021","unstructured":"Yu, W., Xu, H., Yuan, Z., et al.: Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis. Proc. AAAI Conf. Artif. Intell. 35(12), 10790\u201310797 (2021)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"1868_CR9","doi-asserted-by":"crossref","unstructured":"Wu, Z., Gong, Z., Koo, J., et al.: Multimodal multi-loss fusion network for sentiment analysis. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 3588\u20133602 (2024)","DOI":"10.18653\/v1\/2024.naacl-long.197"},{"issue":"02","key":"1868_CR10","first-page":"156","volume":"54","author":"T Changning","year":"2024","unstructured":"Changning, T., Yuzheng, He., Di, W., et al.: Multi-subspace multimodal sentiment analysis method based on Transformer. J. Northwest Univ. (Nat. Sci. Ed.) 54(02), 156\u2013167 (2024)","journal-title":"J. Northwest Univ. (Nat. Sci. Ed.)"},{"issue":"1","key":"1868_CR11","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TII.2024.3452204","volume":"21","author":"R Hu","year":"2024","unstructured":"Hu, R., Yi, J., Chen, L., et al.: Graph reconstruction attention fusion network for multimodal sentiment analysis. IEEE Trans Ind Inform. 21(1), 1\u201310 (2024). https:\/\/doi.org\/10.1109\/TII.2024.3452204","journal-title":"JIEEE Trans Ind Inform."},{"issue":"2","key":"1868_CR12","doi-asserted-by":"publisher","first-page":"143","DOI":"10.1016\/j.joi.2009.01.003","volume":"3","author":"R Prabowo","year":"2009","unstructured":"Prabowo, R., Thelwall, M.: Sentiment analysis: a combined approach. J. Informetr. 3(2), 143\u2013157 (2009)","journal-title":"J. Informetr."},{"key":"1868_CR13","doi-asserted-by":"crossref","unstructured":"Dawei, W., Alfred, R., Obit, J.H., et al.: A literature review on text classification and sentiment analysis approaches. In: Computational Science and Technology: 7th ICCST 2020, Pattaya, Thailand, 29\u201330 August, 2020, pp. 305\u2013323 (2021)","DOI":"10.1007\/978-981-33-4069-5_26"},{"key":"1868_CR14","first-page":"79","volume-title":"Thumbs Up? Sentiment Classification using Machine Learning Techniques","author":"B Pang","year":"2002","unstructured":"Pang, B., Lee, L., Vaithyanathan, S.: Thumbs Up? Sentiment Classification using Machine Learning Techniques, pp. 79\u201386. Association for Computational Linguistics (2002)"},{"issue":"2","key":"1868_CR15","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1023\/A:1009715923555","volume":"2","author":"CJC Burges","year":"1998","unstructured":"Burges, C.J.C.: A tutorial on support vector machines for pattern recognition. Data Min. Knowl. Discov. 2(2), 121\u2013167 (1998)","journal-title":"Data Min. Knowl. Discov."},{"key":"1868_CR16","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1007\/s13042-010-0001-0","volume":"1","author":"Y Zhang","year":"2010","unstructured":"Zhang, Y., Jin, R., Zhou, Z.H.: Understanding bag-of-words model: a statistical framework. Int. J. Mach. Learn. Cybern. 1, 43\u201352 (2010)","journal-title":"Int. J. Mach. Learn. Cybern."},{"key":"1868_CR17","doi-asserted-by":"crossref","unstructured":"Hu, M., Liu, B.: Mining and summarizing customer reviews. In: Proceedings of the Tenth ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, pp. 168\u2013177 (2004)","DOI":"10.1145\/1014052.1014073"},{"key":"1868_CR18","doi-asserted-by":"publisher","first-page":"611","DOI":"10.1007\/s13244-018-0639-9","volume":"9","author":"R Yamashita","year":"2018","unstructured":"Yamashita, R., Nishio, M., Do, R.K.G., et al.: Convolutional neural networks: an overview and application in radiology. Insights Imaging 9, 611\u2013629 (2018)","journal-title":"Insights Imaging"},{"key":"1868_CR19","doi-asserted-by":"crossref","unstructured":"Tang, D., Qin, B., Liu, T.: Document modeling with gated recurrent neural network for sentiment classification. In: Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing, pp. 1422\u20131432 (2015)","DOI":"10.18653\/v1\/D15-1167"},{"key":"1868_CR20","doi-asserted-by":"publisher","first-page":"109259","DOI":"10.1016\/j.patcog.2022.109259","volume":"136","author":"D Wang","year":"2023","unstructured":"Wang, D., Guo, X., Tian, Y., et al.: TETFN: a text enhanced transformer fusion network for multimodal sentiment analysis. Pattern Recognit. 136, 109259 (2023)","journal-title":"Pattern Recognit."},{"key":"1868_CR21","first-page":"313","volume-title":"Multimodal Aspect-Based Sentiment Analysis Under Conditional Relation","author":"X Liu","year":"2025","unstructured":"Liu, X., Li, R., Ye, S., et al.: Multimodal Aspect-Based Sentiment Analysis Under Conditional Relation, pp. 313\u2013323. Association for Computational Linguistics, Abu Dhabi (2025)"},{"key":"1868_CR22","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12021","author":"A Zadeh","year":"2018","unstructured":"Zadeh, A., Liang, P.P., Mazumder, N., et al.: Memory fusion network for multi-view sequential learning. Proc. AAAI Conf. Artif. Intell. (2018). https:\/\/doi.org\/10.1609\/aaai.v32i1.12021","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"1868_CR23","doi-asserted-by":"crossref","unstructured":"Williams, J., Kleinegesse, S., Comanescu, R., et al. Recognizing emotions in video using multimodal DNN feature fusion. In: Proceedings of Grand Challenge and Workshop on Human Multimodal Language (Challenge-HML), pp. 11\u201319 (2018)","DOI":"10.18653\/v1\/W18-3302"},{"key":"1868_CR24","first-page":"606","volume-title":"Investigating Audio, Video, and Text Fusion Methods for End-to-End Automatic Personality Prediction","author":"O Kampman","year":"2018","unstructured":"Kampman, O., Barezi, E.J., Bertero, D., Fung, P.: Investigating Audio, Video, and Text Fusion Methods for End-to-End Automatic Personality Prediction, pp. 606\u2013611. Association for Computational Linguistics, Melbourne (2018)"},{"key":"1868_CR25","doi-asserted-by":"crossref","unstructured":"Han, W., Chen, H., Poria, S.: Improving Multimodal Fusion with Hierarchical Mutual Information Maximization for Multimodal Sentiment Analysis. Online and Punta Cana, Dominican Republic, pp. 9180\u20139192. Association for Computational Linguistics (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.723"},{"key":"1868_CR26","doi-asserted-by":"crossref","unstructured":"Tsai, Y.H.H., Bai, S., Liang, P.P., et al.: Multimodal transformer for unaligned multimodal language sequences. In: Proceedings of the Conference. Association for Computational Linguistics. Meeting. NIH Public Access, p. 6558 (2019)","DOI":"10.18653\/v1\/P19-1656"},{"key":"1868_CR27","doi-asserted-by":"crossref","unstructured":"Zhang, H., Wang, Y., Yin, G., et al.: Learning language-guided adaptive hyper-modality representation for multimodal sentiment analysis. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 756\u2013767 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.49"},{"key":"1868_CR28","first-page":"4611","volume-title":"Modal Feature Optimization Network with Prompt for Multimodal Sentiment Analysis","author":"X Zhang","year":"2025","unstructured":"Zhang, X., Wei, W., Zou, S.: Modal Feature Optimization Network with Prompt for Multimodal Sentiment Analysis, pp. 4611\u20134621. Association for Computational Linguistics, Abu Dhabi (2025)"},{"key":"1868_CR29","doi-asserted-by":"publisher","first-page":"101958","DOI":"10.1016\/j.inffus.2023.101958","volume":"100","author":"C Zhu","year":"2023","unstructured":"Zhu, C., Chen, M., Zhang, S., et al.: SKEAFN: sentiment knowledge enhanced attention fusion network for multimodal sentiment analysis. Inf. Fusion 100, 101958 (2023)","journal-title":"Inf. Fusion"},{"key":"1868_CR30","doi-asserted-by":"publisher","first-page":"111346","DOI":"10.1016\/j.knosys.2023.111346","volume":"285","author":"J Huang","year":"2024","unstructured":"Huang, J., Zhou, J., Tang, Z., et al.: TMBL: Transformer-based multimodal binding learning model for multimodal sentiment analysis. Knowl.-Based Syst. 285, 111346 (2024)","journal-title":"Knowl.-Based Syst."},{"key":"1868_CR31","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1016\/j.inffus.2022.11.022","volume":"92","author":"K Kim","year":"2023","unstructured":"Kim, K., Park, S.: AOBERT: all-modalities-in-One BERT for multimodal sentiment analysis. Inf. Fusion 92, 37\u201345 (2023)","journal-title":"Inf. Fusion"},{"issue":"2","key":"1868_CR32","first-page":"374","volume":"62","author":"Wu Luo Yuanyi","year":"2025","unstructured":"Luo Yuanyi, Wu., Rui, L.J., Xianglong, T.: Multimodal sentiment analysis method for sentimental semantic inconsistency. J. Comput. Res. Dev. 62(2), 374\u2013382 (2025)","journal-title":"J. Comput. Res. Dev."},{"key":"1868_CR33","doi-asserted-by":"publisher","first-page":"110502","DOI":"10.1016\/j.knosys.2023.110502","volume":"269","author":"C Huang","year":"2023","unstructured":"Huang, C., Zhang, J., Wu, X., et al.: TeFNA: text-centered fusion network with crossmodal attention for multimodal sentiment analysis. Knowl.-Based Syst. 269, 110502 (2023)","journal-title":"Knowl.-Based Syst."},{"issue":"6","key":"1868_CR34","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MIS.2016.94","volume":"31","author":"A Zadeh","year":"2016","unstructured":"Zadeh, A., Zellers, R., Pincus, E., et al.: MOSI: multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos. IEEE Intell. Syst. 31(6), 82\u201388 (2016)","journal-title":"IEEE Intell. Syst."},{"key":"1868_CR35","unstructured":"Zadeh, A.A.B, Liang, P.P., Poria, S., et al. Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2236\u20132246 (2018)"},{"key":"1868_CR36","doi-asserted-by":"publisher","first-page":"111982","DOI":"10.1016\/j.knosys.2024.111982","volume":"299","author":"C Gan","year":"2024","unstructured":"Gan, C., Tang, Y., Fu, X., et al.: Video multimodal sentiment analysis using cross-modal feature translation and dynamical propagation. Knowl.-Based Syst. 299, 111982 (2024)","journal-title":"Knowl.-Based Syst."},{"issue":"5","key":"1868_CR37","first-page":"8992","volume":"34","author":"Z Sun","year":"2020","unstructured":"Sun, Z., Sarma, P., Sethares, W., et al.: Learning relationships between text, audio, and video via deep canonical correlation for multimodal language analysis. Proc. AAAI Conf. Artif. Intell. 34(5), 8992\u20138999 (2020)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"issue":"2","key":"1868_CR38","doi-asserted-by":"publisher","first-page":"103229","DOI":"10.1016\/j.ipm.2022.103229","volume":"60","author":"H Lin","year":"2023","unstructured":"Lin, H., Zhang, P., Ling, J., et al.: PS-mixer: a polar-vector and strength-vector mixer model for multimodal sentiment analysis. Inf. Process. Manag. 60(2), 103229 (2023)","journal-title":"Inf. Process. Manag."}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01868-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-01868-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01868-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T09:03:05Z","timestamp":1757926985000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-01868-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,11]]},"references-count":38,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["1868"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-01868-5","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,11]]},"assertion":[{"value":"6 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 May 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"307"}}