{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T04:48:51Z","timestamp":1769921331172,"version":"3.49.0"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"18","license":[{"start":{"date-parts":[[2024,3,27]],"date-time":"2024-03-27T00:00:00Z","timestamp":1711497600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,3,27]],"date-time":"2024-03-27T00:00:00Z","timestamp":1711497600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61672190"],"award-info":[{"award-number":["61672190"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s00521-024-09644-8","type":"journal-article","created":{"date-parts":[[2024,3,27]],"date-time":"2024-03-27T13:03:37Z","timestamp":1711544617000},"page":"10799-10809","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Semantic-specific multimodal relation learning for sentiment analysis"],"prefix":"10.1007","volume":"36","author":[{"given":"Rui","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"YuanYi","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"JiaFeng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"XiangLong","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,3,27]]},"reference":[{"key":"9644_CR1","doi-asserted-by":"crossref","unstructured":"Sahay S, Oku E, Kumar SH, Nachman L (2020) Low rank fusion based transformers for multimodal sequences. In: International conference on association for computational linguistics, pp 29\u201334","DOI":"10.18653\/v1\/2020.challengehml-1.4"},{"key":"9644_CR2","doi-asserted-by":"crossref","unstructured":"Tsai YH, Bai S, Liang PP, Kolter JZ, Morency L, Salakhutdinov R (2019) Multimodal transformer for unaligned multimodal language sequences. In: International conference on association for computational linguistics, pp 6558\u20136569","DOI":"10.18653\/v1\/P19-1656"},{"key":"9644_CR3","doi-asserted-by":"crossref","unstructured":"Sun Z, Sarma P, Sethares W, Liang Y (2020) Learning relationships between text, audio, and video via deep canonical correlation for multimodal language analysis. In: Conference on artificial intelligence, vol 34, pp 8992\u20138999","DOI":"10.1609\/aaai.v34i05.6431"},{"key":"9644_CR4","doi-asserted-by":"publisher","unstructured":"Hazarika D, Zimmermann R, Poria S (2020) MISA: modality-invariant and -specific representations for multimodal sentiment analysis. In: International conference on multimedia (MM), pp 1122\u20131131. https:\/\/doi.org\/10.1145\/3394171.3413678","DOI":"10.1145\/3394171.3413678"},{"key":"9644_CR5","doi-asserted-by":"crossref","unstructured":"Zhu H, Zheng Z, Soleymani M, Nevatia R (2022) Self-supervised learning for sentiment analysis via image-text matching. In: IEEE international conference on acoustics, speech and signal processing, pp 1710\u20131714","DOI":"10.1109\/ICASSP43922.2022.9747819"},{"key":"9644_CR6","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1016\/j.inffus.2022.11.022","volume":"92","author":"K Kim","year":"2023","unstructured":"Kim K, Park S (2023) AOBERT: all-modalities-in-one BERT for multimodal sentiment analysis. Inf Fusion 92:37\u201345","journal-title":"Inf Fusion"},{"key":"9644_CR7","doi-asserted-by":"publisher","unstructured":"Devlin J, Chang MW, Lee K (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: North American chapter of the association for computational linguistics: human language technologies, vol 1, pp 4171\u20134186. https:\/\/doi.org\/10.18653\/v1\/n19-1423","DOI":"10.18653\/v1\/n19-1423"},{"key":"9644_CR8","doi-asserted-by":"publisher","first-page":"1650","DOI":"10.1109\/LSP.2021.3101421","volume":"28","author":"Y Sun","year":"2021","unstructured":"Sun Y, Mai S, Hu H (2021) Learning to balance the learning rates between various modalities via adaptive tracking factor. IEEE Signal Process Lett 28:1650\u20131654","journal-title":"IEEE Signal Process Lett"},{"key":"9644_CR9","unstructured":"Chen Z, Badrinarayanan V, Lee CY, Rabinovich A (2018) Gradient normalization for adaptive loss balancing in deep multitask networks. In: International conference on machine learning, vol 80, pp 794\u2013803"},{"key":"9644_CR10","doi-asserted-by":"crossref","unstructured":"Liu B, Zhang L (2012) A survey of opinion mining and sentiment analysis. In: Mining text data, pp 415\u2013463","DOI":"10.1007\/978-1-4614-3223-4_13"},{"key":"9644_CR11","doi-asserted-by":"publisher","first-page":"110021","DOI":"10.1016\/j.knosys.2022.110021","volume":"258","author":"J Ye","year":"2022","unstructured":"Ye J, Zhou J, Tian J, Wang R, Zhou J, Gui T (2022) Sentiment-aware multimodal pre-training for multimodal sentiment analysis. Knowl Based Syst 258:110021. https:\/\/doi.org\/10.1016\/j.knosys.2022.110021","journal-title":"Knowl Based Syst"},{"issue":"10","key":"9644_CR12","doi-asserted-by":"publisher","first-page":"11184","DOI":"10.1007\/s10489-021-02936-9","volume":"52","author":"W Liao","year":"2022","unstructured":"Liao W, Zeng B, Liu J, Wei P, Fang J (2022) Image-text interaction graph neural network for image-text sentiment analysis. Appl Intell 52(10):11184\u201311198. https:\/\/doi.org\/10.1007\/s10489-021-02936-9","journal-title":"Appl Intell"},{"key":"9644_CR13","doi-asserted-by":"crossref","unstructured":"Cambria E, Hazarika D, Poria S, Hussain A (2017) Benchmarking multimodal sentiment analysis. In: Computational linguistics and intelligent text processing, pp 17\u201323","DOI":"10.1007\/978-3-319-77116-8_13"},{"key":"9644_CR14","doi-asserted-by":"publisher","unstructured":"Liu Z, Shen Y, Lakshminarasimhan VB, Liang PP, Zadeh A, Morency LP (2018) Efficient low-rank multimodal fusion with modality-specific factors. In: Proceedings of the 56th annual meeting of the association for computational linguistics (Long Papers), vol 1, pp 2247\u20132256. https:\/\/doi.org\/10.18653\/v1\/p18-1209","DOI":"10.18653\/v1\/p18-1209"},{"issue":"21","key":"9644_CR15","doi-asserted-by":"publisher","first-page":"18391","DOI":"10.1007\/s00521-022-07451-7","volume":"34","author":"M Salur","year":"2022","unstructured":"Salur M, Aydin I (2022) A soft voting ensemble learning-based approach for multimodal sentiment analysis. Neural Comput Appl 34(21):18391\u201318406. https:\/\/doi.org\/10.1007\/s00521-022-07451-7","journal-title":"Neural Comput Appl"},{"key":"9644_CR16","doi-asserted-by":"publisher","unstructured":"Sahay S, Kumar SH, Xia R, Huang J, Nachman L (2018) Multimodal relational tensor network for sentiment and emotion classification. In: 1st grand challenge and workshop on human multimodal language, pp 20\u201327. https:\/\/doi.org\/10.18653\/v1\/W18-3303","DOI":"10.18653\/v1\/W18-3303"},{"key":"9644_CR17","doi-asserted-by":"publisher","first-page":"124","DOI":"10.1016\/j.knosys.2018.07.041","volume":"161","author":"N Majumder","year":"2018","unstructured":"Majumder N, Hazarika D, Gelbukh A, Cambria E, Poria S (2018) Multimodal sentiment analysis using hierarchical fusion with context modeling. Knowl Based Syst 161:124\u2013133","journal-title":"Knowl Based Syst"},{"key":"9644_CR18","doi-asserted-by":"publisher","unstructured":"ZadehD A, Chen M, Poria S, Cambria E, Morency LP (2017) Tensor fusion network for multimodal sentiment analysis. In: Proceedings of the 2017 conference on empirical methods in natural language processing, pp 1103\u20131114. https:\/\/doi.org\/10.18653\/v1\/d17-1115","DOI":"10.18653\/v1\/d17-1115"},{"key":"9644_CR19","doi-asserted-by":"publisher","unstructured":"Rahman W, Hasan MK, Lee S, Zadeh A, Mao CF, Morency LP, Hoque E (2020) Integrating multimodal information in large pretrained transformers. In: 58th annual meeting of the association-for-computational-linguistics, pp 2359\u20132369. https:\/\/doi.org\/10.48550\/arXiv.1908.05787","DOI":"10.48550\/arXiv.1908.05787"},{"key":"9644_CR20","unstructured":"Sahay S, Kumar SH, Xia R, Huang J, Nachman L (2021) Improving multimodal fusion with hierarchical mutual information maximization for multimodal sentiment analysis. In: Empirical methods in natural language processing, pp 9180\u20139192"},{"key":"9644_CR21","unstructured":"Zijie L, Bin L, Yunfei L, Yixue D, Min Y, Min Z, Ruifeng X (2022) Modeling intra- and inter-modal relations: Hierarchical graph contrastive learning for multimodal sentiment analysis. In: International conference on computational linguistics, pp 7124\u20137135"},{"key":"9644_CR22","first-page":"289","volume":"29","author":"JS Lu","year":"2016","unstructured":"Lu JS, Yang JW, Batra D, Parikh D (2016) Hierarchical question-image co-attention for visual question answering. Adv Neural Inf Process Syst 29:289\u2013297","journal-title":"Adv Neural Inf Process Syst"},{"key":"9644_CR23","doi-asserted-by":"crossref","unstructured":"Yu Z, Yu J, Cui YH, Tao DC, Tian Q (2019) Deep modular co-attention networks for visual question answering. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 6281\u20136290","DOI":"10.1109\/CVPR.2019.00644"},{"key":"9644_CR24","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1007\/s41095-022-0271-y","volume":"8","author":"MH Guo","year":"2022","unstructured":"Guo MH, Xu TX, Liu JJ, Liu ZN, Jiang PT, Mu TJ, Zhang SH, Martin RR, Cheng M, Hu SM (2022) Attention mechanisms in computer vision: a survey. Comput Vis Media 8:331\u2013368","journal-title":"Comput Vis Media"},{"key":"9644_CR25","unstructured":"Vaswani A, Shazeer N, Parmar Nea (2017) Attention is all you need. In: Advances in neural information processing systems, vol 30"},{"key":"9644_CR26","doi-asserted-by":"crossref","unstructured":"Long X, Gan C, Melo G, Liu X, Li YD, Li F, Wen SL (2018) Multimodal keyless attention fusion for video classification. In: 32nd conference on artificial intelligence, pp 7202\u20137209","DOI":"10.1609\/aaai.v32i1.12319"},{"key":"9644_CR27","doi-asserted-by":"publisher","first-page":"2015","DOI":"10.1109\/TASLP.2022.3178204","volume":"30","author":"B Yang","year":"2022","unstructured":"Yang B, Wu LJ, Zhu JH, Shao B, Lin XL, Liu TY (2022) Multimodal sentiment analysis with two-phase multi-task learning. IEEE\/ACM Trans Audio Speech Lang Process 30:2015\u20132024. https:\/\/doi.org\/10.1109\/TASLP.2022.3178204","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"9644_CR28","doi-asserted-by":"publisher","first-page":"2488","DOI":"10.1109\/TMM.2021.3082398","volume":"24","author":"SJ Mai","year":"2022","unstructured":"Mai SJ, Hu HF, Xing SL (2022) A unimodal representation learning and recurrent decomposition fusion structure for utterance-level multimodal embedding learning. IEEE Trans Multimed 24:2488\u20132501. https:\/\/doi.org\/10.1109\/TMM.2021.3082398","journal-title":"IEEE Trans Multimed"},{"issue":"10","key":"9644_CR29","doi-asserted-by":"publisher","first-page":"11539","DOI":"10.1007\/s10489-021-02966-3","volume":"52","author":"Y Wu","year":"2022","unstructured":"Wu Y, Li W (2022) Aspect-level sentiment classification based on location and hybrid multi attention mechanism. Appl Intell 52(10):11539\u201311554. https:\/\/doi.org\/10.1007\/s10489-021-02966-3","journal-title":"Appl Intell"},{"key":"9644_CR30","unstructured":"Ba LJ, Kiros JR, Hinton GE (2016) Layer normalization. arxiv:1607.06450"},{"issue":"6","key":"9644_CR31","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MIS.2016.94","volume":"31","author":"A Zadeh","year":"2016","unstructured":"Zadeh A, Zellers R, Pincus E, Morency LP (2016) Multimodal sentiment intensity analysis in videos: facial gestures and verbal messages. IEEE Intell Syst 31(6):82\u201388. https:\/\/doi.org\/10.1109\/MIS.2016.94","journal-title":"IEEE Intell Syst"},{"key":"9644_CR32","unstructured":"Zadeh A, Liang PP, Vanbriesen J, Poria S, Tong E, Cambria E, Chen MH, Morency LP (2018) Multimodal language analysis in the wild: CMU-MOSEI dataset and interpretable dynamic fusion graph. In: 56th annual meeting of the association for computational linguistics, vol 1, pp 2236\u20132246"},{"key":"9644_CR33","doi-asserted-by":"crossref","unstructured":"Yu W, Xu H, Meng F, Zhu YL, Ma YX, Wu JL, JY, Z, Yang KC (2020) CH-SIMS: a Chinese multimodal sentiment analysis dataset with fine-grained annotation of modality. In: 58th annual meeting of the association for computational linguistics, pp 3718\u20133727","DOI":"10.18653\/v1\/2020.acl-main.343"},{"key":"9644_CR34","doi-asserted-by":"publisher","unstructured":"Yu W, Xu H, Ziqi Y, Jiele W (2021) Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis. In: Proceedings of the conference on artificial intelligence. https:\/\/doi.org\/10.48550\/arXiv.2102.04830","DOI":"10.48550\/arXiv.2102.04830"},{"key":"9644_CR35","doi-asserted-by":"crossref","unstructured":"Mao H, Yuan Z, Xu H, Yu W, Gao K (2022) M-SENA: an integrated platform for multimodal sentiment analysis. In: Association for computational linguistics, pp 204\u2013213","DOI":"10.18653\/v1\/2022.acl-demo.20"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-09644-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-024-09644-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-09644-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,30]],"date-time":"2024-05-30T20:32:21Z","timestamp":1717101141000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-024-09644-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3,27]]},"references-count":35,"journal-issue":{"issue":"18","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["9644"],"URL":"https:\/\/doi.org\/10.1007\/s00521-024-09644-8","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,3,27]]},"assertion":[{"value":"18 June 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 February 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 March 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The data that support the findings of this study are openly available.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and informed consent for data used"}}]}}