{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T08:42:31Z","timestamp":1783068151307,"version":"3.54.6"},"reference-count":65,"publisher":"Springer Science and Business Media LLC","issue":"15","license":[{"start":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T00:00:00Z","timestamp":1759968000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T00:00:00Z","timestamp":1759968000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Social Science Fund of China","award":["24BYY045"],"award-info":[{"award-number":["24BYY045"]}]},{"name":"National Social Science Fund of China","award":["24BYY045"],"award-info":[{"award-number":["24BYY045"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-025-07930-3","type":"journal-article","created":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T13:00:38Z","timestamp":1760014838000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["An efficient utterance-level context-aware fusion architecture for large-scale audio-text sentiment analysis"],"prefix":"10.1007","volume":"81","author":[{"given":"Yuanxi","family":"Wei","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinpeng","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,9]]},"reference":[{"key":"7930_CR1","doi-asserted-by":"crossref","unstructured":"Aattouri I, Mouncif H, Rida M (2023) Call Center Customer Sentiment Analysis Using ML and NLP. In: 2023 14th International Conference on Intelligent Systems: Theories and Applications (SITA). p. 1\u20137","DOI":"10.1109\/SITA60746.2023.10373715"},{"key":"7930_CR2","unstructured":"Arjmand M, Dousti MJ, Moradi H (2021) TEASEL: A Transformer-Based Speech-Prefixed Language Model. CoRR. arXiv:2109.05522"},{"key":"7930_CR3","doi-asserted-by":"crossref","unstructured":"Asritha A, Dusyanth C, Sianyengele O, Suprith BV, Srinivas P, Kumar ML (2023) Efficient Speech Emotion Recognition for Resource-Constrained Devices. In: 2023 Second International Conference on Augmented Intelligence and Sustainable Systems (ICAISS) IEEE. p. 765\u2013772","DOI":"10.1109\/ICAISS58487.2023.10250693"},{"key":"7930_CR4","unstructured":"Baevski A, Hsu WN, Xu Q, Babu A, Gu J, Auli M (2022) Data2vec: A general framework for self-supervised learning in speech, vision and language. In: International Conference on Machine Learning PMLR; p. 1298\u20131312"},{"issue":"6","key":"7930_CR5","doi-asserted-by":"publisher","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","volume":"16","author":"S Chen","year":"2022","unstructured":"Chen S, Wang C, Wu Y, Liu S, Chen Z, Chen Z et al (2022) Wavlm: large-scale self-supervised pre-training for full stack speech processing. IEEE J Sel Top Signal Process 16(6):1505\u20131518","journal-title":"IEEE J Sel Top Signal Process"},{"key":"7930_CR6","unstructured":"Clark K, Luong MT, Le QV, Manning CD (2020) ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators. In: ICLR. https:\/\/openreview.net\/pdf?id=r1xMH1BtvB"},{"issue":"1","key":"7930_CR7","doi-asserted-by":"publisher","first-page":"74","DOI":"10.1109\/TAFFC.2015.2444846","volume":"7","author":"C Clavel","year":"2015","unstructured":"Clavel C, Callejas Z (2015) Sentiment analysis: from opinion mining to human-agent interaction. IEEE Trans Affect Comput 7(1):74\u201393","journal-title":"IEEE Trans Affect Comput"},{"key":"7930_CR8","doi-asserted-by":"crossref","unstructured":"Delbrouck JB, Tits N, Brousmiche M, Dupont S (2020) A transformer-based joint-encoding for emotion recognition and sentiment analysis. Preprint at arXiv:2006.15955","DOI":"10.18653\/v1\/2020.challengehml-1.1"},{"key":"7930_CR9","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K (2019) BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In: Burstein J, Doran C, Solorio T, editors. Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 Minneapolis, Minnesota: Association for Computational Linguistics. p. 4171\u20134186. https:\/\/aclanthology.org\/N19-1423"},{"issue":"1","key":"7930_CR10","first-page":"619","volume":"4","author":"S Ezzat","year":"2012","unstructured":"Ezzat S, Gayar N, Ghanem MM (2012) Sentiment analysis of call centre audio conversations using text classification. Int J Comput Inf Syst Ind Manag Appl 4(1):619\u2013627","journal-title":"Int J Comput Inf Syst Ind Manag Appl"},{"key":"7930_CR11","doi-asserted-by":"crossref","unstructured":"Gadzicki K, Khamsehashari R, Zetzsche C (2020) Early vs late fusion in multimodal convolutional neural networks. In: 2020 IEEE 23rd International Conference on Information Fusion (FUSION) IEEE. p. 1\u20136","DOI":"10.23919\/FUSION45008.2020.9190246"},{"key":"7930_CR12","doi-asserted-by":"publisher","first-page":"424","DOI":"10.1016\/j.inffus.2022.09.025","volume":"91","author":"A Gandhi","year":"2023","unstructured":"Gandhi A, Adhvaryu K, Poria S, Cambria E, Hussain A (2023) Multimodal sentiment analysis: a systematic review of history, datasets, multimodal fusion methods, applications, challenges and future directions. Inf Fusion 91:424\u2013444","journal-title":"Inf Fusion"},{"key":"7930_CR13","doi-asserted-by":"crossref","unstructured":"Glodek M, Tschechne S, Layher G, Schels M, Brosch T, Scherer S, et\u00a0al (2011) Multiple classifier systems for the classification of audio-visual emotional states. In: Affective Computing and Intelligent Interaction: Fourth International Conference, ACII 2011, Memphis, TN, USA, October 9\u201312, 2011, Proceedings, Part II Springer. p. 359\u2013368","DOI":"10.1007\/978-3-642-24571-8_47"},{"issue":"1","key":"7930_CR14","volume":"2015","author":"S Gong","year":"2015","unstructured":"Gong S, Dai Y, Ji J, Wang J, Sun H (2015) Emotion analysis of telephone complaints from customer based on affective computing. Comput Intell Neurosci 2015(1):506905","journal-title":"Comput Intell Neurosci"},{"key":"7930_CR15","doi-asserted-by":"crossref","unstructured":"Han W, Chen H, Poria S (2021) Improving Multimodal Fusion with Hierarchical Mutual Information Maximization for Multimodal Sentiment Analysis. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing Online and Punta Cana, Dominican Republic: Association for Computational Linguistics. p. 9180\u20139192. https:\/\/aclanthology.org\/2021.emnlp-main.723","DOI":"10.18653\/v1\/2021.emnlp-main.723"},{"key":"7930_CR16","doi-asserted-by":"crossref","unstructured":"Hazarika D, Zimmermann R, Poria S (2020) Misa: Modality-invariant and-specific representations for multimodal sentiment analysis. In: Proceedings of the 28th ACM International Conference on Multimedia. p. 1122\u20131131","DOI":"10.1145\/3394171.3413678"},{"issue":"02","key":"7930_CR17","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1142\/S0218488598000094","volume":"6","author":"S Hochreiter","year":"1998","unstructured":"Hochreiter S (1998) The vanishing gradient problem during learning recurrent neural nets and problem solutions. Internat J Uncertain Fuzziness Knowledge-Based Systems 6(02):107\u2013116","journal-title":"Internat J Uncertain Fuzziness Knowledge-Based Systems"},{"key":"7930_CR18","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"WN Hsu","year":"2021","unstructured":"Hsu WN, Bolte B, Tsai YHH, Lakhotia K, Salakhutdinov R, Mohamed A (2021) Hubert: self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans Audio Speech Lang Process 29:3451\u20133460","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"7930_CR19","doi-asserted-by":"crossref","unstructured":"Hu G, Lin TE, Zhao Y, Lu G, Wu Y, Li Y (2022) UniMSE: Towards Unified Multimodal Sentiment Analysis and Emotion Recognition. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing Abu Dhabi, United Arab Emirates: Association for Computational Linguistics. p. 7837\u20137851. https:\/\/aclanthology.org\/2022.emnlp-main.534","DOI":"10.18653\/v1\/2022.emnlp-main.534"},{"key":"7930_CR20","doi-asserted-by":"crossref","unstructured":"Huang J, Tao J, Liu B, Lian Z, Niu M (2020) Multimodal transformer fusion for continuous emotion recognition. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) IEEE. p. 3507\u20133511","DOI":"10.1109\/ICASSP40776.2020.9053762"},{"key":"7930_CR21","first-page":"8002","volume":"34","author":"W Jiao","year":"2020","unstructured":"Jiao W, Lyu M, King I (2020) Real-time emotion recognition via attention gated hierarchical memory network. Proc AAAI Conf Artif Intell 34:8002\u20138009","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"7930_CR22","doi-asserted-by":"crossref","unstructured":"Joshi A, Bhat A, Jain A, Singh A, Modi A (2022) COGMEN: COntextualized GNN based multimodal emotion recognitioN. In: Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies; p. 4148\u20134164","DOI":"10.18653\/v1\/2022.naacl-main.306"},{"issue":"4","key":"7930_CR23","doi-asserted-by":"publisher","first-page":"3893","DOI":"10.1109\/TCSVT.2024.3502134","volume":"35","author":"M Lei","year":"2025","unstructured":"Lei M, Fan J, Shao L, Song H, Xiao D, Ai D et al (2025) Double-shot 3d shape measurement with a dual-branch network for structured light projection profilometry. IEEE Trans Circuits Syst Video Technol 35(4):3893\u20133906. https:\/\/doi.org\/10.1109\/TCSVT.2024.3502134","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"issue":"4","key":"7930_CR24","doi-asserted-by":"publisher","first-page":"2954","DOI":"10.1109\/TAFFC.2023.3234777","volume":"14","author":"Y Lei","year":"2023","unstructured":"Lei Y, Cao H (2023) Audio-visual emotion recognition with preference learning based on intended and multi-modal perceived labels. IEEE Trans Affect Comput 14(4):2954\u20132969","journal-title":"IEEE Trans Affect Comput"},{"key":"7930_CR25","doi-asserted-by":"crossref","unstructured":"Li B, Fei H, Liao L, Zhao Y, Teng C, Chua TS, et\u00a0al (2023) Revisiting disentanglement and fusion on modality and context in conversational multimodal emotion recognition. In: Proceedings of the 31st ACM International Conference on Multimedia; p. 5923\u20135934","DOI":"10.1145\/3581783.3612053"},{"key":"7930_CR26","doi-asserted-by":"crossref","unstructured":"Li B, Dimitriadis D, Stolcke A (2019) Acoustic and lexical sentiment analysis for customer service calls. In: ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) IEEE. p. 5876\u20135880","DOI":"10.1109\/ICASSP.2019.8683679"},{"key":"7930_CR27","doi-asserted-by":"crossref","unstructured":"Li Y, Wang Y, Cui Z (2023) Decoupled multimodal distilling for emotion recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition; p. 6631\u20136640","DOI":"10.1109\/CVPR52729.2023.00641"},{"key":"7930_CR28","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1016\/j.neunet.2023.02.008","volume":"162","author":"Y Lin","year":"2023","unstructured":"Lin Y, Ji P, Chen X, He Z (2023) Lifelong text-audio sentiment analysis learning. Neural Netw 162:162\u2013174","journal-title":"Neural Netw"},{"key":"7930_CR29","doi-asserted-by":"crossref","unstructured":"Liu S, Qi L, Qin H, Shi J, Jia J (2018) Path aggregation network for instance segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition. p. 8759\u20138768","DOI":"10.1109\/CVPR.2018.00913"},{"key":"7930_CR30","unstructured":"Liu Y (2019) Roberta: A robustly optimized bert pretraining approach. Preprint at arXiv:1907.11692"},{"key":"7930_CR31","doi-asserted-by":"crossref","unstructured":"Liu Z, Shen Y, Lakshminarasimhan VB, Liang PP, Bagher\u00a0Zadeh A, Morency LP (2018) Efficient Low-rank Multimodal Fusion With Modality-Specific Factors. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) Melbourne, Australia: Association for Computational Linguistics. p. 2247\u20132256. https:\/\/aclanthology.org\/P18-1209","DOI":"10.18653\/v1\/P18-1209"},{"key":"7930_CR32","doi-asserted-by":"publisher","first-page":"20654","DOI":"10.1109\/ACCESS.2024.3361750","volume":"12","author":"A Mallol-Ragolta","year":"2024","unstructured":"Mallol-Ragolta A, Schuller B (2024) Coupling sentiment and arousal analysis towards an affective dialogue manager. IEEE Access 12:20654\u201320662. https:\/\/doi.org\/10.1109\/ACCESS.2024.3361750","journal-title":"IEEE Access"},{"key":"7930_CR33","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1016\/j.knosys.2016.05.032","volume":"108","author":"A Muhammad","year":"2016","unstructured":"Muhammad A, Wiratunga N, Lothian R (2016) Contextual sentiment analysis for social media genres. Knowl-Based Syst 108:92\u2013101","journal-title":"Knowl-Based Syst"},{"key":"7930_CR34","unstructured":"Ngiam J, Khosla A, Kim M, Nam J, Lee H, Ng AY (2011) Multimodal deep learning. In: Proceedings of the 28th international conference on machine learning (ICML-11); p. 689\u2013696"},{"issue":"2","key":"7930_CR35","doi-asserted-by":"publisher","first-page":"2887","DOI":"10.1007\/s11042-020-08836-3","volume":"80","author":"YR Pandeya","year":"2021","unstructured":"Pandeya YR, Lee J (2021) Deep learning-based late fusion of multimodal information for emotion classification of music video. Multimed Tools Appl 80(2):2887\u20132905","journal-title":"Multimed Tools Appl"},{"issue":"3","key":"7930_CR36","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MIS.2021.3057757","volume":"36","author":"W Peng","year":"2021","unstructured":"Peng W, Hong X, Zhao G (2021) Adaptive modality distillation for separable multimodal sentiment analysis. IEEE Intell Syst 36(3):82\u201389","journal-title":"IEEE Intell Syst"},{"key":"7930_CR37","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1016\/j.inffus.2017.02.003","volume":"37","author":"S Poria","year":"2017","unstructured":"Poria S, Cambria E, Bajpai R, Hussain A (2017) A review of affective computing: from unimodal analysis to multimodal fusion. Inf Fusion 37:98\u2013125. https:\/\/doi.org\/10.1016\/j.inffus.2017.02.003","journal-title":"Inf Fusion"},{"key":"7930_CR38","doi-asserted-by":"crossref","unstructured":"Poria S, Cambria E, Gelbukh A (2015) Deep Convolutional Neural Network Textual Features and Multiple Kernel Learning for Utterance-level Multimodal Sentiment Analysis. In: M\u00e0rquez L, Callison-Burch C, Su J, editors. Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing Lisbon, Portugal: Association for Computational Linguistics. p. 2539\u20132544. https:\/\/aclanthology.org\/D15-1303","DOI":"10.18653\/v1\/D15-1303"},{"key":"7930_CR39","doi-asserted-by":"crossref","unstructured":"Poria S, Cambria E, Hazarika D, Majumder N, Zadeh A, Morency LP (2017) Context-Dependent Sentiment Analysis in User-Generated Videos. In: Barzilay R, Kan MY, editors. Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) Vancouver, Canada: Association for Computational Linguistics. p. 873\u2013883. https:\/\/aclanthology.org\/P17-1081","DOI":"10.18653\/v1\/P17-1081"},{"key":"7930_CR40","doi-asserted-by":"crossref","unstructured":"Praveen RG, Alam J (2024) Recursive Joint Cross-Modal Attention for Multimodal Fusion in Dimensional Emotion Recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. p. 4803\u20134813","DOI":"10.1109\/CVPRW63382.2024.00483"},{"key":"7930_CR41","doi-asserted-by":"crossref","unstructured":"Rahman W, Hasan MK, Lee S, Zadeh A, Mao C, Morency LP, et\u00a0al (2020) Integrating multimodal information in large pretrained transformers. In: Proceedings of the conference. Association for Computational Linguistics. Meeting, vol. 2020 NIH Public Access. p. 2359","DOI":"10.18653\/v1\/2020.acl-main.214"},{"issue":"1","key":"7930_CR42","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1016\/j.ipm.2015.01.005","volume":"52","author":"H Saif","year":"2016","unstructured":"Saif H, He Y, Fernandez M, Alani H (2016) Contextual semantics for sentiment analysis of twitter. Inf Process Manag 52(1):5\u201319","journal-title":"Inf Process Manag"},{"key":"7930_CR43","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102382","volume":"108","author":"L Sun","year":"2024","unstructured":"Sun L, Lian Z, Liu B, Tao J (2024) Hicmae: hierarchical contrastive masked autoencoder for self-supervised audio-visual emotion recognition. Inf Fusion 108:102382","journal-title":"Inf Fusion"},{"key":"7930_CR44","first-page":"8992","volume":"34","author":"Z Sun","year":"2020","unstructured":"Sun Z, Sarma P, Sethares W, Liang Y (2020) Learning relationships between text, audio, and video via deep canonical correlation for multimodal language analysis. Proc AAAI Conf Artif Intell 34:8992\u20138999","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"7930_CR45","doi-asserted-by":"crossref","unstructured":"Syed ZS, Schroeter J, Sidorov KA, Marshall AD (2018) Computational Paralinguistics: Automatic Assessment of Emotions, Mood and Behavioural State from Acoustics of Speech. In: Interspeech. p. 511\u2013515","DOI":"10.21437\/Interspeech.2018-2019"},{"key":"7930_CR46","doi-asserted-by":"crossref","unstructured":"Tsai YHH, Bai S, Liang PP, Kolter JZ, Morency LP, Salakhutdinov R (2019) Multimodal transformer for unaligned multimodal language sequences. In: Proceedings of the conference. Association for computational linguistics. Meeting, vol. 2019 NIH Public Access. p. 6558","DOI":"10.18653\/v1\/P19-1656"},{"key":"7930_CR47","unstructured":"Tsai YH, Liang PP, Zadeh A, Morency L, Salakhutdinov R (2019) Learning Factorized Multimodal Representations. In: ICLR"},{"key":"7930_CR48","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, et\u00a0al (2017) Attention is all you need. Adv Neural Inf Process Syst. 30"},{"issue":"9","key":"7930_CR49","doi-asserted-by":"publisher","first-page":"10745","DOI":"10.1109\/TPAMI.2023.3263585","volume":"45","author":"J Wagner","year":"2023","unstructured":"Wagner J, Triantafyllopoulos A, Wierstorf H, Schmitt M, Burkhardt F, Eyben F et al (2023) Dawn of the transformer era in speech emotion recognition: closing the valence gap. IEEE Trans Pattern Anal Mach Intell 45(9):10745\u201310759. https:\/\/doi.org\/10.1109\/TPAMI.2023.3263585","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"7930_CR50","doi-asserted-by":"crossref","unstructured":"Waligora P, Aslam MH, Zeeshan MO, Belharbi S, Koerich AL, Pedersoli M, et\u00a0al (2024) Joint Multimodal Transformer for Emotion Recognition in the Wild. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition; p. 4625\u20134635","DOI":"10.1109\/CVPRW63382.2024.00465"},{"key":"7930_CR51","doi-asserted-by":"publisher","first-page":"4909","DOI":"10.1109\/TMM.2022.3183830","volume":"25","author":"D Wang","year":"2022","unstructured":"Wang D, Liu S, Wang Q, Tian Y, He L, Gao X (2022) Cross-modal enhancement network for multimodal sentiment analysis. IEEE Trans Multimed 25:4909\u20134921","journal-title":"IEEE Trans Multimed"},{"key":"7930_CR52","doi-asserted-by":"publisher","first-page":"4909","DOI":"10.1109\/TMM.2022.3183830","volume":"25","author":"D Wang","year":"2023","unstructured":"Wang D, Liu S, Wang Q, Tian Y, He L, Gao X (2023) Cross-modal enhancement network for multimodal sentiment analysis. IEEE Trans Multimed 25:4909\u20134921. https:\/\/doi.org\/10.1109\/TMM.2022.3183830","journal-title":"IEEE Trans Multimed"},{"key":"7930_CR53","doi-asserted-by":"crossref","unstructured":"Wolf T, Debut L, Sanh V, Chaumond J, Delangue C, Moi A, et\u00a0al (2020) Transformers: State-of-the-Art Natural Language Processing. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations Online: Association for Computational Linguistics. p. 38\u201345","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"7930_CR54","doi-asserted-by":"crossref","unstructured":"Wu Z, Gong Z, Koo J, Hirschberg J (2024) Multimodal Multi-loss Fusion Network for Sentiment Analysis. In: Duh K, Gomez H, Bethard S, editors. Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers) Mexico City, Mexico: Association for Computational Linguistics. p. 3588\u20133602. https:\/\/aclanthology.org\/2024.naacl-long.197","DOI":"10.18653\/v1\/2024.naacl-long.197"},{"key":"7930_CR55","doi-asserted-by":"crossref","unstructured":"Yang D, Huang S, Kuang H, Du Y, Zhang L (2022) Disentangled representation learning for multimodal emotion recognition. In: Proceedings of the 30th ACM International Conference on Multimedia. p. 1642\u20131651","DOI":"10.1145\/3503161.3547754"},{"key":"7930_CR56","doi-asserted-by":"crossref","unstructured":"Yang J, Wang Y, Yi R, Zhu Y, Rehman A, Zadeh A, et\u00a0al (2021) MTAG: Modal-Temporal Attention Graph for Unaligned Human Multimodal Language Sequences. In: Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies Online: Association for Computational Linguistics. p. 1009\u20131021. https:\/\/aclanthology.org\/2021.naacl-main.79","DOI":"10.18653\/v1\/2021.naacl-main.79"},{"key":"7930_CR57","doi-asserted-by":"crossref","unstructured":"Yang K, Xu H, Gao K (2020) Cm-bert: Cross-modal bert for text-audio sentiment analysis. In: Proceedings of the 28th ACM international conference on multimedia. p. 521\u2013528","DOI":"10.1145\/3394171.3413690"},{"key":"7930_CR58","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2025.3584834","author":"Z Yang","year":"2025","unstructured":"Yang Z, Gao K, Zhang Y, Zhang X, Hu Z, Wang J et al (2025) Dsfuse: a dual-diffusion structure for feature fidelity infrared and visible image fusion. IEEE Trans Neural Netw Learn Syst. https:\/\/doi.org\/10.1109\/TNNLS.2025.3584834","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"7930_CR59","doi-asserted-by":"crossref","unstructured":"Yu T, Gao H, Lin TE, Yang M, Wu Y, Ma W, et\u00a0al (2023) Speech-Text Pre-training for Spoken Dialog Understanding with Explicit Cross-Modal Alignment. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) Toronto, Canada: Association for Computational Linguistics. p. 7900\u20137913. https:\/\/aclanthology.org\/2023.acl-long.438","DOI":"10.18653\/v1\/2023.acl-long.438"},{"key":"7930_CR60","first-page":"10790","volume":"35","author":"W Yu","year":"2021","unstructured":"Yu W, Xu H, Yuan Z, Wu J (2021) Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis. Proc AAAI Conf Artif Intell 35:10790\u201310797","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"7930_CR61","doi-asserted-by":"publisher","first-page":"617","DOI":"10.1007\/s10115-018-1236-4","volume":"60","author":"L Yue","year":"2019","unstructured":"Yue L, Chen W, Li X, Zuo W, Yin M (2019) A survey of sentiment analysis in social media. Knowl Inf Syst 60:617\u2013663","journal-title":"Knowl Inf Syst"},{"key":"7930_CR62","doi-asserted-by":"crossref","unstructured":"Zadeh A, Chen M, Poria S, Cambria E, Morency LP (2017) Tensor fusion network for multimodal sentiment analysis. Preprint at arXiv:1707.07250","DOI":"10.18653\/v1\/D17-1115"},{"key":"7930_CR63","unstructured":"Zadeh A, Zellers R, Pincus E, Morency LP (2016) Mosi: multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos. Preprint at arXiv:1606.06259"},{"key":"7930_CR64","doi-asserted-by":"crossref","unstructured":"Zadeh AB, Liang PP, Poria S, Cambria E, Morency LP (2018) Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). p. 2236\u20132246","DOI":"10.18653\/v1\/P18-1208"},{"issue":"12","key":"7930_CR65","doi-asserted-by":"publisher","first-page":"16332","DOI":"10.1007\/s10489-022-03343-4","volume":"53","author":"Q Zhang","year":"2023","unstructured":"Zhang Q, Shi L, Liu P, Zhu Z, Xu L (2023) Retracted article: icdn: integrating consistency and difference networks by transformer for multimodal sentiment analysis. Appl Intell 53(12):16332\u201316345","journal-title":"Appl Intell"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07930-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-025-07930-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07930-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,10]],"date-time":"2025-10-10T01:02:55Z","timestamp":1760058175000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-025-07930-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,9]]},"references-count":65,"journal-issue":{"issue":"15","published-online":{"date-parts":[[2025,10]]}},"alternative-id":["7930"],"URL":"https:\/\/doi.org\/10.1007\/s11227-025-07930-3","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,9]]},"assertion":[{"value":"10 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 September 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 October 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"1428"}}