{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,19]],"date-time":"2026-08-19T19:02:09Z","timestamp":1787166129518,"version":"build-2736575974"},"reference-count":77,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276073"],"award-info":[{"award-number":["62276073"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61966004"],"award-info":[{"award-number":["61966004"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information Fusion"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.inffus.2026.104436","type":"journal-article","created":{"date-parts":[[2026,4,28]],"date-time":"2026-04-28T07:09:04Z","timestamp":1777360144000},"page":"104436","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["MPF: A multi-level perceiving framework for multimodal sarcasm detection"],"prefix":"10.1016","volume":"135","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-6461-3266","authenticated-orcid":false,"given":"Xingjie","family":"Zhuang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5313-6134","authenticated-orcid":false,"given":"Zhixin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6913-8223","authenticated-orcid":false,"given":"Fengling","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4375-1405","authenticated-orcid":false,"given":"Canlong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5104-8982","authenticated-orcid":false,"given":"Huifang","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.inffus.2026.104436_bib0001","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","first-page":"10006","article-title":"MMoE: enhancing multimodal models with mixtures of multimodal interaction experts","author":"Yu","year":"2024"},{"issue":"1","key":"10.1016\/j.inffus.2026.104436_bib0002","first-page":"1","article-title":"M3GAT: a multi-modal, multi-task interactive graph attention network for conversational sentiment analysis and emotion recognition","volume":"42","author":"Zhang","year":"2023","journal-title":"ACM Trans. Inf. Syst."},{"issue":"5","key":"10.1016\/j.inffus.2026.104436_bib0003","first-page":"1","article-title":"Self-adaptive representation learning model for multi-modal sentiment and sarcasm joint analysis","volume":"20","author":"Zhang","year":"2024","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"10.1016\/j.inffus.2026.104436_bib0004","doi-asserted-by":"crossref","first-page":"282","DOI":"10.1016\/j.inffus.2023.01.005","article-title":"A multitask learning model for multimodal sarcasm, sentiment and emotion recognition in conversations","volume":"93","author":"Zhang","year":"2023","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104436_bib0005","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113029","article-title":"DyCR-Net: a dynamic context-aware routing network for multi-modal sarcasm detection in conversation","volume":"310","author":"Zhuang","year":"2025","journal-title":"Knowl. Based Syst."},{"key":"10.1016\/j.inffus.2026.104436_bib0006","unstructured":"H. Lin, Z. Luo, B. Wang, R. Yang, J. Ma, Goat-bench: safety insights to large multimodal models through meme-based social abuse, arXiv: 2401.01523(2024)."},{"key":"10.1016\/j.inffus.2026.104436_bib0007","series-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics","first-page":"1767","article-title":"Multi-modal sarcasm detection via cross-modal graph convolutional network","author":"Liang","year":"2022"},{"key":"10.1016\/j.inffus.2026.104436_bib0008","series-title":"Proceedings of the 10th CCF International Conference on Natural Language Processing and Chinese Computing","first-page":"822","article-title":"Multi-modal sarcasm detection based on contrastive attention mechanism","author":"Zhang","year":"2021"},{"issue":"1","key":"10.1016\/j.inffus.2026.104436_bib0009","doi-asserted-by":"crossref","first-page":"326","DOI":"10.1109\/TAFFC.2023.3279145","article-title":"A quantum probability driven framework for joint multi-modal sarcasm, sentiment and emotion analysis","volume":"15","author":"Liu","year":"2024","journal-title":"IEEE Trans. Affect. Comput."},{"issue":"3","key":"10.1016\/j.inffus.2026.104436_bib0010","doi-asserted-by":"crossref","first-page":"1349","DOI":"10.1109\/TAI.2023.3298328","article-title":"Learning multi-task commonness and uniqueness for multi-modal sarcasm detection and sentiment analysis in conversation","volume":"5","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Artif. Intell."},{"key":"10.1016\/j.inffus.2026.104436_bib0011","series-title":"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation","first-page":"298","article-title":"Action and reaction go hand in hand! A multi-modal dialogue act aided sarcasm identification","author":"Tomar","year":"2024"},{"key":"10.1016\/j.inffus.2026.104436_bib0012","unstructured":"X. Yang, W. Wu, S. Feng, M. Wang, D. Wang, Y. Li, Q. Sun, Y. Zhang, X. Fu, S. Poria, MM-BigBench: evaluating multimodal models on multimodal content comprehension tasks, arXiv: 2310.09036(2023)."},{"key":"10.1016\/j.inffus.2026.104436_bib0013","unstructured":"Q. Sun, X. Cui, W. Zhou, H. Li, Exploiting GPT-4 vision for zero-shot point cloud understanding, arXiv: 2401.07572(2024)."},{"key":"10.1016\/j.inffus.2026.104436_bib0014","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"26296","article-title":"Improved baselines with visual instruction tuning","author":"Liu","year":"2024"},{"issue":"1","key":"10.1016\/j.inffus.2026.104436_bib0015","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1163\/187847510X488603","article-title":"Multisensory processing in review: from physiology to behaviour","volume":"23","author":"Alais","year":"2010","journal-title":"Seeing Perceiving"},{"key":"10.1016\/j.inffus.2026.104436_bib0016","series-title":"Proceedings of the 33rd ACM International Conference on Information and Knowledge Management","first-page":"3602","article-title":"MV-BART: multi-view BART for multi-modal sarcasm detection","author":"Zhuang","year":"2024"},{"key":"10.1016\/j.inffus.2026.104436_bib0017","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102272","article-title":"Bi-stream graph learning based multimodal fusion for emotion recognition in conversation","volume":"106","author":"Lu","year":"2024","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104436_bib0018","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102590","article-title":"Adversarial alignment and graph fusion via information bottleneck for multimodal emotion recognition in conversations","volume":"112","author":"Shou","year":"2024","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104436_bib0019","series-title":"Proceedings of the 2024 IEEE International Conference on Data Mining","first-page":"971","article-title":"Multi-modal sarcasm detection via dual synergetic perception graph convolutional networks","author":"Zhuang","year":"2024"},{"key":"10.1016\/j.inffus.2026.104436_bib0020","doi-asserted-by":"crossref","unstructured":"Y. Shou, T. Meng, W. Ai, N. Yin, K. Li, A comprehensive survey on multi-modal conversational emotion recognition with deep learning, arXiv: 2312.05735(2023).","DOI":"10.2139\/ssrn.5017731"},{"issue":"12","key":"10.1016\/j.inffus.2026.104436_bib0021","doi-asserted-by":"crossref","first-page":"6472","DOI":"10.1109\/TAI.2024.3445325","article-title":"Deep imbalanced learning for multimodal emotion recognition in conversations","volume":"5","author":"Meng","year":"2024","journal-title":"IEEE Trans. Artif. Intell."},{"key":"10.1016\/j.inffus.2026.104436_bib0022","series-title":"Proceedings of the 21th Annual Meeting of the Special Interest Group on Discourse and Dialogue","first-page":"186","article-title":"Contextualized emotion recognition in conversation as sequence tagging","author":"Wang","year":"2020"},{"key":"10.1016\/j.inffus.2026.104436_bib0023","doi-asserted-by":"crossref","unstructured":"D. Ghosal, N. Majumder, A. Gelbukh, R. Mihalcea, S. Poria, Cosmic: commonsense knowledge for emotion identification in conversations, arXiv: 2010.02795(2020).","DOI":"10.18653\/v1\/2020.findings-emnlp.224"},{"issue":"5","key":"10.1016\/j.inffus.2026.104436_bib0024","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3722115","article-title":"Multi-modal sarcasm detection via knowledge-aware focused graph convolutional networks","volume":"21","author":"Zhuang","year":"2025","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"10.1016\/j.inffus.2026.104436_bib0025","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.127428","article-title":"A survey of automatic sarcasm detection: fundamental theories, formulation, datasets, detection methods, and opportunities","volume":"578","author":"Chen","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.inffus.2026.104436_bib0026","series-title":"Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval","first-page":"1844","article-title":"Affective dependency graph for sarcasm detection","author":"Lou","year":"2021"},{"key":"10.1016\/j.inffus.2026.104436_bib0027","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"4702","article-title":"Augmenting affective dependency graph via iterative incongruity graph learning for sarcasm detection","volume":"37","author":"Wang","year":"2023"},{"key":"10.1016\/j.inffus.2026.104436_bib0028","doi-asserted-by":"crossref","unstructured":"J. Plepi, L. Flek, Perceived and intended sarcasm detection with graph attention networks, arXiv: 2110.04001(2021).","DOI":"10.18653\/v1\/2021.findings-emnlp.408"},{"key":"10.1016\/j.inffus.2026.104436_bib0029","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics","first-page":"2468","article-title":"Dynamic routing transformer network for multimodal sarcasm detection","author":"Tian","year":"2023"},{"key":"10.1016\/j.inffus.2026.104436_bib0030","series-title":"Proceedings of the 32nd ACM International Conference on Multimedia","first-page":"5810","article-title":"Sentiment-oriented sarcasm integration for video sentiment analysis enhancement with sarcasm assistance","author":"Fang","year":"2024"},{"key":"10.1016\/j.inffus.2026.104436_bib0031","unstructured":"K. Ouyang, L. Jing, X. Song, M. Liu, Y. Hu, L. Nie, Sentiment-enhanced graph-based sarcasm explanation in dialogue, arXiv: 2402.03658(2024)."},{"key":"10.1016\/j.inffus.2026.104436_bib0032","first-page":"12449","article-title":"wav2vec 2.0: a framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104436_bib0033","series-title":"Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5206","article-title":"LibriSpeech: an ASR corpus based on public domain audio books","author":"Panayotov","year":"2015"},{"issue":"1","key":"10.1016\/j.inffus.2026.104436_bib0034","first-page":"7","article-title":"Librivox: free public domain audiobooks","volume":"28","author":"Kearns","year":"2014","journal-title":"Ref. Rev."},{"key":"10.1016\/j.inffus.2026.104436_bib0035","series-title":"SciPy","first-page":"18","article-title":"librosa: audio and music signal analysis in Python","author":"McFee","year":"2015"},{"key":"10.1016\/j.inffus.2026.104436_bib0036","doi-asserted-by":"crossref","unstructured":"S. Castro, D. Hazarika, V. P\u00e9rez-Rosas, R. Zimmermann, R. Mihalcea, S. Poria, Towards multimodal sarcasm detection (an _obviously_ perfect paper), arXiv: 1906.01815(2019).","DOI":"10.18653\/v1\/P19-1455"},{"key":"10.1016\/j.inffus.2026.104436_bib0037","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.inffus.2026.104436_bib0038","series-title":"Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics","first-page":"7871","article-title":"BART: denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension","author":"Lewis","year":"2020"},{"key":"10.1016\/j.inffus.2026.104436_bib0039","series-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","first-page":"5998","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.inffus.2026.104436_bib0040","unstructured":"A. Dosovitskiy, An image is worth 16x16 words: transformers for image recognition at scale, arXiv: 2010.11929(2020)."},{"key":"10.1016\/j.inffus.2026.104436_bib0041","first-page":"12116","article-title":"Do vision transformers see like convolutional neural networks?","volume":"34","author":"Raghu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104436_bib0042","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"5807","article-title":"Scratching visual transformer\u2019s back with uniform attention","author":"Hyeon-Woo","year":"2023"},{"key":"10.1016\/j.inffus.2026.104436_bib0043","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102787","article-title":"Multimodal sentiment analysis with unimodal label generation and modality decomposition","volume":"116","author":"Zhu","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104436_bib0044","doi-asserted-by":"crossref","first-page":"726","DOI":"10.1162\/tacl_a_00343","article-title":"Multilingual denoising pre-training for neural machine translation","volume":"8","author":"Liu","year":"2020","journal-title":"Trans. Assoc. Comput. Linguist."},{"issue":"9","key":"10.1016\/j.inffus.2026.104436_bib0045","doi-asserted-by":"crossref","first-page":"5762","DOI":"10.1109\/TCSVT.2022.3155795","article-title":"Adaptive path selection for dynamic image captioning","volume":"32","author":"Xian","year":"2022","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.inffus.2026.104436_bib0046","doi-asserted-by":"crossref","first-page":"6917","DOI":"10.1109\/TIP.2021.3099733","article-title":"HCE: hierarchical context embedding for region-based object detection","volume":"30","author":"Chen","year":"2021","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.inffus.2026.104436_bib0047","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"2074","article-title":"Trar: routing the attention spans in transformer for visual question answering","author":"Zhou","year":"2021"},{"key":"10.1016\/j.inffus.2026.104436_bib0048","series-title":"Proceedings of the 2018 IEEE Symposium Series on Computational Intelligence","first-page":"1292","article-title":"BabelSenticNet: a commonsense reasoning framework for multilingual sentiment analysis","author":"Vilares","year":"2018"},{"key":"10.1016\/j.inffus.2026.104436_bib0049","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2021.107597","article-title":"Multi-view informed attention-based model for irony and satire detection in Spanish variants","volume":"235","author":"Ortega-Bueno","year":"2022","journal-title":"Knowl. Based Syst."},{"key":"10.1016\/j.inffus.2026.104436_bib0050","series-title":"Proceedings of the 2024 International Conference on Information Technology Research and Innovation","first-page":"65","article-title":"Investigating ChatGPT on reddit by using lexicon-based sentiment analysis","author":"Putra","year":"2024"},{"key":"10.1016\/j.inffus.2026.104436_bib0051","series-title":"Proceedings of the 5th International Conference on Language Resources and Evaluation","first-page":"417","article-title":"SentiwordNet: a publicly available lexical resource for opinion mining","author":"Sebastiani","year":"2006"},{"key":"10.1016\/j.inffus.2026.104436_bib0052","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"2486","article-title":"A joint cross-attention model for audio-visual fusion in dimensional emotion recognition","author":"Praveen","year":"2022"},{"key":"10.1016\/j.inffus.2026.104436_bib0053","doi-asserted-by":"crossref","unstructured":"A. Ray, S. Mishra, A. Nunna, P. Bhattacharyya, A multimodal corpus for emotion recognition in sarcasm, arXiv: 2206.02119(2022).","DOI":"10.63317\/3d3nbg2do7ny"},{"key":"10.1016\/j.inffus.2026.104436_bib0054","doi-asserted-by":"crossref","unstructured":"M.K. Hasan, W. Rahman, A. Zadeh, J. Zhong, M.I. Tanveer, L.-P. Morency, et al., UR-FUNNY: a multimodal language dataset for understanding humor, arXiv: 1904.06618(2019).","DOI":"10.18653\/v1\/D19-1211"},{"key":"10.1016\/j.inffus.2026.104436_bib0055","unstructured":"D.P. Kingma, Adam: a method for stochastic optimization, arXiv: 1412.6980(2014)."},{"key":"10.1016\/j.inffus.2026.104436_bib0056","unstructured":"J. Devlin, M.-W. Chang, K. Lee, K. Toutanova, BERT: pre-training of deep bidirectional transformers for language understanding, arXiv: 1810.04805(2018)."},{"issue":"23","key":"10.1016\/j.inffus.2026.104436_bib0057","doi-asserted-by":"crossref","first-page":"17309","DOI":"10.1007\/s00521-020-05102-3","article-title":"A transformer-based approach to irony and sarcasm detection","volume":"32","author":"Potamias","year":"2020","journal-title":"Neural Comput. Appl."},{"key":"10.1016\/j.inffus.2026.104436_bib0058","series-title":"Proceedings of the International Conference on Machine Learning","first-page":"6105","article-title":"EfficientNet: rethinking model scaling for convolutional neural networks","author":"Tan","year":"2019"},{"key":"10.1016\/j.inffus.2026.104436_bib0059","series-title":"Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics","first-page":"4351","article-title":"Sentiment and emotion help sarcasm? a multi-task learning framework for multi-modal sarcasm, sentiment and emotion analysis","author":"Chauhan","year":"2020"},{"key":"10.1016\/j.inffus.2026.104436_bib0060","first-page":"24206","article-title":"Vatt: transformers for multimodal self-supervised learning from raw video, audio and text","volume":"34","author":"Akbari","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104436_bib0061","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"3930","article-title":"Multimodal learning using optimal transport for sarcasm and humor detection","author":"Pramanick","year":"2022"},{"issue":"5","key":"10.1016\/j.inffus.2026.104436_bib0062","doi-asserted-by":"crossref","first-page":"5740","DOI":"10.1109\/TCSS.2024.3388016","article-title":"Hybrid quantum-classical neural network for multimodal multitask sarcasm, emotion, and sentiment analysis","volume":"11","author":"Phukan","year":"2024","journal-title":"IEEE Trans. Comput. Soc. Syst."},{"key":"10.1016\/j.inffus.2026.104436_bib0063","series-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing","first-page":"15933","article-title":"Predict and use: harnessing predicted gaze to improve multimodal sarcasm detection","author":"Tiwari","year":"2023"},{"key":"10.1016\/j.inffus.2026.104436_bib0064","unstructured":"K.-W. Chang, Y.-K. Wang, H. Shen, I.-t. Kang, W.-C. Tseng, S.-W. Li, H.-y. Lee, Speechprompt v2: prompt tuning for speech classification tasks, arXiv: 2303.00733(2023)."},{"key":"10.1016\/j.inffus.2026.104436_bib0065","unstructured":"A. Pandey, D.K. Vishwakarma, VyAnG-Net: a novel multi-modal sarcasm recognition model by uncovering visual, acoustic and glossary features, arXiv: 2408.10246(2024)."},{"key":"10.1016\/j.inffus.2026.104436_bib0066","series-title":"Findings of the Association for Computational Linguistics ACL 2024","first-page":"14392","article-title":"Progressive tuning: towards generic sentiment abilities for large language models","author":"Hou","year":"2024"},{"key":"10.1016\/j.inffus.2026.104436_bib0067","doi-asserted-by":"crossref","unstructured":"W. Han, H. Chen, S. Poria, Improving multimodal fusion with hierarchical mutual information maximization for multimodal sentiment analysis, arXiv: 2109.00412(2021).","DOI":"10.18653\/v1\/2021.emnlp-main.723"},{"issue":"3","key":"10.1016\/j.inffus.2026.104436_bib0068","doi-asserted-by":"crossref","first-page":"2276","DOI":"10.1109\/TAFFC.2022.3172360","article-title":"Hybrid contrastive learning of tri-modal representation for multimodal sentiment analysis","volume":"14","author":"Mai","year":"2022","journal-title":"IEEE Trans. Affect. Comput."},{"key":"10.1016\/j.inffus.2026.104436_bib0069","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"22214","article-title":"Boosting multi-modal model performance with adaptive gradient modulation","author":"Li","year":"2023"},{"key":"10.1016\/j.inffus.2026.104436_bib0070","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102725","article-title":"AtCAF: attention-based causality-aware fusion network for multimodal sentiment analysis","volume":"114","author":"Huang","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104436_bib0071","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102663","article-title":"Triple disentangled representation learning for multimodal affective analysis","volume":"114","author":"Zhou","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104436_bib0072","unstructured":"B. Yao, Y. Zhang, Q. Li, J. Qin, Is sarcasm detection a step-by-step reasoning process in large language models?, arXiv: 2407.12725(2024)."},{"key":"10.1016\/j.inffus.2026.104436_bib0073","unstructured":"A. Yang, B. Yang, B. Hui, B. Zheng, B. Yu, C. Zhou, C. Li, C. Li, D. Liu, F. Huang, et al., Qwen2 technical report, arXiv: 2407.10671(2024)."},{"issue":"11","key":"10.1016\/j.inffus.2026.104436_bib0074","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"Van der Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.inffus.2026.104436_bib0075","doi-asserted-by":"crossref","first-page":"18794","DOI":"10.52202\/075280-0824","article-title":"CMMA: benchmarking multi-affection detection in Chinese multi-modal conversations","volume":"36","author":"Zhang","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"2","key":"10.1016\/j.inffus.2026.104436_bib0076","doi-asserted-by":"crossref","first-page":"1363","DOI":"10.1109\/TAFFC.2021.3083522","article-title":"Multi-modal sarcasm detection and humor classification in code-mixed conversations","volume":"14","author":"Bedi","year":"2021","journal-title":"IEEE Trans. Affect. Comput."},{"key":"10.1016\/j.inffus.2026.104436_bib0077","unstructured":"Y. Zhang, C. Zou, Z. Lian, P. Tiwari, J. Qin, SarcasmBench: towards evaluating large language models on sarcasm understanding, arXiv: 2408.11319(2024)."}],"container-title":["Information Fusion"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1566253526003167?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1566253526003167?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T21:15:44Z","timestamp":1783199744000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1566253526003167"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":77,"alternative-id":["S1566253526003167"],"URL":"https:\/\/doi.org\/10.1016\/j.inffus.2026.104436","relation":{},"ISSN":["1566-2535"],"issn-type":[{"value":"1566-2535","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MPF: A multi-level perceiving framework for multimodal sarcasm detection","name":"articletitle","label":"Article Title"},{"value":"Information Fusion","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.inffus.2026.104436","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104436"}}