{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T08:51:01Z","timestamp":1781772661243,"version":"3.54.5"},"reference-count":42,"publisher":"China Science Publishing & Media Ltd.","issue":"2","content-domain":{"domain":["engine.scichina.com"],"crossmark-restriction":false},"short-container-title":["DI"],"published-print":{"date-parts":[[2026,6,1]]},"DOI":"10.3724\/2096-7004.di.2025.0118","type":"journal-article","created":{"date-parts":[[2025,9,22]],"date-time":"2025-09-22T07:06:29Z","timestamp":1758524789000},"page":"20250118","update-policy":"https:\/\/doi.org\/10.1360\/scp-crossmark-policy-page","source":"Crossref","is-referenced-by-count":0,"title":["CoTMSD: Leveraging Chain-of-Thought for Multimodal Sarcasm Detection"],"prefix":"10.3724","volume":"8","author":[{"given":"Xu","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangdong","family":"Su","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Tian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tian","family":"Lan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guanglai","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"2026","published-online":{"date-parts":[[2025,8,10]]},"reference":[{"key":"null","unstructured":"Farias D. H. and Rosso P., \u201cIrony, sarcasm, and sentiment analysis,\u201d in Sentiment analysis in social networks. Elsevier, 2017, pp. 113\u2013128."},{"key":"null","unstructured":"Khare A., Gangwar A., Singh S., and Prakash S., \u201cSentiment analysis and sarcasm detection in indian general election tweets,\u201d in Research Advances in Intelligent Computing. CRC Press, 2023, pp. 253\u2013268."},{"key":"null","unstructured":"Cai Y., Cai H., and Wan X., \u201cMulti-modal sarcasm detection in Twitter with hierarchical fusion model,\u201d in Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, Korhonen A., Traum D., and M\u00e0rquez L., Eds. Florence, Italy: Association for Computational Linguistics, Jul. 2019, pp. 2506\u20132515. [Online]. Available: https:\/\/aclanthology.org\/P19-1239."},{"key":"null","unstructured":"Prasanna M., Shaila S., and Vadivel A., \u201cPolarity classification on twitter data for classifying sarcasm using clause pattern for sentiment analysis,\u201d Multimedia Tools and Applications, vol. 82, no. 21, pp. 32789\u201332825, 2023."},{"key":"null","unstructured":"Li J., Pan H., Lin Z., Fu P., and Wang W., \u201cSarcasm detection with commonsense knowledge,\u201d IEEE\/ACM Transactions on Audio, Speech, and Language Processing, vol. 29, pp. 3192\u20133201, 2021."},{"key":"null","unstructured":"Veale T. and Hao Y., \u201cDetecting ironic intent in creative comparisons,\u201d in ECAI 2010. IOS Press, 2010, pp. 765\u2013770."},{"key":"null","unstructured":"Qiao Y., Jing L., Song X., Chen X., Zhu L., and Nie L., \u201cMutual-enhanced incongruity learning network for multimodal sarcasm detection,\u201d in Proceedings of the AAAI conference on artificial intelligence, vol. 37, no. 8, 2023, pp. 9507\u20139515."},{"key":"null","unstructured":"Hao J., Zhao J., and Wang Z., \u201cMulti-modal sarcasm detection via graph convolutional network and dynamic network,\u201d in Proceedings of the 33rd ACM International Conference on Information and Knowledge Management, 2024, pp. 789\u2013798."},{"key":"null","unstructured":"Guo D. et al., \u201cMulti-view incongruity learning for multimodal sarcasm detection,\u201d arXiv preprint arXiv:2412.00756, 2024."},{"key":"null","unstructured":"Liang B., Lou C., Li X., Yang M., Gui L., He Y., Pei W., and Xu R., \u201cMulti-modal sarcasm detection via crossmodal graph convolutional network,\u201d in Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), vol. 1. Association for Computational Linguistics, 2022, pp. 1767\u20131777."},{"key":"null","unstructured":"Wei Y., Yuan S., Zhou H., Wang L., Yan Z., Yang R., and Chen M., \u201cG\u02c6 2sam: Graph-based global semantic awareness method for multimodal sarcasm detection,\u201d in Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, no. 8, 2024, pp. 9151\u20139159."},{"key":"null","unstructured":"Liu H., Wei R., Tu G., Lin J., Liu C., and Jiang D., \u201cSarcasm driven by sentiment: A sentiment-aware hierarchical fusion network for multimodal sarcasm detection,\u201d Information Fusion, vol. 108, p. 102353, 2024."},{"key":"null","unstructured":"Cambria E., Li Y., Xing F. Z., Poria S., and Kwok K., \u201cSenticnet 6: Ensemble application of symbolic and subsymbolic ai for sentiment analysis,\u201d in Proceedings of the 29th ACM international conference on information &amp; knowledge management, 2020, pp. 105\u2013114."},{"key":"null","unstructured":"Liu H., Wang W., and Li H., \u201cTowards multi-modal sarcasm detection via hierarchical congruity modeling with knowledge enhancement,\u201d arXiv preprint arXiv:2210.03501, 2022."},{"key":"null","unstructured":"Mokady R., Hertz A., and Bermano A. H., \u201cClipcap: Clip prefix for image captioning,\u201d arXiv preprint arXiv:2111.09734, 2021."},{"key":"null","unstructured":"Radford A., Wu J., Child R., Luan D., Amodei D., Sutskever I. et al., \u201cLanguage models are unsupervised multitask learners,\u201d OpenAI blog, vol. 1, no. 8, p. 9, 2019."},{"key":"null","unstructured":"Niu Z., Xie Z., Xu T., Wang X., Hu Y., Yu Y., and Chen E., \u201cKnowledge-enhanced multi-perspective incongruity perception network for multimodal sarcasm detection,\u201d in 2024 IEEE International Conference on Multimedia and Expo (ICME). IEEE, 2024, pp. 1\u20136."},{"key":"null","unstructured":"Li J., Li D., Savarese S., and Hoi S., \u201cBlip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models,\u201d in International conference on machine learning. PMLR, 2023, pp. 19730\u201319742."},{"key":"null","unstructured":"Wang W. et al., \u201cCogvlm: Visual expert for pretrained language models,\u201d arXiv preprint arXiv:2311.03079, 2023."},{"key":"null","unstructured":"Jia M., Xie C., and Jing L., \u201cDebiasing multimodal sarcasm detection with contrastive learning,\u201d in Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, no. 16, 2024, pp. 18354\u201318362."},{"key":"null","unstructured":"Schifanella R., De Juan P., Tetreault J., and Cao L., \u201cDetecting sarcasm in multimodal social platforms,\u201d in Proceedings of the 24th ACM international conference on Multimedia, 2016, pp. 1136\u20131145."},{"key":"null","unstructured":"Xu N., Zeng Z., and Mao W., \u201cReasoning with multimodal sarcastic tweets via modeling cross-modality contrast and semantic association,\u201d in Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, Jurafsky D., Chai J., Schluter N., and Tetreault J., Eds. Online: Association for Computational Linguistics, Jul. 2020, pp. 3777\u20133786. [Online]. Available: https:\/\/aclanthology.org\/2020.acl-main.349\/."},{"key":"null","unstructured":"Pan H., Lin Z., Fu P., Qi Y., and Wang W., \u201cModeling intra and inter-modality incongruity for multi-modal sarcasm detection,\u201d in Findings of the Association for Computational Linguistics: EMNLP 2020, Cohn T., He Y., and Liu Y., Eds. Online: Association for Computational Linguistics, Nov. 2020, pp. 1383\u20131392. [Online]. Available: https:\/\/aclanthology.org\/2020.findings-emnlp.124\/."},{"key":"null","unstructured":"Wang X., Sun X., Yang T., and Wang H., \u201cBuilding a bridge: A method for image-text sarcasm detection without pretraining on image-text data,\u201d in Proceedings of the First International Workshop on Natural Language Processing Beyond Text, Castellucci G., Filice S., Poria S., Cambria E., and Specia L., Eds. Online: Association for Computational Linguistics, Nov. 2020, pp. 19\u201329. [Online]. Available: https:\/\/aclanthology.org\/2020.nlpbt-1.3\/."},{"key":"null","unstructured":"Liang B., Lou C., Li X., Gui L., Yang M., and Xu R., \u201cMulti-modal sarcasm detection with interactive in-modal and cross-modal graphs,\u201d in Proceedings of the 29th ACM international conference on multimedia, 2021, pp. 4707\u20134715."},{"key":"null","unstructured":"Qin L., Huang S., Chen Q., Cai C., Zhang Y., Liang B., Che W., and Xu R., \u201cMMSD2.0: Towards a reliable multi-modal sarcasm detection system,\u201d in Findings of the Association for Computational Linguistics: ACL 2023, Rogers A., Boyd-Graber J., and Okazaki N., Eds. Toronto, Canada: Association for Computational Linguistics, Jul. 2023, pp. 10834\u201310845. [Online]. Available: https:\/\/aclanthology.org\/2023.findings-acl.689."},{"key":"null","unstructured":"Yu H., Qi Z., Jang L., Salakhutdinov R., Morency L.-P., and Liang P. P., \u201cMmoe: Enhancing multimodal models with mixtures of multimodal interaction experts,\u201d in Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, 2024, pp. 10006\u201310030."},{"key":"null","unstructured":"Tang B., Lin B., Yan H., and Li S., \u201cLeveraging generative large language models with visual instruction and demonstration retrieval for multimodal sarcasm detection,\u201d in Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), 2024, pp. 1732\u20131742."},{"key":"null","unstructured":"He K., Zhang X., Ren S., and Sun J., \u201cDeep residual learning for image recognition,\u201d in Proceedings of the IEEE conference on computer vision and pattern recognition, 2016, pp. 770\u2013778."},{"key":"null","unstructured":"Zhu Z., Shen K., Chen Z., Zhang Y., Chen Y., Jiao X., Wan Z., Xie S., Liu W., Wu X. et al., \u201cDglf: A dual graph-based learning framework for multi-modal sarcasm detection,\u201d in Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, 2024, pp. 2900\u20132912."},{"key":"null","unstructured":"Radford A., Kim J.W., Hallacy C., Ramesh A., Goh G., Agarwal S., Sastry G., Askell A., Mishkin P., Clark J. et al., \u201cLearning transferable visual models from natural language supervision,\u201d in International conference on machine learning. PMLR, 2021, pp. 8748\u20138763."},{"key":"null","unstructured":"Fu J., Xu S., Liu H., Liu Y., Xie N., Wang C.-C., Liu J., Sun Y., and Wang B., \u201cCma-clip: Cross-modality attention clip for text-image classification,\u201d in 2022 IEEE International Conference on Image Processing (ICIP), 2022, pp. 2846\u20132850."},{"key":"null","unstructured":"Vaswani A., \u201cAttention is all you need,\u201d in Adv. Neural Inf. Process. Syst. (NeurIPS), 2017."},{"key":"null","unstructured":"Long X., Gan C., Melo G., Liu X., Li Y., Li F., and Wen S., \u201cMultimodal keyless attention fusion for video classification,\u201d in Proceedings of the aaai conference on artificial intelligence, vol. 32, no. 1, 2018."},{"key":"null","unstructured":"Wolf T., \u201cHuggingface\u2019s transformers: State-of-the-art natural language processing,\u201d arXiv preprint arXiv:1910.03771, 2019."},{"key":"null","unstructured":"Loshchilov I., \u201cDecoupled weight decay regularization,\u201d arXiv preprint arXiv:1711.05101, 2017."},{"key":"null","unstructured":"Dong X. et al., \u201cInternlmxcomposer2: Mastering free-form text-image composition and comprehension in vision-language large model,\u201d arXiv preprint arXiv:2401.16420, 2024."},{"key":"null","unstructured":"Tian Y., Xu N., Zhang R., and Mao W., \u201cDynamic routing transformer network for multimodal sarcasm detection,\u201d in Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2023, pp. 2468\u20132480."},{"key":"null","unstructured":"Hwang J. D., Bhagavatula C., Le Bras R., Da J., Sakaguchi K., Bosselut A., and Choi Y., \u201cOn symbolic and neural commonsense knowledge graphs,\u201d arXiv preprint arXiv:2102.10086, 2021."},{"key":"null","unstructured":"Wang P., Bai S., Tan S., Wang S., Fan Z., Bai J., Chen K., Liu X., Wang J., Ge W., Fan Y., Dang K., Du M., Ren X., Men R., Liu D., Zhou C., Zhou J., and Lin J., \u201cQwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution,\u201d arXiv preprint arXiv:2409.12191, 2024."},{"key":"null","unstructured":"Liu H., Li C., Wu Q., and Lee Y. J., \u201cVisual instruction tuning,\u201d arXiv preprint arXiv:2304.08485, 2021."},{"key":"null","unstructured":"Farabi S., Ranasinghe T., Kanojia D., Kong Y., and Zampieri M., \u201cA survey of multimodal sarcasm detection,\u201d arXiv preprint arXiv:2410.18882, 2024."}],"container-title":["Data Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/26ABF3019C4D4F1DA46A767D0D88E962","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0118","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/26ABF3019C4D4F1DA46A767D0D88E962","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T07:53:32Z","timestamp":1781769212000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0118"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,10]]},"references-count":42,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2025,8,10]]},"published-print":{"date-parts":[[2026,6,1]]}},"URL":"https:\/\/doi.org\/10.3724\/2096-7004.di.2025.0118","relation":{},"ISSN":["2096-7004"],"issn-type":[{"value":"2096-7004","type":"print"}],"subject":[],"published":{"date-parts":[[2025,8,10]]}}}