{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,2]],"date-time":"2025-08-02T00:02:43Z","timestamp":1754092963853,"version":"3.41.2"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"26","license":[{"start":{"date-parts":[[2024,11,11]],"date-time":"2024-11-11T00:00:00Z","timestamp":1731283200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,11]],"date-time":"2024-11-11T00:00:00Z","timestamp":1731283200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-024-20419-0","type":"journal-article","created":{"date-parts":[[2024,11,11]],"date-time":"2024-11-11T02:47:25Z","timestamp":1731293245000},"page":"31429-31443","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["CEFM: CLIP Encoded Fusion Model for multimodal humor recognition on memes"],"prefix":"10.1007","volume":"84","author":[{"given":"Hou","family":"Shuo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5843-4675","authenticated-orcid":false,"given":"Zhang","family":"Yijia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wang","family":"Mengyi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lin","family":"Hongfei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lu","family":"Mingyu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,11]]},"reference":[{"key":"20419_CR1","doi-asserted-by":"publisher","unstructured":"Zannettou S, Caulfield T, Blackburn J, De Cristofaro E, Sirivianos M, Stringhini G, Suarez-Tangil G (2018) On the origins of memes by means of fringe web communities. In: Proceedings of the internet measurement conference 2018. pp 188\u2013202. https:\/\/doi.org\/10.1145\/3278532.3278550","DOI":"10.1145\/3278532.3278550"},{"key":"20419_CR2","doi-asserted-by":"publisher","unstructured":"Sharma C, Bhageria D, Scott W, Pykl S, Das A, Chakraborty T, Pulabaigari V, Gamback B (2020) Semeval-2020 task 8: memotion analysis\u2013the visuo-lingual metaphor! arXiv:2008.03781, https:\/\/doi.org\/10.48550\/arXiv.2008.03781","DOI":"10.48550\/arXiv.2008.03781"},{"key":"20419_CR3","doi-asserted-by":"publisher","unstructured":"Kumar A, Garg G (2019) Sarc-m: sarcasm detection in typo-graphic memes. In: International Conference on Advances in Engineering Science Management & Technology (ICAESMT)-2019. Uttaranchal University, Dehradun, India. https:\/\/doi.org\/10.2139\/ssrn.3384025","DOI":"10.2139\/ssrn.3384025"},{"key":"20419_CR4","unstructured":"Kiela D, Firooz H, Mohan A, Goswami V, Singh A, Ringshia P, Testuggine D (2020) The hateful memes challenge: Detecting hate speech in multimodal memes. Adv Neural Inf Process Syst 33:2611\u20132624. https:\/\/doi.org\/10.48550\/arXiv.2005.04790"},{"key":"20419_CR5","doi-asserted-by":"publisher","unstructured":"Velioglu R, Rose J (2020) Detecting hate speech in memes using multimodal deep learning approaches: prize-winning solution to hateful memes challenge. arXiv:2012.12975, https:\/\/doi.org\/10.48550\/arXiv.2012.12975","DOI":"10.48550\/arXiv.2012.12975"},{"key":"20419_CR6","doi-asserted-by":"publisher","unstructured":"Zhou Y, Chen Z, Yang H (2021) Multimodal learning for hateful memes detection. In: 2021 IEEE International Conference on Multimedia & Expo Workshops (ICMEW). IEEE, pp.\u00a01\u20136. https:\/\/doi.org\/10.1109\/ICMEW53276.2021.9455994","DOI":"10.1109\/ICMEW53276.2021.9455994"},{"key":"20419_CR7","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J et\u00a0al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, PMLR. pp 8748\u20138763"},{"key":"20419_CR8","doi-asserted-by":"crossref","unstructured":"Wu Q, Shen C, Liu L, Dick A, Van Den\u00a0Hengel A (2016) What value do explicit high level concepts have in vision to language problems? In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 203\u2013212","DOI":"10.1109\/CVPR.2016.29"},{"key":"20419_CR9","doi-asserted-by":"publisher","unstructured":"Cai Y, Cai H, Wan X (2019) Multi-modal sarcasm detection in twitter with hierarchical fusion model. In: Proceedings of the 57th Annual meeting of the association for computational linguistics. pp 2506\u20132515. https:\/\/doi.org\/10.18653\/v1\/P19-1239","DOI":"10.18653\/v1\/P19-1239"},{"key":"20419_CR10","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1016\/j.tcs.2018.04.029","volume":"752","author":"Y Zhang","year":"2018","unstructured":"Zhang Y, Song D, Zhang P, Wang P, Li J, Li X, Wang B (2018) A quantum-inspired multimodal sentiment analysis framework. Theoret Comput Sci 752:21\u201340. https:\/\/doi.org\/10.1016\/j.tcs.2018.04.029","journal-title":"Theoret Comput Sci"},{"key":"20419_CR11","doi-asserted-by":"publisher","unstructured":"Zadeh A, Chen M, Poria S, Cambria E, Morency LP (2017) Tensor fusion network for multimodal sentiment analysis. arXiv:1707.07250, https:\/\/doi.org\/10.48550\/arXiv.1707.07250","DOI":"10.48550\/arXiv.1707.07250"},{"key":"20419_CR12","doi-asserted-by":"publisher","unstructured":"Liu Z, Shen Y, Lakshminarasimhan VB, Liang PP, Zadeh A, Morency LP Efficient low-rank multimodal fusion with modality-specific factors. arXiv:1806.00064, https:\/\/doi.org\/10.48550\/arXiv.1806.00064","DOI":"10.48550\/arXiv.1806.00064"},{"key":"20419_CR13","doi-asserted-by":"publisher","unstructured":"Chen YC, Li L, Yu L, El\u00a0Kholy A, Ahmed F, Gan Z, Cheng Y, Liu J (2020) Uniter: Universal image-text representation learning. In: European conference on computer vision, Springer. pp 104\u2013120. https:\/\/doi.org\/10.1007\/978-3-030-58577-8_7","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"20419_CR14","doi-asserted-by":"crossref","unstructured":"Girshick R (2015) Fast r-cnn. In: Proceedings of the IEEE international conference on computer vision. pp 1440\u20131448","DOI":"10.1109\/ICCV.2015.169"},{"key":"20419_CR15","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K (2018) Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv:1810.04805"},{"key":"20419_CR16","doi-asserted-by":"publisher","unstructured":"Mihalcea R, Strapparava C, Pulman S (2010) Computational models for incongruity detection in humour. In: International conference on intelligent text processing and computational linguistics. Springer, pp 364\u2013374. https:\/\/doi.org\/10.1007\/978-3-642-12116-6_30","DOI":"10.1007\/978-3-642-12116-6_30"},{"key":"20419_CR17","doi-asserted-by":"crossref","unstructured":"Raskin V (1979) Semantic mechanisms of humor. In: Annual meeting of the berkeley linguistics society vol.\u00a05. pp 325\u2013335","DOI":"10.3765\/bls.v5i0.2164"},{"key":"20419_CR18","doi-asserted-by":"publisher","unstructured":"Attardo S, Raskin V (1991) Script theory revis (it) ed: joke similarity and joke representation model. https:\/\/doi.org\/10.1515\/humr.1991.4.3-4.293","DOI":"10.1515\/humr.1991.4.3-4.293"},{"key":"20419_CR19","doi-asserted-by":"publisher","unstructured":"Zhang R, Liu N (2014) Recognizing humor on twitter. In: Proceedings of the 23rd ACM international conference on conference on information and knowledge management. pp 889\u2013898. https:\/\/doi.org\/10.1145\/2661829.2661997","DOI":"10.1145\/2661829.2661997"},{"key":"20419_CR20","doi-asserted-by":"publisher","unstructured":"Baziotis C, Pelekis N, Doulkeridis C (2017) Datastories at semeval-2017 task 6: Siamese lstm with attention for humorous text comparison. In: Proceedings of the 11th international workshop on Semantic Evaluation (SemEval-2017). pp 390\u2013395. https:\/\/doi.org\/10.18653\/v1\/S17-2065","DOI":"10.18653\/v1\/S17-2065"},{"key":"20419_CR21","unstructured":"Ortega-Bueno R, Muniz-Cuza CE, Pagola JEM, Rosso P (2018) Uo upv: Deep linguistic humor detection in spanish social media. In: Proceedings of the third workshop on evaluation of human language technologies for Iberian languages (IberEval 2018) co-located with 34th conference of the Spanish society for natural language processing (SEPLN 2018). pp 204\u2013213"},{"key":"20419_CR22","doi-asserted-by":"publisher","unstructured":"Blinov V, Bolotova-Baranova V, Braslavski P (2019) Large dataset and language model fun-tuning for humor recognition. In: Proceedings of the 57th annual meeting of the association for computational linguistics. pp 4027\u20134032. https:\/\/doi.org\/10.18653\/v1\/P19-1394","DOI":"10.18653\/v1\/P19-1394"},{"key":"20419_CR23","doi-asserted-by":"publisher","unstructured":"Bertero D, Fung P (2016) Predicting humor response in dialogues from tv sitcoms. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 5780\u20135784. https:\/\/doi.org\/10.1109\/ICASSP.2016.7472785","DOI":"10.1109\/ICASSP.2016.7472785"},{"key":"20419_CR24","doi-asserted-by":"publisher","unstructured":"Eyben F, Weninger F, Gross F, Schuller B (2013) Recent developments in opensmile, the munich open-source multimedia feature extractor. In: Proceedings of the 21st ACM international conference on Multimedia. pp 835\u2013838. https:\/\/doi.org\/10.1145\/2502081.2502224","DOI":"10.1145\/2502081.2502224"},{"key":"20419_CR25","doi-asserted-by":"crossref","unstructured":"Castro S, Hazarika D, P\u00e9rez-Rosas V, Zimmermann R, Mihalcea R, Poria S (2019) Towards multimodal sarcasm detection (an _obviously_ perfect paper). arXiv:1906.01815","DOI":"10.18653\/v1\/P19-1455"},{"key":"20419_CR26","doi-asserted-by":"publisher","unstructured":"Hasan MK, Lee S, Rahman W, Zadeh A, Mihalcea R, Morency LP, Hoque E (2021) Humor knowledge enriched transformer for understanding multimodal humor. In: Proceedings of the AAAI conference on artificial intelligence vol.\u00a035. pp 12972\u201312980. https:\/\/doi.org\/10.1609\/aaai.v35i14.17534","DOI":"10.1609\/aaai.v35i14.17534"},{"key":"20419_CR27","unstructured":"Jia C, Yang Y, Xia Y, Chen YT, Parekh Z, Pham H, Le Q, Sung YH, Li Z, Duerig T (2021) Scaling up visual and vision-language representation learning with noisy text supervision. In: International conference on machine learning. PMLR , pp 4904\u20134916"},{"key":"20419_CR28","doi-asserted-by":"publisher","unstructured":"Bobicev V, Sokolova M (2017) Inter-annotator agreement in sentiment analysis: machine learning perspective. In: International conference recent advances in natural language processing. pp 97\u2013102. https:\/\/doi.org\/10.26615\/978-954-452-049-6-015","DOI":"10.26615\/978-954-452-049-6-015"},{"key":"20419_CR29","doi-asserted-by":"publisher","unstructured":"Pramanick S, Dimitrov D, Mukherjee R, Sharma S, Akhtar M, Nakov P, Chakraborty T et\u00a0al (2021) Detecting harmful memes and their targets. arXiv preprint arXiv:2110.00413, https:\/\/doi.org\/10.48550\/arXiv.2110.00413","DOI":"10.48550\/arXiv.2110.00413"},{"key":"20419_CR30","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556"},{"key":"20419_CR31","unstructured":"Lan Z, Chen M, Goodman S, Gimpel K, Sharma P, Soricut R (2019) Albert: a lite bert for self-supervised learning of language representations. arXiv:1909.11942"},{"key":"20419_CR32","unstructured":"Kingma DP, Ba J (2014) Adam: A method for stochastic optimization. arXiv:1412.6980"},{"key":"20419_CR33","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der\u00a0Maaten L, Weinberger KQ (2017) Densely connected convolutional networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 4700\u20134708","DOI":"10.1109\/CVPR.2017.243"},{"key":"20419_CR34","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"20419_CR35","doi-asserted-by":"publisher","unstructured":"Deng J, Dong W, Socher R, Li LJ, Li K, Fei-Fei L (2009) Imagenet: a large-scale hierarchical image database. In: 2009 IEEE conference on computer vision and pattern recognition. Ieee, pp 248\u2013255. https:\/\/doi.org\/10.1109\/CVPR.2009.5206848","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"20419_CR36","unstructured":"Kiela D, Bhooshan S, Firooz H, Perez E, Testuggine D (2019) Supervised multimodal bitransformers for classifying images and text. arXiv:1909.02950"},{"key":"20419_CR37","doi-asserted-by":"publisher","unstructured":"Yang X, Feng S, Zhang Y, Wang D (2021) Multimodal sentiment detection based on multi-channel graph neural networks. In: Proceedings of the 59th annual meeting of the association for computational linguistics and the 11th International joint conference on natural language processing (Volume 1: Long Papers). pp 328\u2013339. https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.28","DOI":"10.18653\/v1\/2021.acl-long.28"},{"key":"20419_CR38","unstructured":"Su W, Zhu X, Cao Y, Li B, Lu L, Wei F, Dai J (2019) Vl-bert: pre-training of generic visual-linguistic representations. arXiv:1908.08530"},{"key":"20419_CR39","unstructured":"Li LH, Yatskar M, Yin D, Hsieh CJ, Chang KW (2019) Visualbert: a simple and performant baseline for vision and language. arXiv:1908.03557"},{"key":"20419_CR40","doi-asserted-by":"publisher","unstructured":"Lin TY, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: common objects in context. In: European conference on computer vision. Springer, pp 740\u2013755. https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"20419_CR41","unstructured":"Kim W, Son B, Kim I Vilt: vision-and-language transformer without convolution or region supervision. In: International conference on machine learning. PMLR, pp 5583-5594"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-20419-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-024-20419-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-20419-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,1]],"date-time":"2025-08-01T03:27:15Z","timestamp":1754018835000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-024-20419-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,11]]},"references-count":41,"journal-issue":{"issue":"26","published-online":{"date-parts":[[2025,8]]}},"alternative-id":["20419"],"URL":"https:\/\/doi.org\/10.1007\/s11042-024-20419-0","relation":{},"ISSN":["1573-7721"],"issn-type":[{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2024,11,11]]},"assertion":[{"value":"9 March 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 October 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 October 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 November 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interests"}}]}}