{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T14:00:15Z","timestamp":1783432815490,"version":"3.54.6"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2023,3,24]],"date-time":"2023-03-24T00:00:00Z","timestamp":1679616000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,3,24]],"date-time":"2023-03-24T00:00:00Z","timestamp":1679616000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"published-print":{"date-parts":[[2023,11]]},"DOI":"10.1007\/s10462-023-10459-7","type":"journal-article","created":{"date-parts":[[2023,3,24]],"date-time":"2023-03-24T09:02:57Z","timestamp":1679648577000},"page":"12833-12851","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":22,"title":["Detecting hate speech in memes: a review"],"prefix":"10.1007","volume":"56","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6875-8465","authenticated-orcid":false,"given":"Paulo Cezar de Q.","family":"Hermida","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Eulanda M. dos","family":"Santos","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,3,24]]},"reference":[{"key":"10459_CR1","doi-asserted-by":"publisher","first-page":"1451","DOI":"10.1007\/978-3-030-66840-2","volume-title":"Innovations in smart cities applications","author":"TH Afridi","year":"2021","unstructured":"Afridi TH, Alam A, Khan MN et al (2021) A multimodal memes classification: a survey and open research issues. Innovations in smart cities applications, vol 4. Springer, Berlin, pp 1451\u20131466. https:\/\/doi.org\/10.1007\/978-3-030-66840-2"},{"issue":"6","key":"10459_CR2","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1007\/s00530-010-0182-0","volume":"16","author":"PK Atrey","year":"2010","unstructured":"Atrey PK, Hossain MA, Saddik AE et al (2010) Multimodal fusion for multimedia analysis: a survey. Multimedia systems 16(6):345\u2013379. https:\/\/doi.org\/10.1007\/s00530-010-0182-0","journal-title":"Multimedia systems"},{"issue":"2","key":"10459_CR3","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1109\/tpami.2018.2798607","volume":"41","author":"T Baltrusaitis","year":"2019","unstructured":"Baltrusaitis T, Ahuja C, Morency LP (2019) Multimodal machine learning: a survey and taxonomy. IEEE transactions on pattern analysis and machine intelligence 41(2):423\u2013443. https:\/\/doi.org\/10.1109\/tpami.2018.2798607","journal-title":"IEEE transactions on pattern analysis and machine intelligence"},{"key":"10459_CR4","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1016\/j.eswa.2016.10.065","volume":"72","author":"T Chen","year":"2017","unstructured":"Chen T, Xu R, He Y et al (2017) Improving sentiment analysis via sentence type classification using BiLSTM-CRF and CNN. Expert systems with applications 72:221\u2013230. https:\/\/doi.org\/10.1016\/j.eswa.2016.10.065","journal-title":"Expert systems with applications"},{"key":"10459_CR5","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1007\/978-3-030-58577-8_7","volume-title":"Computer vision - ECCV 2020","author":"YC Chen","year":"2020","unstructured":"Chen YC, Li L, Yu L et al (2020) UNITER: universal image-TExt representation learning. Computer vision - ECCV 2020. Springer International Publishing, Berlin, pp 104\u2013120. https:\/\/doi.org\/10.1007\/978-3-030-58577-8_7"},{"key":"10459_CR6","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-329","volume-title":"Unsupervised cross-lingual representation learning for speech recognition. Interspeech 2021","author":"A Conneau","year":"2021","unstructured":"Conneau A, Baevski A, Collobert R et al (2021) Unsupervised cross-lingual representation learning for speech recognition. Interspeech 2021. ISCA, Dublin. https:\/\/doi.org\/10.21437\/interspeech.2021-329"},{"key":"10459_CR7","unstructured":"Das A, Wahi JS, Li S (2020) Detecting hate speech in multi-modal memes. arXiv preprint arXiv:2012.14891"},{"key":"10459_CR8","unstructured":"Devlin J, Chang MW, Lee K, et\u00a0al (2018) Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"10459_CR9","unstructured":"Facebook (2021) Facebook hate speech policy. https:\/\/www.ilga-europe.org\/what-we-do\/our-advocacy-work\/hate-crime-hate-speech, Accessed 2021 March 14"},{"key":"10459_CR10","doi-asserted-by":"publisher","unstructured":"Fersini E, Gasparini F, Corchs S (2019) Detecting sexist MEME on the web: A study on textual and visual cues. In: 2019 8th International Conference on Affective Computing and Intelligent Interaction Workshops and Demos (ACIIW). IEEE, https:\/\/doi.org\/10.1109\/aciiw.2019.8925199","DOI":"10.1109\/aciiw.2019.8925199"},{"key":"10459_CR11","doi-asserted-by":"publisher","unstructured":"Gallo I, Calefati A, Nawaz S, (2018) Image and encoded text fusion for multi-modal classification. In, et al (2018) Digital Image Computing: Techniques and Applications (DICTA). IEEE. https:\/\/doi.org\/10.1109\/dicta.2018.8615789","DOI":"10.1109\/dicta.2018.8615789"},{"key":"10459_CR12","doi-asserted-by":"publisher","unstructured":"Gomez R, Gibert J, Gomez L, et\u00a0al (2020) Exploring hate speech detection in multimodal publications. In: 2020 IEEE Winter Conference on Applications of Computer Vision (WACV). IEEE, https:\/\/doi.org\/10.1109\/wacv45572.2020.9093414","DOI":"10.1109\/wacv45572.2020.9093414"},{"key":"10459_CR13","unstructured":"Goswami S (2021) Reddit memes dataset. Recovered by https:\/\/www.kagglecom\/sayangoswami\/reddit-memes-dataset"},{"key":"10459_CR14","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1007\/978-3-319-28854-3_2","volume-title":"Image features detection, description and matching. Image feature detectors and descriptors","author":"M Hassaballah","year":"2016","unstructured":"Hassaballah M, Abdelmgeid AA, Alshazly HA (2016) Image features detection, description and matching. Image feature detectors and descriptors. Springer International Publishing, Berlin, pp 11\u201345. https:\/\/doi.org\/10.1007\/978-3-319-28854-3_2"},{"key":"10459_CR15","unstructured":"Hootsuite W (2021) Global digital report 2021. Recovered by https:\/\/digitalreport wearesocial com"},{"key":"10459_CR16","unstructured":"ILGA-Europe (2020) Hate crime & hate speech. https:\/\/www.facebook.com\/help\/212722115425932, Accessed 2021 March 16"},{"key":"10459_CR17","doi-asserted-by":"publisher","unstructured":"Karkkainen K, Joo J (2021) FairFace: Face attribute dataset for balanced race, gender, and age for bias measurement and mitigation. In: 2021 IEEE winter conference on applications of computer vision (WACV). IEEE, https:\/\/doi.org\/10.1109\/wacv48630.2021.00159","DOI":"10.1109\/wacv48630.2021.00159"},{"key":"10459_CR18","unstructured":"Kiela D, Firooz H, Mohan A, et\u00a0al (2020) The hateful memes challenge: Detecting hate speech in multimodal memes. arXiv preprint arXiv:2005.04790"},{"key":"10459_CR19","unstructured":"Li LH, Yatskar M, Yin D, et\u00a0al (2019) Visualbert: A simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557"},{"key":"10459_CR20","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1007\/978-3-030-58577-8_8","volume-title":"Oscar: object-semantics aligned pre-training for vision-language tasks computer vision - ECCV 2020","author":"X Li","year":"2020","unstructured":"Li X, Yin X, Li C et al (2020) Oscar: object-semantics aligned pre-training for vision-language tasks computer vision - ECCV 2020. Springer International Publishing, Berlin, pp 121\u2013137. https:\/\/doi.org\/10.1007\/978-3-030-58577-8_8"},{"key":"10459_CR21","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Microsoft COCO: common objects in context. Computer vsion - ECCV 2014","author":"TY Lin","year":"2014","unstructured":"Lin TY, Maire M, Belongie S et al (2014) Microsoft COCO: common objects in context. Computer vsion - ECCV 2014. Springer International Publishing, Berlin, pp 740\u2013755. https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"10459_CR22","unstructured":"Lippe P, Holla N, Chandra S, et\u00a0al (2020) A multimodal framework for the detection of hateful memes. arXiv preprint arXiv:2012.12871"},{"key":"10459_CR23","unstructured":"Liu Y, Ott M, Goyal N, et\u00a0al (2019) Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692"},{"key":"10459_CR24","doi-asserted-by":"publisher","unstructured":"Lu J, Batra D, Parikh D, et\u00a0al (2019) Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Advances in neural information processing systems, pp 13\u201323. https:\/\/doi.org\/10.21437\/Interspeech.2008-","DOI":"10.21437\/Interspeech.2008-"},{"key":"10459_CR25","unstructured":"McCallum A, Nigam K, et\u00a0al (1998) A comparison of event models for naive bayes text classification. In: AAAI-98 workshop on learning for text categorization, Citeseer, pp 41\u201348"},{"key":"10459_CR26","unstructured":"Muennighoff N (2020) Vilio: State-of-the-art visio-linguistic models applied to hateful memes. arXiv preprint arXiv:2012.07788"},{"key":"10459_CR27","doi-asserted-by":"publisher","unstructured":"Nobata C, Tetreault J, Thomas A, et\u00a0al (2016) Abusive language detection in online user content. In: Proceedings of the 25th International Conference on World Wide Web. International World Wide Web Conferences Steering Committee, https:\/\/doi.org\/10.1145\/2872427.2883062","DOI":"10.1145\/2872427.2883062"},{"issue":"1","key":"10459_CR28","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1080\/00220670209598786","volume":"96","author":"CYJ Peng","year":"2002","unstructured":"Peng CYJ, Lee KL, Ingersoll GM (2002) An introduction to logistic regression analysis and reporting. J Educ Res 96(1):3\u201314. https:\/\/doi.org\/10.1080\/00220670209598786","journal-title":"J Educ Res"},{"key":"10459_CR29","doi-asserted-by":"publisher","unstructured":"Pham V, Pham C, Dang T (2020) Road damage detection and classification with detectron2 and faster r-CNN. In: 2020 IEEE International Conference on Big Data (Big Data). IEEE, https:\/\/doi.org\/10.1109\/bigdata50022.2020.9378027","DOI":"10.1109\/bigdata50022.2020.9378027"},{"key":"10459_CR30","doi-asserted-by":"publisher","unstructured":"Pires T, Schlinger E, Garrette D (2019) How multilingual is multilingual BERT? In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics. Association for Computational Linguistics, https:\/\/doi.org\/10.18653\/v1\/p19-1493","DOI":"10.18653\/v1\/p19-1493"},{"key":"10459_CR31","volume-title":"Self-regulation of fundamental rights? The eu code of conduct on hate speech, related initiatives and beyond. Fundamental rights protection online","author":"T Quintel","year":"2020","unstructured":"Quintel T, Ullrich C (2020) Self-regulation of fundamental rights? The eu code of conduct on hate speech, related initiatives and beyond. Fundamental rights protection online. Edward Elgar Publishing, Cheltenham"},{"key":"10459_CR32","doi-asserted-by":"publisher","unstructured":"Redmon J, Farhadi A (2017) YOLO9000: better, faster, stronger. In: 2017 IEEE conference on computer vsion and pattern recognition (CVPR). IEEE, https:\/\/doi.org\/10.1109\/cvpr.2017.690","DOI":"10.1109\/cvpr.2017.690"},{"issue":"6","key":"10459_CR33","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/tpami.2016.2577031","volume":"39","author":"S Ren","year":"2017","unstructured":"Ren S, He K, Girshick R et al (2017) Faster r-CNN: towards real-time object detection with region proposal networks. IEEE Trans Pattern Anal Mach Intell 39(6):1137\u20131149. https:\/\/doi.org\/10.1109\/tpami.2016.2577031","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"10459_CR34","unstructured":"Sabat BO, Ferrer CC, Giro-i Nieto X (2019) Hate speech in pixels: Detection of offensive memes towards automatic moderation. arXiv preprint arXiv:1910.02334"},{"key":"10459_CR35","unstructured":"Sandulescu V (2020) Detecting hateful memes using a multimodal deep ensemble. arXiv preprint arXiv:2012.13235"},{"key":"10459_CR36","doi-asserted-by":"publisher","unstructured":"Sethy A, Ramabhadran B (2008) Bag-of-word normalized n-gram models. In: Interspeech 2008. ISCA, https:\/\/doi.org\/10.21437\/interspeech.2008-265","DOI":"10.21437\/interspeech.2008-265"},{"key":"10459_CR37","doi-asserted-by":"publisher","unstructured":"Sharma P, Ding N, Goodman S, et\u00a0al (2018) Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). Association for Computational Linguistics, https:\/\/doi.org\/10.18653\/v1\/p18-1238","DOI":"10.18653\/v1\/p18-1238"},{"key":"10459_CR38","doi-asserted-by":"crossref","unstructured":"Sharma C, Bhageria D, Scott W, et\u00a0al (2020) Semeval-2020 task 8: Memotion analysis\u2013the visuo-lingual metaphor! arXiv preprint arXiv:2008.03781","DOI":"10.18653\/v1\/2020.semeval-1.99"},{"key":"10459_CR39","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556"},{"key":"10459_CR40","unstructured":"Srinivasa-Desikan B (2018) Natural language processing and computational linguistics: a practical guide to text analysis with Python, Gensim, spaCy, and Keras. Packt Publishing Ltd"},{"key":"10459_CR41","unstructured":"Suryawanshi S, Chakravarthi BR, Arcan M, et\u00a0al (2020) Multimodal meme dataset (multioff) for identifying offensive content in image and text. In: Proceedings of the Second Workshop on Trolling, Aggression and Cyberbullying, pp 32\u201341"},{"key":"10459_CR42","doi-asserted-by":"publisher","unstructured":"Tan H, Bansal M (2019) LXMERT: Learning cross-modality encoder representations from transformers. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP). Association for Computational Linguistics, https:\/\/doi.org\/10.18653\/v1\/d19-1514","DOI":"10.18653\/v1\/d19-1514"},{"key":"10459_CR43","unstructured":"Tan M, Le Q (2019) Efficientnet: Rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, PMLR, pp 6105\u20136114"},{"key":"10459_CR44","unstructured":"Twitter (2021) Coordinated harmful activity. https:\/\/help.twitter.com\/en\/rules-and-policies\/coordinated-harmful-activity, Accessed 2021 April 26"},{"key":"10459_CR45","unstructured":"Velioglu R, Rose J (2020) Detecting hate speech in memes using multimodal deep learning approaches: Prize-winning solution to hateful memes challenge. arXiv preprint arXiv:2012.12975"},{"issue":"4","key":"10459_CR46","doi-asserted-by":"publisher","first-page":"652","DOI":"10.1109\/tpami.2016.2587640","volume":"39","author":"O Vinyals","year":"2017","unstructured":"Vinyals O, Toshev A, Bengio S et al (2017) Show and tell: lessons learned from the 2015 MSCOCO image captioning challenge. IEEE Trans Pattern Anal Mach Intell 39(4):652\u2013663. https:\/\/doi.org\/10.1109\/tpami.2016.2587640","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"10459_CR47","doi-asserted-by":"publisher","first-page":"288","DOI":"10.4000\/books.aaccademia.7360","volume-title":"UPB @ DANKMEMES: Italian memes analysis - employing visual models and graph convolutional networks for meme identification and hate speech detection. EVALITA evaluation of NLP and speech tools for Italian - December 17th, 2020","author":"GA Vlad","year":"2020","unstructured":"Vlad GA, Zaharia GE, Cercel DC et al (2020) UPB @ DANKMEMES: Italian memes analysis - employing visual models and graph convolutional networks for meme identification and hate speech detection. EVALITA evaluation of NLP and speech tools for Italian - December 17th, 2020. Accademia University Press, Maidenhead, pp 288\u2013293. https:\/\/doi.org\/10.4000\/books.aaccademia.7360"},{"key":"10459_CR48","doi-asserted-by":"publisher","first-page":"7370","DOI":"10.1609\/aaai.v33i01.33017370","volume":"33","author":"L Yao","year":"2019","unstructured":"Yao L, Mao C, Luo Y (2019) Graph convolutional networks for text classification. Proc AAAI Conf Artif Intell 33:7370\u20137377. https:\/\/doi.org\/10.1609\/aaai.v33i01.33017370","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"10459_CR49","unstructured":"YouTube (2021) Hate speech policy. https:\/\/support.google.com\/youtube\/answer\/2801939?hl=en, Accessed 2021 April 26"},{"key":"10459_CR50","unstructured":"Yu F, Tang J, Yin W, et\u00a0al (2020) Ernie-vil: Knowledge enhanced vision-language representations through scene graph. arXiv preprint arXiv:2006.16934 1:12"},{"key":"10459_CR51","doi-asserted-by":"publisher","unstructured":"Zhang Y, Liu Q, Song L (2018) Sentence-state LSTM for text representation. https:\/\/doi.org\/10.18653\/v1\/p18-1030","DOI":"10.18653\/v1\/p18-1030"},{"key":"10459_CR52","unstructured":"Zhang W, Liu G, Li Z, et\u00a0al (2020) Hateful memes detection via complementary visual and linguistic networks. arXiv preprint arXiv:2012.04977"},{"key":"10459_CR53","doi-asserted-by":"publisher","unstructured":"Zhou Y, Chen Z, Yang H (2021) Multimodal learning for hateful memes detection. In: 2021 IEEE International Conference on Multimedia & Expo Workshops (ICMEW). IEEE, https:\/\/doi.org\/10.1109\/icmew53276.2021.9455994","DOI":"10.1109\/icmew53276.2021.9455994"},{"key":"10459_CR54","unstructured":"Zhu R (2020) Enhance multimodal transformer with external label and in-domain pretrain: Hateful meme challenge winning solution. arXiv preprint arXiv:2012.08290"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-023-10459-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-023-10459-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-023-10459-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,23]],"date-time":"2023-09-23T03:37:04Z","timestamp":1695440224000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-023-10459-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3,24]]},"references-count":54,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2023,11]]}},"alternative-id":["10459"],"URL":"https:\/\/doi.org\/10.1007\/s10462-023-10459-7","relation":{},"ISSN":["0269-2821","1573-7462"],"issn-type":[{"value":"0269-2821","type":"print"},{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,3,24]]},"assertion":[{"value":"24 March 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"On behalf of all authors, the corresponding author states that there is no\nconflict of interest. The authors have no relevant financial or non-financial interests to disclose. The authors have no competing interests to declare that are relevant to the\ncontent of this article. All authors certify that they have no affiliations with or involvement in any organization or entity with any financial interest or non-financial interest in the subject matter or materials discussed in this manuscript. The authors have no financial or proprietary interests in any material discussed in this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}