{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,16]],"date-time":"2026-01-16T04:43:38Z","timestamp":1768538618758,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T00:00:00Z","timestamp":1746662400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,8]]},"DOI":"10.1145\/3701716.3718384","type":"proceedings-article","created":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T16:09:41Z","timestamp":1748016581000},"page":"1993-1999","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["CBRCL: A CLIP-BERT with Retrieval-Guided Contrastive Learning Multimodal Approach for Crisis-Driven Hate Speech Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-9689-0689","authenticated-orcid":false,"given":"Ilya","family":"Stepanov","sequence":"first","affiliation":[{"name":"Sejong University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1685-7234","authenticated-orcid":false,"given":"Junaid","family":"Rashid","sequence":"additional","affiliation":[{"name":"Sejong University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9425-2601","authenticated-orcid":false,"given":"Jong Weon","family":"Lee","sequence":"additional","affiliation":[{"name":"Sejong University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8467-2802","authenticated-orcid":false,"given":"Salman","family":"Naseem","sequence":"additional","affiliation":[{"name":"University of Stavanger, Stavanger, Norway"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3930-6600","authenticated-orcid":false,"given":"Imran","family":"Razzak","sequence":"additional","affiliation":[{"name":"MBZUAI, Abu Dhabi, United Arab Emirates and University of New South Wales, Sydney, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,5,23]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589335.3652504"},{"key":"e_1_3_2_2_2_1","unstructured":"Jean-Baptiste Alayrac et al. 2022. Flamingo: a visual language model for fewshot learning. Advances in neural information processing systems 35 23716-- 23736."},{"key":"e_1_3_2_2_3_1","unstructured":"William Berrios Gautam Mittal Tristan Thrush Douwe Kiela and Amanpreet Singh. 2023. Towards language models that can see: computer vision through the lens of natural language. arXiv preprint arXiv:2306.16410."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00193"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589335.3651974"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00063"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.tmaid.2022.102346"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Corinna Cortes. 1995. Support-vector networks. Machine Learning.","DOI":"10.1023\/A:1022627411411"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1609\/icwsm.v11i1.14955"},{"key":"e_1_3_2_2_11_1","unstructured":"Jacob Devlin. 2018. Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/2740908.2742760"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"crossref","unstructured":"Tom Fawcett. 2006. An introduction to roc analysis. Pattern recognition letters 27 8 861--874.","DOI":"10.1016\/j.patrec.2005.10.010"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"crossref","unstructured":"Shreyansh Gandhi Samrat Kokkula Abon Chaudhuri Alessandro Magnani Theban Stanley Behzad Ahmadi Venkatesh Kandaswamy Omer Ovenc and Shie Mannor. 2019. Image matters: scalable detection of offensive and noncompliant content\/logo in product images. arXiv preprint arXiv:1905.02234.","DOI":"10.1109\/WACV45572.2020.9093454"},{"key":"e_1_3_2_2_15_1","unstructured":"Ian Goodfellow. 2016. Deep learning. (2016)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.254"},{"key":"e_1_3_2_2_18_1","unstructured":"Matthew Henderson Rami Al-Rfou Brian Strope Yun-Hsuan Sung L\u00e1szl\u00f3 Luk\u00e1cs Ruiqi Guo Sanjiv Kumar Balint Miklos and Ray Kurzweil. 2017. Efficient natural language response suggestion for smart reply. arXiv preprint arXiv:1705.00652."},{"key":"e_1_3_2_2_19_1","volume-title":"Proceedings of the 2nd conference of the asia-pacific","author":"Hossain Eftekhar","unstructured":"Eftekhar Hossain, Omar Sharif, and Mohammed Moshiul Hoque. 2022. Mute: a multimodal dataset for detecting hateful memes. In Proceedings of the 2nd conference of the asia-pacific chapter of the association for computational linguistics and the 12th international joint conference on natural language processing: student research workshop, 32--39."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3587427"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TBDATA.2019.2921572"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/383952.384045"},{"key":"e_1_3_2_2_24_1","unstructured":"Douwe Kiela Suvrat Bhooshan Hamed Firooz Ethan Perez and Davide Testuggine. 2019. Supervised multimodal bitransformers for classifying images and text. arXiv preprint arXiv:1909.02950."},{"key":"e_1_3_2_2_25_1","unstructured":"Douwe Kiela Hamed Firooz Aravind Mohan Vedanuj Goswami Amanpreet Singh Pratik Ringshia and Davide Testuggine. 2020. The hateful memes challenge: detecting hate speech in multimodal memes. Advances in neural information processing systems 33 2611--2624."},{"key":"e_1_3_2_2_26_1","volume-title":"International conference on machine learning. PMLR, 5583--5594","author":"Kim Wonjae","year":"2021","unstructured":"Wonjae Kim, Bokyung Son, and Ildoo Kim. 2021. Vilt: vision-and-language transformer without convolution or region supervision. In International conference on machine learning. PMLR, 5583--5594."},{"key":"e_1_3_2_2_27_1","unstructured":"Gokul Karthik Kumar and Karthik Nandakumar. 2022. Hate-clipper: multimodal hateful meme classification based on cross-modal interaction of clip features. arXiv preprint arXiv:2210.05916."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"e_1_3_2_2_29_1","unstructured":"Liunian Harold Li Mark Yatskar Da Yin Cho-Jui Hsieh and Kai-Wei Chang. 2019. Visualbert: a simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557."},{"key":"e_1_3_2_2_30_1","volume-title":"Proceedings, Part XXX 16","author":"Xiujun","unstructured":"Xiujun Li et al. 2020. Oscar: object-semantics aligned pre-training for visionlanguage tasks. In Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part XXX 16. Springer, 121--137."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.476"},{"key":"e_1_3_2_2_33_1","unstructured":"I Loshchilov. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101."},{"key":"e_1_3_2_2_34_1","unstructured":"Jiasen Lu Dhruv Batra Devi Parikh and Stefan Lee. 2019. Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in neural information processing systems 32."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.291"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539597.3570450"},{"key":"e_1_3_2_2_37_1","volume-title":"Companion Proceedings of the ACM on Web Conference 2024","author":"Xian Ng Lynnette Hui","year":"2024","unstructured":"Lynnette Hui Xian Ng, Adrian Xuan Wei Lim, and Roy Ka-Wei Lee. 2024. Love-hate dataset: a multi-modal multi-platform dataset depicting emotions in the 2023 israel-hamas war. In Companion Proceedings of the ACM on Web Conference 2024, 1807--1815."},{"key":"e_1_3_2_2_38_1","article-title":"Evaluation: From precision, recall and F-measure to ROC, informedness, markedness and correlation","author":"Powers David Martin","year":"2011","unstructured":"David Martin Powers. 2011. Evaluation: From precision, recall and F-measure to ROC, informedness, markedness and correlation. Journal of Machine Learning Technologies.","journal-title":"Journal of Machine Learning Technologies."},{"key":"e_1_3_2_2_39_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Alec","unstructured":"Alec Radford et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/IC3I.2016.7918035"},{"key":"e_1_3_2_2_41_1","volume-title":"Faster r-cnn: towards real-time object detection with region proposal networks","author":"Ren Shaoqing","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2016. Faster r-cnn: towards real-time object detection with region proposal networks. IEEE transactions on pattern analysis and machine intelligence, 39, 6, 1137--1149."},{"key":"e_1_3_2_2_42_1","unstructured":"V Sanh. 2019. Distilbert a distilled version of bert: smaller faster cheaper and lighter. arXiv preprint arXiv:1910.01108."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-1101"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"e_1_3_2_2_45_1","unstructured":"Alexander Shevtsov Christos Tzagkarakis Despoina Antonakaki Polyvios Pratikakis and Sotiris Ioannidis. 2022. Twitter dataset on the russo-ukrainian war. arXiv preprint arXiv:2204.08530."},{"key":"e_1_3_2_2_46_1","unstructured":"Karen Simonyan. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556."},{"key":"e_1_3_2_2_47_1","volume-title":"Proceedings of the WILDRE5--5th workshop on indian language data: resources and evaluation, 7--13","author":"Suryawanshi Shardul","year":"2020","unstructured":"Shardul Suryawanshi, Bharathi Raja Chakravarthi, Pranav Verma, Mihael Arcan, John Philip McCrae, and Paul Buitelaar. 2020. A dataset for troll classification of tamilmemes. In Proceedings of the WILDRE5--5th workshop on indian language data: resources and evaluation, 7--13."},{"key":"e_1_3_2_2_48_1","volume-title":"2nd. newton, ma.","author":"Van Rijsbergen Cornelius Joost","year":"1979","unstructured":"Cornelius Joost Van Rijsbergen. 1979. Information retrieval. 2nd. newton, ma. (1979)."},{"key":"e_1_3_2_2_49_1","volume-title":"Proceedings of the second workshop on language in social media, 19--26","author":"Julia Hirschberg WilliamWarner","year":"2012","unstructured":"WilliamWarner and Julia Hirschberg. 2012. Detecting hate speech on the world wide web. In Proceedings of the second workshop on language in social media, 19--26."},{"key":"e_1_3_2_2_50_1","volume-title":"Hate speech on twitter: a pragmatic approach to collect hateful and offensive expressions and perform hate speech detection","author":"Bouazizi Mondher","unstructured":"HajimeWatanabe, Mondher Bouazizi, and Tomoaki Ohtsuki. 2018. Hate speech on twitter: a pragmatic approach to collect hateful and offensive expressions and perform hate speech detection. IEEE access, 6, 13825--13835."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01760"}],"event":{"name":"WWW '25: The ACM Web Conference 2025","location":"Sydney NSW Australia","acronym":"WWW '25","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Companion Proceedings of the ACM on Web Conference 2025"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3701716.3718384","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3701716.3718384","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,8]],"date-time":"2025-10-08T02:04:23Z","timestamp":1759889063000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3701716.3718384"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,8]]},"references-count":51,"alternative-id":["10.1145\/3701716.3718384","10.1145\/3701716"],"URL":"https:\/\/doi.org\/10.1145\/3701716.3718384","relation":{},"subject":[],"published":{"date-parts":[[2025,5,8]]},"assertion":[{"value":"2025-05-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}