{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:18:59Z","timestamp":1784179139143,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681521","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:27Z","timestamp":1729925967000},"page":"7493-7502","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":23,"title":["MultiHateClip: A Multilingual Benchmark Dataset for Hateful Video Detection on YouTube and Bilibili"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-4486-0693","authenticated-orcid":false,"given":"Han","family":"Wang","sequence":"first","affiliation":[{"name":"Information Systems Technology and Design, Singapore University of Technology and Design, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1325-5888","authenticated-orcid":false,"given":"Tan Rui","family":"Yang","sequence":"additional","affiliation":[{"name":"Information Systems Technology and Design, Singapore University of Technology and Design, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0191-7171","authenticated-orcid":false,"given":"Usman","family":"Naseem","sequence":"additional","affiliation":[{"name":"School of Computing, Macquarie University, Sydney, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1986-7750","authenticated-orcid":false,"given":"Roy Ka-Wei","family":"Lee","sequence":"additional","affiliation":[{"name":"Information Systems Technology and Design, Singapore University of Technology and Design, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Average number of monthly active users of Bilibili Inc. from 4th quarter 2019 to 4th quarter","year":"2023","unstructured":"2024. Average number of monthly active users of Bilibili Inc. from 4th quarter 2019 to 4th quarter 2023. https:\/\/www.statista.com\/statistics\/1109108\/bilibiliaverage-monthly-active-users"},{"key":"e_1_3_2_2_2_1","unstructured":"2024. YouTube Statistics For 2024 (Users Facts & More). https:\/\/www. demandsage.com\/youtube-stats\/"},{"key":"e_1_3_2_2_3_1","volume-title":"GPT-4 technical report. arXiv preprint arXiv:2303.08774","author":"Achiam J.","year":"2023","unstructured":"J. Achiam, S. Adler, S. Agarwal, L. Ahmad, I. Akkaya, F. L. Aleman, and B. McGrew. 2023. GPT-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_2_4_1","volume-title":"Proceedings of the Twelfth Language Resources and Evaluation Conference. 4309--4319","author":"Alc\u00e2ntara C.","unstructured":"C. Alc\u00e2ntara, V. Moreira, and D. Feijo. 2020. Offensive video detection: dataset and baseline results. In Proceedings of the Twelfth Language Resources and Evaluation Conference. 4309--4319."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"Anurag Arnab Mostafa Dehghani Georg Heigold Chen Sun Mario Lucic and Cordelia Schmid. 2021. ViViT: A Video Vision Transformer. arXiv:2103.15691 [cs.CV]","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"e_1_3_2_2_6_1","volume-title":"Localization, Text Reading, and Beyond. arXiv preprint arXiv:2308.12966","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, ShijieWang, Sinan Tan, PengWang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond. arXiv preprint arXiv:2308.12966 (2023)."},{"key":"e_1_3_2_2_7_1","volume-title":"International Journal of Nonlinear Analysis and Applications 12, Special Issue","author":"Banuroopa K.","year":"2021","unstructured":"K. Banuroopa and D. Shanmuga Priyaa. 2021. MFCC based hybrid fingerprinting method for audio classification through LSTM. International Journal of Nonlinear Analysis and Applications 12, Special Issue (2021), 2125--2136."},{"key":"e_1_3_2_2_8_1","volume-title":"Seamless: Multilingual Expressive and Streaming Speech Translation. arXiv preprint arXiv:2312.05187","author":"Barrault L.","year":"2023","unstructured":"L. Barrault, Y. A. Chung, M. C. Meglioli, D. Dale, N. Dong, M. Duppenthaler, and M. Williamson. 2023. Seamless: Multilingual Expressive and Streaming Speech Translation. arXiv preprint arXiv:2312.05187 (2023). arXiv:2312.05187 [cs.CL]"},{"key":"e_1_3_2_2_9_1","volume-title":"CEUR Workshop proceedings","volume":"2253","author":"Bassignana E.","unstructured":"E. Bassignana, V. Basile, and V. Patti. 2018. Hurtlex: A multilingual lexicon of words to hurt. In CEUR Workshop proceedings, Vol. 2253. CEUR-WS, 1--6."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612498"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.22"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00530-023-01051-8"},{"key":"e_1_3_2_2_13_1","volume-title":"Hatemm: A Multi-Modal Dataset for Hate Video Classification. In Proceedings of the International AAAI Conference on Web and Social Media","volume":"17","author":"Das M.","unstructured":"M. Das, R. Raj, P. Saha, B. Mathew, M. Gupta, and A. Mukherjee. 2023. Hatemm: A Multi-Modal Dataset for Hate Video Classification. In Proceedings of the International AAAI Conference on Web and Social Media, Vol. 17. 1014--1023."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1609\/icwsm.v11i1.14955"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_2_2_16_1","volume-title":"COLD: A benchmark for Chinese offensive language detection. arXiv preprint arXiv:2201.06025","author":"Deng J.","year":"2022","unstructured":"J. Deng, J. Zhou, H. Sun, C. Zheng, F. Mi, H. Meng, and M. Huang. 2022. COLD: A benchmark for Chinese offensive language detection. arXiv preprint arXiv:2201.06025 (2022)."},{"key":"e_1_3_2_2_17_1","volume-title":"BERT: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin J.","year":"2018","unstructured":"J. Devlin, M.-W. Chang, K. Lee, and K. Toutanova. 2018. BERT: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_2_18_1","unstructured":"A. Dosovitskiy L. Beyer A. Kolesnikov D. Weissenborn X. Zhai T. Unterthiner M. Dehghani M. Minderer G. Heigold S. Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_2_19_1","volume-title":"Overview of the EVALITA 2018 task on automatic misogyny identification (AMI). In CEUR Workshop Proceedings","volume":"2263","author":"Fersini Elisabetta","year":"2018","unstructured":"Elisabetta Fersini, Debora Nozza, and Paolo Rosso. 2018. Overview of the EVALITA 2018 task on automatic misogyny identification (AMI). In CEUR Workshop Proceedings, Vol. 2263. 1--9."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3232676"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1609\/icwsm.v12i1.14991"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.23919\/FUSION45008.2020.9190246"},{"key":"e_1_3_2_2_23_1","volume-title":"Detecting online hate speech using contextaware models. arXiv preprint arXiv:1710.07395","author":"Gao Lei","year":"2017","unstructured":"Lei Gao and Ruihong Huang. 2017. Detecting online hate speech using contextaware models. arXiv preprint arXiv:1710.07395 (2017)."},{"key":"e_1_3_2_2_24_1","volume-title":"Decoding the underlying meaning of multimodal hateful memes. arXiv preprint arXiv:2305.17678","author":"Hee Ming Shan","year":"2023","unstructured":"Ming Shan Hee, Wen-Haw Chong, and Roy Ka-Wei Lee. 2023. Decoding the underlying meaning of multimodal hateful memes. arXiv preprint arXiv:2305.17678 (2023)."},{"key":"e_1_3_2_2_25_1","volume-title":"Long short-term memory. Neural computation 9, 8","author":"Hochreiter Sepp","year":"1997","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long short-term memory. Neural computation 9, 8 (1997), 1735--1780."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.osnem.2021.100182"},{"key":"e_1_3_2_2_27_1","unstructured":"Jiasen Lei Licheng Li Li Zhou Zhe Gan Tamara L Berg Mohit Bansal and Jianfeng Liu. 2021. Less is More: ClipBERT for Video-and-Language Learning via Sparse Sampling. In CVPR. 7331--7341."},{"key":"e_1_3_2_2_28_1","volume-title":"Microsoft COCO: Common Objects in Context. In European Conference on Computer Vision.","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In European Conference on Computer Vision."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"crossref","unstructured":"J. Lu B. Xu X. Zhang C. Min L. Yang and H. Lin. 2023. Facilitating finegrained detection of Chinese toxic language: Hierarchical taxonomy resources and benchmarks. arXiv preprint arXiv:2305.04446 (2023).","DOI":"10.18653\/v1\/2023.acl-long.898"},{"key":"e_1_3_2_2_30_1","volume-title":"UniVL: A Unified Video and Language Pre-training Model for Multimodal Understanding and Generation. arXiv preprint arXiv:2002.06353","author":"Luo Huaishao","year":"2020","unstructured":"Huaishao Luo, Lei Ji, Baoyuan Shi, Hao Huang, Nan Duan, Tao Li,..., and Ming Zhou. 2020. UniVL: A Unified Video and Language Pre-training Model for Multimodal Understanding and Generation. arXiv preprint arXiv:2002.06353 (2020)."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"M. Mozafari R. Farahbakhsh and N. Crespi. 2020. A BERT-based transfer learning approach for hate speech detection in online social media. In Complex Networks and Their Applications VIII: Volume 1 Proceedings of the Eighth International Conference on Complex Networks and Their Applications COMPLEX NETWORKS 2019 8. Springer International Publishing 928--940.","DOI":"10.1007\/978-3-030-36687-2_77"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.35940\/ijeat.C3392.0211322"},{"key":"e_1_3_2_2_33_1","unstructured":"J. Redmon and A. Farhadi. 2018. YOLOv3: An Incremental Improvement. arXiv preprint arXiv:1804.02767 (2018)."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2022.01.001"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-1101"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/781"},{"key":"e_1_3_2_2_37_1","unstructured":"S. Singh S. Dewangan G. S. Krishna V. Tyagi S. Reddy and P. R. Medi. 2022. Video vision transformers for violence detection. arXiv preprint arXiv:2209.03561 (2022)."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00756"},{"key":"e_1_3_2_2_39_1","volume-title":"Proceedings of the 31st ACM conference on hypertext and social media. 139--140","author":"Trujillo Milo","unstructured":"Milo Trujillo, Maur\u00edcio Gruppi, Cody Buntain, and Benjamin D. Horne. 2020. What is BitChute? Characterizing the. In Proceedings of the 31st ACM conference on hypertext and social media. 139--140."},{"key":"e_1_3_2_2_40_1","volume-title":"IAPR Workshop on Artificial Neural Networks in Pattern Recognition. Springer International Publishing, Cham, 121--128","author":"Velankar A.","unstructured":"A. Velankar, H. Patil, and R. Joshi. 2022. Mono vs multilingual BERT for hate speech detection and text classification: A case study in Marathi. In IAPR Workshop on Artificial Neural Networks in Pattern Recognition. Springer International Publishing, Cham, 121--128."},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0243300"},{"key":"e_1_3_2_2_42_1","volume-title":"2021 12th International Conference on Computing Communication and Networking Technologies (ICCCNT). IEEE, 1--4.","author":"Vimal B.","unstructured":"B. Vimal, M. Surya, V. S. Sridhar, and A. Ashok. 2021. MFCC based audio classification using machine learning. In 2021 12th International Conference on Computing Communication and Networking Technologies (ICCCNT). IEEE, 1--4."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.5555\/2390374.2390377"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N16-2013"},{"key":"e_1_3_2_2_46_1","volume-title":"2020 International Conference on Computational Science and Computational Intelligence (CSCI). IEEE, 585--590","author":"Wu C. S.","unstructured":"C. S. Wu and U. Bhandary. 2020. Detection of hate speech in videos using machine learning. In 2020 International Conference on Computational Science and Computational Intelligence (CSCI). IEEE, 585--590."},{"key":"e_1_3_2_2_47_1","volume-title":"Vlm: Task-agnostic video-language model pre-training for video understanding. arXiv preprint arXiv:2105.09996","author":"Xu H.","year":"2021","unstructured":"H. Xu, G. Ghosh, P. Y. Huang, P. Arora, M. Aminzadeh, C. Feichtenhofer, et al. 2021. Vlm: Task-agnostic video-language model pre-training for video understanding. arXiv preprint arXiv:2105.09996 (2021)."},{"key":"e_1_3_2_2_48_1","volume-title":"Pacific Rim Conference on Multimedia. Springer, 566--574","author":"Xu M.","unstructured":"M. Xu, L.-Y. Duan, J. Cai, L.-T. Chia, C. Xu, and Q. Tian. 2004. HMM-based audio keyword generation. In Pacific Rim Conference on Multimedia. Springer, 566--574."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3109102"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-93417-4_48"},{"key":"e_1_3_2_2_51_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision. 14830--14840","author":"Zhao D.","unstructured":"D. Zhao, A. Wang, and O. Russakovsky. 2021. Understanding and evaluating racial biases in image captioning. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 14830--14840."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681521","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681521","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:48Z","timestamp":1750294668000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681521"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":51,"alternative-id":["10.1145\/3664647.3681521","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681521","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}