{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T18:03:48Z","timestamp":1779991428021,"version":"3.53.1"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","funder":[{"name":"Austrian Science Fund","award":["Intent-aware Music Recommender Systems"],"award-info":[{"award-number":["Intent-aware Music Recommender Systems"]}]},{"name":"Bilateral Artificial Intelligence","award":["10.55776&#x5c;&#x2f;COE12"],"award-info":[{"award-number":["10.55776&#x5c;&#x2f;COE12"]}]},{"name":"Human-Centered Artificial Intelligence","award":["10.55776&#x5c;&#x2f;DFH23"],"award-info":[{"award-number":["10.55776&#x5c;&#x2f;DFH23"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,29]]},"DOI":"10.1145\/3774905.3795610","type":"proceedings-article","created":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T17:14:56Z","timestamp":1779988496000},"page":"273-278","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Robust Harmful Meme Detection under Missing Modalities via Shared Representation Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-4351-5588","authenticated-orcid":false,"given":"Felix","family":"Breiteneder","sequence":"first","affiliation":[{"name":"Johannes Kepler University, Linz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6665-5873","authenticated-orcid":false,"given":"Mohammad","family":"Belal","sequence":"additional","affiliation":[{"name":"Aalto University, Espoo, Finland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0893-9499","authenticated-orcid":false,"given":"Muhammad Saad","family":"Saeed","sequence":"additional","affiliation":[{"name":"University of Michigan, Flint, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2747-0386","authenticated-orcid":false,"given":"Shahed","family":"Masoudian","sequence":"additional","affiliation":[{"name":"Johannes Kepler University, Linz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0191-7171","authenticated-orcid":false,"given":"Usman","family":"Naseem","sequence":"additional","affiliation":[{"name":"Macquarie University, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4375-4641","authenticated-orcid":false,"given":"Kulshrestha","family":"Juhi","sequence":"additional","affiliation":[{"name":"Aalto University, Espoo, Finland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1706-3406","authenticated-orcid":false,"given":"Markus","family":"Schedl","sequence":"additional","affiliation":[{"name":"Johannes Kepler University, Linz, Austria and Linz Institute of Technology, Linz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7715-4409","authenticated-orcid":false,"given":"Shah","family":"Nawaz","sequence":"additional","affiliation":[{"name":"Johannes Kepler University, Linz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,5,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Shaden Shaar, Hamed Firooz, and Preslav Nakov.","author":"Alam Firoj","year":"2022","unstructured":"Firoj Alam, Stefano Cresci, Tanmoy Chakraborty, Fabrizio Silvestri, Dimiter Dimitrov, Giovanni Da San Martino, Shaden Shaar, Hamed Firooz, and Preslav Nakov. 2022. A Survey on Multimodal Disinformation Detection. In Proceedings of the 29th International Conference on Computational Linguistics, Nicoletta Calzolari, Chu-Ren Huang, Hansaem Kim, James Pustejovsky, Leo Wanner, Key-Sun Choi, Pum-Mo Ryu, Hsin-Hsi Chen, Lucia Donatelli, Heng Ji, Sadao Kurohashi, Patrizia Paggio, Nianwen Xue, Seokhwan Kim, Younggyun Hahm, Zhong He, Tony Kyungil Lee, Enrico Santus, Francis Bond, and Seung-Hoon Na (Eds.). International Committee on Computational Linguistics, Gyeongju, Republic of Korea, 6625-6643. https:\/\/aclanthology.org\/2022.coling-1.576\/"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612498"},{"key":"e_1_3_2_1_3_1","volume-title":"Unveiling Misogyny Memes: A Multimodal Analysis of Modality Effects on Identification. In Companion Proceedings of the ACM on Web Conference 2024","author":"Chen Shijing","year":"2024","unstructured":"Shijing Chen, Usman Naseem, Imran Razzak, and Flora Salim. 2024. Unveiling Misogyny Memes: A Multimodal Analysis of Modality Effects on Identification. In Companion Proceedings of the ACM on Web Conference 2024. 1864-1871."},{"key":"e_1_3_2_1_4_1","volume-title":"UNITER: UNiversal Image-TExt Representation Learning. In European Conference on Computer Vision. https:\/\/api.semanticscholar.org\/CorpusID:216080982","author":"Chen Yen-Chun","year":"2019","unstructured":"Yen-Chun Chen, Linjie Li, Licheng Yu, Ahmed El Kholy, Faisal Ahmed, Zhe Gan, Yu Cheng, and Jingjing Liu. 2019. UNITER: UNiversal Image-TExt Representation Learning. In European Conference on Computer Vision. https:\/\/api.semanticscholar.org\/CorpusID:216080982"},{"key":"e_1_3_2_1_5_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In North American","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In North American Chapter of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusID:52967399"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.136"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3648146"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3587427"},{"key":"e_1_3_2_1_9_1","volume-title":"Supervised multimodal bitransformers for classifying images and text. arXiv preprint arXiv:1909.02950","author":"Kiela Douwe","year":"2019","unstructured":"Douwe Kiela, Suvrat Bhooshan, Hamed Firooz, Ethan Perez, and Davide Testuggine. 2019. Supervised multimodal bitransformers for classifying images and text. arXiv preprint arXiv:1909.02950 (2019)."},{"key":"e_1_3_2_1_10_1","first-page":"2611","volume-title":"Lin (Eds.)","volume":"33","author":"Kiela Douwe","year":"2020","unstructured":"Douwe Kiela, Hamed Firooz, Aravind Mohan, Vedanuj Goswami, Amanpreet Singh, Pratik Ringshia, and Davide Testuggine. 2020a. The Hateful Memes Challenge: Detecting Hate Speech in Multimodal Memes. In Advances in Neural Information Processing Systems, H. Larochelle, M. Ranzato, R. Hadsell, M.F. Balcan, and H. Lin (Eds.), Vol. 33. Curran Associates, Inc., 2611-2624. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/file\/1b84c4cee2b8b3d823b30e2d604b1878-Paper.pdf"},{"key":"e_1_3_2_1_11_1","volume-title":"The hateful memes challenge: Detecting hate speech in multimodal memes. Advances in neural information processing systems","author":"Kiela Douwe","year":"2020","unstructured":"Douwe Kiela, Hamed Firooz, Aravind Mohan, Vedanuj Goswami, Amanpreet Singh, Pratik Ringshia, and Davide Testuggine. 2020b. The hateful memes challenge: Detecting hate speech in multimodal memes. Advances in neural information processing systems, Vol. 33 (2020), 2611-2624."},{"key":"e_1_3_2_1_12_1","first-page":"14943","volume-title":"Multimodal Prompting with Missing Modalities for Visual Recognition. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Lee Yi-Lun","year":"2023","unstructured":"Yi-Lun Lee, Yi-Hsuan Tsai, Wei-Chen Chiu, and Chen-Yu Lee. 2023. Multimodal Prompting with Missing Modalities for Visual Recognition. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2023), 14943-14952. https:\/\/api.semanticscholar.org\/CorpusID:257365349"},{"key":"e_1_3_2_1_13_1","volume-title":"Visualbert: A simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557","author":"Li Liunian Harold","year":"2019","unstructured":"Liunian Harold Li, Mark Yatskar, Da Yin, Cho-Jui Hsieh, and Kai-Wei Chang. 2019. Visualbert: A simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557 (2019)."},{"key":"e_1_3_2_1_14_1","volume-title":"Salman Khan, Elisabeth Andre, and Markus Schedl.","author":"Liaqat Muhammad Irzam","year":"2025","unstructured":"Muhammad Irzam Liaqat, Qaiser Abbas, Shah Nawaz, Zaigham Zaheer, Marta Moscati, Yufang Hou, Muhammad Haris Khan, Salman Khan, Elisabeth Andre, and Markus Schedl. 2025a. Multimodal Learning Under Imperfect Data Conditions: A Survey. Authorea Preprints (2025)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/s13735-025-00370-y"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.611"},{"key":"e_1_3_2_1_17_1","volume-title":"Paul Pu Liang, Amir Zadeh, and Louis-Philippe Morency.","author":"Liu Zhun","year":"2018","unstructured":"Zhun Liu, Ying Shen, Varun Bharadhwaj Lakshminarasimhan, Paul Pu Liang, Amir Zadeh, and Louis-Philippe Morency. 2018. Efficient Low-rank Multimodal Fusion With Modality-Specific Factors. ArXiv, Vol. abs\/1806.00064 (2018). https:\/\/api.semanticscholar.org\/CorpusID:44131945"},{"key":"e_1_3_2_1_18_1","unstructured":"Jiasen Lu Dhruv Batra Devi Parikh and Stefan Lee. 2019. ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks. In Neural Information Processing Systems. https:\/\/api.semanticscholar.org\/CorpusID:199453025"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01764"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01764"},{"key":"e_1_3_2_1_21_1","volume-title":"SMIL: Multimodal Learning with Severely Missing Modality. ArXiv","author":"Ma Mengmeng","year":"2021","unstructured":"Mengmeng Ma, Jian Ren, Long Zhao, S. Tulyakov, Cathy Wu, and Xi Peng. 2021. SMIL: Multimodal Learning with Severely Missing Modality. ArXiv, Vol. abs\/2103.05677 (2021). https:\/\/api.semanticscholar.org\/CorpusID:232170317"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539597.3570450"},{"key":"e_1_3_2_1_23_1","first-page":"439","volume-title":"Convolutional MKL Based Multimodal Emotion Recognition and Sentiment Analysis. 2016 IEEE 16th International Conference on Data Mining (ICDM)","author":"Poria Soujanya","year":"2016","unstructured":"Soujanya Poria, Iti Chaturvedi, E. Cambria, and Amir Hussain. 2016. Convolutional MKL Based Multimodal Emotion Recognition and Sentiment Analysis. 2016 IEEE 16th International Conference on Data Mining (ICDM) (2016), 439-448. https:\/\/api.semanticscholar.org\/CorpusID:5749615"},{"key":"e_1_3_2_1_24_1","volume-title":"Preslav Nakov, and Tanmoy Chakraborty.","author":"Pramanick Shraman","year":"2021","unstructured":"Shraman Pramanick, Dimitar Dimitrov, Rituparna Mukherjee, Shivam Sharma, Md Shad Akhtar, Preslav Nakov, and Tanmoy Chakraborty. 2021a. Detecting harmful memes and their targets. arXiv preprint arXiv:2110.00413 (2021)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.379"},{"key":"e_1_3_2_1_26_1","volume-title":"Muhammad Haris Khan, Karthik Nandakumar, Muhammad Haroon Yousaf, Hassan Sajjad, Tom De Schepper, and Markus Schedl.","author":"Saeed Muhammad Saad","year":"2024","unstructured":"Muhammad Saad Saeed, Shah Nawaz, Muhammad Zaigham Zaheer, Muhammad Haris Khan, Karthik Nandakumar, Muhammad Haroon Yousaf, Hassan Sajjad, Tom De Schepper, and Markus Schedl. 2024. Modality Invariant Multimodal Learning to Handle Missing Modalities: A Single-Branch Approach. arXiv e-prints (2024), arXiv-2408."},{"key":"e_1_3_2_1_27_1","volume-title":"Detecting Hateful Memes Using a Multimodal Deep Ensemble. ArXiv","author":"Sandulescu Vlad","year":"2020","unstructured":"Vlad Sandulescu. 2020. Detecting Hateful Memes Using a Multimodal Deep Ensemble. ArXiv, Vol. abs\/2012.13235 (2020). https:\/\/api.semanticscholar.org\/CorpusID:229371500"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/781"},{"key":"e_1_3_2_1_29_1","unstructured":"Shardul Suryawanshi and Bharathi Raja Chakravarthi. 2021. Findings of the Shared Task on Troll Meme Classification in Tamil. In DRAVIDIANLANGTECH. https:\/\/api.semanticscholar.org\/CorpusID:233365114"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"e_1_3_2_1_31_1","volume-title":"Detecting Hate Speech in Memes Using Multimodal Deep Learning Approaches: Prize-winning solution to Hateful Memes Challenge. ArXiv","author":"Velioglu Riza","year":"2020","unstructured":"Riza Velioglu and Jewgeni Rose. 2020. Detecting Hate Speech in Memes Using Multimodal Deep Learning Approaches: Prize-winning solution to Hateful Memes Challenge. ArXiv, Vol. abs\/2012.12975 (2020). https:\/\/api.semanticscholar.org\/CorpusID:229371544"},{"key":"e_1_3_2_1_32_1","first-page":"949","volume-title":"2017 IEEE International Conference on Multimedia and Expo (ICME)","author":"Wang Haohan","year":"2016","unstructured":"Haohan Wang, Aaksha Meghawat, Louis-Philippe Morency, and Eric P. Xing. 2016. Select-additive learning: Improving generalization in multimodal sentiment analysis. 2017 IEEE International Conference on Multimedia and Expo (ICME) (2016), 949-954. https:\/\/api.semanticscholar.org\/CorpusID:6765227"},{"key":"e_1_3_2_1_33_1","volume-title":"Tensor Fusion Network for Multimodal Sentiment Analysis. In Conference on Empirical Methods in Natural Language Processing. https:\/\/api.semanticscholar.org\/CorpusID:950292","author":"Zadeh Amir","year":"2017","unstructured":"Amir Zadeh, Minghai Chen, Soujanya Poria, E. Cambria, and Louis-Philippe Morency. 2017. Tensor Fusion Network for Multimodal Sentiment Analysis. In Conference on Empirical Methods in Natural Language Processing. https:\/\/api.semanticscholar.org\/CorpusID:950292"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3532064"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.203"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW53276.2021.9455994"}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Companion Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774905.3795610","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T17:17:57Z","timestamp":1779988677000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774905.3795610"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,28]]},"references-count":36,"alternative-id":["10.1145\/3774905.3795610","10.1145\/3774905"],"URL":"https:\/\/doi.org\/10.1145\/3774905.3795610","relation":{},"subject":[],"published":{"date-parts":[[2026,5,28]]},"assertion":[{"value":"2026-05-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}