{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T18:05:13Z","timestamp":1779991513776,"version":"3.53.1"},"publisher-location":"New York, NY, USA","reference-count":60,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,29]]},"DOI":"10.1145\/3774905.3795465","type":"proceedings-article","created":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T17:14:56Z","timestamp":1779988496000},"page":"935-944","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Can Thinking Models Think to Detect Hateful Memes?"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-4115-6057","authenticated-orcid":false,"given":"Mohamed Bayan","family":"Kmainasi","sequence":"first","affiliation":[{"name":"Qatar University, Doha, Qatar"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5660-4992","authenticated-orcid":false,"given":"Mucahid","family":"Kutlu","sequence":"additional","affiliation":[{"name":"Qatar University, Doha, Qatar"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3918-7471","authenticated-orcid":false,"given":"Ali","family":"Ezzat Shahroor","sequence":"additional","affiliation":[{"name":"Qatar Computing Research Institute, Doha, Qatar"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2748-8221","authenticated-orcid":false,"given":"Abul","family":"Hasnat","sequence":"additional","affiliation":[{"name":"APAVI.AI, Noisy Le Grand, France and Blackbird.AI, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7172-1997","authenticated-orcid":false,"given":"Firoj","family":"Alam","sequence":"additional","affiliation":[{"name":"Qatar Computing Research Institute, Doha, Qatar"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,5,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.11114\/smc.v13i2.7482","article-title":"The role of memes in shaping political discourse on social media","volume":"13","author":"AlAfnan Mohammad Awad","year":"2025","unstructured":"Mohammad Awad AlAfnan. 2025. The role of memes in shaping political discourse on social media. Studies in Media and Communication, Vol. 13, 2 (2025), 1-10.","journal-title":"Studies in Media and Communication"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.1173"},{"key":"e_1_3_2_1_3_1","first-page":"1","article-title":"Meme opinion categorization by using optical character recognition (OCR) and na\u00efve bayes algorithm. In 2018 third international conference on informatics and computing (ICIC)","author":"Amalia Amalia","year":"2018","unstructured":"Amalia Amalia, Amer Sharif, Fikri Haisar, Dani Gunawan, and Benny B Nasution. 2018. Meme opinion categorization by using optical character recognition (OCR) and na\u00efve bayes algorithm. In 2018 third international conference on informatics and computing (ICIC). IEEE, 1-5.","journal-title":"IEEE"},{"key":"e_1_3_2_1_4_1","unstructured":"Shuai Bai Yuxuan Cai Ruizhe Chen Keqin Chen Xionghui Chen Zesen Cheng Lianghao Deng Wei Ding Chang Gao Chunjiang Ge Wenbin Ge Zhifang Guo Qidong Huang Jie Huang Fei Huang Binyuan Hui Shutong Jiang Zhaohai Li Mingsheng Li Mei Li Kaixin Li Zicheng Lin Junyang Lin Xuejing Liu Jiawei Liu Chenglong Liu Yang Liu Dayiheng Liu Shixuan Liu Dunjie Lu Ruilin Luo Chenxu Lv Rui Men Lingchen Meng Xuancheng Ren Xingzhang Ren Sibo Song Yuchong Sun Jun Tang Jianhong Tu Jianqiang Wan Peng Wang Pengfei Wang Qiuyue Wang Yuxuan Wang Tianbao Xie Yiheng Xu Haiyang Xu Jin Xu Zhibo Yang Mingkun Yang Jianxin Yang An Yang Bowen Yu Fei Zhang Hang Zhang Xi Zhang Bo Zheng Humen Zhong Jingren Zhou Fan Zhou Jing Zhou Yuanzhi Zhu and Ke Zhu. 2025. Qwen3-VL Technical Report. arXiv preprint arXiv:2511.21631 (2025). https:\/\/arxiv.org\/abs\/2511.21631"},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005a. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72."},{"key":"e_1_3_2_1_6_1","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005b. METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments. In Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization Jade Goldstein Alon Lavie Chin-Yew Lin and Clare Voss (Eds.). Association for Computational Linguistics Ann Arbor Michigan 65-72. https:\/\/aclanthology.org\/W05-0909\/"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the Fourteenth Workshop on Semantic Evaluation. 1190-1194","author":"Boinepelli Sravani","year":"2020","unstructured":"Sravani Boinepelli, Manish Shrivastava, and Vasudeva Varma. 2020. Sis@ iiith at semeval-2020 task 8: An overview of simple text classification methods for meme analysis. In Proceedings of the Fourteenth Workshop on Semantic Evaluation. 1190-1194."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW60793.2023.00303"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612498"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.22"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3648145"},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing. 9097-9110","author":"Chen Jiajun","year":"2025","unstructured":"Jiajun Chen and Yik-Cheung Tam. 2025. Predicate-Guided Generation for Mathematical Reasoning. In Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing. 9097-9110."},{"key":"e_1_3_2_1_13_1","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:2507.06261 (2025)."},{"key":"e_1_3_2_1_14_1","unstructured":"DeepSeek-AI Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi Xiaokang Zhang Xingkai Yu Yu Wu Z. F. Wu Zhibin Gou Zhihong Shao Zhuoshu Li Ziyi Gao Aixin Liu Bing Xue Bingxuan Wang Bochao Wu Bei Feng Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan Damai Dai Deli Chen Dongjie Ji Erhang Li Fangyun Lin Fucong Dai Fuli Luo Guangbo Hao Guanting Chen Guowei Li H. Zhang Han Bao Hanwei Xu Haocheng Wang Honghui Ding Huajian Xin Huazuo Gao Hui Qu Hui Li Jianzhong Guo Jiashi Li Jiawei Wang Jingchang Chen Jingyang Yuan Junjie Qiu Junlong Li J. L. Cai Jiaqi Ni Jian Liang Jin Chen Kai Dong Kai Hu Kaige Gao Kang Guan Kexin Huang Kuai Yu Lean Wang Lecong Zhang Liang Zhao Litong Wang Liyue Zhang Lei Xu Leyi Xia Mingchuan Zhang Minghua Zhang Minghui Tang Meng Li Miaojun Wang Mingming Li Ning Tian Panpan Huang Peng Zhang Qiancheng Wang Qinyu Chen Qiushi Du Ruiqi Ge Ruisong Zhang Ruizhe Pan Runji Wang R. J. Chen R. L. Jin Ruyi Chen Shanghao Lu Shangyan Zhou Shanhuang Chen Shengfeng Ye Shiyu Wang Shuiping Yu Shunfeng Zhou Shuting Pan S. S. Li Shuang Zhou Shaoqing Wu Shengfeng Ye Tao Yun Tian Pei Tianyu Sun T. Wang Wangding Zeng Wanjia Zhao Wen Liu Wenfeng Liang Wenjun Gao Wenqin Yu Wentao Zhang W. L. Xiao Wei An Xiaodong Liu Xiaohan Wang Xiaokang Chen Xiaotao Nie Xin Cheng Xin Liu Xin Xie Xingchao Liu Xinyu Yang Xinyuan Li Xuecheng Su Xuheng Lin X. Q. Li Xiangyue Jin Xiaojin Shen Xiaosha Chen Xiaowen Sun Xiaoxiang Wang Xinnan Song Xinyi Zhou Xianzu Wang Xinxia Shan Y. K. Li Y. Q. Wang Y. X. Wei Yang Zhang Yanhong Xu Yao Li Yao Zhao Yaofeng Sun Yaohui Wang Yi Yu Yichao Zhang Yifan Shi Yiliang Xiong Ying He Yishi Piao Yisong Wang Yixuan Tan Yiyang Ma Yiyuan Liu Yongqiang Guo Yuan Ou Yuduan Wang Yue Gong Yuheng Zou Yujia He Yunfan Xiong Yuxiang Luo Yuxiang You Yuxuan Liu Yuyang Zhou Y. X. Zhu Yanhong Xu Yanping Huang Yaohui Li Yi Zheng Yuchen Zhu Yunxian Ma Ying Tang Yukun Zha Yuting Yan Z. Z. Ren Zehui Ren Zhangli Sha Zhe Fu Zhean Xu Zhenda Xie Zhengyan Zhang Zhewen Hao Zhicheng Ma Zhigang Yan Zhiyu Wu Zihui Gu Zijia Zhu Zijun Liu Zilin Li Ziwei Xie Ziyang Song Zizheng Pan Zhen Huang Zhipeng Xu Zhongyu Zhang and Zhen Zhang. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. arXiv:2501.12948 [cs.CL] https:\/\/arxiv.org\/abs\/2501.12948"},{"key":"e_1_3_2_1_15_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The Llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.semeval-1.74"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102269"},{"key":"e_1_3_2_1_18_1","volume-title":"Bridging Modalities: Enhancing Cross-Modality Hate Speech Detection with Few-Shot In-Context Learning","author":"Hee MS","year":"2024","unstructured":"MS Hee, A Kumaresan, and RKW Lee. [n.d.]. Bridging Modalities: Enhancing Cross-Modality Hate Speech Detection with Few-Shot In-Context Learning (2024)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1037\/0021-9010.69.1.85"},{"key":"e_1_3_2_1_20_1","volume-title":"Reinforcement Learning: A Survey. arXiv:cs\/9605103 [cs.AI] https:\/\/arxiv.org\/abs\/cs\/9605103","author":"Kaelbling L. P.","year":"1996","unstructured":"L. P. Kaelbling, M. L. Littman, and A. W. Moore. 1996. Reinforcement Learning: A Survey. arXiv:cs\/9605103 [cs.AI] https:\/\/arxiv.org\/abs\/cs\/9605103"},{"key":"e_1_3_2_1_21_1","unstructured":"Timo Kaufmann Paul Weng Viktor Bengs and Eyke H\u00fcllermeier. 2024. A survey of reinforcement learning from human feedback. (2024)."},{"key":"e_1_3_2_1_22_1","volume-title":"The hateful memes challenge: Detecting hate speech in multimodal memes. Advances in neural information processing systems","author":"Kiela Douwe","year":"2020","unstructured":"Douwe Kiela, Hamed Firooz, Aravind Mohan, Vedanuj Goswami, Amanpreet Singh, Pratik Ringshia, and Davide Testuggine. 2020. The hateful memes challenge: Detecting hate speech in multimodal memes. Advances in neural information processing systems, Vol. 33 (2020), 2611-2624."},{"key":"e_1_3_2_1_23_1","unstructured":"Douwe Kiela Hamed Firooz Aravind Mohan Vedanuj Goswami Amanpreet Singh Pratik Ringshia and Davide Testuggine. 2021. The Hateful Memes Challenge: Detecting Hate Speech in Multimodal Memes. arXiv:2005.04790 [cs.AI] https:\/\/arxiv.org\/abs\/2005.04790"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.1539"},{"key":"e_1_3_2_1_25_1","volume-title":"M3hop-cot: Misogynous meme identification with multimodal multi-hop chain-of-thought. arXiv preprint arXiv:2410.09220","author":"Kumari Gitanjali","year":"2024","unstructured":"Gitanjali Kumari, Kirtan Jain, and Asif Ekbal. 2024. M3hop-cot: Misogynous meme identification with multimodal multi-hop chain-of-thought. arXiv preprint arXiv:2410.09220 (2024)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","unstructured":"Ehsan Latif Yifan Zhou Shuchen Guo Yizhu Gao Lehong Shi Matthew Nyaaba Arne Bewerdorff Xiantong Yang Xiaoming Zhai et al. 2025. Comparative evaluation of OpenAI O1 and human performance in higher order cognition. Scientific Reports (2025). doi:10.1038\/s41598-025-33629-9","DOI":"10.1038\/s41598-025-33629-9"},{"key":"e_1_3_2_1_27_1","unstructured":"Mingxin Li Yanzhao Zhang Dingkun Long Keqin Chen Sibo Song Shuai Bai Zhibo Yang Pengjun Xie An Yang Dayiheng Liu et al. 2026. Qwen3-VL-Embedding and Qwen3-VL-Reranker: A Unified Framework for State-of-the-Art Multimodal Retrieval and Ranking. arXiv preprint arXiv:2601.04720 (2026)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645381"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.611"},{"key":"e_1_3_2_1_30_1","volume-title":"MIND: A Multi-agent Framework for Zero-shot Harmful Meme Detection. arXiv preprint arXiv:2507.06908","author":"Liu Ziyan","year":"2025","unstructured":"Ziyan Liu, Chunxiao Fan, Haoran Lou, Yuexin Wu, and Kaiwei Deng. 2025. MIND: A Multi-agent Framework for Zero-shot Harmful Meme Detection. arXiv preprint arXiv:2507.06908 (2025)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730014"},{"key":"e_1_3_2_1_32_1","volume-title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts. In International Conference on Learning Representations (ICLR). https:\/\/arxiv.org\/abs\/2310","author":"Lu Pan","year":"2024","unstructured":"Pan Lu, Hritik Bansal, Tony Xia, et al., 2024. MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts. In International Conference on Learning Representations (ICLR). https:\/\/arxiv.org\/abs\/2310.02255"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","first-page":"201","DOI":"10.18653\/v1\/2021.woah-1.21","volume-title":"Proceedings of the 5th Workshop on Online Abuse and Harms (WOAH","author":"Mathias Lambert","year":"2021","unstructured":"Lambert Mathias, Shaoliang Nie, Aida Mostafazadeh Davani, Douwe Kiela, Vinodkumar Prabhakaran, Bertie Vidgen, and Zeerak Talat. 2021a. Findings of the WOAH 5 shared task on fine grained hateful memes detection. In Proceedings of the 5th Workshop on Online Abuse and Harms (WOAH 2021). 201-206."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.woah-1.21"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.291"},{"key":"e_1_3_2_1_36_1","volume-title":"ExPO-HM: Learning to Explain-then-Detect for Hateful Meme Detection. arXiv preprint arXiv:2510.08630","author":"Mei Jingbiao","year":"2025","unstructured":"Jingbiao Mei, Mingsheng Sun, Jinghong Chen, Pengda Qin, Yuhong Li, Da Chen, and Bill Byrne. 2025. ExPO-HM: Learning to Explain-then-Detect for Hateful Meme Detection. arXiv preprint arXiv:2510.08630 (2025)."},{"key":"e_1_3_2_1_37_1","first-page":"205630512412962","article-title":"Never Mess With the ''Memers'': How Meme Creators Are Redefining Contemporary Politics","volume":"10","author":"Mih\u0103ilescu Mihaela-Georgiana","year":"2024","unstructured":"Mihaela-Georgiana Mih\u0103ilescu. 2024. Never Mess With the ''Memers'': How Meme Creators Are Redefining Contemporary Politics. Social Media Society, Vol. 10, 4 (2024), 20563051241296256.","journal-title":"Social Media Society"},{"key":"e_1_3_2_1_38_1","volume-title":"GPT-4 technical report. arXiv","author":"R","year":"2023","unstructured":"R OpenAI. 2023. GPT-4 technical report. arXiv (2023), 2303-08774."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Long Ouyang Jeffrey Wu Xu Jiang Diogo Almeida Carroll Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray et al. 2022. Training language models to follow instructions with human feedback. Advances in neural information processing systems Vol. 35 (2022) 27730-27744.","DOI":"10.52202\/068431-2011"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.osnem.2025.100317"},{"key":"e_1_3_2_1_41_1","volume-title":"Preslav Nakov, and Tanmoy Chakraborty.","author":"Pramanick Shraman","year":"2021","unstructured":"Shraman Pramanick, Shivam Sharma, Dimitar Dimitrov, Md Shad Akhtar, Preslav Nakov, and Tanmoy Chakraborty. 2021. MOMENTA: A multimodal framework for detecting harmful memes and their targets. arXiv preprint arXiv:2109.05184 (2021)."},{"key":"e_1_3_2_1_42_1","volume-title":"Direct preference optimization: Your language model is secretly a reward model. Advances in neural information processing systems","author":"Rafailov Rafael","year":"2023","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2023. Direct preference optimization: Your language model is secretly a reward model. Advances in neural information processing systems, Vol. 36 (2023), 53728-53741."},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers). 11566-11582","author":"Ranaldi Leonardo","year":"2025","unstructured":"Leonardo Ranaldi and Giulia Pucci. 2025. Multilingual Reasoning via Self-training. In Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers). 11566-11582."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"crossref","first-page":"1588","DOI":"10.1177\/14614448231198169","article-title":"Humorous hate speech on social media: A mixed-methods investigation of users' perceptions and processing of hateful memes","volume":"27","author":"Schmid Ursula Kristin","year":"2025","unstructured":"Ursula Kristin Schmid. 2025. Humorous hate speech on social media: A mixed-methods investigation of users' perceptions and processing of hateful memes. New Media & Society, Vol. 27, 3 (2025), 1588-1606.","journal-title":"New Media & Society"},{"key":"e_1_3_2_1_45_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_1_46_1","unstructured":"Zhihong Shao Peiyi Wang Qihao Zhu Runxin Xu Junxiao Song Xiao Bi Haowei Zhang Mingchuan Zhang Y. K. Li Y. Wu and Daya Guo. 2024. DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models. arXiv:2402.03300 [cs.CL] https:\/\/arxiv.org\/abs\/2402.03300"},{"key":"e_1_3_2_1_47_1","volume-title":"Dimitar Dimitrov, Giovanni Da San Martino, Hamed Firooz, Alon Halevy, Fabrizio Silvestri, Preslav Nakov, and Tanmoy Chakraborty.","author":"Sharma Shivam","year":"2022","unstructured":"Shivam Sharma, Firoj Alam, Md Shad Akhtar, Dimitar Dimitrov, Giovanni Da San Martino, Hamed Firooz, Alon Halevy, Fabrizio Silvestri, Preslav Nakov, and Tanmoy Chakraborty. 2022. Detecting and understanding harmful memes: A survey. arXiv preprint arXiv:2205.04274 (2022)."},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the Fourteenth Workshop on Semantic Evaluation. 891-900","author":"Shrestha Ingroj","year":"2020","unstructured":"Ingroj Shrestha and Jonathan Rusert. 2020. NLP_UIOWA at SemEval-2020 Task 8: You're not the only one cursed with knowledge-multi branch model memotion analysis. In Proceedings of the Fourteenth Workshop on Semantic Evaluation. 891-900."},{"key":"e_1_3_2_1_49_1","volume-title":"2025 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV). IEEE, 8617-8626","author":"Sun Li","year":"2025","unstructured":"Li Sun, Chaitanya Ahuja, Peng Chen, Matt D'Zmura, Kayhan Batmanghelich, and Philip Bontrager. 2025. Multi-Modal Large Language Models are Effective Vision Learners. In 2025 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV). IEEE, 8617-8626."},{"key":"e_1_3_2_1_50_1","unstructured":"Gemma Team Aishwarya Kamath Johan Ferret Shreya Pathak Nino Vieillard Ramona Merhej Sarah Perrin Tatiana Matejovicova Alexandre Ram\u00e9 Morgane Rivi\u00e8re et al. 2025. Gemma 3 technical report. arXiv preprint arXiv:2503.19786 (2025)."},{"key":"e_1_3_2_1_51_1","unstructured":"Lai Wei Yuting Li Chen Wang Yue Wang Linghe Kong Weiran Huang and Lichao Sun. 2025. First SFT Second RL Third UPT: Continual Improving Multi-Modal LLM Reasoning via Unsupervised Post-Training. arXiv:2505.22453 [cs.CL] https:\/\/arxiv.org\/abs\/2505.22453"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.3390\/electronics13142780"},{"key":"e_1_3_2_1_53_1","first-page":"69925","article-title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks","volume":"37","author":"Wu Jiannan","year":"2024","unstructured":"Jiannan Wu, Muyan Zhong, Sen Xing, Zeqiang Lai, Zhaoyang Liu, Zhe Chen, Wenhai Wang, Xizhou Zhu, Lewei Lu, Tong Lu, et al., 2024b. Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks. Advances in Neural Information Processing Systems, Vol. 37 (2024), 69925-69975.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_54_1","volume-title":"Sailing by the Stars: A Survey on Reward Models and Learning Strategies for Learning from Rewards. arXiv preprint arXiv:2505.02686","author":"Xiaobao Wu.","year":"2025","unstructured":"Xiaobao Wu. 2025. Sailing by the Stars: A Survey on Reward Models and Learning Strategies for Learning from Rewards. arXiv preprint arXiv:2505.02686 (2025)."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.6"},{"key":"e_1_3_2_1_56_1","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chang Gao Chengen Huang Chenxu Lv et al. 2025. Qwen3 technical report. arXiv preprint arXiv:2505.09388 (2025)."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.239"},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). https:\/\/arxiv.org\/abs\/2311","author":"Yue Xiang","year":"2024","unstructured":"Xiang Yue, Yuansheng Ni, Kai Zhang, et al., 2024. MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). https:\/\/arxiv.org\/abs\/2311.16502"},{"key":"e_1_3_2_1_59_1","unstructured":"Tianyi Zhang Varsha Kishore Felix Wu Kilian Q. Weinberger and Yoav Artzi. 2020. BERTScore: Evaluating Text Generation with BERT. arXiv:1904.09675 [cs.CL] https:\/\/arxiv.org\/abs\/1904.09675"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106594"}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Companion Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774905.3795465","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T17:27:21Z","timestamp":1779989241000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774905.3795465"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,28]]},"references-count":60,"alternative-id":["10.1145\/3774905.3795465","10.1145\/3774905"],"URL":"https:\/\/doi.org\/10.1145\/3774905.3795465","relation":{},"subject":[],"published":{"date-parts":[[2026,5,28]]},"assertion":[{"value":"2026-05-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}