{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:11:11Z","timestamp":1784139071832,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"funder":[{"name":"National Natural Science Foundation of China","award":["62576120"],"award-info":[{"award-number":["62576120"]}]},{"name":"Research Grants Council of the Hong Kong Special Administrative Region, China","award":["No. T41-517&#x5c;&#x2f;25-N"],"award-info":[{"award-number":["No. T41-517&#x5c;&#x2f;25-N"]}]},{"name":"Research Grants Council of the Hong Kong Special Administrative Region, China","award":["PolyU&#x5c;&#x2f;25200821"],"award-info":[{"award-number":["PolyU&#x5c;&#x2f;25200821"]}]},{"name":"Innovation and Technology Fund","award":["No. PRP&#x5c;&#x2f;047&#x5c;&#x2f;22FX"],"award-info":[{"award-number":["No. PRP&#x5c;&#x2f;047&#x5c;&#x2f;22FX"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809880","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"4256-4260","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Negotiating the Punchline: Contextual Meme Understanding via Discrete Semantic Energy Minimization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0283-8400","authenticated-orcid":false,"given":"Bingbing","family":"Wang","sequence":"first","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China and The Hong Kong Polytechnic University, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2206-6355","authenticated-orcid":false,"given":"Zihan","family":"Wang","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2333-6105","authenticated-orcid":false,"given":"Zhengda","family":"Jin","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8044-2284","authenticated-orcid":false,"given":"Jing","family":"Li","sequence":"additional","affiliation":[{"name":"The Hong Kong Polytechnic University, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4009-5679","authenticated-orcid":false,"given":"Ruifeng","family":"Xu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China, Shenzhen Loop Area Institute, Shenzhen, China, and Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3895-5510","authenticated-orcid":false,"given":"Min","family":"Zhang","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657950"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al., 2024. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198."},{"key":"e_1_3_2_1_3_1","first-page":"1","article-title":"Scaling instruction-finetuned language models","volume":"25","author":"Chung Hyung Won","year":"2024","unstructured":"Hyung Won Chung, Le Hou, Shayne Longpre, Barret Zoph, Yi Tay, William Fedus, Yunxuan Li, Xuezhi Wang, Mostafa Dehghani, Siddhartha Brahma, et al., 2024. Scaling instruction-finetuned language models. Journal of Machine Learning Research, V25, 70 (2024), 1-53.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_4_1","volume-title":"Selfish genes and selfish memes. The mind's I: Fantasies and reflections on self and soul","author":"Dawkins Richard","year":"1981","unstructured":"Richard Dawkins. 1981. Selfish genes and selfish memes. The mind's I: Fantasies and reflections on self and soul (1981), 124-144."},{"key":"e_1_3_2_1_5_1","volume-title":"MemeMind: A Large-Scale Multimodal Dataset with Chain-of-Thought Reasoning for Harmful Meme Detection. arXiv preprint arXiv:2506.18919","author":"Gu Hexiang","year":"2025","unstructured":"Hexiang Gu, Qifan Yu, Saihui Hou, Zhiqin Fang, Huijia Wu, and Zhaofeng He. 2025. MemeMind: A Large-Scale Multimodal Dataset with Chain-of-Thought Reasoning for Harmful Meme Detection. arXiv preprint arXiv:2506.18919 (2025)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.89"},{"key":"e_1_3_2_1_7_1","volume-title":"V33","author":"Kiela Douwe","year":"2020","unstructured":"Douwe Kiela, Hamed Firooz, Aravind Mohan, Vedanuj Goswami, Amanpreet Singh, Pratik Ringshia, and Davide Testuggine. 2020. The hateful memes challenge: Detecting hate speech in multimodal memes. Advances in neural information processing systems, V33 (2020), 2611-2624."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.1234"},{"key":"e_1_3_2_1_9_1","volume-title":"Llava-onevision: Easy visual task transfer. arXiv preprint arXiv:2408.03326","author":"Li Bo","year":"2024","unstructured":"Bo Li, Yuanhan Zhang, Dong Guo, Renrui Zhang, Feng Li, Hao Zhang, Kaichen Zhang, Peiyuan Zhang, Yanwei Li, Ziwei Liu, et al., 2024. Llava-onevision: Easy visual task transfer. arXiv preprint arXiv:2408.03326 (2024)."},{"key":"e_1_3_2_1_10_1","unstructured":"Jiaze Li Jingyang Chen Yuxun Qu Shijie Xu Zhenru Lin Junyou Zhu Boshen Xu Wenhui Tan Pei Fu Jianzhong Ju et al. 2025. Xiaomi MiMo-VL-Miloco Technical Report. arXiv preprint arXiv:2512.17436 (2025)."},{"key":"e_1_3_2_1_11_1","unstructured":"Microsoft: Abdelrahman Abouelenin Atabak Ashfaq Adam Atkinson Hany Awadalla Nguyen Bach Jianmin Bao Alon Benhaim Martin Cai Vishrav Chaudhary Congcong Chen Dong Chen Dongdong Chen Junkun Chen Weizhu Chen Yen-Chun Chen Yi ling Chen Qi Dai Xiyang Dai Ruchao Fan Mei Gao Min Gao Amit Garg Abhishek Goswami Junheng Hao Amr Hendy Yuxuan Hu Xin Jin Mahmoud Khademi Dongwoo Kim Young Jin Kim Gina Lee Jinyu Li Yunsheng Li Chen Liang Xihui Lin Zeqi Lin Mengchen Liu Yang Liu Gilsinia Lopez Chong Luo Piyush Madan Vadim Mazalov Arindam Mitra Ali Mousavi Anh Nguyen Jing Pan Daniel Perez-Becker Jacob Platin Thomas Portet Kai Qiu Bo Ren Liliang Ren Sambuddha Roy Ning Shang Yelong Shen Saksham Singhal Subhojit Som Xia Song Tetyana Sych Praneetha Vaddamanu Shuohang Wang Yiming Wang Zhenghao Wang Haibin Wu Haoran Xu Weijian Xu Yifan Yang Ziyi Yang Donghan Yu Ishmam Zabir Jianwen Zhang Li Lyna Zhang Yunan Zhang and Xiren Zhou. 2025. Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs. arXiv:2503.01743 [cs.CL] https:\/\/arxiv.org\/abs\/2503.01743"},{"key":"e_1_3_2_1_12_1","volume-title":"Preslav Nakov, and Tanmoy Chakraborty.","author":"Pramanick Shraman","year":"2021","unstructured":"Shraman Pramanick, Shivam Sharma, Dimitar Dimitrov, Md Shad Akhtar, Preslav Nakov, and Tanmoy Chakraborty. 2021. MOMENTA: A Multimodal Framework for Detecting Harmful Memes and Their Targets. In Findings of the Association for Computational Linguistics: EMNLP 2021. 4439-4455."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Stephen Robertson Hugo Zaragoza et al. 2009. The probabilistic relevance framework: BM25 and beyond. Foundations and trends\u00ae in information retrieval V3 4 (2009) 333-389.","DOI":"10.1561\/1500000019"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.semeval-1.99"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i8.26166"},{"key":"e_1_3_2_1_16_1","volume-title":"MemHateCaptioning: Enhancing Hate Speech Detection in Memes with Context-Aware Captioning and Chain-of-Thought. In Companion Proceedings of the ACM on Web Conference 2025. 2034","author":"Sood Rishik","year":"2025","unstructured":"Rishik Sood, Ali Anaissi, Weidong Huang, and Ali Braytee. 2025. MemHateCaptioning: Enhancing Hate Speech Detection in Memes with Context-Aware Captioning and Chain-of-Thought. In Companion Proceedings of the ACM on Web Conference 2025. 2034-2041."},{"key":"e_1_3_2_1_17_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3532019"},{"key":"e_1_3_2_1_19_1","unstructured":"Jin Xu Zhifang Guo Jinzheng He Hangrui Hu Ting He Shuai Bai Keqin Chen Jialin Wang Yang Fan Kai Dang et al. 2025. Qwen2. 5-omni technical report. arXiv preprint arXiv:2503.20215 (2025)."},{"key":"e_1_3_2_1_20_1","unstructured":"An Yang Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoyan Huang Jiandong Jiang Jianhong Tu Jianwei Zhang Jingren Zhou et al. 2025. Qwen2. 5-1m technical report. arXiv preprint arXiv:2501.15383 (2025)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657740"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3767695.3769486"},{"key":"e_1_3_2_1_23_1","first-page":"188","volume-title":"Proceedings of the internet measurement conference","author":"Zannettou Savvas","year":"2018","unstructured":"Savvas Zannettou, Tristan Caulfield, Jeremy Blackburn, Emiliano De Cristofaro, Michael Sirivianos, Gianluca Stringhini, and Guillermo Suarez-Tangil. 2018. On the origins of memes by means of fringe web communities. In Proceedings of the internet measurement conference 2018. 188-202."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Zhengyi Zhao Shubo Zhang Yuxi Zhang Yanxi Zhao Yifan Zhang Zezhong Wang Huimin Wang Yutian Zhao Bin Liang Yefeng Zheng et al. 2025. MemeReaCon: Probing Contextual Meme Understanding in Large Vision-Language Models. arXiv preprint arXiv:2505.17433 (2025).","DOI":"10.18653\/v1\/2025.emnlp-main.176"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:30:52Z","timestamp":1784136652000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809880"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":24,"alternative-id":["10.1145\/3805712.3809880","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809880","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}