{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T09:06:37Z","timestamp":1784279197321,"version":"3.55.0"},"reference-count":40,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23A20315"],"award-info":[{"award-number":["U23A20315"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62572097"],"award-info":[{"award-number":["62572097"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100018542","name":"Natural Science Foundation of Sichuan Province","doi-asserted-by":"publisher","award":["24NSFSC1771"],"award-info":[{"award-number":["24NSFSC1771"]}],"id":[{"id":"10.13039\/501100018542","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100018542","name":"Natural Science Foundation of Sichuan Province","doi-asserted-by":"publisher","award":["2026NSFSC0428"],"award-info":[{"award-number":["2026NSFSC0428"]}],"id":[{"id":"10.13039\/501100018542","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113781","type":"journal-article","created":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T15:22:42Z","timestamp":1776871362000},"page":"113781","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["CapSRA: LMM-guided caption semantic retrieval aggregation for hateful meme detection"],"prefix":"10.1016","volume":"179","author":[{"given":"Pengfei","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haoyu","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yili","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jin","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kunpeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8038-8150","authenticated-orcid":false,"given":"Fan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113781_b1","doi-asserted-by":"crossref","unstructured":"H. Lin, Z. Luo, W. Gao, J. Ma, B. Wang, R. Yang, Towards explainable harmful meme detection through multimodal debate between large language models, in: Proceedings of the ACM Web Conference 2024, 2024, pp. 2359\u20132370.","DOI":"10.1145\/3589334.3645381"},{"key":"10.1016\/j.patcog.2026.113781_b2","doi-asserted-by":"crossref","unstructured":"R. Cao, R.K.-W. Lee, J. Jiang, Modularized networks for few-shot hateful meme detection, in: Proceedings of the ACM Web Conference 2024, 2024, pp. 4575\u20134584.","DOI":"10.1145\/3589334.3648145"},{"key":"10.1016\/j.patcog.2026.113781_b3","first-page":"2611","article-title":"The hateful memes challenge: Detecting hate speech in multimodal memes","volume":"33","author":"Kiela","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113781_b4","doi-asserted-by":"crossref","unstructured":"M.S. Hee, Z. Gao, Y. Wang, X. Chu, R.K.-W. Lee, Z. Qin, Contrastive instruction fine-tuning large multimodal model for hateful meme classification, in: Proceedings of the International AAAI Conference on Web and Social Media, vol. 19, 2025, pp. 760\u2013773.","DOI":"10.1609\/icwsm.v19i1.35844"},{"key":"10.1016\/j.patcog.2026.113781_b5","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113781_b6","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"issue":"11","key":"10.1016\/j.patcog.2026.113781_b7","doi-asserted-by":"crossref","first-page":"1886","DOI":"10.1177\/1461444814535194","article-title":"Memes as genre: A structurational analysis of the memescape","volume":"17","author":"Wiggins","year":"2015","journal-title":"New Media & Soc."},{"key":"10.1016\/j.patcog.2026.113781_b8","doi-asserted-by":"crossref","unstructured":"S. Zannettou, T. Caulfield, J. Blackburn, E. De Cristofaro, M. Sirivianos, G. Stringhini, G. Suarez-Tangil, On the origins of memes by means of fringe web communities, in: Proceedings of the Internet Measurement Conference 2018, 2018, pp. 188\u2013202.","DOI":"10.1145\/3278532.3278550"},{"key":"10.1016\/j.patcog.2026.113781_b9","doi-asserted-by":"crossref","unstructured":"J. Mei, J. Chen, W. Lin, B. Byrne, M. Tomalin, Improving Hateful Meme Detection through Retrieval-Guided Contrastive Learning, in: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, 2024, pp. 5333\u20135347.","DOI":"10.18653\/v1\/2024.acl-long.291"},{"key":"10.1016\/j.patcog.2026.113781_b10","doi-asserted-by":"crossref","unstructured":"M.S. Hee, W.-H. Chong, R.K.-W. Lee, Decoding the underlying meaning of multimodal hateful memes, in: Proceedings of the Thirty-Second International Joint Conference on Artificial Intelligence, 2023, pp. 5995\u20136003.","DOI":"10.24963\/ijcai.2023\/665"},{"key":"10.1016\/j.patcog.2026.113781_b11","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2021","first-page":"4439","article-title":"MOMENTA: A multimodal framework for detecting harmful memes and their targets","author":"Pramanick","year":"2021"},{"key":"10.1016\/j.patcog.2026.113781_b12","series-title":"Proceedings of the Second Workshop on NLP for Positive Impact","first-page":"171","article-title":"Hate-CLIPper: Multimodal hateful meme classification based on cross-modal interaction of CLIP features","author":"Kumar","year":"2022"},{"key":"10.1016\/j.patcog.2026.113781_b13","doi-asserted-by":"crossref","unstructured":"G. Burbi, A. Baldrati, L. Agnolucci, M. Bertini, A. Del Bimbo, Mapping memes to words for multimodal hateful meme classification, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 2832\u20132836.","DOI":"10.1109\/ICCVW60793.2023.00303"},{"key":"10.1016\/j.patcog.2026.113781_b14","doi-asserted-by":"crossref","unstructured":"S.B. Shah, S. Shiwakoti, M. Chaudhary, H. Wang, MemeCLIP: Leveraging CLIP Representations for Multimodal Meme Classification, in: Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, 2024, pp. 17320\u201317332.","DOI":"10.18653\/v1\/2024.emnlp-main.959"},{"key":"10.1016\/j.patcog.2026.113781_b15","doi-asserted-by":"crossref","unstructured":"R. Cao, R.K.-W. Lee, W.-H. Chong, J. Jiang, Prompting for Multimodal Hateful Meme Classification, in: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, 2022, pp. 321\u2013332.","DOI":"10.18653\/v1\/2022.emnlp-main.22"},{"key":"10.1016\/j.patcog.2026.113781_b16","doi-asserted-by":"crossref","unstructured":"R. Cao, M.S. Hee, A. Kuek, W.-H. Chong, R.K.-W. Lee, J. Jiang, Pro-cap: Leveraging a frozen vision-language model for hateful meme detection, in: Proceedings of the 31st ACM International Conference on Multimedia, 2023, pp. 5244\u20135252.","DOI":"10.1145\/3581783.3612498"},{"issue":"1","key":"10.1016\/j.patcog.2026.113781_b17","doi-asserted-by":"crossref","first-page":"45","DOI":"10.1016\/S0306-4573(02)00021-3","article-title":"An information-theoretic perspective of tf\u2013idf measures","volume":"39","author":"Aizawa","year":"2003","journal-title":"Inf. Process. Manage."},{"issue":"4","key":"10.1016\/j.patcog.2026.113781_b18","first-page":"333","article-title":"The probabilistic relevance framework: BM25 and beyond","volume":"3","author":"Robertson","year":"2009","journal-title":"Found. Trends\u00ae Inf. Retr."},{"key":"10.1016\/j.patcog.2026.113781_b19","doi-asserted-by":"crossref","unstructured":"V. Karpukhin, B. Oguz, S. Min, P.S. Lewis, L. Wu, S. Edunov, D. Chen, W.-t. Yih, Dense Passage Retrieval for Open-Domain Question Answering., in: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing, 2020, pp. 6769\u20136781.","DOI":"10.18653\/v1\/2020.emnlp-main.550"},{"key":"10.1016\/j.patcog.2026.113781_b20","article-title":"A novel hypergraph neural network combining multi-view learning with density awareness","author":"Liao","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113781_b21","doi-asserted-by":"crossref","unstructured":"J. Huang, H. Lin, L. Ziyan, Z. Luo, G. Chen, J. Ma, Towards low-resource harmful meme detection with LMM agents, in: Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, 2024, pp. 2269\u20132293.","DOI":"10.18653\/v1\/2024.emnlp-main.136"},{"key":"10.1016\/j.patcog.2026.113781_b22","doi-asserted-by":"crossref","unstructured":"J. Mei, J. Chen, G. Yang, W. Lin, B. Byrne, Robust adaptation of large multimodal models for retrieval augmented hateful meme detection, in: Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, 2025, pp. 23817\u201323839.","DOI":"10.18653\/v1\/2025.emnlp-main.1215"},{"key":"10.1016\/j.patcog.2026.113781_b23","article-title":"An efficient community-aware pre-training method for graph neural networks","author":"Huang","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113781_b24","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109859","article-title":"Graph-based pattern recognition on spectral reduced graphs","volume":"144","author":"Gillioz","year":"2023","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113781_b25","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2020.107544","article-title":"Relational graph neural network for situation recognition","volume":"108","author":"Jing","year":"2020","journal-title":"Pattern Recognit."},{"issue":"7","key":"10.1016\/j.patcog.2026.113781_b26","doi-asserted-by":"crossref","first-page":"2866","DOI":"10.1109\/TCSVT.2020.3030656","article-title":"Learning dual semantic relations with graph attention for image-text matching","volume":"31","author":"Wen","year":"2020","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.113781_b27","doi-asserted-by":"crossref","unstructured":"S. Kim, N. Lee, J. Lee, D. Hyun, C. Park, Heterogeneous graph learning for multi-modal medical data analysis, in: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, 2023, pp. 5141\u20135150.","DOI":"10.1609\/aaai.v37i4.25643"},{"key":"10.1016\/j.patcog.2026.113781_b28","series-title":"2021 IEEE 17th International Conference on EScience (EScience)","first-page":"186","article-title":"KnowMeme: A knowledge-enriched graph neural network solution to offensive meme detection","author":"Shang","year":"2021"},{"key":"10.1016\/j.patcog.2026.113781_b29","doi-asserted-by":"crossref","unstructured":"C. Yang, F. Zhu, J. Han, S. Hu, Invariant meets specific: A scalable harmful memes detection framework, in: Proceedings of the 31st ACM International Conference on Multimedia, 2023, pp. 4788\u20134797.","DOI":"10.1145\/3581783.3611761"},{"key":"10.1016\/j.patcog.2026.113781_b30","series-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing","first-page":"3982","article-title":"Sentence-BERT: Sentence embeddings using siamese BERT-networks","author":"Reimers","year":"2019"},{"issue":"1","key":"10.1016\/j.patcog.2026.113781_b31","doi-asserted-by":"crossref","first-page":"185","DOI":"10.1007\/s11263-023-01858-y","article-title":"Correlation information bottleneck: Towards adapting pretrained multimodal models for robust visual question answering","volume":"132","author":"Jiang","year":"2024","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.patcog.2026.113781_b32","doi-asserted-by":"crossref","first-page":"4121","DOI":"10.1109\/TMM.2022.3171679","article-title":"Multimodal information bottleneck: Learning minimal sufficient unimodal and multimodal representations","volume":"25","author":"Mai","year":"2022","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.113781_b33","series-title":"Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021","first-page":"2783","article-title":"Detecting harmful memes and their targets","author":"Pramanick","year":"2021"},{"key":"10.1016\/j.patcog.2026.113781_b34","doi-asserted-by":"crossref","unstructured":"E. Fersini, F. Gasparini, G. Rizzi, A. Saibene, B. Chulvi, P. Rosso, A. Lees, J. Sorensen, SemEval-2022 Task 5: Multimedia automatic misogyny identification, in: Proceedings of the 16th International Workshop on Semantic Evaluation, SemEval-2022, 2022, pp. 533\u2013549.","DOI":"10.18653\/v1\/2022.semeval-1.74"},{"key":"10.1016\/j.patcog.2026.113781_b35","series-title":"International Conference on Machine Learning","first-page":"5583","article-title":"Vilt: Vision-and-language transformer without convolution or region supervision","author":"Kim","year":"2021"},{"key":"10.1016\/j.patcog.2026.113781_b36","doi-asserted-by":"crossref","unstructured":"M. Tzelepi, V. Mezaris, Improving multimodal hateful meme detection exploiting LMM-generated knowledge, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2025, pp. 202\u2013211.","DOI":"10.1109\/CVPRW67362.2025.00025"},{"key":"10.1016\/j.patcog.2026.113781_b37","doi-asserted-by":"crossref","DOI":"10.3389\/fphy.2025.1614267","article-title":"A retrieval-augmented prompting network for hateful meme detection","volume":"13","author":"Kuang","year":"2025","journal-title":"Front. Phys."},{"key":"10.1016\/j.patcog.2026.113781_b38","series-title":"Qwen2. 5-vl technical report","author":"Bai","year":"2025"},{"key":"10.1016\/j.patcog.2026.113781_b39","doi-asserted-by":"crossref","unstructured":"H. Liu, C. Li, Y. Li, Y.J. Lee, Improved baselines with visual instruction tuning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 26296\u201326306.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"10.1016\/j.patcog.2026.113781_b40","series-title":"International Conference on Machine Learning","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326007466?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326007466?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T08:31:35Z","timestamp":1784277095000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326007466"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":40,"alternative-id":["S0031320326007466"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113781","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"CapSRA: LMM-guided caption semantic retrieval aggregation for hateful meme detection","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113781","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113781"}}