{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,2]],"date-time":"2025-08-02T17:05:50Z","timestamp":1754154350597,"version":"3.41.2"},"reference-count":38,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,12,13]],"date-time":"2024-12-13T00:00:00Z","timestamp":1734048000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,13]],"date-time":"2024-12-13T00:00:00Z","timestamp":1734048000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,12,13]]},"DOI":"10.1109\/hpcc64274.2024.00148","type":"proceedings-article","created":{"date-parts":[[2025,7,23]],"date-time":"2025-07-23T18:33:48Z","timestamp":1753295628000},"page":"1098-1105","source":"Crossref","is-referenced-by-count":0,"title":["GenHMD: Enhancing Hateful Meme Detection with Generated Rationale from Multimodal Large Language Models"],"prefix":"10.1109","author":[{"given":"Haimei","family":"Qin","sequence":"first","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Information Engineering,Beijing,China,100093"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiwei","family":"Yang","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Information Engineering,Beijing,China,100093"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaodong","family":"Tong","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Information Engineering,Beijing,China,100093"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lei","family":"Jiang","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Information Engineering,Beijing,China,100093"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.246"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3587819.3592545"},{"article-title":"Visualbert: A simple and performant baseline for vision and language","year":"2019","author":"Li","key":"ref3"},{"key":"ref4","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref5","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611761"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3695869"},{"key":"ref8","first-page":"2611","article-title":"The hateful memes challenge: Detecting hate speech in multimodal memes","volume":"33","author":"Kiela","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.nlp4pi-1.20"},{"article-title":"Evolver: Chain-of-evolution prompting to boost large multimodal models for hateful meme detection","year":"2024","author":"Huang","key":"ref10"},{"article-title":"Mathqa: Towards interpretable math word problem solving with operation-based formalisms","year":"2019","author":"Amini","key":"ref11"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.734"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612498"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645381"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1423"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/s13042-024-02443-6"},{"article-title":"Gpt-4 technical report","year":"2023","author":"Achiam","key":"ref17"},{"article-title":"Openflamingo: An open-source framework for training large autoregressive vision-language models","year":"2023","author":"Awadalla","key":"ref18"},{"article-title":"Gemini: a family of highly capable multimodal models","year":"2023","author":"Anil","key":"ref19"},{"article-title":"Visual instruction tuning","volume-title":"NeurIPS","author":"Liu","key":"ref20"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"ref22","first-page":"19 730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"International conference on machine learning","author":"Li"},{"article-title":"Llama: Open and efficient foundation language models","year":"2023","author":"Touvron","key":"ref23"},{"article-title":"Chain-of-thought prompt distillation for multi-modal named entity and multimodal relation extraction","year":"2023","author":"Chen","key":"ref24"},{"key":"ref25","first-page":"22 199","article-title":"Large language models are zero-shot reasoners","volume":"35","author":"Kojima","year":"2022","journal-title":"Advances in neural information processing systems"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681339"},{"key":"ref27","article-title":"Attention is all you need","author":"Vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/3571730"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"article-title":"Decoupled weight decay regularization","year":"2017","author":"Loshchilov","key":"ref30"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","year":"2020","author":"Dosovitskiy","key":"ref32"},{"key":"ref33","article-title":"Faster r-cnn: Towards real-time object detection with region proposal networks","volume":"28","author":"Ren","year":"2015","journal-title":"Advances in neural information processing systems"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.379"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.22"},{"key":"ref38","first-page":"32","article-title":"Multi-modal meme dataset (multioff) for identifying offensive content in image and text","volume-title":"Proceedings of the second workshop on trolling, aggression and cyberbullying","author":"Suryawanshi"}],"event":{"name":"2024 IEEE International Conference on High Performance Computing and Communications (HPCC)","start":{"date-parts":[[2024,12,13]]},"location":"Wuhan, China","end":{"date-parts":[[2024,12,15]]}},"container-title":["2024 IEEE International Conference on High Performance Computing and Communications (HPCC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11083115\/11083125\/11083262.pdf?arnumber=11083262","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T04:49:28Z","timestamp":1753332568000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11083262\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,13]]},"references-count":38,"URL":"https:\/\/doi.org\/10.1109\/hpcc64274.2024.00148","relation":{},"subject":[],"published":{"date-parts":[[2024,12,13]]}}}