{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T13:03:33Z","timestamp":1785503013031,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","funder":[{"name":"Zhejiang Leading Innovative and Entrepreneur Team Introduction Program","award":["2024R01007"],"award-info":[{"award-number":["2024R01007"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,9]]},"DOI":"10.1145\/3770854.3780277","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:07:40Z","timestamp":1785499660000},"page":"1275-1286","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Medusa: Cross-Modal Transferable Adversarial Attacks on Multimodal Medical Retrieval-Augmented Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-2116-1892","authenticated-orcid":false,"given":"Yingjia","family":"Shang","sequence":"first","affiliation":[{"name":"Westlake University, Hangzhou, China and Heilongjiang University, Harbin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0811-6150","authenticated-orcid":false,"given":"Yi","family":"Liu","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6147-8310","authenticated-orcid":false,"given":"Huimin","family":"Wang","sequence":"additional","affiliation":[{"name":"Tencent YouTu Lab, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6352-2052","authenticated-orcid":false,"given":"Furong","family":"Li","sequence":"additional","affiliation":[{"name":"Westlake University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8830-2136","authenticated-orcid":false,"given":"Wenfang","family":"Sun","sequence":"additional","affiliation":[{"name":"Westlake University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0450-8649","authenticated-orcid":false,"given":"Chengyu","family":"Wu","sequence":"additional","affiliation":[{"name":"Westlake University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2195-2847","authenticated-orcid":false,"given":"Yefeng","family":"Zheng","sequence":"additional","affiliation":[{"name":"Westlake University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,20]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"[n.d.]. Main Page. https:\/\/mdwiki.org\/wiki\/Main_Page. Accessed: 2024-04-01."},{"key":"e_1_3_2_2_2_1","volume-title":"The Free Encyclopedia. https:\/\/www.wikipedia.org\/. Accessed: 2024-04-01","unstructured":"[n.d.]. Wikipedia, The Free Encyclopedia. https:\/\/www.wikipedia.org\/. Accessed: 2024-04-01."},{"key":"e_1_3_2_2_3_1","volume-title":"Torr","author":"Arnab Anurag","year":"2018","unstructured":"Anurag Arnab, Ondrej Miksik, and Philip H.S. Torr. 2018. On the Robustness of Semantic Segmentation Models to Adversarial Attacks. In Proc. of CVPR."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/293347.293348"},{"key":"e_1_3_2_2_5_1","volume-title":"Haystack: Open-Source Framework for Building Production-Ready LLM Applications. https:\/\/haystack.deepset.ai\/. Accessed: 2024-04-05.","year":"2024","unstructured":"deepset. 2024. Haystack: Open-Source Framework for Building Production-Ready LLM Applications. https:\/\/haystack.deepset.ai\/. Accessed: 2024-04-05."},{"key":"e_1_3_2_2_6_1","volume-title":"Proc. of USENIX security.","author":"Demontis Ambra","year":"2019","unstructured":"Ambra Demontis, Marco Melis, Maura Pintor, Matthew Jagielski, Battista Biggio, Alina Oprea, Cristina Nita-Rotaru, and Fabio Roli. 2019. Why do adversarial attacks transfer? explaining transferability of evasion and poisoning attacks. In Proc. of USENIX security."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3690624.3709296"},{"key":"e_1_3_2_2_8_1","volume-title":"FAISS: A Library for Efficient Similarity Search. https:\/\/github.com\/facebookresearch\/faiss. Accessed: 2023-10-01.","author":"Research Facebook AI","year":"2023","unstructured":"Facebook AI Research. 2023. FAISS: A Library for Efficient Similarity Search. https:\/\/github.com\/facebookresearch\/faiss. Accessed: 2023-10-01."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00291"},{"key":"e_1_3_2_2_10_1","volume-title":"Explaining and harnessing adversarial examples. arXiv preprint arXiv:1412.6572","author":"Goodfellow Ian J","year":"2014","unstructured":"Ian J Goodfellow, Jonathon Shlens, and Christian Szegedy. 2014. Explaining and harnessing adversarial examples. arXiv preprint arXiv:1412.6572 (2014)."},{"key":"e_1_3_2_2_11_1","volume-title":"MM-PoisonRAG: Disrupting Multimodal RAG with Local and Global Poisoning Attacks. arXiv preprint arXiv:2502.17832","author":"Ha Hyeonjeong","year":"2025","unstructured":"Hyeonjeong Ha, Qiusi Zhan, Jeonghwan Kim, Dimitrios Bralios, Saikrishna Sanniboina, Nanyun Peng, Kai-Wei Chang, Daniel Kang, and Heng Ji. 2025. MM-PoisonRAG: Disrupting Multimodal RAG with Local and Global Poisoning Attacks. arXiv preprint arXiv:2502.17832 (2025)."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Tianyu Han Sven Nebelung Firas Khader Tianci Wang Gustav M\u00fcller-Franzes Christiane Kuhl Sebastian F\u00f6rsch Jens Kleesiek Christoph Haarburger Keno K Bressem et al. 2024. Medical large language models are susceptible to targeted misinformation attacks. NPJ digital medicine 7 1 (2024) 288.","DOI":"10.1038\/s41746-024-01282-7"},{"key":"e_1_3_2_2_13_1","unstructured":"InfiniFlow. 2024. RAGFlow: A Visual LLM Pipeline for Document Intelligence and Retrieval-Augmented Generation. https:\/\/github.com\/infiniflow\/ragflow. Accessed: 2024-06-15."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00624"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671644"},{"key":"e_1_3_2_2_16_1","volume-title":"a de-identified publicly available database of chest radiographs with free-text reports. Scientific data 6, 1","author":"Johnson Alistair EW","year":"2019","unstructured":"Alistair EW Johnson, Tom J Pollard, Seth J Berkowitz, Nathaniel R Greenbaum, Matthew P Lungren, Chih-ying Deng, Roger G Mark, and Steven Horng. 2019. MIMIC-CXR, a de-identified publicly available database of chest radiographs with free-text reports. Scientific data 6, 1 (2019), 317."},{"key":"e_1_3_2_2_17_1","volume-title":"Roxana Daneshjou, and Su-In Lee.","author":"Kim Chanwoo","year":"2024","unstructured":"Chanwoo Kim, Soham U Gadgil, Alex J DeGrave, Jesutofunmi A Omiye, Zhuo Ran Cai, Roxana Daneshjou, and Su-In Lee. 2024. Transparent medical image AI via an image-text foundation model grounded in medical literature. Nature medicine 30, 4 (2024), 1154-1165."},{"key":"e_1_3_2_2_18_1","volume-title":"Proc. of NeurIPS.","author":"Lewis Patrick","year":"2020","unstructured":"Patrick Lewis, Ethan Perez, Aleksandra Piktus, Fabio Petroni, Vladimir Karpukhin, Naman Goyal, Heinrich K\u00fcttler, Mike Lewis,Wen-tau Yih, Tim Rockt\u00e4schel, et al. 2020. Retrieval-augmented generation for knowledge-intensive nlp tasks. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_19_1","volume-title":"Proc. of NeurIPS.","author":"Li Chunyuan","year":"2023","unstructured":"Chunyuan Li, Cliff Wong, Sheng Zhang, Naoto Usuyama, Haotian Liu, Jianwei Yang, Tristan Naumann, Hoifung Poon, and Jianfeng Gao. 2023. LLaVA-Med: Training a large language-and-vision assistant for biomedicine in one day. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_20_1","volume-title":"Proc. of ECCV.","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In Proc. of ECCV."},{"key":"e_1_3_2_2_21_1","volume-title":"Proc. of NeurIPS.","author":"Lin Weizhe","year":"2023","unstructured":"Weizhe Lin, Jinghong Chen, Jingbiao Mei, Alexandru Coca, and Bill Byrne. 2023. Fine-grained Late-interaction Multi-modal Retrieval for Retrieval Augmented Visual Question Answering. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-43993-3_51"},{"key":"e_1_3_2_2_23_1","unstructured":"Aixin Liu Bei Feng Bing Xue BingxuanWang BochaoWu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan et al. 2024. Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437 (2024)."},{"key":"e_1_3_2_2_24_1","volume-title":"Proc. of NeurIPS.","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual instruction tuning. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_25_1","volume-title":"Proc. of ICLR.","author":"Liu Yanpei","year":"2017","unstructured":"Yanpei Liu, Xinyun Chen, Chang Liu, and Dawn Song. 2017. Delving into Transferable Adversarial Examples and Black-box Attacks. In Proc. of ICLR."},{"key":"e_1_3_2_2_26_1","volume-title":"Proc. of CVPR.","author":"Majumdar Arjun","unstructured":"Arjun Majumdar, Anurag Ajay, Xiaohan Zhang, Pranav Putta, and et al. 2024. OpenEQA: Embodied Question Answering in the Era of Foundation Models. In Proc. of CVPR."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19809-0_39"},{"key":"e_1_3_2_2_28_1","volume-title":"Proc. of ICML.","author":"Nie Weili","year":"2022","unstructured":"Weili Nie, Brandon Guo, Yujia Huang, Chaowei Xiao, Arash Vahdat, and Animashree Anandkumar. 2022. Diffusion Models for Adversarial Purification. In Proc. of ICML."},{"key":"e_1_3_2_2_29_1","unstructured":"OpenAI. 2021. CLIP ViT-B\/16: Pretrained Model for Vision and Language Tasks. https:\/\/huggingface.co\/openai\/clip-vit-base-patch16. Accessed: 2024-04-01."},{"key":"e_1_3_2_2_30_1","volume-title":"Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, et al.","author":"Singhal Karan","year":"2023","unstructured":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, et al. 2023. Large language models encode clinical knowledge. Nature 620, 7972 (2023), 172-180."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890248"},{"key":"e_1_3_2_2_32_1","unstructured":"OpenAI Team. 2024. GPT-4o System Card. arXiv:2410.21276"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP54263.2024.00102"},{"key":"e_1_3_2_2_34_1","volume-title":"Proc. of ICML.","author":"Wang Tongzhou","year":"2020","unstructured":"Tongzhou Wang and Phillip Isola. 2020. Understanding contrastive representation learning through alignment and uniformity on the hypersphere. In Proc. of ICML."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3319535.3363253"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.256"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3330769"},{"key":"e_1_3_2_2_38_1","unstructured":"WinterSchool. 2024. MedificsDataset: A Dataset of Conversations on Radiology and Skin Cancer Images. https:\/\/huggingface.co\/datasets\/WinterSchool\/ MedificsDataset. Accessed: 2024-04-01."},{"key":"e_1_3_2_2_39_1","volume-title":"Proc. of ICLR.","author":"Xia Peng","year":"2025","unstructured":"Peng Xia, Kangyu Zhu, Haoran Li, Tianze Wang, Weijia Shi, Sheng Wang, Linjun Zhang, James Zou, and Huaxiu Yao. 2025. MMed-RAG: Versatile Multimodal RAG System for Medical Vision Language Models. In Proc. of ICLR."},{"key":"e_1_3_2_2_40_1","volume-title":"Proc. of ICLR.","author":"Xie Cihang","year":"2018","unstructured":"Cihang Xie, Jianyu Wang, Zhishuai Zhang, Zhou Ren, and Alan Yuille. 2018. Mitigating Adversarial Effects Through Randomization. In Proc. of ICLR."},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.372"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01456"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMI.2022.3156268"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681538"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.14722\/ndss.2018.23198"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2025.3590930"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714756"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3690624.3709440"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547801"},{"key":"e_1_3_2_2_50_1","unstructured":"Sheng Zhang Yanbo Xu Naoto Usuyama Hanwen Xu Jaspreet Bagga Robert Tinn Sam Preston Rajesh Rao Mu Wei Naveen Valluri et al. 2023. Biomed- CLIP: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs. arXiv preprint arXiv:2303.00915 (2023)."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72120-5_43"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714782"},{"key":"e_1_3_2_2_53_1","volume-title":"Proc. of NeurIPS.","author":"Zhao Yunqing","year":"2023","unstructured":"Yunqing Zhao, Tianyu Pang, Chao Du, Xiao Yang, Chongxuan LI, Ngai-Man (Man) Cheung, and Min Lin. 2023. On Evaluating Adversarial Robustness of Large Vision-Language Models. In Proc. of NeurIPS."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2023.3261910"}],"event":{"name":"KDD '26: The 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Jeju Island Republic of Korea","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3770854.3780277","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:18:35Z","timestamp":1785500315000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3770854.3780277"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,20]]},"references-count":54,"alternative-id":["10.1145\/3770854.3780277","10.1145\/3770854"],"URL":"https:\/\/doi.org\/10.1145\/3770854.3780277","relation":{},"subject":[],"published":{"date-parts":[[2026,4,20]]},"assertion":[{"value":"2026-04-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}