{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T13:20:24Z","timestamp":1783171224208,"version":"3.54.6"},"reference-count":89,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100002367","name":"Chinese Academy of Sciences","doi-asserted-by":"publisher","award":["XDB0690303"],"award-info":[{"award-number":["XDB0690303"]}],"id":[{"id":"10.13039\/501100002367","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computers &amp; Security"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.cose.2026.104931","type":"journal-article","created":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T16:38:17Z","timestamp":1777307897000},"page":"104931","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Privacy-preserving for user-uploaded images and text in Vision\u2013Language Models"],"prefix":"10.1016","volume":"168","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-3106-7122","authenticated-orcid":false,"given":"Zixiang","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuguang","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9362-4460","authenticated-orcid":false,"given":"Weilong","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaojie","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peizhuo","family":"Lv","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cose.2026.104931_b1","doi-asserted-by":"crossref","unstructured":"Abadi, M., Chu, A., Goodfellow, I., McMahan, H.B., Mironov, I., Talwar, K., Zhang, L., 2016. Deep learning with differential privacy. In: Proceedings of the 2016 ACM SIGSAC Conference on Computer and Communications Security. pp. 308\u2013318.","DOI":"10.1145\/2976749.2978318"},{"key":"10.1016\/j.cose.2026.104931_b2","doi-asserted-by":"crossref","unstructured":"Aditya, P., Sen, R., Druschel, P., Joon Oh, S., Benenson, R., Fritz, M., Schiele, B., Bhattacharjee, B., Wu, T.T., 2016. I-pic: A platform for privacy-compliant image capture. In: Proceedings of the 14th Annual International Conference on Mobile Systems, Applications, and Services. pp. 235\u2013248.","DOI":"10.1145\/2906388.2906412"},{"key":"10.1016\/j.cose.2026.104931_b3","unstructured":"Agustsson, E., Timofte, R.N., 2016. challenge on single image super-resolution: Dataset and study. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops, Honolulu, HA, USA. pp. 21\u201326."},{"key":"10.1016\/j.cose.2026.104931_b4","series-title":"European Conference on Computer Vision","first-page":"469","article-title":"Face recognition with local binary patterns","author":"Ahonen","year":"2004"},{"key":"10.1016\/j.cose.2026.104931_b5","series-title":"Readings in Computer Vision","first-page":"671","article-title":"The Laplacian pyramid as a compact image code","author":"Burt","year":"1987"},{"key":"10.1016\/j.cose.2026.104931_b6","series-title":"The-x: Privacy-preserving transformer inference with homomorphic encryption","author":"Chen","year":"2022"},{"key":"10.1016\/j.cose.2026.104931_b7","series-title":"Microsoft coco captions: Data collection and evaluation server","author":"Chen","year":"2015"},{"key":"10.1016\/j.cose.2026.104931_b8","series-title":"Hide and seek (has): A lightweight framework for prompt privacy protection","author":"Chen","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b9","series-title":"European Conference on Computer Vision","first-page":"1","article-title":"Instructir: High-quality image restoration following human instructions","author":"Conde","year":"2025"},{"key":"10.1016\/j.cose.2026.104931_b10","series-title":"Instructblip: Towards general-purpose vision-language models with instruction tuning. arxiv 2023","author":"Dai","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b11","doi-asserted-by":"crossref","unstructured":"Dave, I.R., Chen, C., Shah, M., 2022. Spact: Self-supervised privacy preservation for action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 20164\u201320173.","DOI":"10.1109\/CVPR52688.2022.01953"},{"issue":"7","key":"10.1016\/j.cose.2026.104931_b12","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3716628","article-title":"Ai agents under threat: A survey of key security challenges and future pathways","volume":"57","author":"Deng","year":"2025","journal-title":"ACM Comput. Surv."},{"issue":"5","key":"10.1016\/j.cose.2026.104931_b13","doi-asserted-by":"crossref","first-page":"872","DOI":"10.1109\/JAS.2025.125498","article-title":"Exploring DeepSeek: A survey on advances, applications, challenges and future directions","volume":"12","author":"Deng","year":"2025","journal-title":"IEEE\/CAA J. Autom. Sin."},{"key":"10.1016\/j.cose.2026.104931_b14","doi-asserted-by":"crossref","DOI":"10.1109\/TIFS.2025.3581103","article-title":"Hardening llm fine-tuning: From differentially private data selection to trustworthy model quantization","author":"Deng","year":"2025","journal-title":"IEEE Trans. Inf. Forensics Secur."},{"key":"10.1016\/j.cose.2026.104931_b15","doi-asserted-by":"crossref","DOI":"10.1109\/TDSC.2025.3532957","article-title":"Leakage-resilient and carbon-neutral aggregation featuring the federated ai-enabled critical infrastructure","author":"Deng","year":"2025","journal-title":"IEEE Trans. Dependable Secur. Comput."},{"key":"10.1016\/j.cose.2026.104931_b16","series-title":"Delving into differentially private transformer","author":"Ding","year":"2024"},{"key":"10.1016\/j.cose.2026.104931_b17","doi-asserted-by":"crossref","unstructured":"Du, M., Yue, X., Chow, S.S., Wang, T., Huang, C., Sun, H., 2023. Dp-forward: Fine-tuning and inference on language models with differential privacy in forward pass. In: Proceedings of the 2023 ACM SIGSAC Conference on Computer and Communications Security. pp. 2665\u20132679.","DOI":"10.1145\/3576915.3616592"},{"key":"10.1016\/j.cose.2026.104931_b18","series-title":"The llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"10.1016\/j.cose.2026.104931_b19","series-title":"International Colloquium on Automata, Languages, and Programming","first-page":"1","article-title":"Differential privacy","author":"Dwork","year":"2006"},{"key":"10.1016\/j.cose.2026.104931_b20","article-title":"Learning to confuse: Generating training time adversarial data with auto-encoder","volume":"32","author":"Feng","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cose.2026.104931_b21","doi-asserted-by":"crossref","unstructured":"Feyisetan, O., Balle, B., Drake, T., Diethe, T., 2020. Privacy-and utility-preserving textual analysis via calibrated multivariate perturbations. In: Proceedings of the 13th International Conference on Web Search and Data Mining. pp. 178\u2013186.","DOI":"10.1145\/3336191.3371856"},{"key":"10.1016\/j.cose.2026.104931_b22","doi-asserted-by":"crossref","unstructured":"Fredrikson, M., Jha, S., Ristenpart, T., 2015. Model inversion attacks that exploit confidence information and basic countermeasures. In: Proceedings of the 22nd ACM SIGSAC Conference on Computer and Communications Security. pp. 1322\u20131333.","DOI":"10.1145\/2810103.2813677"},{"key":"10.1016\/j.cose.2026.104931_b23","series-title":"Retrieval-augmented generation for large language models: A survey","author":"Gao","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b24","doi-asserted-by":"crossref","unstructured":"Girdhar, R., El-Nouby, A., Liu, Z., Singh, M., Alwala, K.V., Joulin, A., Misra, I., 2023. Imagebind: One embedding space to bind them all. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 15180\u201315190.","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"10.1016\/j.cose.2026.104931_b25","doi-asserted-by":"crossref","first-page":"15718","DOI":"10.52202\/068431-1143","article-title":"Iron: Private inference on transformers","volume":"35","author":"Hao","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cose.2026.104931_b26","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2021","first-page":"1178","article-title":"Learning and evaluating a differentially private pre-trained language model","author":"Hoory","year":"2021"},{"key":"10.1016\/j.cose.2026.104931_b27","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2020","first-page":"1368","article-title":"TextHide: Tackling data privacy in language understanding tasks","author":"Huang","year":"2020"},{"key":"10.1016\/j.cose.2026.104931_b28","series-title":"European Conference on Computer Vision","first-page":"475","article-title":"Privacy-preserving face recognition with learnable privacy budgets in frequency domain","author":"Ji","year":"2022"},{"key":"10.1016\/j.cose.2026.104931_b29","unstructured":"Jin, S., Wang, H., Wang, Z., Xiao, F., Hu, J., He, Y., Zhang, W., Ba, Z., Fang, W., Yuan, S., et al., 2024. {FaceObfuscator}: Defending Deep Learning-based Privacy Attacks with Gradient Descent-resistant Features in Face Recognition. In: 33rd USENIX Security Symposium (USENIX Security 24). pp. 6849\u20136866."},{"issue":"12","key":"10.1016\/j.cose.2026.104931_b30","doi-asserted-by":"crossref","first-page":"2088","DOI":"10.4249\/scholarpedia.2088","article-title":"Signal-to-noise ratio","volume":"1","author":"Johnson","year":"2006","journal-title":"Scholarpedia"},{"key":"10.1016\/j.cose.2026.104931_b31","series-title":"Protecting user privacy in remote conversational systems: A privacy-preserving framework based on text sanitization","author":"Kan","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b32","series-title":"Differentially private language models benefit from public pre-training","author":"Kerrigan","year":"2020"},{"key":"10.1016\/j.cose.2026.104931_b33","series-title":"International Conference on Machine Learning","first-page":"5583","article-title":"Vilt: Vision-and-language transformer without convolution or region supervision","author":"Kim","year":"2021"},{"key":"10.1016\/j.cose.2026.104931_b34","doi-asserted-by":"crossref","first-page":"194","DOI":"10.21552\/edpl\/2020\/2\/7","article-title":"ISO\/IEC 27701 standard: Threats and opportunities for GDPR certification","volume":"6","author":"Lachaud","year":"2020","journal-title":"Eur. Data Prot. L. Rev."},{"key":"10.1016\/j.cose.2026.104931_b35","series-title":"Findings of the Association for Computational Linguistics: ACL 2024","first-page":"692","article-title":"Large language models can share images, too!","author":"Lee","year":"2024"},{"key":"10.1016\/j.cose.2026.104931_b36","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b37","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cose.2026.104931_b38","series-title":"Privacy-preserving prompt tuning for large language model services","author":"Li","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b39","series-title":"You can use but cannot recognize: Preserving visual privacy in deep neural networks","author":"Li","year":"2024"},{"key":"10.1016\/j.cose.2026.104931_b40","series-title":"Text Summarization Branches Out","first-page":"74","article-title":"ROUGE: A package for automatic evaluation of summaries","author":"Lin","year":"2004"},{"key":"10.1016\/j.cose.2026.104931_b41","series-title":"Deepseek-v3 technical report","author":"Liu","year":"2024"},{"key":"10.1016\/j.cose.2026.104931_b42","doi-asserted-by":"crossref","unstructured":"Liu, X., Jia, X., Xun, Y., Liang, S., Cao, X., 2024. Multimodal unlearnable examples: Protecting data against multimodal contrastive learning. In: Proceedings of the 32nd ACM International Conference on Multimedia. pp. 8024\u20138033.","DOI":"10.1145\/3664647.3680708"},{"key":"10.1016\/j.cose.2026.104931_b43","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J., 2024b. Improved baselines with visual instruction tuning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 26296\u201326306.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"10.1016\/j.cose.2026.104931_b44","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cose.2026.104931_b45","series-title":"Llms can understand encrypted prompt: Towards privacy-computing friendly transformers","author":"Liu","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b46","series-title":"European Conference on Computer Vision","first-page":"38","article-title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","author":"Liu","year":"2024"},{"key":"10.1016\/j.cose.2026.104931_b47","article-title":"Bumblebee: Secure two-party inference framework for large transformers","author":"Lu","year":"2023","journal-title":"Cryptol. EPrint Arch."},{"key":"10.1016\/j.cose.2026.104931_b48","series-title":"2023 IEEE Symposium on Security and Privacy","first-page":"346","article-title":"Analyzing leakage of personally identifiable information in language models","author":"Lukas","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b49","series-title":"Split-and-denoise: Protect large language model inference with local differential privacy","author":"Mai","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b50","series-title":"No images, no problem: Retaining knowledge in continual VQA with questions-only memory","author":"Marouf","year":"2025"},{"key":"10.1016\/j.cose.2026.104931_b51","article-title":"Towards a human in the loop approach to preserve privacy in images","volume":"vol. 2947","author":"Mauri","year":"2021"},{"key":"10.1016\/j.cose.2026.104931_b52","doi-asserted-by":"crossref","unstructured":"Mi, Y., Huang, Y., Ji, J., Liu, H., Xu, X., Ding, S., Zhou, S., 2022. Duetface: Collaborative privacy-preserving face recognition via channel splitting in the frequency domain. In: Proceedings of the 30th ACM International Conference on Multimedia. pp. 6755\u20136764.","DOI":"10.1145\/3503161.3548303"},{"key":"10.1016\/j.cose.2026.104931_b53","doi-asserted-by":"crossref","unstructured":"Mi, Y., Huang, Y., Ji, J., Zhao, M., Wu, J., Xu, X., Ding, S., Zhou, S., 2023. Privacy-preserving face recognition using random frequency components. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 19673\u201319684.","DOI":"10.1109\/ICCV51070.2023.01802"},{"key":"10.1016\/j.cose.2026.104931_b54","series-title":"Proceedings of the 14th International Joint Conference on Natural Language Processing and the 4th Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics","first-page":"2877","article-title":"Revision: A dataset and baseline VLM for privacy-preserving task-oriented visual instruction rewriting","author":"Mishra","year":"2025"},{"key":"10.1016\/j.cose.2026.104931_b55","doi-asserted-by":"crossref","unstructured":"Orekondy, T., Fritz, M., Schiele, B., 2018. Connecting pixels to privacy and utility: Automatic redaction of private information in images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 8466\u20138475.","DOI":"10.1109\/CVPR.2018.00883"},{"key":"10.1016\/j.cose.2026.104931_b56","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J., 2002. BLEU: a method for automatic evaluation of machine translation. In: Proceedings of ACL. pp. 311\u2013318.","DOI":"10.3115\/1073083.1073135"},{"key":"10.1016\/j.cose.2026.104931_b57","doi-asserted-by":"crossref","unstructured":"Qu, C., Kong, W., Yang, L., Zhang, M., Bendersky, M., Najork, M., 2021. Natural language understanding with privacy-preserving bert. In: Proceedings of the 30th ACM International Conference on Information & Knowledge Management. pp. 1488\u20131497.","DOI":"10.1145\/3459637.3482281"},{"key":"10.1016\/j.cose.2026.104931_b58","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.cose.2026.104931_b59","first-page":"2016","article-title":"Regulation (EU) 2016\/679 of the European parliament and of the council","volume":"679","author":"Regulation","year":"2016","journal-title":"Regul. (Eu)"},{"issue":"6","key":"10.1016\/j.cose.2026.104931_b60","doi-asserted-by":"crossref","first-page":"3913","DOI":"10.3390\/e17063913","article-title":"Entropy-based privacy against profiling of user mobility","volume":"17","author":"Rodriguez-Carrion","year":"2015","journal-title":"Entropy"},{"key":"10.1016\/j.cose.2026.104931_b61","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B., 2022. High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10684\u201310695.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"10.1016\/j.cose.2026.104931_b62","series-title":"Differentially private representation learning via image captioning","author":"Sander","year":"2024"},{"key":"10.1016\/j.cose.2026.104931_b63","doi-asserted-by":"crossref","unstructured":"Schroff, F., Kalenichenko, D., Philbin, J., 2015. Facenet: A unified embedding for face recognition and clustering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 815\u2013823.","DOI":"10.1109\/CVPR.2015.7298682"},{"issue":"3","key":"10.1016\/j.cose.2026.104931_b64","doi-asserted-by":"crossref","first-page":"379","DOI":"10.1002\/j.1538-7305.1948.tb01338.x","article-title":"A mathematical theory of communication","volume":"27","author":"Shannon","year":"1948","journal-title":"Bell Syst. Tech. J."},{"key":"10.1016\/j.cose.2026.104931_b65","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R., 2018. Conceptual Captions: A Cleaned, Hypernymed, Image Alt-text Dataset For Automatic Image Captioning. In: Proceedings of ACL.","DOI":"10.18653\/v1\/P18-1238"},{"key":"10.1016\/j.cose.2026.104931_b66","series-title":"A split-and-privatize framework for large language model fine-tuning","author":"Shen","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b67","series-title":"Gemma 3","author":"Team","year":"2025"},{"key":"10.1016\/j.cose.2026.104931_b68","unstructured":"Timofte, R., Gu, S., Wu, J., Van Gool, L.N., 2018. challenge on single image super-resolution: Methods and results. In: Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 852\u2013863."},{"key":"10.1016\/j.cose.2026.104931_b69","series-title":"Privacy-preserving personalized federated prompt learning for multimodal large language models","author":"Tran","year":"2025"},{"issue":"4","key":"10.1016\/j.cose.2026.104931_b70","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3528223.3530068","article-title":"Clipasso: Semantically-aware object sketching","volume":"41","author":"Vinker","year":"2022","journal-title":"ACM Trans. Graph."},{"issue":"4","key":"10.1016\/j.cose.2026.104931_b71","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1145\/103085.103089","article-title":"The JPEG still picture compression standard","volume":"34","author":"Wallace","year":"1991","journal-title":"Commun. ACM"},{"key":"10.1016\/j.cose.2026.104931_b72","series-title":"Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution","author":"Wang","year":"2024"},{"issue":"4","key":"10.1016\/j.cose.2026.104931_b73","doi-asserted-by":"crossref","first-page":"600","DOI":"10.1109\/TIP.2003.819861","article-title":"Image quality assessment: from error visibility to structural similarity","volume":"13","author":"Wang","year":"2004","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cose.2026.104931_b74","first-page":"2558","article-title":"Privacy-preserving face recognition in the frequency domain","volume":"vol. 36","author":"Wang","year":"2022"},{"key":"10.1016\/j.cose.2026.104931_b75","doi-asserted-by":"crossref","unstructured":"Wang, X., Xie, L., Dong, C., Shan, Y., 2021. Real-esrgan: Training real-world blind super-resolution with pure synthetic data. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 1905\u20131914.","DOI":"10.1109\/ICCVW54120.2021.00217"},{"key":"10.1016\/j.cose.2026.104931_b76","doi-asserted-by":"crossref","first-page":"197","DOI":"10.1016\/j.neucom.2022.06.039","article-title":"Identitydp: Differential private identification protection for face images","volume":"501","author":"Wen","year":"2022","journal-title":"Neurocomputing"},{"key":"10.1016\/j.cose.2026.104931_b77","doi-asserted-by":"crossref","unstructured":"Xu, K., Qin, M., Sun, F., Wang, Y., Chen, Y.-K., Ren, F., 2020. Learning in the frequency domain. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 1740\u20131749.","DOI":"10.1109\/CVPR42600.2020.00181"},{"key":"10.1016\/j.cose.2026.104931_b78","series-title":"Qwen3 technical report","author":"Yang","year":"2025"},{"key":"10.1016\/j.cose.2026.104931_b79","doi-asserted-by":"crossref","unstructured":"Ye, R., Wang, W., Chai, J., Li, D., Li, Z., Xu, Y., Du, Y., Wang, Y., Chen, S., 2024. Openfedllm: Training large language models on decentralized private data via federated learning. In: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining. pp. 6137\u20136147.","DOI":"10.1145\/3637528.3671582"},{"key":"10.1016\/j.cose.2026.104931_b80","doi-asserted-by":"crossref","DOI":"10.1136\/bmj-2022-072619","article-title":"China\u2019s personal information protection law","volume":"379","author":"Yin","year":"2022","journal-title":"BMJ"},{"key":"10.1016\/j.cose.2026.104931_b81","series-title":"Differentially private fine-tuning of language models","author":"Yu","year":"2021"},{"key":"10.1016\/j.cose.2026.104931_b82","series-title":"Coca: Contrastive captioners are image-text foundation models","author":"Yu","year":"2022"},{"key":"10.1016\/j.cose.2026.104931_b83","series-title":"Mm-vet: Evaluating large multimodal models for integrated capabilities","author":"Yu","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b84","series-title":"Secure transformer inference","author":"Yuan","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b85","doi-asserted-by":"crossref","unstructured":"Zhao, Z., Duan, J., Xu, K., Wang, C., Zhang, R., Du, Z., Guo, Q., Hu, X., 2024. Can Protective Perturbation Safeguard Personal Data from Being Exploited by Stable Diffusion?. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 24398\u201324407.","DOI":"10.1109\/CVPR52733.2024.02303"},{"key":"10.1016\/j.cose.2026.104931_b86","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 3: System Demonstrations)","article-title":"LlamaFactory: Unified efficient fine-tuning of 100+ language models","author":"Zheng","year":"2024"},{"key":"10.1016\/j.cose.2026.104931_b87","doi-asserted-by":"crossref","unstructured":"Zhou, X., Lu, J., Gui, T., Ma, R., Fei, Z., Wang, Y., Ding, Y., Cheung, Y., Zhang, Q., Huang, X.-J., 2022. Textfusion: Privacy-preserving pre-trained model inference via token fusion. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing. pp. 8360\u20138371.","DOI":"10.18653\/v1\/2022.emnlp-main.572"},{"key":"10.1016\/j.cose.2026.104931_b88","series-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","year":"2023"},{"key":"10.1016\/j.cose.2026.104931_b89","series-title":"Converting transformers to polynomial form for secure inference over homomorphic encryption","author":"Zimerman","year":"2023"}],"container-title":["Computers &amp; Security"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167404826001070?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167404826001070?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T12:53:08Z","timestamp":1783169588000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167404826001070"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":89,"alternative-id":["S0167404826001070"],"URL":"https:\/\/doi.org\/10.1016\/j.cose.2026.104931","relation":{},"ISSN":["0167-4048"],"issn-type":[{"value":"0167-4048","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Privacy-preserving for user-uploaded images and text in Vision\u2013Language Models","name":"articletitle","label":"Article Title"},{"value":"Computers & Security","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cose.2026.104931","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104931"}}