{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T05:25:04Z","timestamp":1755926704547,"version":"3.40.3"},"publisher-location":"Cham","reference-count":52,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031198359"},{"type":"electronic","value":"9783031198366"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-19836-6_1","type":"book-chapter","created":{"date-parts":[[2022,10,21]],"date-time":"2022-10-21T09:04:58Z","timestamp":1666343098000},"page":"1-18","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Most and\u00a0Least Retrievable Images in\u00a0Visual-Language Query Systems"],"prefix":"10.1007","author":[{"given":"Liuwan","family":"Zhu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rui","family":"Ning","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiang","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunsheng","family":"Xin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongyi","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,10,22]]},"reference":[{"key":"1_CR1","doi-asserted-by":"crossref","unstructured":"Acar, G., Eubank, C., Englehardt, S., Juarez, M., Narayanan, A., Diaz, C.: The web never forgets: persistent tracking mechanisms in the wild. In: Proceedings of the ACM SIGSAC Conference on Computer and Communications Security(CCS), pp. 674\u2013689 (2014)","DOI":"10.1145\/2660267.2660347"},{"key":"1_CR2","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition(CVPR), pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"1_CR3","unstructured":"Thomas, A.: Ogiz Elibol: defense against adversarial attack-rank3. \u2018github.com\/anlthms\/nips-2017\/tree\/master\/mmd\u2019 (2017)"},{"key":"1_CR4","doi-asserted-by":"crossref","unstructured":"Antol, S., et al.: VQA: visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision(ICCV), pp. 2425\u20132433 (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"1_CR5","unstructured":"Bachman, P., Hjelm, R.D., Buchwalter, W.: Learning representations by maximizing mutual information across views. In: Proceedings of the International Conference on Neural Information Processing Systems (NeurIPS) (2019)"},{"key":"1_CR6","unstructured":"Belghazi, M.I., et al.: Mutual information neural estimation. In: Proceedings of the International Conference on Machine Learning (ICML), pp. 531\u2013540 (2018)"},{"key":"1_CR7","unstructured":"Benjamin, E.: False and deceptive display ads at yahoo\u2019s right media. www.benedelman.org\/rightmedia-deception (2009)"},{"key":"1_CR8","doi-asserted-by":"crossref","unstructured":"Carlini, N., Wagner, D.: Towards evaluating the robustness of neural networks. In: Proceedings of the IEEE Symposium on Security and Privacy (S&P), pp. 39\u201357 (2017)","DOI":"10.1109\/SP.2017.49"},{"key":"1_CR9","doi-asserted-by":"crossref","unstructured":"Changpinyo, S., Sharma, P., Ding, N., Soricut, R.: Conceptual 12m: pushing web-scale image-text pre-training to recognize long-tail visual concepts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3558\u20133568 (2021)","DOI":"10.1109\/CVPR46437.2021.00356"},{"key":"1_CR10","doi-asserted-by":"crossref","unstructured":"Chen, H., Zhang, H., Chen, P.Y., Yi, J., Hsieh, C.J.: Attacking visual language grounding with adversarial examples: a case study on neural image captioning. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics (ACL) (2018)","DOI":"10.18653\/v1\/P18-1241"},{"key":"1_CR11","doi-asserted-by":"crossref","unstructured":"Chen, Y.C., et al.: Uniter: universal image-text representation learning. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 104\u2013120 (2020)","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"1_CR12","unstructured":"Croce, F., Hein, M.: Reliable evaluation of adversarial robustness with an ensemble of diverse parameter-free attacks. In: Proceedings of the International Conference on Machine Learning(ICML), pp. 2206\u20132216 (2020)"},{"key":"1_CR13","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition(CVPR), pp. 248\u2013255 (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1_CR14","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"1_CR15","doi-asserted-by":"crossref","unstructured":"Dzabraev, M., Kalashnikov, M., Komkov, S., Petiushko, A.: MDMMT: multidomain multimodal transformer for video retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3354\u20133363 (2021)","DOI":"10.1109\/CVPRW53098.2021.00374"},{"key":"1_CR16","doi-asserted-by":"crossref","unstructured":"Gao, J., Lanchantin, J., Soffa, M.L., Qi, Y.: Black-box generation of adversarial text sequences to evade deep learning classifiers. In: Proceedings of the IEEE Security and Privacy Workshops (SPW), pp. 50\u201356 (2018)","DOI":"10.1109\/SPW.2018.00016"},{"key":"1_CR17","unstructured":"Goodfellow, I.J., Shlens, J., Szegedy, C.: Explaining and harnessing adversarial examples. In: Proceedings of the International Conference on Learning Representations (ICLR) (2015)"},{"key":"1_CR18","unstructured":"Guo, C., Rana, M., Ciss\u00e9, M., van der Maaten, L.: Countering adversarial images using input transformations. In: 6th International Conference on Learning Representations, ICLR (2018)"},{"key":"1_CR19","doi-asserted-by":"crossref","unstructured":"Han, Y., Shen, Y.: Accurate spear phishing campaign attribution and early detection. In: Proceedings of the Annual ACM Symposium on Applied Computing(SAC), pp. 2079\u20132086 (2016)","DOI":"10.1145\/2851613.2851801"},{"key":"1_CR20","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition(CVPR), pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"1_CR21","doi-asserted-by":"crossref","unstructured":"Ji, J., et al.: Attacking image captioning towards accuracy-preserving target words removal. In: Proceedings of the ACM International Conference on Multimedia(ACMMM), pp. 4226\u20134234 (2020)","DOI":"10.1145\/3394171.3414009"},{"key":"1_CR22","doi-asserted-by":"crossref","unstructured":"Li, G., Duan, N., Fang, Y., Gong, M., Jiang, D.: Unicoder-VL: a universal encoder for vision and language by cross-modal pre-training. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11336\u201311344 (2020)","DOI":"10.1609\/aaai.v34i07.6795"},{"key":"1_CR23","doi-asserted-by":"crossref","unstructured":"Li, J., Ji, R., Liu, H., Hong, X., Gao, Y., Tian, Q.: Universal perturbation attack against image retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4899\u20134908 (2019)","DOI":"10.1109\/ICCV.2019.00500"},{"key":"1_CR24","doi-asserted-by":"crossref","unstructured":"Li, L., Ma, R., Guo, Q., Xue, X., Qiu, X.: BERT-ATTACK: adversarial attack against BERT using BERT. In: Proceedings of the IEEE Conference on Empirical Methods in Natural Language Processing (EMNLP) (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.500"},{"key":"1_CR25","unstructured":"Li, L.H., Yatskar, M., Yin, D., Hsieh, C.J., Chang, K.W.: VisualBERT: a simple and performant baseline for vision and language. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics (ACL) (2019)"},{"key":"1_CR26","doi-asserted-by":"crossref","unstructured":"Li, L.H., Yatskar, M., Yin, D., Hsieh, C.J., Chang, K.W.: What does BERT with vision look at? In: Proceedings of the Annual Meeting of the Association for Computational Linguistics(ACL), pp. 5265\u20135275 (2020)","DOI":"10.18653\/v1\/2020.acl-main.469"},{"key":"1_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"1_CR28","doi-asserted-by":"crossref","unstructured":"Liu, Z., Zhao, Z., Larson, M.: Who\u2019s afraid of adversarial queries? the impact of image modifications on content-based image retrieval. In: Proceedings of the Annual ACM International Conference on Multimedia Retrieval(ICMR), pp. 306\u2013314 (2019)","DOI":"10.1145\/3323873.3325052"},{"key":"1_CR29","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: VILBERT: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Proceedings of the Advances in Neural Information Processing Systems(NeurIPS), vol. 32 (2019)"},{"issue":"86","key":"1_CR30","first-page":"2579","volume":"9","author":"L van der Maaten","year":"2008","unstructured":"van der Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9(86), 2579\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."},{"key":"1_CR31","unstructured":"Madry, A., Makelov, A., Schmidt, L., Tsipras, D., Vladu, A.: Towards deep learning models resistant to adversarial attacks. In: Proceedings of the International Conference on Learning Representations (ICLR) (2018)"},{"key":"1_CR32","unstructured":"Mosbach, M., Andriushchenko, M., Trost, T., Hein, M., Klakow, D.: Logit pairing methods can fool gradient-based attacks. In: Proceedings of the NeurIPS Workshop on Security in Machine Learning (2018)"},{"key":"1_CR33","doi-asserted-by":"crossref","unstructured":"Philbin, J., Chum, O., Isard, M., Sivic, J., Zisserman, A.: Lost in quantization: improving particular object retrieval in large scale image databases. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition(CVPR), pp. 1\u20138. IEEE (2008)","DOI":"10.1109\/CVPR.2008.4587635"},{"key":"1_CR34","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. arXiv preprint arXiv:2103.00020 (2021)"},{"key":"1_CR35","doi-asserted-by":"crossref","unstructured":"Reznichenko, A., Francis, P.: Private-by-design advertising meets the real world. In: Proceedings of the ACM SIGSAC Conference on Computer and Communications Security(CCS), pp. 116\u2013128 (2014)","DOI":"10.1145\/2660267.2660305"},{"key":"1_CR36","unstructured":"Sayak Paul, P.Y.C.: Vision transformers are robust learners. arXiv preprint arXiv:2105.07581 (2021)"},{"key":"1_CR37","doi-asserted-by":"crossref","unstructured":"Sharma, V., Kalra, A., Vaibhav, Chaudhary, S., Patel, L., Morency, L.: Attend and attack: attention guided adversarial attacks on visual question answering models. In: Proceedings of the Advances in Neural Information Processing Systems (NeurIPS) (2018)","DOI":"10.18653\/v1\/P17-3008"},{"key":"1_CR38","unstructured":"Su, W., et al.: VL-BERT: pre-training of generic visual-linguistic representations. In: Proceedings of the International Conference on Learning Representations (ICLR) (2020)"},{"key":"1_CR39","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: LXMERT: learning cross-modality encoder representations from transformers. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing (EMNLP) (2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"1_CR40","unstructured":"Tan, M., Le, Q.: EfficientNet: rethinking model scaling for convolutional neural networks. In: Proceedings of the International Conference on Machine Learning(ICML), pp. 6105\u20136114 (2019)"},{"key":"1_CR41","unstructured":"The New York Times: clearview ai\u2019s facial recognition app called illegal in canada. www.nytimes.com\/2021\/02\/03\/technology\/clearview-ai-illegal-canada.html (2021). Accessed 03 Feb 2021"},{"key":"1_CR42","doi-asserted-by":"crossref","unstructured":"Tian, Y., Krishnan, D., Isola, P.: Contrastive multiview coding. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 776\u2013794 (2020)","DOI":"10.1007\/978-3-030-58621-8_45"},{"key":"1_CR43","doi-asserted-by":"crossref","unstructured":"Tolias, G., Radenovic, F., Chum, O.: Targeted mismatch adversarial attack: query with a flower to retrieve the tower. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision(ICCV) pp. 5037\u20135046 (2019)","DOI":"10.1109\/ICCV.2019.00514"},{"key":"1_CR44","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Proceedings of the 31st International Conference on Neural Information Processing Systems, pp. 6000\u20136010 (2017)"},{"key":"1_CR45","unstructured":"Xie, C., Wang, J., Zhang, Z., Ren, Z., Yuille, A.L.: Mitigating adversarial effects through randomization. In: 6th International Conference on Learning Representations, ICLR (2018)"},{"key":"1_CR46","doi-asserted-by":"crossref","unstructured":"Xie, C., et al.: Improving transferability of adversarial examples with input diversity. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2730\u20132739 (2019)","DOI":"10.1109\/CVPR.2019.00284"},{"key":"1_CR47","doi-asserted-by":"crossref","unstructured":"Xu, W., Evans, D., Qi, Y.: Feature squeezing: detecting adversarial examples in deep neural networks. In: 25th Annual Network and Distributed System Security Symposium NDSS (2018)","DOI":"10.14722\/ndss.2018.23198"},{"key":"1_CR48","doi-asserted-by":"crossref","unstructured":"Xu, X., Chen, X., Liu, C., Rohrbach, A., Darrell, T., Song, D.: Fooling vision and language models despite localization and attention mechanism. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition(CVPR), pp. 4951\u20134961 (2018)","DOI":"10.1109\/CVPR.2018.00520"},{"key":"1_CR49","doi-asserted-by":"crossref","unstructured":"Xu, Y., et al.: Exact adversarial attack to image captioning via structured output learning with latent variables. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4135\u20134144 (2019)","DOI":"10.1109\/CVPR.2019.00426"},{"key":"1_CR50","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young, P., Lai, A., Hodosh, M., Hockenmaier, J.: From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. Trans. Assoc. Comput. Linguistics 2, 67\u201378 (2014)","journal-title":"Trans. Assoc. Comput. Linguistics"},{"key":"1_CR51","doi-asserted-by":"crossref","unstructured":"Zellers, R., Bisk, Y., Farhadi, A., Choi, Y.: From recognition to cognition: visual commonsense reasoning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition(CVPR), pp. 6720\u20136731 (2019)","DOI":"10.1109\/CVPR.2019.00688"},{"key":"1_CR52","doi-asserted-by":"publisher","unstructured":"Zhao, G., Zhang, M., Liu, J., Li, Y., Wen, J.-R.: AP-GAN: adversarial patch attack on content-based image retrieval systems. GeoInformatica, 1\u201331 (2020). https:\/\/doi.org\/10.1007\/s10707-020-00418-7","DOI":"10.1007\/s10707-020-00418-7"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-19836-6_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,24]],"date-time":"2022-10-24T23:05:21Z","timestamp":1666652721000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-19836-6_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031198359","9783031198366"],"references-count":52,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-19836-6_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"22 October 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}