{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T13:56:16Z","timestamp":1769090176657,"version":"3.49.0"},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,1,21]],"date-time":"2026-01-21T00:00:00Z","timestamp":1768953600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,1,21]],"date-time":"2026-01-21T00:00:00Z","timestamp":1768953600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-026-21196-8","type":"journal-article","created":{"date-parts":[[2026,1,21]],"date-time":"2026-01-21T21:57:57Z","timestamp":1769032677000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Settle for the next best option: improving similarity identification of CLIP for impossible text-image retrieval"],"prefix":"10.1007","volume":"85","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-1678-1844","authenticated-orcid":false,"given":"Yidan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Naoto","family":"Naka","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shin\u2019ichi","family":"Satoh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,21]]},"reference":[{"issue":"1","key":"21196_CR1","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1006\/jvci.1999.0413","volume":"10","author":"Y Rui","year":"1999","unstructured":"Rui Y, Huang TS, Chang S-F (1999) Image retrieval: current techniques, promising directions, and open issues. J Vis Commun Image Represent 10(1):39\u201362","journal-title":"J Vis Commun Image Represent"},{"key":"21196_CR2","doi-asserted-by":"publisher","unstructured":"Chang NS, Fu KS (1980) A relational database system for images. Springer, Berlin, Heidelberg, pp 288\u2013321. https:\/\/doi.org\/10.1007\/3-540-09757-0_11","DOI":"10.1007\/3-540-09757-0_11"},{"key":"21196_CR3","doi-asserted-by":"publisher","unstructured":"Kato T (1992) Database architecture for content-based image retrieval. In: Jamberdino AA, Niblack CW (eds) Image storage and retrieval systems, vol 1662. SPIE, pp 112\u2013123.https:\/\/doi.org\/10.1117\/12.58497, International Society for Optics and Photonics","DOI":"10.1117\/12.58497"},{"key":"21196_CR4","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) ImageNet classification with deep convolutional neural networks. In: Proceedings of the 25th International Conference on Neural Information Processing Systems - vol 1. NIPS\u201912. Curran Associates Inc., Red Hook, NY, USA, pp 1097\u20131105"},{"key":"21196_CR5","doi-asserted-by":"publisher","unstructured":"Bradshaw B (2000) Semantic based image retrieval: a probabilistic approach. In: Proceedings of the Eighth ACM international conference on multimedia. MULTIMEDIA \u201900. Association for Computing Machinery, New York, NY, USA, pp 167\u2013176. https:\/\/doi.org\/10.1145\/354384.354456","DOI":"10.1145\/354384.354456"},{"key":"21196_CR6","unstructured":"Wang K, Yin Q, Wang W, Wu S, Wang L (2016) A comprehensive survey on cross-modal retrieval. arXiv:1607.06215"},{"key":"21196_CR7","doi-asserted-by":"publisher","unstructured":"Vo N, Jiang L, Sun C, Murphy K, Li L-J, Fei-Fei L, Hays J (2019) Composing text and image for image retrieval - an empirical odyssey. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). pp 6432\u20136441. https:\/\/doi.org\/10.1109\/CVPR.2019.00660","DOI":"10.1109\/CVPR.2019.00660"},{"issue":"3","key":"21196_CR8","doi-asserted-by":"publisher","first-page":"216","DOI":"10.1145\/321033.321035","volume":"7","author":"ME Maron","year":"1960","unstructured":"Maron ME, Kuhns JL (1960) On relevance, probabilistic indexing and information retrieval. J ACM (JACM) 7(3):216\u2013244","journal-title":"J ACM (JACM)"},{"issue":"5","key":"21196_CR9","doi-asserted-by":"publisher","first-page":"217","DOI":"10.1016\/0020-0271(71)90051-9","volume":"7","author":"N Jardine","year":"1971","unstructured":"Jardine N, Rijsbergen CJ (1971) The use of hierarchic clustering in information retrieval. Inf Storage Retrieval 7(5):217\u2013240","journal-title":"Inf Storage Retrieval"},{"issue":"6","key":"21196_CR10","doi-asserted-by":"publisher","first-page":"329","DOI":"10.1016\/0020-0271(72)90021-6","volume":"8","author":"J Minker","year":"1972","unstructured":"Minker J, Wilson GA, Zimmerman BH (1972) An evaluation of query expansion by the addition of clustered terms for a document retrieval system. Inf Storage and Retrieval 8(6):329\u2013348. https:\/\/doi.org\/10.1016\/0020-0271(72)90021-6","journal-title":"Inf Storage and Retrieval"},{"key":"21196_CR11","first-page":"313","volume-title":"The smart retrieval system - experiments in automatic document processing","author":"JJ Rocchio","year":"1971","unstructured":"Rocchio JJ (1971) Relevance feedback in information retrieval. In: Salton G (ed) The smart retrieval system - experiments in automatic document processing. Prentice-Hall, Englewood Cliffs, NJ, pp 313\u2013323"},{"key":"21196_CR12","unstructured":"Wang H, Zhan Y, Liu L, Ding L, Yang Y, Yu J (2024) Towards alleviating text-to-image retrieval hallucination for CLIP in zero-shot learning. arXiv:2402.18400"},{"key":"21196_CR13","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J et al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning. PMLR, pp 8748\u20138763"},{"key":"21196_CR14","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J, Krueger G, Sutskever I (2021) Learning transferable visual models from natural language supervision. In: Meila M, Zhang T (eds) Proceedings of the 38th international conference on machine learning. Proceedings of Machine Learning Research, vol 139. PMLR, pp 8748\u20138763. https:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"21196_CR15","doi-asserted-by":"crossref","unstructured":"Saito K, Sohn K, Zhang X, Li C-L, Lee C-Y, Saenko K, Pfister T (2023) Pic2word: mapping pictures to words for zero-shot composed image retrieval. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). pp 19305\u201319314","DOI":"10.1109\/CVPR52729.2023.01850"},{"key":"21196_CR16","unstructured":"Li J, Li D, Xiong C, Hoi S (2022) BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: ICML"},{"key":"21196_CR17","doi-asserted-by":"publisher","unstructured":"Biancalana C, Gasparetti F, Micarelli A, Sansonetti G (2013) Social semantic query expansion. ACM Trans Intell Syst Technol 4(4). https:\/\/doi.org\/10.1145\/2508037.2508041","DOI":"10.1145\/2508037.2508041"},{"key":"21196_CR18","doi-asserted-by":"publisher","first-page":"614","DOI":"10.1016\/j.ins.2016.07.046","volume":"369","author":"MR Bouadjenek","year":"2016","unstructured":"Bouadjenek MR, Hacid H, Bouzeghoub M, Vakali A (2016) PerSaDoR: personalized social document representation for improving web search. Inf Sci 369:614\u2013633. https:\/\/doi.org\/10.1016\/j.ins.2016.07.046","journal-title":"Inf Sci"},{"issue":"4","key":"21196_CR19","doi-asserted-by":"publisher","first-page":"381","DOI":"10.1007\/s11257-012-9124-1","volume":"23","author":"MR Ghorab","year":"2013","unstructured":"Ghorab MR, Zhou D, O\u2019connor A, Wade V (2013) Personalised information retrieval: survey and classification. User Model User-Adap Inter 23(4):381\u2013443. https:\/\/doi.org\/10.1007\/s11257-012-9124-1","journal-title":"User Model User-Adap Inter"},{"key":"21196_CR20","doi-asserted-by":"crossref","unstructured":"Paiss R, Ephrat A, Tov O, Zada S, Mosseri I, Irani M, Dekel T (2023) Teaching clip to count to ten. arXiv:2302.12066","DOI":"10.1109\/ICCV51070.2023.00294"},{"key":"21196_CR21","doi-asserted-by":"publisher","unstructured":"Ranasinghe K, McKinzie B, Ravi S, Yang Y, Toshev A, Shlens J (2023) Perceptual grouping in contrastive vision-language models. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV). pp 5548\u20135561. https:\/\/doi.org\/10.1109\/ICCV51070.2023.00513","DOI":"10.1109\/ICCV51070.2023.00513"},{"key":"21196_CR22","doi-asserted-by":"crossref","unstructured":"Zhong Y, Yang J, Zhang P, Li C, Codella N, Li LH, Zhou L, Dai X, Yuan L, Li Y et al (2022) Regionclip: region-based language-image pretraining. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 16793\u201316803","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"21196_CR23","doi-asserted-by":"publisher","unstructured":"Yi C, Ren L, Zhan D-C, Ye H-J (2024) Leveraging cross-modal neighbor representation for improved CLIP classification . In: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE Computer Society, Los Alamitos, CA, USA. pp 27392\u201327401. https:\/\/doi.org\/10.1109\/CVPR52733.2024.02587, https:\/\/doi.ieeecomputersociety.org\/10.1109\/CVPR52733.2024.02587","DOI":"10.1109\/CVPR52733.2024.02587"},{"key":"21196_CR24","doi-asserted-by":"crossref","unstructured":"Bianchi L, Carrara F, Messina N, Gennaro C, Falchi F (2024) The devil is in the fine-grained details: evaluating open-vocabulary object detectors for fine-grained understanding. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 22520\u201322529","DOI":"10.1109\/CVPR52733.2024.02125"},{"key":"21196_CR25","doi-asserted-by":"crossref","unstructured":"Bianchi L, Carrara F, Messina N, Falchi F (2024) Is CLIP the main roadblock for fine-grained open-world perception?","DOI":"10.1109\/CBMI62980.2024.10859215"},{"key":"21196_CR26","doi-asserted-by":"publisher","unstructured":"Cao Z, Qin T, Liu T-Y, Tsai M-F, Li H (2007) Learning to rank: from pairwise approach to listwise approach. In: Proceedings of the 24th international conference on machine learning. ICML \u201907. Association for Computing Machinery, New York, NY, USA, pp 129\u2013136. https:\/\/doi.org\/10.1145\/1273496.1273513","DOI":"10.1145\/1273496.1273513"},{"key":"21196_CR27","unstructured":"Krizhevsky A (2009) Learning multiple layers of features from tiny images. Technical report, University of Toronto"},{"key":"21196_CR28","unstructured":"Wah C, Branson S, Welinder P, Perona P, Belongie S (2011) The Caltech-UCSD birds-200-2011 dataset"},{"key":"21196_CR29","unstructured":"Li J, Li D, Xiong C, Hoi S (2022) BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. arXiv:2201.12086"},{"key":"21196_CR30","unstructured":"Li J, Li D, Savarese S, Hoi S (2023) BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv:2301.12597"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21196-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-026-21196-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21196-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,21]],"date-time":"2026-01-21T21:58:00Z","timestamp":1769032680000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-026-21196-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,21]]},"references-count":30,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,1]]}},"alternative-id":["21196"],"URL":"https:\/\/doi.org\/10.1007\/s11042-026-21196-8","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,21]]},"assertion":[{"value":"11 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 September 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 November 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 January 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}}],"article-number":"17"}}