{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,26]],"date-time":"2025-05-26T09:10:10Z","timestamp":1748250610659,"version":"3.41.0"},"publisher-location":"Cham","reference-count":62,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031915840","type":"print"},{"value":"9783031915857","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-91585-7_2","type":"book-chapter","created":{"date-parts":[[2025,5,26]],"date-time":"2025-05-26T08:30:04Z","timestamp":1748248204000},"page":"17-33","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Rethinking Sparse Lexical Representations for\u00a0Image Retrieval in\u00a0the\u00a0Age of\u00a0Rising Multi-modal Large Language Models"],"prefix":"10.1007","author":[{"given":"Kengo","family":"Nakata","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daisuke","family":"Miyashita","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Youyang","family":"Ng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yasuto","family":"Hoshi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Deguchi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"2_CR1","doi-asserted-by":"crossref","unstructured":"Baldrati, A., Bertini, M., Uricchio, T., Del\u00a0Bimbo, A.: Conditioned and composed image retrieval combining and partially fine-tuning clip-based features. In: CVPRW (2022)","DOI":"10.1109\/CVPRW56347.2022.00543"},{"key":"2_CR2","doi-asserted-by":"crossref","unstructured":"Baldrati, A., Bertini, M., Uricchio, T., Del\u00a0Bimbo, A.: Effective conditioned and composed image retrieval combining clip-based features. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.02080"},{"key":"2_CR3","unstructured":"Berrios, W., Mittal, G., Thrush, T., Kiela, D., Singh, A.: Towards language models that can see: computer vision through the lens of natural language. arXiv abs\/2306.16410 (2023)"},{"key":"2_CR4","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: NeurIPS (2020)"},{"key":"2_CR5","doi-asserted-by":"crossref","unstructured":"Chen, C., et al.: STAIR: learning sparse text and image representation in grounded tokens. In: EMNLP (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.932"},{"key":"2_CR6","doi-asserted-by":"crossref","unstructured":"Chen, Y.C., et al.: UNITER: universal image-text representation learning. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"2_CR7","doi-asserted-by":"crossref","unstructured":"Chua, T.S., Tang, J., Hong, R., Li, H., Luo, Z., Zheng, Y.T.: NUS-WIDE: a real-world web image database from national university of Singapore. In: ACM International Conference on Image and Video Retrieval (CIVR) (2009)","DOI":"10.1145\/1646396.1646452"},{"key":"2_CR8","unstructured":"Dai, W., et al.: InstructBLIP: towards general-purpose vision-language models with instruction tuning. In: NeurIPS (2023)"},{"issue":"1","key":"2_CR9","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1007\/s11263-014-0733-5","volume":"111","author":"M Everingham","year":"2015","unstructured":"Everingham, M., Eslami, S., Van Gool, L., Williams, C., Winn, J., Zisserman, A.: The pascal visual object classes challenge: a retrospective. IJCV 111(1), 98\u2013136 (2015)","journal-title":"IJCV"},{"key":"2_CR10","doi-asserted-by":"crossref","unstructured":"Gordo, A., Almaz\u00e1n, J., Revaud, J., Larlus, D.: Deep image retrieval: learning global representations for image search. In: ECCV (2016)","DOI":"10.1007\/978-3-319-46466-4_15"},{"key":"2_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Dollar, P., Girshick, R.: Mask R-CNN. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"2_CR12","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2_CR13","doi-asserted-by":"crossref","unstructured":"Hessel, J., Holtzman, A., Forbes, M., Le\u00a0Bras, R., Choi, Y.: CLIPScore: a reference-free evaluation metric for image captioning. In: EMNLP (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"2_CR14","unstructured":"Jaderberg, M., Simonyan, K., Zisserman, A., kavukcuoglu, k.: Spatial transformer networks. In: NIPS (2015)"},{"key":"2_CR15","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: ICML (2021)"},{"key":"2_CR16","doi-asserted-by":"crossref","unstructured":"Jiang, Q., Li, W.: Deep cross-modal hashing. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.348"},{"key":"2_CR17","doi-asserted-by":"crossref","unstructured":"Ke, L., Pei, W., Li, R., Shen, X., Tai, Y.W.: Reflective decoding network for image captioning. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00898"},{"key":"2_CR18","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: ImageNet classification with deep convolutional neural networks. In: NIPS (2012)"},{"key":"2_CR19","doi-asserted-by":"crossref","unstructured":"Li, C., Ge, Y., Mao, J., Li, D., Shan, Y.: TagGPT: large language models are zero-shot multimodal taggers. arXiv abs\/2304.03022 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.459"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Li, C., et al.: mPLUG: effective and efficient vision-language learning by cross-modal skip-connections. In: EMNLP (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.488"},{"key":"2_CR21","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: ICML (2023)"},{"key":"2_CR22","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: ICML (2022)"},{"key":"2_CR23","doi-asserted-by":"crossref","unstructured":"Lin, J., Ma, X., Lin, S.C., Yang, J.H., Pradeep, R., Nogueira, R.: Pyserini: a python toolkit for reproducible information retrieval research with sparse and dense representations. In: International ACM SIGIR Conference on Research and Development in Information Retrieval (2021)","DOI":"10.1145\/3404835.3463238"},{"key":"2_CR24","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., et al.: Microsoft COCO: common objects in context. In: ECCV (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. In: NeurIPS 2023 Workshop on Instruction Tuning and Instruction Following (2023)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"2_CR26","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: NeurIPS (2023)"},{"key":"2_CR27","doi-asserted-by":"crossref","unstructured":"Luo, Z., et al.: LexLIP: lexicon-bottlenecked language-image pre-training for large-scale image-text sparse retrieval. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01029"},{"key":"2_CR28","unstructured":"Mikolov, T., Sutskever, I., Chen, K., Corrado, G.S., Dean, J.: Distributed representations of words and phrases and their compositionality. In: NIPS (2013)"},{"key":"2_CR29","unstructured":"Nguyen, T., Hendriksen, M., Yates, A.: Multimodal learned sparse retrieval for image suggestion. arXiv abs\/2402.07736 (2024)"},{"key":"2_CR30","unstructured":"OpenAI: GPT-4 technical report. arXiv abs\/2303.08774 (2023)"},{"key":"2_CR31","doi-asserted-by":"crossref","unstructured":"Otani, M., et al.: Toward verifiable and reproducible human evaluation for text-to-image generation. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01372"},{"issue":"1","key":"2_CR32","doi-asserted-by":"publisher","first-page":"74","DOI":"10.1007\/s11263-016-0965-7","volume":"123","author":"BA Plummer","year":"2017","unstructured":"Plummer, B.A., Wang, L., Cervantes, C.M., Caicedo, J.C., Hockenmaier, J., Lazebnik, S.: Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. IJCV 123(1), 74\u201393 (2017)","journal-title":"IJCV"},{"key":"2_CR33","doi-asserted-by":"crossref","unstructured":"Potapov, A., Zhdanov, I., Scherbakov, O., Skorobogatko, N., Latapie, H., Fenoglio, E.: Semantic image retrieval by uniting deep neural networks and cognitive architectures. In: Artificial General Intelligence, pp. 196\u2013206. Springer International Publishing (2018)","DOI":"10.1007\/978-3-319-97676-1_19"},{"key":"2_CR34","unstructured":"Qi, D., Su, L., Song, J., Cui, E., Bharti, T., Sacheti, A.: ImageBERT: Cross-modal pre-training with large-scale weak-supervised image-text data. arXiv abs\/2001.07966 (2020)"},{"key":"2_CR35","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"2_CR36","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.91"},{"key":"2_CR37","unstructured":"Redmon, J., Farhadi, A.: YOLOv3: An incremental improvement. arXiv (2018)"},{"key":"2_CR38","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: NIPS (2015)"},{"key":"2_CR39","doi-asserted-by":"crossref","unstructured":"Robertson, S.E., Walker, S., Jones, S., Hancock-Beaulieu, M., Gatford, M.: Okapi at TREC-3. In: Proceedings of The Third Text REtrieval Conference, TREC. vol. 500-225, pp. 109\u2013126. National Institute of Standards and Technology (NIST) (1994)","DOI":"10.6028\/NIST.SP.500-225.routing-city"},{"key":"2_CR40","doi-asserted-by":"crossref","unstructured":"Sammut, C., Webb, G.I. (eds.): TF\u2013IDF, pp. 986\u2013987. Springer US (2010)","DOI":"10.1007\/978-0-387-30164-8_832"},{"key":"2_CR41","doi-asserted-by":"crossref","unstructured":"Sarto, S., Barraco, M., Cornia, M., Baraldi, L., Cucchiara, R.: Positive-augmented contrastive learning for image and video captioning evaluation. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00668"},{"key":"2_CR42","doi-asserted-by":"crossref","unstructured":"Sciavolino, C., Zhong, Z., Lee, J., Chen, D.: Simple entity-centric questions challenge dense retrievers. In: EMNLP (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.496"},{"key":"2_CR43","unstructured":"Shuster, K., et al.: BlenderBot 3: a deployed conversational agent that continually learns to responsibly engage. arXiv abs\/2208.03188 (2022)"},{"key":"2_CR44","doi-asserted-by":"crossref","unstructured":"Singh, A., Hu, R., Goswami, V., Couairon, G., Galuba, W., Rohrbach, M., Kiela, D.: FLAVA: a foundational language and vision alignment model. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01519"},{"key":"2_CR45","unstructured":"Su, H., et al.: BRIGHT: A realistic and challenging benchmark for reasoning-intensive retrieval. arXiv abs\/2407.12883 (2024)"},{"key":"2_CR46","doi-asserted-by":"crossref","unstructured":"Tan, F., Yuan, J., Ordonez, V.: Instance-level image retrieval using reranking transformers. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01189"},{"key":"2_CR47","unstructured":"Tan, Z., et al.: Large language models for data annotation: A survey. arXiv abs\/2402.13446 (2024)"},{"key":"2_CR48","unstructured":"Thakur, N., Reimers, N., R\u00fcckl\u00e9, A., Srivastava, A., Gurevych, I.: BEIR: a heterogeneous benchmark for zero-shot evaluation of information retrieval models. In: Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2) (2021)"},{"key":"2_CR49","unstructured":"Thoppilan, R., et al.: LaMDA: Language models for dialog applications. arXiv abs\/2201.08239 (2022)"},{"key":"2_CR50","unstructured":"Touvron, H., et al.: LLaMA: Open and efficient foundation language models. arXiv abs\/2302.13971 (2023)"},{"key":"2_CR51","unstructured":"Touvron, H., et al.: LLaMA 2: Open foundation and fine-tuned chat models. arXiv abs\/2307.09288 (2023)"},{"key":"2_CR52","doi-asserted-by":"crossref","unstructured":"Tushabe, F., Wilkinson, M.H.F.: Content-based image retrieval using combined 2D attribute pattern spectra. In: Advances in Multilingual and Multimodal Information Retrieval (2008)","DOI":"10.1007\/978-3-540-85760-0_69"},{"key":"2_CR53","unstructured":"Wolf, T., et al.: Transformers: state-of-the-art natural language processing. In: EMNLP: System Demonstrations (2020)"},{"key":"2_CR54","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2022.105090","volume":"114","author":"Y Xie","year":"2022","unstructured":"Xie, Y., Zeng, X., Wang, T., Xu, L., Wang, D.: Multiple deep neural networks with multiple labels for cross-modal hashing retrieval. Eng. Appl. Artif. Intell. 114, 105090 (2022)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"2_CR55","doi-asserted-by":"crossref","unstructured":"Xu, J., Shi, C., Qi, C., Wang, C., Xiao, B.: Unsupervised part-based weighting aggregation of deep convolutional features for image retrieval (2018)","DOI":"10.1609\/aaai.v32i1.12231"},{"key":"2_CR56","unstructured":"Yang, J., Zhang, H., Li, F., Zou, X., Li, C., Gao, J.: Set-of-mark prompting unleashes extraordinary visual grounding in GPT-4V. arXiv abs\/2310.11441 (2023)"},{"issue":"11","key":"2_CR57","doi-asserted-by":"publisher","first-page":"5288","DOI":"10.1109\/TIP.2018.2845136","volume":"27","author":"J Yang","year":"2018","unstructured":"Yang, J., Liang, J., Shen, H., Wang, K., Rosin, P.L., Yang, M.H.: Dynamic match kernel with deep convolutional features for image retrieval. IEEE Trans. Image Process. 27(11), 5288\u20135302 (2018)","journal-title":"IEEE Trans. Image Process."},{"key":"2_CR58","unstructured":"Yang, Z., et al.: The dawn of LMMs: Preliminary explorations with GPT-4V(ision). arXiv abs\/2309.17421 (2023)"},{"key":"2_CR59","doi-asserted-by":"crossref","unstructured":"Ye, L., Rochan, M., Liu, Z., Wang, Y.: Cross-modal self-attention network for referring image segmentation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01075"},{"issue":"3","key":"2_CR60","doi-asserted-by":"publisher","first-page":"2385","DOI":"10.1016\/j.eswa.2011.08.086","volume":"39","author":"E Yildizer","year":"2012","unstructured":"Yildizer, E., Balci, A.M., Hassan, M., Alhajj, R.: Efficient content-based image retrieval using multiple support vector machines ensemble. Expert Syst. Appl. 39(3), 2385\u20132396 (2012)","journal-title":"Expert Syst. Appl."},{"key":"2_CR61","unstructured":"Yu, J., Wang, Z., Vasudevan, V., Yeung, L., Seyedhosseini, M., Wu, Y.: CoCa: contrastive captioners are image-text foundation models. Transactions on Machine Learning Research (2022)"},{"key":"2_CR62","unstructured":"Zhou, J., Li, X., Shang, L., Jiang, X., Liu, Q., Chen, L.: Retrieval-based disentangled representation learning with natural language supervision. In: ICLR (2024)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-91585-7_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,26]],"date-time":"2025-05-26T08:30:42Z","timestamp":1748248242000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-91585-7_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031915840","9783031915857"],"references-count":62,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-91585-7_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}