{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T06:45:56Z","timestamp":1785653156979,"version":"3.56.0"},"publisher-location":"Cham","reference-count":39,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032316653","type":"print"},{"value":"9783032316660","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-31666-0_18","type":"book-chapter","created":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:46:17Z","timestamp":1785649577000},"page":"267-281","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["What Matters for Grocery Product Retrieval with Open Source Vision Language Models"],"prefix":"10.1007","author":[{"given":"Emmanuel G.","family":"Maminta","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rowel O.","family":"Atienza","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,3]]},"reference":[{"key":"18_CR1","unstructured":"Bai, Y.: Products-10k: a large-scale product recognition dataset, (2020). arXiv:2008.10545 arXiv preprint"},{"key":"18_CR2","unstructured":"Bolya, D.: Perception encoder: the best visual embeddings are not at the output of the network. In: Advances in Neural Information Processing Systems, vol. 38, pp. 60884\u201360937. Curran Associates, Inc (2025)"},{"key":"18_CR3","doi-asserted-by":"crossref","unstructured":"Chen, J., Yu, Q., Shen, X., Yuille, A., Chen, L.C.: Vitamin: designing scalable vision models in the vision-language era. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12954\u201312966 (2024)","DOI":"10.1109\/CVPR52733.2024.01231"},{"key":"18_CR4","unstructured":"Chen, X.: Pali: a jointly-scaled multilingual language-image model. In: International Conference on Learning Representations (ICLR, (2023)"},{"key":"18_CR5","unstructured":"Chuang, Y.S.: Meta clip 2: a worldwide scaling recipe. In: Advances in Neural Information Processing Systems, vol. 38, pp. 48009\u201348036 (2026)"},{"key":"18_CR6","unstructured":"Czerwinska, U., Bircanoglu, C., Chamoux, J.: Benchmarking image embeddings for e-commerce: evaluating off-the shelf foundation models, fine-tuning strategies and practical trade-offs. arXiv preprint arXiv:2504.07567 (2025)"},{"key":"18_CR7","unstructured":"Dehghani, M., Tay, Y., Arnab, A., Beyer, L., Vaswani, A.: The efficiency misnomer. In: International Conference on Learning Representations (2022)"},{"key":"18_CR8","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth $$16\\times 16$$ words: transformers for image recognition at scale. In: International Conference on Learning Representations (ICLR) (2021)"},{"key":"18_CR9","unstructured":"Fan, Q., Li, W., Miao, S., Ma, S.: The 4th groceryvision challenge: ICCV25 retailvision workshop (2025). https:\/\/grocery-vision.github.io\/past_challenge\/iccv2025.html. Accessed 27 Nov 2025"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Gadre, S.Y.: DataComp: in search of the next generation of multimodal datasets. In: Advances in Neural Information Processing Systems (NeurIPS), vol. 36, pp. 27092\u201327112 (2023)","DOI":"10.52202\/075280-1179"},{"key":"18_CR11","unstructured":"Grattafiori, A., et\u00a0al.: The llama 3 herd of models (2024). https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"18_CR12","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"18_CR13","doi-asserted-by":"crossref","unstructured":"Hoffmann, J.: Training compute-optimal large language models. In: Proceedings of the 36th International Conference on Neural Information Processing Systems, pp. 30016\u201330030 (2022)","DOI":"10.52202\/068431-2176"},{"key":"18_CR14","doi-asserted-by":"publisher","unstructured":"Ilharco, G., et al.: OpenCLIP, July 2021. https:\/\/doi.org\/10.5281\/zenodo.5143773","DOI":"10.5281\/zenodo.5143773"},{"key":"18_CR15","doi-asserted-by":"crossref","unstructured":"Khattab, O., Zaharia, M.: ColBERT: efficient and effective passage search via contextualized late interaction over BERT. In: Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 39\u201348 (2020)","DOI":"10.1145\/3397271.3401075"},{"key":"18_CR16","doi-asserted-by":"crossref","unstructured":"Leutenegger, S., Chli, M., Siegwart, R.Y.: Brisk: binary robust invariant scalable keypoints. In: 2011 International Conference on Computer Vision, pp. 2548\u20132555. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126542"},{"key":"18_CR17","doi-asserted-by":"crossref","unstructured":"Li, C.: Elevater: a benchmark and toolkit for evaluating language-augmented visual models. In: Advances in Neural Information Processing Systems, vol. 35, pp. 9287\u20139301 (2022)","DOI":"10.52202\/068431-0675"},{"key":"18_CR18","doi-asserted-by":"crossref","unstructured":"Li, X., Wang, Z., Xie, C.: An inverse scaling law for clip training. In: Advances in Neural Information Processing Systems, vol. 36, pp. 49068\u201349087 (2023)","DOI":"10.52202\/075280-2132"},{"key":"18_CR19","unstructured":"Liu, Y.: RoBERTa: a robustly optimized BERT pretraining approach. arXiv preprint arXiv:1907.11692 (2019)"},{"key":"18_CR20","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.Y., Feichtenhofer, C., Darrell, T., Xie, S.: A convnet for the 2020s. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 11976\u201311986 (2022)","DOI":"10.1109\/CVPR52688.2022.01167"},{"issue":"2","key":"18_CR21","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1023\/B:VISI.0000029664.99615.94","volume":"60","author":"DG Lowe","year":"2004","unstructured":"Lowe, D.G.: Distinctive image features from scale-invariant keypoints. Int. J. Comput. Vision 60(2), 91\u2013110 (2004)","journal-title":"Int. J. Comput. Vision"},{"key":"18_CR22","unstructured":"Marafioti, A.: SmolVLM: redefining small and efficient multimodal models. arXiv preprint arXiv:2504.05299 (2025)"},{"key":"18_CR23","unstructured":"Nogueira, R., Cho, K.: Passage re-ranking with BERT. arXiv preprint arXiv:1901.04085 (2019)"},{"key":"18_CR24","unstructured":"Oord, A.v.d., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)"},{"key":"18_CR25","unstructured":"Peng, J., Xiao, C., Li, Y.: RP2K: a large-scale retail product dataset for fine-grained image classification. arXiv preprint arXiv:2006.12634 (2020)"},{"key":"18_CR26","unstructured":"Radford, A.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"18_CR27","doi-asserted-by":"crossref","unstructured":"Schuhmann, C.: Laion-5b: an open large-scale dataset for training next generation image-text models. In: Advances in Neural Information Processing Systems, vol. 35, pp. 25278\u201325294 (2022)","DOI":"10.52202\/068431-1833"},{"key":"18_CR28","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1007\/978-3-030-50347-5_8","volume-title":"Image Analysis and Recognition","author":"MM Srivastava","year":"2020","unstructured":"Srivastava, M.M.: Bag of tricks for retail product image classification. In: Campilho, A., Karray, F., Wang, Z. (eds.) ICIAR 2020. LNCS, vol. 12131, pp. 71\u201382. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-50347-5_8"},{"key":"18_CR29","doi-asserted-by":"crossref","unstructured":"Srivastava, M.M.: RetailKLIP: finetuning OpenCLIP backbone using metric learning on a single GPU for zero-shot retail product image classification. arXiv preprint arXiv:2312.10282 (2023)","DOI":"10.5220\/0012576000003660"},{"key":"18_CR30","doi-asserted-by":"crossref","unstructured":"Srivastava, S., Wu, K.: SGBD: sharpness-aware mirror gradient with blip-based denoising for robust multimodal product recommendation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) Workshops, pp. 2380\u20132389 (2025)","DOI":"10.1109\/ICCVW69036.2025.00248"},{"key":"18_CR31","unstructured":"Sun, Q., Fang, Y., Wu, L., Wang, X., Cao, Y.: Eva-clip: improved training techniques for clip at scale. arXiv preprint arXiv:2303.15389 (2023)"},{"issue":"2","key":"18_CR32","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1145\/2812802","volume":"59","author":"B Thomee","year":"2016","unstructured":"Thomee, B., et al.: Yfcc100m: the new data in multimedia research. Commun. ACM 59(2), 64\u201373 (2016)","journal-title":"Commun. ACM"},{"key":"18_CR33","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1016\/j.cviu.2019.03.005","volume":"182","author":"A Tonioni","year":"2019","unstructured":"Tonioni, A., Di Stefano, L.: Domain invariant hierarchical embedding for grocery products recognition. Comput. Vis. Image Underst. 182, 81\u201392 (2019)","journal-title":"Comput. Vis. Image Underst."},{"key":"18_CR34","unstructured":"Tschannen, M.: SigLIP 2: multilingual vision-language encoders with improved semantic understanding, localization, and dense features. arXiv preprint arXiv:2502.14786 (2025)"},{"key":"18_CR35","doi-asserted-by":"crossref","unstructured":"Vasu, P.K.A., Pouransari, H., Faghri, F., Vemulapalli, R., Tuzel, O.: MobileCLIP: fast image-text models through multi-modal reinforced training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 15963\u201315974 (2024)","DOI":"10.1109\/CVPR52733.2024.01511"},{"key":"18_CR36","unstructured":"Visheratin, A.: NLLB-CLIP - train performant multilingual image retrieval model on a budget. arXiv preprint arXiv:2309.01859 (2023)"},{"key":"18_CR37","unstructured":"Xu, H., et al.: Demystifying clip data. In: International Conference on Learning Representations (ICLR) (2024)"},{"key":"18_CR38","unstructured":"Yu, J., Wang, Z., Vasudevan, V., Yeung, L., Seyedhosseini, M., Wu, Y.: CoCa: contrastive captioners are image-text foundation models. Trans. Mach. Learn. Res. (2022)"},{"key":"18_CR39","doi-asserted-by":"crossref","unstructured":"Zhai, X., Mustafa, B., Kolesnikov, A., Beyer, L.: Sigmoid loss for language image pre-training. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 11975\u201311986 (2023)","DOI":"10.1109\/ICCV51070.2023.01100"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-31666-0_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:46:20Z","timestamp":1785649580000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-31666-0_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,3]]},"ISBN":["9783032316653","9783032316660"],"references-count":39,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-31666-0_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,3]]},"assertion":[{"value":"3 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lyon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 August 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}