{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T16:30:49Z","timestamp":1773937849246,"version":"3.50.1"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729829","type":"print"},{"value":"9783031729836","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T00:00:00Z","timestamp":1730160000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T00:00:00Z","timestamp":1730160000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72983-6_7","type":"book-chapter","created":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T09:34:20Z","timestamp":1730108060000},"page":"111-127","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Enhancing Recipe Retrieval with\u00a0Foundation Models: A Data Augmentation Perspective"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5390-4115","authenticated-orcid":false,"given":"Fangzhou","family":"Song","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9213-2611","authenticated-orcid":false,"given":"Bin","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0695-1566","authenticated-orcid":false,"given":"Yanbin","family":"Hao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4881-9344","authenticated-orcid":false,"given":"Shuo","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,29]]},"reference":[{"key":"7_CR1","unstructured":"Achiam, J., et\u00a0al.: GPT-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"7_CR2","unstructured":"Bommasani, R., et\u00a0al.: On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258 (2021)"},{"key":"7_CR3","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems, vol. 33, pp. 1877\u20131901 (2020)"},{"key":"7_CR4","doi-asserted-by":"crossref","unstructured":"Carvalho, M., Cad\u00e8ne, R., Picard, D., Soulier, L., Thome, N., Cord, M.: Cross-modal retrieval in the cooking context: learning semantic text-image embeddings. In: The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval, pp. 35\u201344 (2018)","DOI":"10.1145\/3209978.3210036"},{"key":"7_CR5","doi-asserted-by":"publisher","first-page":"1514","DOI":"10.1109\/TIP.2020.3045639","volume":"30","author":"J Chen","year":"2020","unstructured":"Chen, J., Zhu, B., Ngo, C.W., Chua, T.S., Jiang, Y.G.: A study of multi-task and region-wise deep learning for food ingredient recognition. IEEE Trans. Image Process. 30, 1514\u20131526 (2020)","journal-title":"IEEE Trans. Image Process."},{"key":"7_CR6","doi-asserted-by":"crossref","unstructured":"Fu, H., Wu, R., Liu, C., Sun, J.: MCEN: bridging cross-modal gap between cooking recipes and dish images with latent variable model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14570\u201314580 (2020)","DOI":"10.1109\/CVPR42600.2020.01458"},{"key":"7_CR7","doi-asserted-by":"crossref","unstructured":"Guerrero, R., Pham, H.X., Pavlovic, V.: Cross-modal retrieval and synthesis (X-MRS): closing the modality gap in shared subspace learning. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 3192\u20133201 (2021)","DOI":"10.1145\/3474085.3475465"},{"key":"7_CR8","unstructured":"Houlsby, N., et al.: Parameter-efficient transfer learning for NLP. In: International Conference on Machine Learning, pp. 2790\u20132799. PMLR (2019)"},{"key":"7_CR9","doi-asserted-by":"crossref","unstructured":"Huang, X., Liu, J., Zhang, Z., Xie, Y.: Improving cross-modal recipe retrieval with component-aware prompted clip embedding. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 529\u2013537 (2023)","DOI":"10.1145\/3581783.3612193"},{"key":"7_CR10","doi-asserted-by":"crossref","unstructured":"Ji, W., Li, J., Bi, Q., Li, W., Cheng, L.: Segment anything is not always perfect: an investigation of SAM on different real-world applications. arXiv preprint arXiv:2304.05750 (2023)","DOI":"10.1007\/s11633-024-1526-0"},{"key":"7_CR11","doi-asserted-by":"crossref","unstructured":"Karras, T., Laine, S., Aittala, M., Hellsten, J., Lehtinen, J., Aila, T.: Analyzing and improving the image quality of StyleGAN. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), June 2020 (2020)","DOI":"10.1109\/CVPR42600.2020.00813"},{"key":"7_CR12","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"7_CR13","unstructured":"Kiros, R., et al.: Skip-thought vectors. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"7_CR14","doi-asserted-by":"crossref","unstructured":"Li, J., Sun, J., Xu, X., Yu, W., Shen, F.: Cross-modal image-recipe retrieval via intra-and inter-modality hybrid fusion. In: Proceedings of the 2021 International Conference on Multimedia Retrieval, pp. 173\u2013182 (2021)","DOI":"10.1145\/3460426.3463618"},{"key":"7_CR15","doi-asserted-by":"crossref","unstructured":"Li, L., Li, M., Zan, Z., Xie, Q., Liu, J.: Multi-subspace implicit alignment for cross-modal retrieval on cooking recipes and food images. In: Proceedings of the 30th ACM International Conference on Information & Knowledge Management, pp. 3211\u20133215 (2021)","DOI":"10.1145\/3459637.3482149"},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"Liu, G., Jiao, Y., Chen, J., Zhu, B., Jiang, Y.G.: From canteen food to daily meals: generalizing food recognition to more practical scenarios. IEEE Trans. Multimedia (2024)","DOI":"10.1109\/TMM.2024.3371212"},{"key":"7_CR17","doi-asserted-by":"crossref","unstructured":"Ma, J., Wang, B.: Segment anything in medical images. arXiv preprint arXiv:2304.12306 (2023)","DOI":"10.1038\/s41467-024-44824-z"},{"key":"7_CR18","unstructured":"Mikolov, T., Chen, K., Corrado, G., Dean, J.: Efficient estimation of word representations in vector space. arXiv preprint arXiv:1301.3781 (2013)"},{"issue":"5","key":"7_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3329168","volume":"52","author":"W Min","year":"2019","unstructured":"Min, W., Jiang, S., Liu, L., Rui, Y., Jain, R.: A survey on food computing. ACM Comput. Surv. (CSUR) 52(5), 1\u201336 (2019)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"7_CR20","doi-asserted-by":"publisher","first-page":"9932","DOI":"10.1109\/TPAMI.2023.3237871","volume":"45","author":"W Min","year":"2023","unstructured":"Min, W., et al.: Large scale visual food recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45, 9932\u20139949 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"7_CR21","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1007\/978-3-319-73600-6_12","volume-title":"MultiMedia Modeling","author":"Z-Y Ming","year":"2018","unstructured":"Ming, Z.-Y., Chen, J., Cao, Yu., Forde, C., Ngo, C.-W., Chua, T.S.: Food photo recognition for dietary tracking: system and experiment. In: Schoeffmann, K., et al. (eds.) MMM 2018, Part II 24. LNCS, vol. 10705, pp. 129\u2013141. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-319-73600-6_12"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Papadopoulos, D.P., Mora, E., Chepurko, N., Huang, K.W., Ofli, F., Torralba, A.: Learning program representations for food images and cooking recipes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16559\u201316569 (2022)","DOI":"10.1109\/CVPR52688.2022.01606"},{"key":"7_CR23","doi-asserted-by":"crossref","unstructured":"Pham, H.X., Guerrero, R., Pavlovic, V., Li, J.: CHEF: cross-modal hierarchical embeddings for food domain retrieval. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035, pp. 2423\u20132430 (2021)","DOI":"10.1609\/aaai.v35i3.16343"},{"key":"7_CR24","doi-asserted-by":"crossref","unstructured":"Sahoo, D., et al.: FoodAI: food image recognition via deep learning for smart food logging. In: Proceedings of the 25th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 2260\u20132268 (2019)","DOI":"10.1145\/3292500.3330734"},{"key":"7_CR25","doi-asserted-by":"crossref","unstructured":"Salvador, A., Gundogdu, E., Bazzani, L., Donoser, M.: Revamping cross-modal recipe retrieval with hierarchical transformers and self-supervised learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15475\u201315484 (2021)","DOI":"10.1109\/CVPR46437.2021.01522"},{"key":"7_CR26","doi-asserted-by":"crossref","unstructured":"Salvador, A., et al.: Learning cross-modal embeddings for cooking recipes and food images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3020\u20133028 (2017)","DOI":"10.1109\/CVPR.2017.327"},{"key":"7_CR27","doi-asserted-by":"crossref","unstructured":"Shukor, M., Couairon, G., Grechka, A., Cord, M.: Transformer decoders with multimodal regularization for cross-modal food retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4567\u20134578 (2022)","DOI":"10.1109\/CVPRW56347.2022.00503"},{"key":"7_CR28","doi-asserted-by":"crossref","unstructured":"Shukor, M., Thome, N., Cord, M.: Vision and structured-language pretraining for cross-modal food retrieval. Available at SSRN 4511116 (2023)","DOI":"10.2139\/ssrn.4511116"},{"key":"7_CR29","doi-asserted-by":"crossref","unstructured":"Sun, Y., et al.: Circle loss: a unified perspective of pair similarity optimization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6398\u20136407 (2020)","DOI":"10.1109\/CVPR42600.2020.00643"},{"key":"7_CR30","doi-asserted-by":"crossref","unstructured":"Sung, Y.L., Cho, J., Bansal, M.: Vl-adapter: parameter-efficient transfer learning for vision-and-language tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5227\u20135237 (2022)","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"7_CR31","unstructured":"Gemini Team Google, et\u00a0al.: Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)"},{"key":"7_CR32","unstructured":"Touvron, H., et\u00a0al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"7_CR33","unstructured":"Trabucco, B., Doherty, K., Gurinas, M., Salakhutdinov, R.: Effective data augmentation with diffusion models. arXiv preprint arXiv:2302.07944 (2023)"},{"key":"7_CR34","unstructured":"Voutharoja, B.P., Wang, P., Wang, L., Guan, V.: MALM: mask augmentation based local matching for food-recipe retrieval. arXiv preprint arXiv:2305.11327 (2023)"},{"key":"7_CR35","doi-asserted-by":"crossref","unstructured":"Wahed, M., Zhou, X., Yu, T., Lourentzou, I.: Fine-grained alignment for cross-modal recipe retrieval. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 5584\u20135593 (2024)","DOI":"10.1109\/WACV57701.2024.00549"},{"key":"7_CR36","doi-asserted-by":"crossref","unstructured":"Wang, H., Lin, G., Hoi, S., Miao, C.: Paired cross-modal data augmentation for fine-grained image-to-text retrieval. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 5517\u20135526 (2022)","DOI":"10.1145\/3503161.3547809"},{"issue":"3","key":"7_CR37","first-page":"3363","volume":"45","author":"H Wang","year":"2022","unstructured":"Wang, H., Lin, G., Hoi, S.C., Miao, C.: Learning structural representations for recipe generation and food retrieval. IEEE Trans. Pattern Anal. Mach. Intell. 45(3), 3363\u20133377 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"7_CR38","doi-asserted-by":"crossref","unstructured":"Wang, H., Sahoo, D., Liu, C., Lim, E., Hoi, S.C.: Learning cross-modal embeddings with adversarial networks for cooking recipes and food images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11572\u201311581 (2019)","DOI":"10.1109\/CVPR.2019.01184"},{"issue":"1","key":"7_CR39","first-page":"1","volume":"17","author":"W Wang","year":"2021","unstructured":"Wang, W., Duan, L.Y., Jiang, H., Jing, P., Song, X., Nie, L.: Market2Dish: health-aware food recommendation. ACM Trans. Multimedia Comput. Commun. Appl. (TOMM) 17(1), 1\u201319 (2021)","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl. (TOMM)"},{"key":"7_CR40","doi-asserted-by":"crossref","unstructured":"Whitehouse, C., Choudhury, M., Aji, A.F.: LLM-powered data augmentation for enhanced crosslingual performance. arXiv preprint arXiv:2305.14288 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.44"},{"key":"7_CR41","doi-asserted-by":"publisher","first-page":"3304","DOI":"10.1109\/TSC.2021.3098834","volume":"15","author":"Z Xie","year":"2021","unstructured":"Xie, Z., Liu, L., Wu, Y., Li, L., Zhong, L.: Learning TFIDF enhanced joint embedding for recipe-image cross-modal retrieval service. IEEE Trans. Serv. Comput. 15, 3304\u20133316 (2021)","journal-title":"IEEE Trans. Serv. Comput."},{"key":"7_CR42","doi-asserted-by":"crossref","unstructured":"Zan, Z., Li, L., Liu, J., Zhou, D.: Sentence-based and noise-robust cross-modal retrieval on cooking recipes and food images. In: Proceedings of the 2020 International Conference on Multimedia Retrieval, pp. 117\u2013125 (2020)","DOI":"10.1145\/3372278.3390681"},{"key":"7_CR43","unstructured":"Zhang, C., et al.: A comprehensive survey on segment anything model for vision and beyond. arXiv preprint arXiv:2305.08196 (2023)"},{"key":"7_CR44","doi-asserted-by":"publisher","unstructured":"Zhang, Y., Zhou, T., Wang, S., Liang, P., Zhang, Y., Chen, D.Z.: Input augmentation with SAM: boosting medical image segmentation with segmentation foundation model. In: Celebi, M.E., et al. (eds.) Medical Image Computing and Computer Assisted Intervention, MICCAI 2023 Workshops, MICCAI 2023. LNCS, vol. 14393, pp. 129\u2013139. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-47401-9_13","DOI":"10.1007\/978-3-031-47401-9_13"},{"key":"7_CR45","doi-asserted-by":"publisher","first-page":"1175","DOI":"10.1109\/TMM.2021.3123474","volume":"24","author":"B Zhu","year":"2021","unstructured":"Zhu, B., Ngo, C.W., Chan, W.K.: Learning from web recipe-image pairs for food recognition: problem, baselines and performance. IEEE Trans. Multimedia 24, 1175\u20131185 (2021)","journal-title":"IEEE Trans. Multimedia"},{"key":"7_CR46","doi-asserted-by":"crossref","unstructured":"Zhu, B., Ngo, C.W., Chen, J.: Cross-domain cross-modal food transfer. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 3762\u20133770 (2020)","DOI":"10.1145\/3394171.3413809"},{"key":"7_CR47","doi-asserted-by":"crossref","unstructured":"Zhu, B., Ngo, C.W., Chen, J., Hao, Y.: R$$^2$$GAN: cross-modal recipe retrieval with generative adversarial network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11477\u201311486 (2019)","DOI":"10.1109\/CVPR.2019.01174"},{"key":"7_CR48","doi-asserted-by":"publisher","first-page":"33283","DOI":"10.1109\/ACCESS.2024.3370158","volume":"12","author":"Z Zou","year":"2024","unstructured":"Zou, Z., Zhu, X., Zhu, Q., Liu, Y., Zhu, L.: CREAMY: cross-modal recipe retrieval by avoiding matching imperfectly. IEEE Access 12, 33283\u201333295 (2024)","journal-title":"IEEE Access"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72983-6_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T09:55:29Z","timestamp":1730109329000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72983-6_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,29]]},"ISBN":["9783031729829","9783031729836"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72983-6_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,29]]},"assertion":[{"value":"29 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}