{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T03:43:35Z","timestamp":1743047015288,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":33,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819601271"},{"type":"electronic","value":"9789819601288"}],"license":[{"start":{"date-parts":[[2024,11,12]],"date-time":"2024-11-12T00:00:00Z","timestamp":1731369600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,12]],"date-time":"2024-11-12T00:00:00Z","timestamp":1731369600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-0128-8_7","type":"book-chapter","created":{"date-parts":[[2024,11,16]],"date-time":"2024-11-16T18:17:30Z","timestamp":1731781050000},"page":"79-90","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Fine-Grained Modalities Interaction for\u00a0Cross-Modal Recipe Retrieval"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-8571-1273","authenticated-orcid":false,"given":"Fangying","family":"Qu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2575-3171","authenticated-orcid":false,"given":"Yuqing","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7887-8582","authenticated-orcid":false,"given":"Zhuo","family":"Yao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6672-7948","authenticated-orcid":false,"given":"Fan","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,12]]},"reference":[{"key":"7_CR1","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"446","DOI":"10.1007\/978-3-319-10599-4_29","volume-title":"Computer Vision \u2013 ECCV 2014","author":"L Bossard","year":"2014","unstructured":"Bossard, L., Guillaumin, M., Van Gool, L.: Food-101 \u2013 mining discriminative components with random forests. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8694, pp. 446\u2013461. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10599-4_29"},{"key":"7_CR2","doi-asserted-by":"publisher","unstructured":"Cao, M., Li, S., Li, J., Nie, L., Zhang, M.: Image-text retrieval: a survey on recent research and development. In: Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence, IJCAI 2022, pp. 5410\u20135417 (July 2022). https:\/\/doi.org\/10.24963\/ijcai.2022\/759","DOI":"10.24963\/ijcai.2022\/759"},{"key":"7_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1007\/978-3-030-58577-8_7","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Y-C Chen","year":"2020","unstructured":"Chen, Y.-C., et al.: UNITER: universal image-text representation learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 104\u2013120. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_7"},{"key":"7_CR4","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, vol.\u00a01, pp. 4171\u20134186 (2019)"},{"issue":"10","key":"7_CR5","doi-asserted-by":"publisher","first-page":"2993","DOI":"10.1016\/j.patcog.2015.04.005","volume":"48","author":"S Ding","year":"2015","unstructured":"Ding, S., Lin, L., Wang, G., Chao, H.: Deep feature learning with relative distance comparison for person re-identification. Pattern Recogn. 48(10), 2993\u20133003 (2015)","journal-title":"Pattern Recogn."},{"key":"7_CR6","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale. In: Proceedings of the International Conference on Learning Representations (ICLR) (2021)"},{"key":"7_CR7","unstructured":"Frome, A., et al.: Devise: a deep visual-semantic embedding model. In: Burges, C., Bottou, L., Welling, M., Ghahramani, Z., Weinberger, K. (eds.) Advances in Neural Information Processing Systems, vol.\u00a026. Curran Associates, Inc. (2013)"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Guerrero, R., Pham, H.X., Pavlovic, V.: Cross-modal retrieval and synthesis (x-mrs): closing the modality gap in shared subspace learning. In: Proceedings of the ACM International Conference on Multimedia (ACM MM), pp. 3192\u20133201. ACM (2021)","DOI":"10.1145\/3474085.3475465"},{"key":"7_CR9","doi-asserted-by":"publisher","unstructured":"Hadsell, R., Chopra, S., LeCun, Y.: Dimensionality reduction by learning an invariant mapping. In: 2006 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2006), vol.\u00a02, pp. 1735\u20131742 (2006). https:\/\/doi.org\/10.1109\/CVPR.2006.100","DOI":"10.1109\/CVPR.2006.100"},{"key":"7_CR10","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"7_CR11","unstructured":"Hermans, A., Beyer, L., Leibe, B.: In defense of the triplet loss for person re-identification. arXiv preprint arXiv:1703.07737 (2017)"},{"key":"7_CR12","doi-asserted-by":"crossref","unstructured":"Huang, X., Liu, J., Zhang, Z., Xie, Y.: Improving cross-modal recipe retrieval with component-aware prompted clip embedding. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 529\u2013537 (2023)","DOI":"10.1145\/3581783.3612193"},{"key":"7_CR13","doi-asserted-by":"publisher","unstructured":"Ji, Z., Wang, H., Han, J., Pang, Y.: Saliency-guided attention network for image-sentence matching. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 5753\u20135762 (2019). https:\/\/doi.org\/10.1109\/ICCV.2019.00585","DOI":"10.1109\/ICCV.2019.00585"},{"key":"7_CR14","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference on Machine Learning, pp. 4904\u20134916. PMLR (2021)"},{"key":"7_CR15","unstructured":"Kiros, R., et al.: Skip-thought vectors. In: Proceedings of the 28th International Conference on Neural Information Processing Systems, vol. 2, pp. 3294\u20133302. MIT Press, Cambridge, MA, USA (2015)"},{"key":"7_CR16","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"212","DOI":"10.1007\/978-3-030-01225-0_13","volume-title":"Computer Vision \u2013 ECCV 2018","author":"K-H Lee","year":"2018","unstructured":"Lee, K.-H., Chen, X., Hua, G., Hu, H., He, X.: Stacked cross attention for image-text matching. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11208, pp. 212\u2013228. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01225-0_13"},{"key":"7_CR17","doi-asserted-by":"crossref","unstructured":"Li, L., Li, M., Zan, Z., Xie, Q., Liu, J.: Multi-subspace implicit alignment for cross-modal retrieval on cooking recipes and food images. In: Proceedings of the ACM International Conference on Multimedia (ICM), pp. 3211\u20133215. ACM (2021)","DOI":"10.1145\/3459637.3482149"},{"key":"7_CR18","unstructured":"Mikolov, T., Chen, K., Corrado, G., Dean, J.: Efficient estimation of word representations in vector space. In: ICLR (2013)"},{"issue":"4","key":"7_CR19","doi-asserted-by":"publisher","first-page":"950","DOI":"10.1109\/TMM.2017.2759499","volume":"20","author":"W Min","year":"2018","unstructured":"Min, W., Bao, B.K., Mei, S., Zhu, Y., Rui, Y., Jiang, S.: You are what you eat: exploring rich recipe information for cross-region food analysis. IEEE Trans. Multimedia 20(4), 950\u2013964 (2018). https:\/\/doi.org\/10.1109\/TMM.2017.2759499","journal-title":"IEEE Trans. Multimedia"},{"key":"7_CR20","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"7_CR21","doi-asserted-by":"crossref","unstructured":"Salvador, A., Gundogdu, E., Bazzani, L., Donoser, M.: Revamping cross-modal recipe retrieval with hierarchical transformers and self-supervised learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15475\u201315484 (2021)","DOI":"10.1109\/CVPR46437.2021.01522"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Salvador, A., et al.: Learning cross-modal embeddings for cooking recipes and food images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (July 2017)","DOI":"10.1109\/CVPR.2017.327"},{"key":"7_CR23","doi-asserted-by":"crossref","unstructured":"Schroff, F., Kalenichenko, D., Philbin, J.: Facenet: a unified embedding for face recognition and clustering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 815\u2013823 (2015)","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"7_CR24","unstructured":"Shukor, M., Couairon, G., Cord, M.: Efficient vision-language pretraining with visual concepts and hierarchical alignment. In: 33rd British Machine Vision Conference (BMVC) (2022)"},{"key":"7_CR25","doi-asserted-by":"crossref","unstructured":"Shukor, M., Couairon, G., Grechka, A., Cord, M.: Transformer decoders with multimodal regularization for cross-modal food retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4567\u20134578 (2022)","DOI":"10.1109\/CVPRW56347.2022.00503"},{"key":"7_CR26","doi-asserted-by":"crossref","unstructured":"Song, F., Zhu, B., Hao, Y., Wang, S.: Enhancing recipe retrieval with foundation models: A data augmentation perspective (2024). https:\/\/arxiv.org\/abs\/2312.04763","DOI":"10.1007\/978-3-031-72983-6_7"},{"key":"7_CR27","unstructured":"Vaswani, A., et al.: Attention is all you need. Adv. Neural Inform. Process. Syst. (2017)"},{"key":"7_CR28","unstructured":"Voutharoja, B.P., Wang, P., Wang, L., Guan, V.: Malm: Mask augmentation based local matching for food-recipe retrieval (2023). https:\/\/arxiv.org\/abs\/2305.11327"},{"key":"7_CR29","doi-asserted-by":"crossref","unstructured":"Wahed, M., Zhou, X., Yu, T., Lourentzou, I.: Fine-grained alignment for cross-modal recipe retrieval. In: 2024 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 5572\u20135581. IEEE (2024)","DOI":"10.1109\/WACV57701.2024.00549"},{"key":"7_CR30","doi-asserted-by":"crossref","unstructured":"Wang, H., Sahoo, D., Liu, C., Lim, E.p., Hoi, S.C.: Learning cross-modal embeddings with adversarial networks for cooking recipes and food images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11572\u201311581 (2019)","DOI":"10.1109\/CVPR.2019.01184"},{"key":"7_CR31","doi-asserted-by":"publisher","unstructured":"Wang, H., et al.: Cross-modal food retrieval: Learning a joint embedding of food images and recipes with semantic consistency and attention mechanism. IEEE Trans. Multimedia 24, 2515\u20132525 (2022). https:\/\/doi.org\/10.1109\/TMM.2021.3083109","DOI":"10.1109\/TMM.2021.3083109"},{"key":"7_CR32","doi-asserted-by":"crossref","unstructured":"Xie, Z., Liu, L., Wu, Y., Li, L., Zhong, L.: Learning tf-idf enhanced joint embedding for recipe-image cross-modal retrieval service. IEEE Trans. Serv. Comput. (2021)","DOI":"10.1109\/TSC.2021.3098834"},{"issue":"4","key":"7_CR33","first-page":"1","volume":"40","author":"Z Xie","year":"2021","unstructured":"Xie, Z., Liu, L., Wu, Y., Zhong, L., Li, L.: Learning text-image joint embedding for efficient cross-modal retrieval with deep feature engineering. ACM Trans. Inform. Syst. (TOIS) 40(4), 1\u201327 (2021)","journal-title":"ACM Trans. Inform. Syst. (TOIS)"}],"container-title":["Lecture Notes in Computer Science","PRICAI 2024: Trends in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-0128-8_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,16]],"date-time":"2024-11-16T19:23:18Z","timestamp":1731784998000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-0128-8_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,12]]},"ISBN":["9789819601271","9789819601288"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-0128-8_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,11,12]]},"assertion":[{"value":"12 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRICAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pacific Rim International Conference on Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kyoto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Japan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 November 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 November 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"pricai2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.pricai.org\/2024\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}